A numbers float file adapts decimal literals to a fixed operand or slot, and refuses to promote a computed int to a float implicitly: there it is almost always float bits. Explicit float(x) is always allowed. render3d's numbers are float, converted by tools/migrate/floatbits.py - a whole-program inference of which ints carried IEEE bits (union-find over flows, calls, returns, buffers, nested buffers and lexical scopes) and a rewriter to operators, Math.* and float literals, with float_bits / float_from_bits left only where bits really cross (runtime scratch buffers, mixed buffers). Seed regenerated. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1914 lines
98 KiB
Text
1914 lines
98 KiB
Text
# gpu_vk_draw.ludic — the Vulkan backend's drawing: programs as manifest variants, pipelines
|
|
# built from what OpenGL decides at the draw, uniform blocks and descriptor sets per draw, and the
|
|
# frame's passes with dynamic rendering, through to the present and the screenshot.
|
|
#
|
|
# It reads render3d's own records - the SPIR-V manifest (gpu_manifest.ludic), a Mesh's recorded
|
|
# vertex layout, gpu.ludic's texture and framebuffer records - so it is compiled only inside
|
|
# render3d, after gpu_vk.ludic and gpu_vk_res.ludic.
|
|
|
|
# ---- programs -----------------------------------------------------------------------------
|
|
# A program handle is a manifest variant on Vulkan: its two SPIR-V modules, a descriptor set
|
|
# layout (binding 0 the vertex stage's uniform block, 1 the fragment stage's, the samplers at
|
|
# the manifest's bindings from 2) and the pipeline layout over it. Programs are never compiled
|
|
# here; one whose variant is not in the manifest cannot draw, and says so once.
|
|
var gvk_prog_var: []GpuVariant = null
|
|
var gvk_prog_vs: []long = null
|
|
var gvk_prog_fs: []long = null
|
|
var gvk_prog_dsl: []long = null
|
|
var gvk_prog_layout: []long = null
|
|
var gvk_spv_dir: string = ""
|
|
|
|
function gvk_read_spv(path: string) -> bytes {
|
|
let f = file_open(path, "rb")
|
|
if f == null { return null }
|
|
file_seek(f, 0, 2)
|
|
let n = file_tell(f)
|
|
file_seek(f, 0, 0)
|
|
let b = bytes(n + 4)
|
|
file_read(f, b, n)
|
|
file_close(f)
|
|
gvk_spv_len = n
|
|
return b
|
|
}
|
|
var gvk_spv_len: int = 0
|
|
|
|
function gvk_module(path: string) -> long {
|
|
let zero: long = 0
|
|
let spv = gvk_read_spv(path)
|
|
if spv == null { print(`r3d: vulkan: no SPIR-V at {path}`); return zero }
|
|
let smci = bytes(VkShaderModuleCreateInfo_sizeof)
|
|
Vk.zero(smci, VkShaderModuleCreateInfo_sizeof)
|
|
Vk.put_i32(smci, VkShaderModuleCreateInfo_sType, VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO)
|
|
let code_size: long = gvk_spv_len
|
|
Vk.put_i64(smci, VkShaderModuleCreateInfo_codeSize, code_size)
|
|
Vk.put_ptr(smci, VkShaderModuleCreateInfo_pCode, spv)
|
|
let out = bytes(8)
|
|
let r = Vk.create_shader_module(gvk_dev, smci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateShaderModule {path}`, r); return zero }
|
|
return gvk_handle(out)
|
|
}
|
|
|
|
# The Vulkan side of a program handle the renderer already made (gpu_program): the handle's
|
|
# manifest key finds the variant. Returns false when there is no such variant.
|
|
# a program whose first stage is a mesh shader (its first file is *.mesh): its pipeline has no vertex
|
|
# input, and its first stage's bindings are the mesh stage's
|
|
var gvk_prog_mesh: []int = null
|
|
function gvk_stage_first(p: int) -> int {
|
|
if gvk_prog_mesh != null and p < len(gvk_prog_mesh) and gvk_prog_mesh[p] == 1 { return VK_SHADER_STAGE_MESH_BIT_EXT }
|
|
return VK_SHADER_STAGE_VERTEX_BIT
|
|
}
|
|
function gvk_program(p: int, key: string, spv_dir: string) -> bool {
|
|
let zero: long = 0
|
|
if gvk_prog_var == null {
|
|
gvk_prog_var = new []GpuVariant; gvk_prog_vs = new []long; gvk_prog_fs = new []long
|
|
gvk_prog_dsl = new []long; gvk_prog_layout = new []long; gvk_prog_mesh = new []int
|
|
}
|
|
while len(gvk_prog_var) <= p {
|
|
push(gvk_prog_var, null); push(gvk_prog_vs, zero); push(gvk_prog_fs, zero); push(gvk_prog_dsl, zero); push(gvk_prog_layout, zero); push(gvk_prog_mesh, 0)
|
|
}
|
|
let parts = Text.split(key, "|")
|
|
var defs = ""
|
|
if len(parts) > 2 { defs = parts[2] }
|
|
let v = gpu_variant_find_key(parts[0], parts[1], defs)
|
|
if v == null { print(`r3d: vulkan: no SPIR-V variant for {key}`); return false }
|
|
let vs = gvk_module(`{spv_dir}/{v.id}.vert.spv`)
|
|
if Text.ends_with(v.vs, ".mesh") { gvk_prog_mesh[p] = 1 } else { gvk_prog_mesh[p] = 0 }
|
|
let fs = gvk_module(`{spv_dir}/{v.id}.frag.spv`)
|
|
if vs == 0 or fs == 0 { return false }
|
|
|
|
let nt = len(v.t_name)
|
|
var nb = nt
|
|
if v.vblock >= 0 { nb += 1 }
|
|
if v.fblock >= 0 { nb += 1 }
|
|
let bw = VkDescriptorSetLayoutBinding_sizeof
|
|
let binds = bytes(bw * (nb + 1))
|
|
Vk.zero(binds, bw * (nb + 1))
|
|
var k = 0
|
|
if v.vblock >= 0 {
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, 0)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p))
|
|
k += 1
|
|
}
|
|
if v.fblock >= 0 {
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_FRAGMENT_BIT)
|
|
k += 1
|
|
}
|
|
for t in 0 .. nt {
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, v.t_bind[t])
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p) | VK_SHADER_STAGE_FRAGMENT_BIT)
|
|
k += 1
|
|
}
|
|
let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.zero(dslci, VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_bindingCount, nb)
|
|
Vk.put_ptr(dslci, VkDescriptorSetLayoutCreateInfo_pBindings, binds)
|
|
let dsl = bytes(8)
|
|
var r = Vk.create_descriptor_set_layout(gvk_dev, dslci, null, dsl)
|
|
if r != VK_SUCCESS { return gvk_fail(`vkCreateDescriptorSetLayout for {key}`, r) }
|
|
let plci = bytes(VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.zero(plci, VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_setLayoutCount, 1)
|
|
Vk.put_ptr(plci, VkPipelineLayoutCreateInfo_pSetLayouts, dsl)
|
|
let out = bytes(8)
|
|
r = Vk.create_pipeline_layout(gvk_dev, plci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail(`vkCreatePipelineLayout for {key}`, r) }
|
|
gvk_prog_var[p] = v; gvk_prog_vs[p] = vs; gvk_prog_fs[p] = fs
|
|
gvk_prog_dsl[p] = Vk.get_i64(dsl, 0); gvk_prog_layout[p] = gvk_handle(out)
|
|
return true
|
|
}
|
|
|
|
# ---- compute ------------------------------------------------------------------------------
|
|
# A compute program is <name>.comp.spv beside the manifest, with bindings fixed by convention:
|
|
# 0 the parameter block (a uniform buffer, copied into the frame's ring at the dispatch) and
|
|
# 1 .. n storage buffers. Handles start at 1.
|
|
var gvk_cp_pipe: []long = null
|
|
var gvk_cp_layout: []long = null
|
|
var gvk_cp_dsl: []long = null
|
|
var gvk_cp_nbuf: []int = null
|
|
|
|
function gvk_compute_new(name: string, n_bufs: int) -> int {
|
|
let zero: long = 0
|
|
if gvk_cp_pipe == null {
|
|
gvk_cp_pipe = new []long; gvk_cp_layout = new []long; gvk_cp_dsl = new []long; gvk_cp_nbuf = new []int
|
|
push(gvk_cp_pipe, zero); push(gvk_cp_layout, zero); push(gvk_cp_dsl, zero); push(gvk_cp_nbuf, 0)
|
|
}
|
|
let module = gvk_module(`{gvk_spv_dir}/{name}.comp.spv`)
|
|
if module == 0 { return 0 }
|
|
let nb = n_bufs + 1
|
|
let bw = VkDescriptorSetLayoutBinding_sizeof
|
|
let binds = bytes(bw * nb)
|
|
Vk.zero(binds, bw * nb)
|
|
for k in 0 .. nb {
|
|
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
|
|
if k == 0 { kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, k)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, kind)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_COMPUTE_BIT)
|
|
}
|
|
let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.zero(dslci, VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_bindingCount, nb)
|
|
Vk.put_ptr(dslci, VkDescriptorSetLayoutCreateInfo_pBindings, binds)
|
|
let dsl = bytes(8)
|
|
var r = Vk.create_descriptor_set_layout(gvk_dev, dslci, null, dsl)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateDescriptorSetLayout for compute {name}`, r); return 0 }
|
|
let plci = bytes(VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.zero(plci, VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_setLayoutCount, 1)
|
|
Vk.put_ptr(plci, VkPipelineLayoutCreateInfo_pSetLayouts, dsl)
|
|
let out = bytes(8)
|
|
r = Vk.create_pipeline_layout(gvk_dev, plci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreatePipelineLayout for compute {name}`, r); return 0 }
|
|
let layout = gvk_handle(out)
|
|
let cpci = bytes(VkComputePipelineCreateInfo_sizeof)
|
|
Vk.zero(cpci, VkComputePipelineCreateInfo_sizeof)
|
|
Vk.put_i32(cpci, VkComputePipelineCreateInfo_sType, VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO)
|
|
let so = VkComputePipelineCreateInfo_stage
|
|
Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
|
|
Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_COMPUTE_BIT)
|
|
Vk.put_i64(cpci, so + VkPipelineShaderStageCreateInfo_module, module)
|
|
Vk.put_ptr(cpci, so + VkPipelineShaderStageCreateInfo_pName, "main")
|
|
Vk.put_i64(cpci, VkComputePipelineCreateInfo_layout, layout)
|
|
r = Vk.create_compute_pipelines(gvk_dev, zero, 1, cpci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateComputePipelines {name}`, r); return 0 }
|
|
push(gvk_cp_pipe, gvk_handle(out)); push(gvk_cp_layout, layout); push(gvk_cp_dsl, Vk.get_i64(dsl, 0)); push(gvk_cp_nbuf, n_bufs)
|
|
return len(gvk_cp_pipe) - 1
|
|
}
|
|
|
|
# Record a dispatch of compute program c into the frame, between passes: params (n_params
|
|
# bytes, std140) as binding 0 and bufs[0 .. n) as bindings 1 .. n.
|
|
function gvk_dispatch(c: int, params: pointer, n_params: int, bufs: words, gx: int, gy: int, gz: int) -> void {
|
|
let zero: long = 0
|
|
if gvk_cp_pipe == null or c <= 0 or c >= len(gvk_cp_pipe) { return }
|
|
gvk_pass_end()
|
|
let cb = gvk_frame_cb()
|
|
let nb = gvk_cp_nbuf[c]
|
|
let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof)
|
|
Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO)
|
|
Vk.put_i64(dsai, VkDescriptorSetAllocateInfo_descriptorPool, gvk_dpool)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_descriptorSetCount, 1)
|
|
let layouts = bytes(8)
|
|
Vk.put_i64(layouts, 0, gvk_cp_dsl[c])
|
|
Vk.put_ptr(dsai, VkDescriptorSetAllocateInfo_pSetLayouts, layouts)
|
|
let sets = bytes(8)
|
|
let r = Vk.allocate_descriptor_sets(gvk_dev, dsai, sets)
|
|
if r != VK_SUCCESS { gvk_fail("vkAllocateDescriptorSets (compute)", r); return }
|
|
let set = Vk.get_i64(sets, 0)
|
|
let at = gvk_ring_put(params, n_params)
|
|
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return }
|
|
let ww = VkWriteDescriptorSet_sizeof
|
|
let bw = VkDescriptorBufferInfo_sizeof
|
|
let writes = bytes(ww * (nb + 1))
|
|
Vk.zero(writes, ww * (nb + 1))
|
|
let infos = bytes(bw * (nb + 1))
|
|
Vk.zero(infos, bw * (nb + 1))
|
|
for k in 0 .. nb + 1 {
|
|
var buf = gvk_ring_buf
|
|
var off: long = 0
|
|
var range: long = 0
|
|
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
|
|
if k == 0 { off = at; range = n_params; kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
|
|
else { buf = bufs[k - 1]; range = gvk_buf_size[buf]; gvk_buf_used[buf] = gvk_frame_no }
|
|
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_buffer, gvk_buf[buf])
|
|
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_offset, off)
|
|
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_range, range)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
|
|
Vk.put_i64(writes, k * ww + VkWriteDescriptorSet_dstSet, set)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_dstBinding, k)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorCount, 1)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorType, kind)
|
|
Vk.put_ptr(writes, k * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(infos, k * bw))
|
|
}
|
|
Vk.update_descriptor_sets(gvk_dev, nb + 1, writes, 0, null)
|
|
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_pipe[c])
|
|
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_layout[c], 0, 1, sets, 0, null)
|
|
Vk.cmd_dispatch(cb, gx, gy, gz)
|
|
}
|
|
|
|
# R3D_VK_PROBE=1: one dispatch over a GPU-owned buffer, read back - the compute path end to end
|
|
function gvk_compute_probe() -> void {
|
|
let c = gvk_compute_new("probe", 1)
|
|
if c == 0 { print("r3d: vulkan compute probe: FAILED (no program)"); return }
|
|
let n = 1000
|
|
let b = gvk_buf_new()
|
|
let data = bytes(n * 4)
|
|
for i in 0 .. n { Vk.put_i32(data, i * 4, float_bits(float(i))) }
|
|
gvk_buf_upload(b, n * 4, data)
|
|
gvk_buf_gpu_owned(b)
|
|
let pr = bytes(8)
|
|
Vk.put_i32(pr, 0, n)
|
|
Vk.put_i32(pr, 4, float_bits(3.0))
|
|
let bufs = words(1)
|
|
bufs[0] = b
|
|
gvk_dispatch(c, pr, 8, bufs, (n + 63) / 64, 1, 1)
|
|
gvk_flush()
|
|
var bad = 0
|
|
let mp = gvk_buf_map[b]
|
|
for i in 0 .. n { if float_from_bits(Vk.get_i32(mp, i * 4)) != float(i * 3) { bad += 1 } }
|
|
if bad == 0 { print(`r3d: vulkan compute probe OK ({n} values)`) } else { print(`r3d: vulkan compute probe: FAILED ({bad} of {n} wrong)`) }
|
|
}
|
|
|
|
# ---- pipelines ----------------------------------------------------------------------------
|
|
# A pipeline is everything OpenGL decides at the draw: the program, the vertex layout the mesh
|
|
# recorded, the render state the renderer set, and the formats and sample count of the pass it
|
|
# draws into. Each distinct combination is built once, the first time it is drawn.
|
|
property GvkState {
|
|
depth_test: int = 0,
|
|
depth_write: int = 1,
|
|
depth_func: int = 0x0201, # GL_LESS
|
|
blend: int = 0,
|
|
blend_src: int = 1, # GL_ONE
|
|
blend_dst: int = 0, # GL_ZERO
|
|
cull: int = 0,
|
|
cull_face: int = 0x0405, # GL_BACK
|
|
color_write: int = 1,
|
|
a2c: int = 0,
|
|
bias: int = 0,
|
|
bias_factor: int = 0, # float bits
|
|
bias_units: int = 0, # float bits
|
|
wireframe: int = 0
|
|
}
|
|
var gvk_pipe_keys: []string = null
|
|
var gvk_pipe: []long = null
|
|
|
|
# does the mesh leave shader input `loc` unfed (no attribute recorded there)?
|
|
function gvk_input_unfed(m: Mesh, loc: int) -> bool {
|
|
if m == null or m.attrs == null or loc < 0 or loc >= m.n_attrs { return true }
|
|
return m.attrs[loc * GPU_ATTR_W + 1] == 0
|
|
}
|
|
# the float format a zero-fed shader input reads, from its manifest type
|
|
function gvk_input_format(t: string) -> int {
|
|
if t == "float" { return VK_FORMAT_R32_SFLOAT }
|
|
if t == "vec2" { return VK_FORMAT_R32G32_SFLOAT }
|
|
if t == "vec3" { return VK_FORMAT_R32G32B32_SFLOAT }
|
|
return VK_FORMAT_R32G32B32A32_SFLOAT
|
|
}
|
|
# 64 KB of zeros, the buffer every unfed input reads (one vec4 per instance, up to 4096)
|
|
var gvk_zero_vbuf: int = 0
|
|
var gvk_skip_said: words = null # per program: its skipped draws have been reported
|
|
function gvk_zero_vbuf_get() -> int {
|
|
if gvk_zero_vbuf == 0 {
|
|
gvk_zero_vbuf = gvk_buf_new()
|
|
let z = bytes(65536)
|
|
Vk.zero(z, 65536)
|
|
gvk_buf_upload(gvk_zero_vbuf, 65536, z)
|
|
}
|
|
return gvk_zero_vbuf
|
|
}
|
|
|
|
function gvk_attr_format(comps: int, type: int, normalized: int) -> int {
|
|
if type == GPU_U8 {
|
|
if normalized == 1 { if comps == 4 { return VK_FORMAT_R8G8B8A8_UNORM }; if comps == 3 { return VK_FORMAT_R8G8B8_UNORM }; if comps == 2 { return VK_FORMAT_R8G8_UNORM }; return VK_FORMAT_R8_UNORM }
|
|
# not normalised, read by a float input (a_joints is a vec4): OpenGL converts the integer to a
|
|
# float, and Vulkan's form of that is USCALED - a UINT format against a float input is invalid
|
|
if comps == 4 { return VK_FORMAT_R8G8B8A8_USCALED }; if comps == 2 { return VK_FORMAT_R8G8_USCALED }; return VK_FORMAT_R8_USCALED
|
|
}
|
|
if type == GPU_U16 {
|
|
if normalized == 1 { if comps == 4 { return VK_FORMAT_R16G16B16A16_UNORM }; if comps == 2 { return VK_FORMAT_R16G16_UNORM }; return VK_FORMAT_R16_UNORM }
|
|
if comps == 4 { return VK_FORMAT_R16G16B16A16_USCALED }; if comps == 2 { return VK_FORMAT_R16G16_USCALED }; return VK_FORMAT_R16_USCALED
|
|
}
|
|
if comps == 4 { return VK_FORMAT_R32G32B32A32_SFLOAT }
|
|
if comps == 3 { return VK_FORMAT_R32G32B32_SFLOAT }
|
|
if comps == 2 { return VK_FORMAT_R32G32_SFLOAT }
|
|
return VK_FORMAT_R32_SFLOAT
|
|
}
|
|
function gvk_blend_factor(f: int) -> int {
|
|
if f == GL_ONE { return VK_BLEND_FACTOR_ONE }
|
|
if f == GL_SRC_ALPHA { return VK_BLEND_FACTOR_SRC_ALPHA }
|
|
if f == GL_ONE_MINUS_SRC_ALPHA { return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA }
|
|
if f == GL_DST_ALPHA { return VK_BLEND_FACTOR_DST_ALPHA }
|
|
if f == GL_ONE_MINUS_DST_ALPHA { return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA }
|
|
if f == GL_SRC_COLOR { return VK_BLEND_FACTOR_SRC_COLOR }
|
|
if f == GL_ONE_MINUS_SRC_COLOR { return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR }
|
|
return VK_BLEND_FACTOR_ZERO
|
|
}
|
|
function gvk_depth_op(f: int) -> int {
|
|
if f == GL_LEQUAL { return VK_COMPARE_OP_LESS_OR_EQUAL }
|
|
if f == GL_EQUAL { return VK_COMPARE_OP_EQUAL }
|
|
if f == GL_ALWAYS { return VK_COMPARE_OP_ALWAYS }
|
|
if f == GL_GREATER { return VK_COMPARE_OP_GREATER }
|
|
if f == GL_GEQUAL { return VK_COMPARE_OP_GREATER_OR_EQUAL }
|
|
if f == GL_NEVER { return VK_COMPARE_OP_NEVER }
|
|
if f == GL_NOTEQUAL { return VK_COMPARE_OP_NOT_EQUAL }
|
|
return VK_COMPARE_OP_LESS
|
|
}
|
|
# the vertex layout a mesh recorded (gpu.ludic's attrs), as part of a pipeline key
|
|
# Buffers are named by the order they are first read in, not by handle: a scatter mesh re-pointed
|
|
# at another instance buffer keeps its layout, and so its pipeline.
|
|
function gvk_layout_key(m: Mesh) -> string {
|
|
if m == null or m.attrs == null { return "none" }
|
|
var k = ""
|
|
let seen = words(GPU_MAX_ATTRS)
|
|
var ns = 0
|
|
for i in 0 .. m.n_attrs {
|
|
let o = i * GPU_ATTR_W
|
|
if m.attrs[o + 1] == 0 { continue }
|
|
var bi = -1
|
|
for q in 0 .. ns { if seen[q] == m.attrs[o] and bi < 0 { bi = q } }
|
|
if bi < 0 { bi = ns; seen[ns] = m.attrs[o]; ns += 1 }
|
|
k = k + `{i}:{bi}:{m.attrs[o + 1]}:{m.attrs[o + 2]}:{m.attrs[o + 3]}:{m.attrs[o + 4]}:{m.attrs[o + 5]}:{m.attrs[o + 6]};`
|
|
}
|
|
return k
|
|
}
|
|
|
|
# The pipeline for program p drawing mesh m (null for a draw without vertex input, like the
|
|
# full-screen triangle) with state st into a pass of n_color colour attachments of format
|
|
# color_fmt, a depth attachment of depth_fmt (VK_FORMAT_UNDEFINED for none) at `samples`.
|
|
function gvk_pipeline(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: int, depth_fmt: int, samples: int) -> long {
|
|
let zero: long = 0
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return zero }
|
|
let key = `{p}|{gvk_layout_key(m)}|{st.depth_test},{st.depth_write},{st.depth_func},{st.blend},{st.blend_src},{st.blend_dst},{st.cull},{st.cull_face},{st.color_write},{st.a2c},{st.bias},{st.bias_factor},{st.bias_units},{st.wireframe}|{n_color},{color_fmt},{depth_fmt},{samples}`
|
|
if gvk_pipe_keys == null { gvk_pipe_keys = new []string; gvk_pipe = new []long }
|
|
for i in 0 .. len(gvk_pipe_keys) { if gvk_pipe_keys[i] == key { return gvk_pipe[i] } }
|
|
|
|
let ss = VkPipelineShaderStageCreateInfo_sizeof
|
|
let stages = bytes(ss * 2)
|
|
Vk.zero(stages, ss * 2)
|
|
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
|
|
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_stage, gvk_stage_first(p))
|
|
Vk.put_i64(stages, VkPipelineShaderStageCreateInfo_module, gvk_prog_vs[p])
|
|
Vk.put_ptr(stages, VkPipelineShaderStageCreateInfo_pName, "main")
|
|
Vk.put_i32(stages, ss + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
|
|
Vk.put_i32(stages, ss + VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_FRAGMENT_BIT)
|
|
Vk.put_i64(stages, ss + VkPipelineShaderStageCreateInfo_module, gvk_prog_fs[p])
|
|
Vk.put_ptr(stages, ss + VkPipelineShaderStageCreateInfo_pName, "main")
|
|
|
|
# vertex input: one binding per distinct buffer the mesh's attributes read, in the order met;
|
|
# an attribute the shader does not read is left out
|
|
let aw = VkVertexInputAttributeDescription_sizeof
|
|
let bdw = VkVertexInputBindingDescription_sizeof
|
|
let attrs = bytes(aw * (GPU_MAX_ATTRS + 1))
|
|
let bnds = bytes(bdw * (GPU_MAX_VBUFS + 1))
|
|
Vk.zero(attrs, aw * (GPU_MAX_ATTRS + 1))
|
|
Vk.zero(bnds, bdw * (GPU_MAX_VBUFS + 1))
|
|
var na = 0
|
|
var nbd = 0
|
|
if m != null and m.attrs != null {
|
|
let bufs = words(GPU_MAX_VBUFS)
|
|
for i in 0 .. m.n_attrs {
|
|
let o = i * GPU_ATTR_W
|
|
if m.attrs[o + 1] == 0 { continue }
|
|
var wanted = false
|
|
for q in 0 .. len(v.i_loc) { if v.i_loc[q] == i { wanted = true } }
|
|
if not wanted { continue }
|
|
var bi = -1
|
|
for q in 0 .. nbd { if bufs[q] == m.attrs[o] and bi < 0 { bi = q } }
|
|
if bi < 0 and nbd < GPU_MAX_VBUFS {
|
|
bi = nbd
|
|
bufs[nbd] = m.attrs[o]
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_binding, nbd)
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_stride, m.attrs[o + 3])
|
|
if m.attrs[o + 6] == 1 { Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_inputRate, VK_VERTEX_INPUT_RATE_INSTANCE) }
|
|
nbd += 1
|
|
}
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_location, i)
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_binding, bi)
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_format, gvk_attr_format(m.attrs[o + 1], m.attrs[o + 2], m.attrs[o + 5]))
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_offset, m.attrs[o + 4])
|
|
na += 1
|
|
}
|
|
}
|
|
# A shader input the mesh does not provide (a_joints on a mesh without a skin) reads zeros from a
|
|
# shared buffer, as OpenGL's disabled attribute reads its default. Vulkan requires every input the
|
|
# vertex stage declares to be fed (VUID-VkGraphicsPipelineCreateInfo-Input-07904).
|
|
var need_zero = false
|
|
for q in 0 .. len(v.i_loc) {
|
|
if not gvk_input_unfed(m, v.i_loc[q]) or na >= GPU_MAX_ATTRS { continue }
|
|
if not need_zero {
|
|
need_zero = true
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_binding, nbd)
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_stride, 16)
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_inputRate, VK_VERTEX_INPUT_RATE_INSTANCE)
|
|
}
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_location, v.i_loc[q])
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_binding, nbd)
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_format, gvk_input_format(v.i_type[q]))
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_offset, 0)
|
|
na += 1
|
|
}
|
|
if need_zero { nbd += 1 }
|
|
let vin = bytes(VkPipelineVertexInputStateCreateInfo_sizeof)
|
|
Vk.zero(vin, VkPipelineVertexInputStateCreateInfo_sizeof)
|
|
Vk.put_i32(vin, VkPipelineVertexInputStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO)
|
|
Vk.put_i32(vin, VkPipelineVertexInputStateCreateInfo_vertexBindingDescriptionCount, nbd)
|
|
Vk.put_ptr(vin, VkPipelineVertexInputStateCreateInfo_pVertexBindingDescriptions, bnds)
|
|
Vk.put_i32(vin, VkPipelineVertexInputStateCreateInfo_vertexAttributeDescriptionCount, na)
|
|
Vk.put_ptr(vin, VkPipelineVertexInputStateCreateInfo_pVertexAttributeDescriptions, attrs)
|
|
|
|
let ias = bytes(VkPipelineInputAssemblyStateCreateInfo_sizeof)
|
|
Vk.zero(ias, VkPipelineInputAssemblyStateCreateInfo_sizeof)
|
|
Vk.put_i32(ias, VkPipelineInputAssemblyStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO)
|
|
Vk.put_i32(ias, VkPipelineInputAssemblyStateCreateInfo_topology, VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST)
|
|
let vps = bytes(VkPipelineViewportStateCreateInfo_sizeof)
|
|
Vk.zero(vps, VkPipelineViewportStateCreateInfo_sizeof)
|
|
Vk.put_i32(vps, VkPipelineViewportStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO)
|
|
Vk.put_i32(vps, VkPipelineViewportStateCreateInfo_viewportCount, 1)
|
|
Vk.put_i32(vps, VkPipelineViewportStateCreateInfo_scissorCount, 1)
|
|
|
|
let rs = bytes(VkPipelineRasterizationStateCreateInfo_sizeof)
|
|
Vk.zero(rs, VkPipelineRasterizationStateCreateInfo_sizeof)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO)
|
|
if st.wireframe == 1 { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_polygonMode, VK_POLYGON_MODE_LINE) } else { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_polygonMode, VK_POLYGON_MODE_FILL) }
|
|
if st.cull == 1 {
|
|
if st.cull_face == GL_FRONT { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_cullMode, VK_CULL_MODE_FRONT_BIT) } else { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_cullMode, VK_CULL_MODE_BACK_BIT) }
|
|
}
|
|
# no y flip between the APIs' clip spaces and their framebuffers' rows, so a triangle that is
|
|
# counter-clockwise on OpenGL's bottom-up window is clockwise in Vulkan's top-down one
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_frontFace, VK_FRONT_FACE_CLOCKWISE)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_lineWidth, 0x3F800000)
|
|
if st.bias == 1 {
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_depthBiasEnable, 1)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_depthBiasSlopeFactor, st.bias_factor)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_depthBiasConstantFactor, st.bias_units)
|
|
}
|
|
let ms = bytes(VkPipelineMultisampleStateCreateInfo_sizeof)
|
|
Vk.zero(ms, VkPipelineMultisampleStateCreateInfo_sizeof)
|
|
Vk.put_i32(ms, VkPipelineMultisampleStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO)
|
|
Vk.put_i32(ms, VkPipelineMultisampleStateCreateInfo_rasterizationSamples, samples)
|
|
# OpenGL ignores alpha to coverage without a multisampled target; Vulkan with one sample would
|
|
# instead drop every fragment under half alpha, which erased the meadow's flowers
|
|
if samples > 1 { Vk.put_i32(ms, VkPipelineMultisampleStateCreateInfo_alphaToCoverageEnable, st.a2c) }
|
|
let ds = bytes(VkPipelineDepthStencilStateCreateInfo_sizeof)
|
|
Vk.zero(ds, VkPipelineDepthStencilStateCreateInfo_sizeof)
|
|
Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO)
|
|
if depth_fmt != VK_FORMAT_UNDEFINED {
|
|
Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_depthTestEnable, st.depth_test)
|
|
# OpenGL writes no depth with the test off, whatever the mask says
|
|
if st.depth_test == 1 { Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_depthWriteEnable, st.depth_write) }
|
|
Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_depthCompareOp, gvk_depth_op(st.depth_func))
|
|
}
|
|
let cbw = VkPipelineColorBlendAttachmentState_sizeof
|
|
let cba = bytes(cbw * (n_color + 1))
|
|
Vk.zero(cba, cbw * (n_color + 1))
|
|
for c in 0 .. n_color {
|
|
if st.color_write == 1 { Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_colorWriteMask, 15) }
|
|
if st.blend == 1 {
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_blendEnable, 1)
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_srcColorBlendFactor, gvk_blend_factor(st.blend_src))
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_dstColorBlendFactor, gvk_blend_factor(st.blend_dst))
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_srcAlphaBlendFactor, gvk_blend_factor(st.blend_src))
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_dstAlphaBlendFactor, gvk_blend_factor(st.blend_dst))
|
|
}
|
|
}
|
|
let cbs = bytes(VkPipelineColorBlendStateCreateInfo_sizeof)
|
|
Vk.zero(cbs, VkPipelineColorBlendStateCreateInfo_sizeof)
|
|
Vk.put_i32(cbs, VkPipelineColorBlendStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO)
|
|
Vk.put_i32(cbs, VkPipelineColorBlendStateCreateInfo_attachmentCount, n_color)
|
|
Vk.put_ptr(cbs, VkPipelineColorBlendStateCreateInfo_pAttachments, cba)
|
|
let dyn_states = bytes(8)
|
|
Vk.put_i32(dyn_states, 0, VK_DYNAMIC_STATE_VIEWPORT)
|
|
Vk.put_i32(dyn_states, 4, VK_DYNAMIC_STATE_SCISSOR)
|
|
let dys = bytes(VkPipelineDynamicStateCreateInfo_sizeof)
|
|
Vk.zero(dys, VkPipelineDynamicStateCreateInfo_sizeof)
|
|
Vk.put_i32(dys, VkPipelineDynamicStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO)
|
|
Vk.put_i32(dys, VkPipelineDynamicStateCreateInfo_dynamicStateCount, 2)
|
|
Vk.put_ptr(dys, VkPipelineDynamicStateCreateInfo_pDynamicStates, dyn_states)
|
|
let formats = bytes(4 * (n_color + 1))
|
|
for c in 0 .. n_color { Vk.put_i32(formats, c * 4, color_fmt) }
|
|
let prci = bytes(VkPipelineRenderingCreateInfo_sizeof)
|
|
Vk.zero(prci, VkPipelineRenderingCreateInfo_sizeof)
|
|
Vk.put_i32(prci, VkPipelineRenderingCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_RENDERING_CREATE_INFO)
|
|
Vk.put_i32(prci, VkPipelineRenderingCreateInfo_colorAttachmentCount, n_color)
|
|
Vk.put_ptr(prci, VkPipelineRenderingCreateInfo_pColorAttachmentFormats, formats)
|
|
Vk.put_i32(prci, VkPipelineRenderingCreateInfo_depthAttachmentFormat, depth_fmt)
|
|
|
|
let gpci = bytes(VkGraphicsPipelineCreateInfo_sizeof)
|
|
Vk.zero(gpci, VkGraphicsPipelineCreateInfo_sizeof)
|
|
Vk.put_i32(gpci, VkGraphicsPipelineCreateInfo_sType, VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pNext, prci)
|
|
Vk.put_i32(gpci, VkGraphicsPipelineCreateInfo_stageCount, 2)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pStages, stages)
|
|
# a mesh-shader pipeline has neither: the mesh stage makes its own vertices
|
|
if gvk_stage_first(p) == VK_SHADER_STAGE_VERTEX_BIT {
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pVertexInputState, vin)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pInputAssemblyState, ias)
|
|
}
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pViewportState, vps)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pRasterizationState, rs)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pMultisampleState, ms)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pDepthStencilState, ds)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pColorBlendState, cbs)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pDynamicState, dys)
|
|
Vk.put_i64(gpci, VkGraphicsPipelineCreateInfo_layout, gvk_prog_layout[p])
|
|
let out = bytes(8)
|
|
let r = Vk.create_graphics_pipelines(gvk_dev, zero, 1, gpci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateGraphicsPipelines for {v.vs} + {v.fs}`, r); return zero }
|
|
let pipe = gvk_handle(out)
|
|
# R3D_VK_PROF names each pipeline as it is made: one made during play is a stall a warm-up missed
|
|
if gvk_prof() { gvk_n_pipe_new += 1; print(`r3d: vulkan pipeline {len(gvk_pipe) + 1}: {v.vs} + {v.fs}, {n_color} colour format {color_fmt}, depth {depth_fmt}, {samples}x`) }
|
|
push(gvk_pipe_keys, key)
|
|
push(gvk_pipe, pipe)
|
|
return pipe
|
|
}
|
|
|
|
# ---- uniforms -----------------------------------------------------------------------------
|
|
# On Vulkan a loose uniform is a place in its program's uniform block: glslang's relaxed mode
|
|
# gathered each stage's loose uniforms into one block and the manifest says where each one sits.
|
|
# A uniform "location" is program * 4096 + the index of its first manifest entry, so -1 still
|
|
# means "not in this program". Writing one writes every stage that declares it; the blocks go to
|
|
# the GPU at the draw.
|
|
var gvk_ublk_v: []bytes = null # per program: the vertex stage's block as last set
|
|
var gvk_ublk_f: []bytes = null
|
|
var gvk_prog_tex: []words = null # per program: the texture bound to each manifest sampler
|
|
var gvk_prog_unit: []words = null # per program: the unit + 1 each sampler was pointed at with u_i
|
|
var gvk_prog_dirty: []int = null # per program: its blocks or textures changed since its last set
|
|
var gvk_set_last: []long = null # per program: the descriptor set its last draw used
|
|
var gvk_set_frame: []int = null # per program: the frame that set belongs to
|
|
var gvk_set_tex: []words = null # per program: the textures that set was filled with
|
|
|
|
function gvk_same(a: pointer, b: pointer, n: int) -> bool {
|
|
var i = 0
|
|
while i < n { if Vk.get_i32(a, i) != Vk.get_i32(b, i) { return false }; i += 4 }
|
|
return true
|
|
}
|
|
|
|
function gvk_uniform_blocks(p: int) -> void {
|
|
if gvk_ublk_v == null {
|
|
gvk_ublk_v = new []bytes; gvk_ublk_f = new []bytes; gvk_prog_tex = new []words; gvk_prog_unit = new []words
|
|
gvk_prog_dirty = new []int; gvk_set_last = new []long; gvk_set_frame = new []int; gvk_set_tex = new []words
|
|
}
|
|
let zero: long = 0
|
|
while len(gvk_ublk_v) <= p {
|
|
push(gvk_ublk_v, null); push(gvk_ublk_f, null); push(gvk_prog_tex, null); push(gvk_prog_unit, null)
|
|
push(gvk_prog_dirty, 1); push(gvk_set_last, zero); push(gvk_set_frame, 0); push(gvk_set_tex, null)
|
|
}
|
|
let v = gvk_prog_var[p]
|
|
if v == null or gvk_ublk_v[p] != null or gvk_prog_tex[p] != null { return }
|
|
if v.vblock > 0 { let b = bytes(v.vblock); Vk.zero(b, v.vblock); gvk_ublk_v[p] = b }
|
|
if v.fblock > 0 { let b = bytes(v.fblock); Vk.zero(b, v.fblock); gvk_ublk_f[p] = b }
|
|
let nt = len(v.t_name) + 1
|
|
let t = words(nt)
|
|
let u = words(nt)
|
|
for i in 0 .. nt { t[i] = 0; u[i] = 0 }
|
|
gvk_prog_tex[p] = t
|
|
gvk_prog_unit[p] = u
|
|
let st = words(nt)
|
|
for i in 0 .. nt { st[i] = -1 }
|
|
gvk_set_tex[p] = st
|
|
}
|
|
|
|
function gvk_uniform(p: int, name: string) -> int {
|
|
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) { return -1 }
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return -1 }
|
|
for i in 0 .. len(v.u_name) { if v.u_name[i] == name { return p * 4096 + i } }
|
|
# a sampler: OpenGL code points it at a texture unit once with u_i and then only binds units
|
|
# (the actors do), so its location is kept apart and u_i on it records the unit
|
|
for i in 0 .. len(v.t_name) { if v.t_name[i] == name { return p * 4096 + 2048 + i } }
|
|
return -1
|
|
}
|
|
|
|
# n elements of `size` bytes each from src into every block that declares the uniform at loc
|
|
function gvk_u_set(loc: int, src: pointer, size: int, n: int) -> void {
|
|
if loc < 0 { return }
|
|
let p = loc / 4096
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return }
|
|
gvk_uniform_blocks(p)
|
|
if loc % 4096 >= 2048 {
|
|
let si = loc % 4096 - 2048
|
|
if si < len(v.t_name) and gvk_prog_unit[p][si] != Vk.get_i32(src, 0) + 1 { gvk_prog_unit[p][si] = Vk.get_i32(src, 0) + 1; gvk_prog_dirty[p] = 1 }
|
|
return
|
|
}
|
|
let name = v.u_name[loc % 4096]
|
|
for j in 0 .. len(v.u_name) {
|
|
if v.u_name[j] != name { continue }
|
|
var blk = gvk_ublk_v[p]
|
|
var cap = v.vblock
|
|
if v.u_stage[j] == 1 { blk = gvk_ublk_f[p]; cap = v.fblock }
|
|
if blk == null { continue }
|
|
var stride = v.u_stride[j]
|
|
if stride == 0 { stride = size }
|
|
var count = n
|
|
if count > v.u_count[j] { count = v.u_count[j] }
|
|
for k in 0 .. count {
|
|
let at = v.u_off[j] + k * stride
|
|
# the same value again (most per-draw uniforms repeat) leaves the program's set reusable
|
|
if at + size <= cap and not gvk_same(mem_off(blk, at), mem_off(src, k * size), size) {
|
|
mem_copy(mem_off(blk, at), mem_off(src, k * size), size)
|
|
gvk_prog_dirty[p] = 1
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
# a sampler uniform: the texture for the manifest binding with that name
|
|
function gvk_bind_texture(p: int, name: string, tex: int) -> void {
|
|
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) { return }
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return }
|
|
gvk_uniform_blocks(p)
|
|
for i in 0 .. len(v.t_name) { if v.t_name[i] == name and gvk_prog_tex[p][i] != tex { gvk_prog_tex[p][i] = tex; gvk_prog_dirty[p] = 1 } }
|
|
}
|
|
|
|
# ---- per-frame uniform ring and descriptor sets -------------------------------------------
|
|
# Each draw's blocks are copied into one host-visible ring buffer at the device's alignment and
|
|
# its descriptor set comes from a pool that is reset with the frame. Both are rewound at
|
|
# gvk_frame_reset.
|
|
const GVK_RING_BYTES: int = 64 * 1024 * 1024
|
|
var gvk_ring_buf: int = 0 # a gvk_buf handle
|
|
var gvk_ring_off: int = 0
|
|
var gvk_ring_align: int = 256
|
|
var gvk_dpool: long = 0
|
|
var gvk_white: int = 0 # a 1x1 white texture for a sampler nothing was bound to
|
|
var gvk_msaa_max: int = 0 # samples the scene may use (gvk_frame_init reads the device's limits)
|
|
|
|
function gvk_frame_init() -> bool {
|
|
let props = bytes(VkPhysicalDeviceProperties_sizeof)
|
|
Vk.get_physical_device_properties(gvk_pd, props)
|
|
let al = Vk.get_i64(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_minUniformBufferOffsetAlignment)
|
|
gvk_ring_align = Text.to_int(string(al))
|
|
if gvk_ring_align < 16 { gvk_ring_align = 16 }
|
|
# MSAA: the most samples (up to the 4 the scene asks for) both a colour and a depth target can take
|
|
let lim = VkPhysicalDeviceProperties_limits
|
|
let counts = Vk.get_i32(props, lim + VkPhysicalDeviceLimits_framebufferColorSampleCounts) & Vk.get_i32(props, lim + VkPhysicalDeviceLimits_framebufferDepthSampleCounts)
|
|
gvk_msaa_max = 1
|
|
if (counts & VK_SAMPLE_COUNT_2_BIT) != 0 { gvk_msaa_max = 2 }
|
|
if (counts & VK_SAMPLE_COUNT_4_BIT) != 0 { gvk_msaa_max = 4 }
|
|
gvk_ring_buf = gvk_buf_new()
|
|
if not gvk_buf_reserve(gvk_ring_buf, GVK_RING_BYTES) { return false }
|
|
let sizes = bytes(VkDescriptorPoolSize_sizeof * 3)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_descriptorCount, 32768)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_descriptorCount, 131072)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_STORAGE_BUFFER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_descriptorCount, 16384)
|
|
let dpci = bytes(VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.zero(dpci, VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_maxSets, 16384)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 3)
|
|
Vk.put_ptr(dpci, VkDescriptorPoolCreateInfo_pPoolSizes, sizes)
|
|
let out = bytes(8)
|
|
let r = Vk.create_descriptor_pool(gvk_dev, dpci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateDescriptorPool", r) }
|
|
gvk_dpool = gvk_handle(out)
|
|
if not gvk_kpool_make() { return false }
|
|
gvk_white = gvk_tex_new()
|
|
let px = bytes(4)
|
|
px[0] = 255; px[1] = 255; px[2] = 255; px[3] = 255
|
|
return gvk_tex_storage(gvk_white, false, GL_RGBA8, 1, 1, 1, false) and gvk_tex_upload(gvk_white, GL_RGBA8, 1, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, px)
|
|
}
|
|
|
|
function gvk_frame_reset() -> void {
|
|
gvk_ring_off = 0
|
|
Vk.reset_descriptor_pool(gvk_dev, gvk_dpool, 0)
|
|
}
|
|
|
|
# a block into the ring; its offset, or -1 when the frame has used the whole ring
|
|
function gvk_ring_put(blk: pointer, n: int) -> int {
|
|
let at = (gvk_ring_off + gvk_ring_align - 1) / gvk_ring_align * gvk_ring_align
|
|
if at + n > GVK_RING_BYTES { return -1 }
|
|
mem_copy(mem_off(gvk_buf_map[gvk_ring_buf], at), blk, n)
|
|
gvk_ring_off = at + n
|
|
return at
|
|
}
|
|
|
|
# ---- descriptor sets that survive the frame -------------------------------------------------
|
|
# A draw's set carries its textures; its uniform blocks are dynamic uniform buffers into the
|
|
# frame's ring, bound with this draw's offsets. So one set serves every draw of a program with the
|
|
# same textures, this frame and the frames after it, and a draw that changes only uniforms allocates
|
|
# and writes no set at all - which was about half of a Vulkan frame's CPU time (8.5 ms at the camp).
|
|
# The sets live in a pool of their own that is only reset when it fills. The key holds each
|
|
# texture's handle and generation (bumped when the image behind a handle is replaced) and its
|
|
# sampler, so a set is never matched against a texture it no longer describes.
|
|
const GVK_KPOOL_SETS: int = 8192
|
|
var gvk_kpool: long = 0
|
|
var gvk_sc_prog: []int = null # per cached set: its program
|
|
var gvk_sc_koff: []int = null # per cached set: where its key starts in gvk_sc_keys
|
|
var gvk_sc_set: []long = null
|
|
var gvk_sc_next: []int = null # the program's next cached set, -1 at the end
|
|
var gvk_sc_keys: []long = null # per texture of a key: handle * 65536 + generation, sampler
|
|
var gvk_sc_head: words = null # per program (< 4096): its most recently made set, -1 if none
|
|
var gvk_sc_tmp: []long = null # this draw's key while it is looked up
|
|
var gvk_set_offs: bytes = null # this draw's dynamic offsets, in binding order
|
|
var gvk_set_ndyn: int = 0
|
|
var gvk_ub_frame: []int = null # per program: the frame its blocks were last copied into the ring
|
|
var gvk_ub_offv: []int = null
|
|
var gvk_ub_offf: []int = null
|
|
var gvk_sc_made: int = 0 # sets made since start (R3D_VK_PROF)
|
|
var gvk_default_smp: long = 0 # linear, clamped: for a texture slot nothing was bound to
|
|
|
|
function gvk_kpool_make() -> bool {
|
|
let sizes = bytes(VkDescriptorPoolSize_sizeof * 2)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_descriptorCount, GVK_KPOOL_SETS * 2)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_descriptorCount, GVK_KPOOL_SETS * 8)
|
|
let dpci = bytes(VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.zero(dpci, VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_maxSets, GVK_KPOOL_SETS)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 2)
|
|
Vk.put_ptr(dpci, VkDescriptorPoolCreateInfo_pPoolSizes, sizes)
|
|
let out = bytes(8)
|
|
let r = Vk.create_descriptor_pool(gvk_dev, dpci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateDescriptorPool (kept sets)", r) }
|
|
gvk_kpool = gvk_handle(out)
|
|
gvk_sc_clear()
|
|
return true
|
|
}
|
|
|
|
function gvk_sc_clear() -> void {
|
|
gvk_sc_prog = new []int; gvk_sc_koff = new []int; gvk_sc_set = new []long; gvk_sc_next = new []int
|
|
gvk_sc_keys = new []long
|
|
if gvk_sc_head == null { gvk_sc_head = words(4096) }
|
|
for i in 0 .. 4096 { gvk_sc_head[i] = -1 }
|
|
}
|
|
|
|
# the pool is full: finish the frame recorded so far (it may use sets from it), then start again
|
|
function gvk_kpool_reset() -> void {
|
|
gvk_flush()
|
|
Vk.reset_descriptor_pool(gvk_dev, gvk_kpool, 0)
|
|
gvk_sc_clear()
|
|
}
|
|
|
|
function gvk_draw_set(p: int, tx: words, tx_w: int, tx_cap: int) -> long {
|
|
let zero: long = 0
|
|
let v = gvk_prog_var[p]
|
|
gvk_uniform_blocks(p)
|
|
if gvk_sc_tmp == null {
|
|
gvk_sc_tmp = new []long
|
|
gvk_set_offs = bytes(16)
|
|
gvk_ub_frame = new []int; gvk_ub_offv = new []int; gvk_ub_offf = new []int
|
|
}
|
|
while len(gvk_ub_frame) <= p { push(gvk_ub_frame, 0); push(gvk_ub_offv, 0); push(gvk_ub_offf, 0) }
|
|
let nt = len(v.t_name)
|
|
# the key: every texture as this draw resolves it, and its sampler
|
|
while len(gvk_sc_tmp) < nt * 2 + 2 { push(gvk_sc_tmp, zero) }
|
|
for t in 0 .. nt {
|
|
var tex = gvk_prog_tex[p][t]
|
|
# nothing bound by name: the texture on the unit the sampler was pointed at, as OpenGL reads it
|
|
let unit1 = gvk_prog_unit[p][t]
|
|
if tex <= 0 and unit1 > 0 and unit1 <= 32 and gpu_unit_2d != null { tex = gpu_unit_2d[unit1 - 1] }
|
|
if tex <= 0 or tex >= len(gvk_tex_image) or gvk_tex_image[tex] == 0 { tex = gvk_white }
|
|
var smp: long = 0
|
|
let o = tex * tx_w
|
|
if tex != gvk_white and tex < tx_cap {
|
|
smp = gvk_tex_sampler(tex, tx[o + 5], tx[o + 6], tx[o + 7], tx[o + 8], tx[o + 9], tx[o + 11])
|
|
} else {
|
|
# the sampler for a slot nothing was bound to, made once: looking it up by its string key on
|
|
# every draw was the costly part of the key for the shadow and depth variants
|
|
if gvk_default_smp == 0 { gvk_default_smp = gvk_sampler(GL_LINEAR, GL_LINEAR, GL_CLAMP_TO_EDGE, GL_CLAMP_TO_EDGE, 0, 0) }
|
|
smp = gvk_default_smp
|
|
}
|
|
let tg: long = tex * 65536 + gvk_tex_gen[tex]
|
|
gvk_sc_tmp[t * 2] = tg
|
|
gvk_sc_tmp[t * 2 + 1] = smp
|
|
}
|
|
var set: long = 0
|
|
var e = -1
|
|
if p < 4096 { e = gvk_sc_head[p] }
|
|
while e >= 0 and set == 0 {
|
|
let ko = gvk_sc_koff[e]
|
|
var same = true
|
|
var t = 0
|
|
while same and t < nt * 2 { if gvk_sc_keys[ko + t] != gvk_sc_tmp[t] { same = false }; t += 1 }
|
|
if same { set = gvk_sc_set[e] } else { e = gvk_sc_next[e] }
|
|
}
|
|
if set == 0 {
|
|
set = gvk_sc_make(p, nt)
|
|
if set == 0 { return zero }
|
|
}
|
|
# the uniform blocks: copied into the ring once per frame, and again only when a value changed
|
|
if gvk_prog_dirty[p] == 1 or gvk_ub_frame[p] != gvk_frame_no {
|
|
if gvk_ublk_v[p] != null and v.vblock > 0 {
|
|
let at = gvk_ring_put(gvk_ublk_v[p], v.vblock)
|
|
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return zero }
|
|
gvk_ub_offv[p] = at
|
|
}
|
|
if gvk_ublk_f[p] != null and v.fblock > 0 {
|
|
let at = gvk_ring_put(gvk_ublk_f[p], v.fblock)
|
|
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return zero }
|
|
gvk_ub_offf[p] = at
|
|
}
|
|
gvk_ub_frame[p] = gvk_frame_no
|
|
gvk_prog_dirty[p] = 0
|
|
}
|
|
gvk_set_ndyn = 0
|
|
if v.vblock >= 0 { Vk.put_i32(gvk_set_offs, gvk_set_ndyn * 4, gvk_ub_offv[p]); gvk_set_ndyn += 1 }
|
|
if v.fblock >= 0 { Vk.put_i32(gvk_set_offs, gvk_set_ndyn * 4, gvk_ub_offf[p]); gvk_set_ndyn += 1 }
|
|
gvk_buf_used[gvk_ring_buf] = gvk_frame_no
|
|
return set
|
|
}
|
|
|
|
# a new kept set for program p with the textures in gvk_sc_tmp, written and cached
|
|
function gvk_sc_make(p: int, nt: int) -> long {
|
|
let zero: long = 0
|
|
let v = gvk_prog_var[p]
|
|
let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof)
|
|
let layouts = bytes(8)
|
|
let sets = bytes(8)
|
|
var r = 0
|
|
var tries = 0
|
|
while tries < 2 {
|
|
Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO)
|
|
Vk.put_i64(dsai, VkDescriptorSetAllocateInfo_descriptorPool, gvk_kpool)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_descriptorSetCount, 1)
|
|
Vk.put_i64(layouts, 0, gvk_prog_dsl[p])
|
|
Vk.put_ptr(dsai, VkDescriptorSetAllocateInfo_pSetLayouts, layouts)
|
|
r = Vk.allocate_descriptor_sets(gvk_dev, dsai, sets)
|
|
if r == VK_SUCCESS { tries = 2 } else { gvk_kpool_reset(); tries += 1 }
|
|
}
|
|
if r != VK_SUCCESS { gvk_fail("vkAllocateDescriptorSets (kept sets)", r); return zero }
|
|
let set = Vk.get_i64(sets, 0)
|
|
let ww = VkWriteDescriptorSet_sizeof
|
|
let writes = bytes(ww * (nt + 3))
|
|
Vk.zero(writes, ww * (nt + 3))
|
|
let bis = bytes(VkDescriptorBufferInfo_sizeof * 2)
|
|
let iis = bytes(VkDescriptorImageInfo_sizeof * (nt + 1))
|
|
var nw = 0
|
|
var bi = 0
|
|
for stage in 0 .. 2 {
|
|
var size = v.vblock
|
|
if stage == 1 { size = v.fblock }
|
|
if size < 0 { continue }
|
|
let at = bi * VkDescriptorBufferInfo_sizeof
|
|
let off0: long = 0
|
|
var range: long = size
|
|
if size == 0 { range = 16 }
|
|
Vk.put_i64(bis, at + VkDescriptorBufferInfo_buffer, gvk_buf[gvk_ring_buf])
|
|
Vk.put_i64(bis, at + VkDescriptorBufferInfo_offset, off0)
|
|
Vk.put_i64(bis, at + VkDescriptorBufferInfo_range, range)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
|
|
Vk.put_i64(writes, nw * ww + VkWriteDescriptorSet_dstSet, set)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_dstBinding, stage)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorCount, 1)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_ptr(writes, nw * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(bis, at))
|
|
nw += 1
|
|
bi += 1
|
|
}
|
|
let iw = VkDescriptorImageInfo_sizeof
|
|
for t in 0 .. nt {
|
|
let tex = Text.to_int(string(gvk_sc_tmp[t * 2] / 65536))
|
|
Vk.put_i64(iis, t * iw + VkDescriptorImageInfo_sampler, gvk_sc_tmp[t * 2 + 1])
|
|
Vk.put_i64(iis, t * iw + VkDescriptorImageInfo_imageView, gvk_tex_view[tex])
|
|
Vk.put_i32(iis, t * iw + VkDescriptorImageInfo_imageLayout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
|
|
Vk.put_i64(writes, nw * ww + VkWriteDescriptorSet_dstSet, set)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_dstBinding, v.t_bind[t])
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorCount, 1)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_ptr(writes, nw * ww + VkWriteDescriptorSet_pImageInfo, mem_off(iis, t * iw))
|
|
nw += 1
|
|
}
|
|
if nw > 0 { Vk.update_descriptor_sets(gvk_dev, nw, writes, 0, null) }
|
|
let e = len(gvk_sc_set)
|
|
push(gvk_sc_prog, p); push(gvk_sc_koff, len(gvk_sc_keys)); push(gvk_sc_set, set)
|
|
for t in 0 .. nt * 2 { push(gvk_sc_keys, gvk_sc_tmp[t]) }
|
|
var head = -1
|
|
if p < 4096 { head = gvk_sc_head[p]; gvk_sc_head[p] = e }
|
|
push(gvk_sc_next, head)
|
|
gvk_sc_made += 1
|
|
return set
|
|
}
|
|
|
|
# ---- the frame and its passes -------------------------------------------------------------
|
|
# One command buffer records the whole frame. Binding a framebuffer ends the pass in progress;
|
|
# the next one begins at its first clear or draw, so a clear that comes first becomes the pass's
|
|
# load op. For the length of a pass its attachments sit in the attachment layouts, and between
|
|
# passes every image is back in SHADER_READ_ONLY where samplers expect it. The present submits
|
|
# and waits: the frame is finished before the next one starts, as it is on OpenGL.
|
|
#
|
|
# Framebuffer 0 is the screen: a colour and a depth image made at gvk_screen_make. Headless that
|
|
# is all the screen there is; with a window the present copies it to the swapchain.
|
|
var gvk_cb: pointer = null
|
|
var gvk_in_pass: bool = false
|
|
var gvk_fb_cur: int = 0
|
|
var gvk_fb_read: int = 0
|
|
var gvk_fb_draw: int = 0
|
|
var gvk_pass_w: int = 0
|
|
var gvk_pass_h: int = 0
|
|
var gvk_pass_ncolor: int = 0
|
|
var gvk_pass_cfmt: int = 0
|
|
var gvk_pass_dfmt: int = 0
|
|
var gvk_pass_samples: int = 1 # the attachments' sample count, which every pipeline in the pass must match
|
|
var gvk_pass_col: words = null # the pass's colour attachments: texture, layer + 1
|
|
var gvk_pass_dep: words = null # its depth attachment: texture, layer + 1
|
|
var gvk_clear_bits: int = 0 # GL_COLOR_BUFFER_BIT / GL_DEPTH_BUFFER_BIT waiting for the pass
|
|
var gvk_clear_rgba: floats = null # float bits
|
|
var gvk_vp: words = null # x, y, w, h
|
|
var gvk_sc: words = null # scissor on, x, y (OpenGL rows), w, h
|
|
var gvk_screen_color: int = 0
|
|
var gvk_screen_depth: int = 0
|
|
var gvk_screen_w: int = 0
|
|
var gvk_screen_h: int = 0
|
|
var gvk_layer_views: []string = null # "tex:layer" -> index into gvk_layer_view
|
|
var gvk_layer_view: []long = null
|
|
var gvk_fb_ncolor: words = null # per framebuffer: colour slots drawn (draw buffers; 0 = none)
|
|
|
|
# ---- HDR output ------------------------------------------------------------------------------
|
|
# HDR10 (ST 2084 over BT.2020) when the player asks for it and the surface offers it. The screen
|
|
# image and the tonemap's LDR image go 10-bit with it; the tonemap and the overlay switch to their
|
|
# HDR10 variants (gpu_hdr_active). Headless there is no surface, so never.
|
|
var gvk_hdr_want: bool = false # r3d_hdr: the setting
|
|
var gvk_hdr_on: bool = false # the swapchain is HDR10 now
|
|
var gvk_has_colorspace: bool = false # the instance took VK_EXT_swapchain_colorspace
|
|
var gvk_has_hdr_meta: bool = false # the device took VK_EXT_hdr_metadata
|
|
# R3D_HDR=1 / 0 overrides the setting, for a desktop test
|
|
function r3d_hdr(on: bool) -> void {
|
|
var want = on
|
|
if r3d_env_has("R3D_HDR") { want = Text.to_int(r3d_env("R3D_HDR")) != 0 }
|
|
if want == gvk_hdr_want { return }
|
|
gvk_hdr_want = want
|
|
if gvk_swap != 0 { gvk_swap_stale = true }
|
|
}
|
|
function gpu_hdr_active() -> bool { return gpu_kind == GPU_VK and gvk_hdr_on }
|
|
function gvk_screen_fmt() -> int { if gvk_hdr_on { return GL_RGB10_A2 }; return GL_RGBA8 }
|
|
|
|
# The player's calibration of their display, in nits (float bits): the brightest it shows, where the
|
|
# picture's and the interface's white sit, and how far the darkest shade is lifted. Displays differ by
|
|
# an order of magnitude - a 400-nit monitor and a 2000-nit television - and one fixed curve either
|
|
# clips the first's highlights flat or leaves the second dim.
|
|
var r3d_hdr_peak: float = 0.0 # 0 until r3d_hdr_calibrate: 1000
|
|
var r3d_hdr_paper: float = 0.0 # 200
|
|
var r3d_hdr_black: float = 0.0 # 0
|
|
function r3d_hdr_peak_nits() -> float { if r3d_hdr_peak == 0.0 { return 1000.0 }; return r3d_hdr_peak }
|
|
function r3d_hdr_paper_nits() -> float { if r3d_hdr_paper == 0.0 { return 200.0 }; return r3d_hdr_paper }
|
|
function r3d_hdr_black_nits() -> float { return r3d_hdr_black }
|
|
function r3d_hdr_calibrate(peak: float, paper: float, black: float) -> void {
|
|
var pk = Math.clamp(peak, 100.0, 10000.0)
|
|
let pp = Math.clamp(paper, 80.0, 1000.0)
|
|
if pk < pp { pk = pp }
|
|
let bl = Math.clamp(black, 0.0, 5.0)
|
|
if pk == r3d_hdr_peak and pp == r3d_hdr_paper and bl == r3d_hdr_black { return }
|
|
r3d_hdr_peak = pk; r3d_hdr_paper = pp; r3d_hdr_black = bl
|
|
# the display is told what the picture now reaches
|
|
if gvk_hdr_on and gvk_has_hdr_meta and gvk_swap != 0 { gvk_hdr_metadata() }
|
|
}
|
|
|
|
# what the picture is: graded in BT.709 around D65, highlights to the calibrated peak, paper white average
|
|
function gvk_hdr_metadata() -> void {
|
|
let md = bytes(VkHdrMetadataEXT_sizeof)
|
|
Vk.zero(md, VkHdrMetadataEXT_sizeof)
|
|
Vk.put_i32(md, VkHdrMetadataEXT_sType, VK_STRUCTURE_TYPE_HDR_METADATA_EXT)
|
|
Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryRed + VkXYColorEXT_x, float_bits(0.64)); Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryRed + VkXYColorEXT_y, float_bits(0.33))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryGreen + VkXYColorEXT_x, float_bits(0.30)); Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryGreen + VkXYColorEXT_y, float_bits(0.60))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryBlue + VkXYColorEXT_x, float_bits(0.15)); Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryBlue + VkXYColorEXT_y, float_bits(0.06))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_whitePoint + VkXYColorEXT_x, float_bits(0.3127)); Vk.put_i32(md, VkHdrMetadataEXT_whitePoint + VkXYColorEXT_y, float_bits(0.3290))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_maxLuminance, float_bits(r3d_hdr_peak_nits()))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_minLuminance, float_bits(0.001))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_maxContentLightLevel, float_bits(r3d_hdr_peak_nits()))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_maxFrameAverageLightLevel, float_bits(r3d_hdr_paper_nits()))
|
|
let chains = bytes(8)
|
|
Vk.put_i64(chains, 0, gvk_swap)
|
|
Vk.set_hdr_metadata_ext(gvk_dev, 1, chains, md)
|
|
}
|
|
|
|
function gvk_screen_make(w: int, h: int) -> bool {
|
|
if gvk_vp == null {
|
|
gvk_vp = words(4); gvk_sc = words(5); gvk_clear_rgba = floats(4)
|
|
gvk_pass_col = words(4); gvk_pass_dep = words(2)
|
|
gvk_fb_ncolor = words(4096)
|
|
for i in 0 .. 4096 { gvk_fb_ncolor[i] = 1 }
|
|
for i in 0 .. 5 { gvk_sc[i] = 0 }
|
|
}
|
|
gvk_screen_w = w
|
|
gvk_screen_h = h
|
|
if gvk_screen_color == 0 { gvk_screen_color = gvk_tex_new(); gvk_screen_depth = gvk_tex_new() }
|
|
gvk_vp[0] = 0; gvk_vp[1] = 0; gvk_vp[2] = w; gvk_vp[3] = h
|
|
return gvk_tex_storage(gvk_screen_color, false, gvk_screen_fmt(), w, h, 1, false) and gvk_tex_storage(gvk_screen_depth, false, GL_DEPTH_COMPONENT32F, w, h, 1, false)
|
|
}
|
|
|
|
function gvk_frame_cb() -> pointer {
|
|
if gvk_cb == null {
|
|
gvk_frame_reset()
|
|
gvk_cb = gvk_once_begin()
|
|
}
|
|
return gvk_cb
|
|
}
|
|
|
|
# The view a pass draws into: level 0 only (an attachment view has exactly one level, and a target
|
|
# the exposure measure mipmaps has several), and one layer of an array image for a cascade drawn
|
|
# on its own. Keyed by the image's generation, so a replaced image never reuses a stale view.
|
|
function gvk_view_of(tex: int, layer1: int) -> long {
|
|
if layer1 == 0 and gvk_tex_levels[tex] <= 1 and gvk_tex_layers[tex] <= 1 { return gvk_tex_view[tex] }
|
|
let key = `{tex}:{gvk_tex_gen[tex]}:{layer1}`
|
|
if gvk_layer_views == null { gvk_layer_views = new []string; gvk_layer_view = new []long }
|
|
for i in 0 .. len(gvk_layer_views) { if gvk_layer_views[i] == key { return gvk_layer_view[i] } }
|
|
let depth = gvk_tex_vkfmt[tex] == VK_FORMAT_D32_SFLOAT
|
|
let vci = bytes(VkImageViewCreateInfo_sizeof)
|
|
Vk.zero(vci, VkImageViewCreateInfo_sizeof)
|
|
Vk.put_i32(vci, VkImageViewCreateInfo_sType, VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO)
|
|
Vk.put_i64(vci, VkImageViewCreateInfo_image, gvk_tex_image[tex])
|
|
Vk.put_i32(vci, VkImageViewCreateInfo_viewType, VK_IMAGE_VIEW_TYPE_2D)
|
|
Vk.put_i32(vci, VkImageViewCreateInfo_format, gvk_tex_vkfmt[tex])
|
|
let sr = VkImageViewCreateInfo_subresourceRange
|
|
if depth { Vk.put_i32(vci, sr + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT) } else { Vk.put_i32(vci, sr + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT) }
|
|
Vk.put_i32(vci, sr + VkImageSubresourceRange_levelCount, 1)
|
|
if layer1 > 0 { Vk.put_i32(vci, sr + VkImageSubresourceRange_baseArrayLayer, layer1 - 1) }
|
|
Vk.put_i32(vci, sr + VkImageSubresourceRange_layerCount, 1)
|
|
let out = bytes(8)
|
|
let zero: long = 0
|
|
if Vk.create_image_view(gvk_dev, vci, null, out) != VK_SUCCESS { return zero }
|
|
push(gvk_layer_views, key)
|
|
push(gvk_layer_view, gvk_handle(out))
|
|
return gvk_handle(out)
|
|
}
|
|
# the layer range a barrier for an attachment covers: the whole image unless one layer is drawn
|
|
function gvk_att_barrier(cb: pointer, tex: int, layer1: int, depth: bool, old_layout: int, new_layout: int) -> void {
|
|
# an attachment whose image was never made (no memory for it) has nothing to transition
|
|
if tex <= 0 or tex >= len(gvk_tex_image) or gvk_tex_image[tex] == 0 { return }
|
|
let b = bytes(VkImageMemoryBarrier_sizeof)
|
|
Vk.zero(b, VkImageMemoryBarrier_sizeof)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_sType, VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_srcAccessMask, gvk_layout_access(old_layout))
|
|
Vk.put_i32(b, VkImageMemoryBarrier_dstAccessMask, gvk_layout_access(new_layout))
|
|
Vk.put_i32(b, VkImageMemoryBarrier_oldLayout, old_layout)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_newLayout, new_layout)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_srcQueueFamilyIndex, VK_QUEUE_FAMILY_IGNORED)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_dstQueueFamilyIndex, VK_QUEUE_FAMILY_IGNORED)
|
|
Vk.put_i64(b, VkImageMemoryBarrier_image, gvk_tex_image[tex])
|
|
let r = VkImageMemoryBarrier_subresourceRange
|
|
if depth { Vk.put_i32(b, r + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT) } else { Vk.put_i32(b, r + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT) }
|
|
Vk.put_i32(b, r + VkImageSubresourceRange_levelCount, 1)
|
|
if layer1 == 0 { Vk.put_i32(b, r + VkImageSubresourceRange_layerCount, gvk_tex_layers[tex]) }
|
|
else { Vk.put_i32(b, r + VkImageSubresourceRange_baseArrayLayer, layer1 - 1); Vk.put_i32(b, r + VkImageSubresourceRange_layerCount, 1) }
|
|
Vk.cmd_pipeline_barrier(cb, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, null, 0, null, 1, b)
|
|
}
|
|
|
|
# fb's attachments from gpu.ludic's record (colour 0, colour 1, depth texture, depth layer + 1,
|
|
# colour rb, depth rb, samples, colour layer + 1), framebuffer 0 being the screen
|
|
function gvk_pass_collect(fb: int, rec: words, rec_o: int) -> void {
|
|
for i in 0 .. 4 { gvk_pass_col[i] = 0 }
|
|
gvk_pass_dep[0] = 0; gvk_pass_dep[1] = 0
|
|
gvk_pass_ncolor = 0
|
|
if fb == 0 {
|
|
gvk_pass_col[0] = gvk_screen_color; gvk_pass_ncolor = 1
|
|
gvk_pass_dep[0] = gvk_screen_depth
|
|
return
|
|
}
|
|
if rec_o < 0 { return }
|
|
var want = 1
|
|
if fb < 4096 { want = gvk_fb_ncolor[fb] }
|
|
for slot in 0 .. 2 {
|
|
if slot < want and rec[rec_o + slot] > 0 {
|
|
gvk_pass_col[gvk_pass_ncolor * 2] = rec[rec_o + slot]
|
|
if slot == 0 { gvk_pass_col[gvk_pass_ncolor * 2 + 1] = rec[rec_o + 7] }
|
|
gvk_pass_ncolor += 1
|
|
}
|
|
}
|
|
gvk_pass_dep[0] = rec[rec_o + 2]
|
|
gvk_pass_dep[1] = rec[rec_o + 3]
|
|
}
|
|
|
|
function gvk_pass_begin(rec: words, rec_o: int) -> void {
|
|
if gvk_in_pass { return }
|
|
let cb = gvk_frame_cb()
|
|
gvk_pass_collect(gvk_fb_cur, rec, rec_o)
|
|
let aw = VkRenderingAttachmentInfo_sizeof
|
|
let catt = bytes(aw * 3)
|
|
Vk.zero(catt, aw * 3)
|
|
gvk_pass_w = 0
|
|
gvk_pass_h = 0
|
|
gvk_pass_cfmt = VK_FORMAT_UNDEFINED
|
|
gvk_pass_dfmt = VK_FORMAT_UNDEFINED
|
|
gvk_pass_samples = 1
|
|
for c in 0 .. gvk_pass_ncolor {
|
|
let tex = gvk_pass_col[c * 2]
|
|
let layer1 = gvk_pass_col[c * 2 + 1]
|
|
gvk_att_barrier(cb, tex, layer1, false, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO)
|
|
Vk.put_i64(catt, c * aw + VkRenderingAttachmentInfo_imageView, gvk_view_of(tex, layer1))
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_imageLayout, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE)
|
|
if (gvk_clear_bits & GL_COLOR_BUFFER_BIT) != 0 {
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_CLEAR)
|
|
for k in 0 .. 4 { Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_clearValue + k * 4, float_bits(gvk_clear_rgba[k])) }
|
|
} else { Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) }
|
|
gvk_pass_cfmt = gvk_tex_vkfmt[tex]
|
|
gvk_pass_samples = gvk_tex_samples_of(tex)
|
|
if gvk_pass_w == 0 { gvk_pass_w = gvk_tex_w(tex); gvk_pass_h = gvk_tex_h(tex) }
|
|
}
|
|
let datt = bytes(aw)
|
|
Vk.zero(datt, aw)
|
|
let dtex = gvk_pass_dep[0]
|
|
if dtex > 0 {
|
|
gvk_att_barrier(cb, dtex, gvk_pass_dep[1], true, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO)
|
|
Vk.put_i64(datt, VkRenderingAttachmentInfo_imageView, gvk_view_of(dtex, gvk_pass_dep[1]))
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_imageLayout, VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE)
|
|
if (gvk_clear_bits & GL_DEPTH_BUFFER_BIT) != 0 {
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_CLEAR)
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_clearValue, 0x3F800000)
|
|
} else { Vk.put_i32(datt, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) }
|
|
gvk_pass_dfmt = VK_FORMAT_D32_SFLOAT
|
|
gvk_pass_samples = gvk_tex_samples_of(dtex)
|
|
if gvk_pass_w == 0 { gvk_pass_w = gvk_tex_w(dtex); gvk_pass_h = gvk_tex_h(dtex) }
|
|
}
|
|
gvk_clear_bits = 0
|
|
let ri = bytes(VkRenderingInfo_sizeof)
|
|
Vk.zero(ri, VkRenderingInfo_sizeof)
|
|
Vk.put_i32(ri, VkRenderingInfo_sType, VK_STRUCTURE_TYPE_RENDERING_INFO)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_width, gvk_pass_w)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_height, gvk_pass_h)
|
|
Vk.put_i32(ri, VkRenderingInfo_layerCount, 1)
|
|
Vk.put_i32(ri, VkRenderingInfo_colorAttachmentCount, gvk_pass_ncolor)
|
|
Vk.put_ptr(ri, VkRenderingInfo_pColorAttachments, catt)
|
|
if dtex > 0 { Vk.put_ptr(ri, VkRenderingInfo_pDepthAttachment, datt) }
|
|
Vk.cmd_begin_rendering(cb, ri)
|
|
gvk_in_pass = true
|
|
}
|
|
|
|
function gvk_pass_end() -> void {
|
|
if not gvk_in_pass { return }
|
|
let cb = gvk_cb
|
|
Vk.cmd_end_rendering(cb)
|
|
for c in 0 .. gvk_pass_ncolor {
|
|
gvk_att_barrier(cb, gvk_pass_col[c * 2], gvk_pass_col[c * 2 + 1], false, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
if gvk_pass_dep[0] > 0 {
|
|
gvk_att_barrier(cb, gvk_pass_dep[0], gvk_pass_dep[1], true, VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
gvk_in_pass = false
|
|
}
|
|
|
|
# a clear: the load op of a pass that has not begun, an explicit clear inside one that has
|
|
function gvk_clear(mask: int, rec: words, rec_o: int) -> void {
|
|
if not gvk_in_pass { gvk_clear_bits = gvk_clear_bits | mask; return }
|
|
let cb = gvk_cb
|
|
let caw = VkClearAttachment_sizeof
|
|
let atts = bytes(caw * 4)
|
|
Vk.zero(atts, caw * 4)
|
|
var n = 0
|
|
if (mask & GL_COLOR_BUFFER_BIT) != 0 {
|
|
for c in 0 .. gvk_pass_ncolor {
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_colorAttachment, c)
|
|
for k in 0 .. 4 { Vk.put_i32(atts, n * caw + VkClearAttachment_clearValue + k * 4, float_bits(gvk_clear_rgba[k])) }
|
|
n += 1
|
|
}
|
|
}
|
|
if (mask & GL_DEPTH_BUFFER_BIT) != 0 and gvk_pass_dep[0] > 0 {
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT)
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_clearValue, 0x3F800000)
|
|
n += 1
|
|
}
|
|
if n == 0 { return }
|
|
let rect = bytes(VkClearRect_sizeof)
|
|
Vk.zero(rect, VkClearRect_sizeof)
|
|
Vk.put_i32(rect, VkClearRect_rect + VkRect2D_extent + VkExtent2D_width, gvk_pass_w)
|
|
Vk.put_i32(rect, VkClearRect_rect + VkRect2D_extent + VkExtent2D_height, gvk_pass_h)
|
|
Vk.put_i32(rect, VkClearRect_layerCount, 1)
|
|
Vk.cmd_clear_attachments(cb, n, atts, 1, rect)
|
|
}
|
|
|
|
function gvk_tex_w(tex: int) -> int { return gvk_tex_dims_w[tex] }
|
|
function gvk_tex_h(tex: int) -> int { return gvk_tex_dims_h[tex] }
|
|
|
|
# viewport and scissor for the draw; OpenGL's rows count from the bottom and so do a Vulkan
|
|
# target's here (no y flip), so both pass straight through
|
|
function gvk_set_view(cb: pointer) -> void {
|
|
let vp = bytes(VkViewport_sizeof)
|
|
Vk.zero(vp, VkViewport_sizeof)
|
|
Vk.put_i32(vp, VkViewport_x, float_bits(float(gvk_vp[0])))
|
|
Vk.put_i32(vp, VkViewport_y, float_bits(float(gvk_vp[1])))
|
|
Vk.put_i32(vp, VkViewport_width, float_bits(float(gvk_vp[2])))
|
|
Vk.put_i32(vp, VkViewport_height, float_bits(float(gvk_vp[3])))
|
|
Vk.put_i32(vp, VkViewport_maxDepth, 0x3F800000)
|
|
Vk.cmd_set_viewport(cb, 0, 1, vp)
|
|
let sc = bytes(VkRect2D_sizeof)
|
|
Vk.zero(sc, VkRect2D_sizeof)
|
|
if gvk_sc[0] == 1 {
|
|
Vk.put_i32(sc, VkRect2D_offset + VkOffset2D_x, gvk_sc[1])
|
|
Vk.put_i32(sc, VkRect2D_offset + VkOffset2D_y, gvk_sc[2])
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_width, gvk_sc[3])
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_height, gvk_sc[4])
|
|
} else {
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_width, gvk_pass_w)
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_height, gvk_pass_h)
|
|
}
|
|
Vk.cmd_set_scissor(cb, 0, 1, sc)
|
|
}
|
|
|
|
# A draw of mesh m with program p and state st: the pipeline, the view, the set, the buffers.
|
|
# first / count select vertices or indices; count 0 means the mesh's own count. instances >= 1.
|
|
function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instances: int, tx: words, tx_w: int, tx_cap: int, rec: words, rec_o: int) -> void {
|
|
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) or gvk_prog_var[p] == null { return }
|
|
let prof = gvk_prof()
|
|
var t0: long = 0
|
|
if prof { t0 = gl_now_us() }
|
|
gvk_pass_begin(rec, rec_o)
|
|
let cb = gvk_cb
|
|
let samples = gvk_pass_samples
|
|
let pipe = gvk_pipeline_fast(p, m, st, gvk_pass_ncolor, gvk_pass_cfmt, gvk_pass_dfmt, samples)
|
|
if pipe == 0 {
|
|
# Say so, once per program: a driver that refuses a pipeline other drivers accept (MoltenVK
|
|
# rejected every skinned one) otherwise leaves a whole layer missing from the frame in silence.
|
|
if gvk_skip_said == null { gvk_skip_said = words(4096); for i in 0 .. 4096 { gvk_skip_said[i] = 0 } }
|
|
if p < 4096 and gvk_skip_said[p] == 0 {
|
|
gvk_skip_said[p] = 1
|
|
print(`r3d: vulkan: no pipeline for {gpu_program_key(p)}; its draws are skipped`)
|
|
}
|
|
return
|
|
}
|
|
var t1: long = 0
|
|
if prof { t1 = gl_now_us(); gvk_us_pipe = gvk_us_pipe + (t1 - t0) }
|
|
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, pipe)
|
|
gvk_set_view(cb)
|
|
let set = gvk_draw_set(p, tx, tx_w, tx_cap)
|
|
if set == 0 { return }
|
|
if prof { gvk_us_set = gvk_us_set + (gl_now_us() - t1) }
|
|
let sets = bytes(8)
|
|
Vk.put_i64(sets, 0, set)
|
|
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, gvk_prog_layout[p], 0, 1, sets, gvk_set_ndyn, gvk_set_offs)
|
|
let v = gvk_prog_var[p]
|
|
if m != null and m.attrs != null {
|
|
# the same bindings, in the same order, as gvk_pipeline gave the layout
|
|
let bufs = bytes(8 * (GPU_MAX_VBUFS + 1))
|
|
let offs = bytes(8 * (GPU_MAX_VBUFS + 1))
|
|
Vk.zero(offs, 8 * (GPU_MAX_VBUFS + 1))
|
|
let seen = words(GPU_MAX_VBUFS)
|
|
var nbd = 0
|
|
for i in 0 .. m.n_attrs {
|
|
let o = i * GPU_ATTR_W
|
|
if m.attrs[o + 1] == 0 { continue }
|
|
var wanted = false
|
|
for q in 0 .. len(v.i_loc) { if v.i_loc[q] == i { wanted = true } }
|
|
if not wanted { continue }
|
|
var known = false
|
|
for q in 0 .. nbd { if seen[q] == m.attrs[o] { known = true } }
|
|
if not known and nbd < GPU_MAX_VBUFS {
|
|
seen[nbd] = m.attrs[o]
|
|
Vk.put_i64(bufs, nbd * 8, gvk_buf[m.attrs[o]])
|
|
gvk_buf_used[m.attrs[o]] = gvk_frame_no
|
|
nbd += 1
|
|
}
|
|
}
|
|
var unfed = false
|
|
for q in 0 .. len(v.i_loc) { if gvk_input_unfed(m, v.i_loc[q]) { unfed = true } }
|
|
if unfed and nbd <= GPU_MAX_VBUFS {
|
|
let zb = gvk_zero_vbuf_get()
|
|
Vk.put_i64(bufs, nbd * 8, gvk_buf[zb])
|
|
gvk_buf_used[zb] = gvk_frame_no
|
|
nbd += 1
|
|
}
|
|
if nbd > 0 { Vk.cmd_bind_vertex_buffers(cb, 0, nbd, bufs, offs) }
|
|
}
|
|
var n = count
|
|
if n == 0 and m != null { n = m.count }
|
|
if gvk_mesh_x > 0 {
|
|
# a mesh-shader dispatch (gpu_draw_mesh_tasks): the device's command, through a pointer
|
|
Vk.sl_call_piii(gvk_mesh_fn, cb, gvk_mesh_x, gvk_mesh_y, gvk_mesh_z)
|
|
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
|
|
return
|
|
}
|
|
if m != null and m.ebo != 0 {
|
|
let zero: long = 0
|
|
var itype = VK_INDEX_TYPE_UINT32
|
|
if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 }
|
|
Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype)
|
|
gvk_buf_used[m.ebo] = gvk_frame_no
|
|
if gvk_ind_buf > 0 {
|
|
# the draws are records in a buffer (gvk_draw_indirect_now); the GPU may have written them
|
|
let ioff: long = gvk_ind_off
|
|
gvk_buf_used[gvk_ind_buf] = gvk_frame_no
|
|
if gvk_ind_cbuf > 0 and gvk_has_dic {
|
|
let coff: long = gvk_ind_coff
|
|
gvk_buf_used[gvk_ind_cbuf] = gvk_frame_no
|
|
Vk.cmd_draw_indexed_indirect_count(cb, gvk_buf[gvk_ind_buf], ioff, gvk_buf[gvk_ind_cbuf], coff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof)
|
|
} else {
|
|
Vk.cmd_draw_indexed_indirect(cb, gvk_buf[gvk_ind_buf], ioff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof)
|
|
}
|
|
} else {
|
|
Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0)
|
|
}
|
|
} else {
|
|
Vk.cmd_draw(cb, n, instances, first, 0)
|
|
}
|
|
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
|
|
}
|
|
|
|
# The frame so far, submitted and waited for, so work that submits on its own - an upload, a
|
|
# read-back, a new or freed image or buffer - happens after the draws recorded before it, in the
|
|
# order OpenGL would have done them. Costs a submit per such call while the backend comes up.
|
|
function gvk_flush() -> void {
|
|
if gvk_cb == null { return }
|
|
gvk_n_flush += 1
|
|
gvk_pass_end()
|
|
gvk_once_end(gvk_cb)
|
|
gvk_cb = null
|
|
gvk_frame_no += 1
|
|
gvk_retire_flush()
|
|
}
|
|
|
|
# The finished frame: submitted and waited for. With a window, the screen image is blitted into
|
|
# the swapchain's next image first - flipped, since the screen image keeps OpenGL's bottom-up rows
|
|
# - and that image is presented.
|
|
function gvk_present() -> void {
|
|
gvk_prof_frame()
|
|
if gvk_swap != 0 { gvk_present_window(); gvk_frame_no += 1; gvk_retire_flush(); return }
|
|
if gvk_cb == null { return }
|
|
gvk_pass_end()
|
|
gvk_once_end(gvk_cb)
|
|
gvk_cb = null
|
|
gvk_frame_no += 1
|
|
gvk_retire_flush()
|
|
}
|
|
|
|
# The screen as a binary PPM, top row first. The frame so far is finished first, then read back;
|
|
# row 0 of the image is OpenGL's bottom row, so rows are written last to first, as gl_screenshot does.
|
|
function gvk_screenshot(path: string) -> bool {
|
|
# the screen image is PQ-encoded 10-bit while HDR is on, which an 8-bit PPM would misread
|
|
if gvk_hdr_on { print("r3d: vulkan: screenshots are SDR only - turn HDR output off to take one"); return false }
|
|
gvk_present()
|
|
let w = gvk_screen_w
|
|
let h = gvk_screen_h
|
|
let px = bytes(w * h * 4)
|
|
if not gvk_tex_read(gvk_screen_color, GL_RGBA8, w, h, GL_RGBA, GL_UNSIGNED_BYTE, px) { return false }
|
|
let f = file_open(path, "wb")
|
|
if f == null { return false }
|
|
let hdr = `P6\n{w} {h}\n255\n`
|
|
file_write(f, hdr, len(hdr))
|
|
let row = bytes(w * 3)
|
|
var y = h - 1
|
|
while y >= 0 {
|
|
for x in 0 .. w {
|
|
let o = (y * w + x) * 4
|
|
row[x * 3] = px[o]; row[x * 3 + 1] = px[o + 1]; row[x * 3 + 2] = px[o + 2]
|
|
}
|
|
file_write(f, row, w * 3)
|
|
y -= 1
|
|
}
|
|
file_close(f)
|
|
return true
|
|
}
|
|
|
|
# ---- what gpu.ludic's Vulkan branches call -----------------------------------------------------
|
|
var gvk_fb_counter: int = 0
|
|
var gvk_prog_counter: int = 0
|
|
var gvk_wireframe: int = 0
|
|
var gvk_state: GvkState = null
|
|
|
|
# the manifest the programs are looked up in, from the renderer's own shader directory
|
|
function gvk_manifest() -> bool {
|
|
r3d_find_root()
|
|
gvk_spv_dir = `{r3d_root}/shaders/spv`
|
|
return gpu_manifest_load(`{gvk_spv_dir}/manifest.txt`) > 0 and len(gpu_variants) > 0
|
|
}
|
|
|
|
# The screen: a colour and a depth image. Headless they are the asked-for size; with a window
|
|
# they are its client area in pixels, and the swapchain is made on it.
|
|
function gvk_open(w: int, h: int, title: string) -> bool {
|
|
gl_w = w
|
|
gl_h = h
|
|
if is_windowed() {
|
|
win_open(w, h, 1, title) # the window exists before main: this retitles it
|
|
win_gl_resize(w, h)
|
|
if gvk_size_buf == null { gvk_size_buf = words(4) }
|
|
win_gl_drawable(gvk_size_buf)
|
|
if gvk_size_buf[0] > 0 and gvk_size_buf[1] > 0 { gl_w = gvk_size_buf[0]; gl_h = gvk_size_buf[1] }
|
|
gl_scale = win_gl_scale()
|
|
}
|
|
if not gvk_frame_init() { return false }
|
|
if not gvk_screen_make(gl_w, gl_h) { return false }
|
|
if r3d_env_has("R3D_VK_PROBE") { gvk_compute_probe() }
|
|
if is_windowed() { return gvk_swap_make(gl_w, gl_h) }
|
|
return true
|
|
}
|
|
|
|
# a program handle for a variant; the key is gpu_program's, so gpu_program_key works on both
|
|
function gvk_program_new(vs: string, fs: string, defines: string) -> int {
|
|
gvk_prog_counter += 1
|
|
let p = gvk_prog_counter
|
|
let key = `{vs}|{fs}|{Text.replace(defines, "\n", ";")}`
|
|
if gpu_prog_ids == null { gpu_prog_ids = new []int; gpu_prog_keys = new []string }
|
|
push(gpu_prog_ids, p)
|
|
push(gpu_prog_keys, key)
|
|
if not gvk_program(p, key, gvk_spv_dir) { return 0 }
|
|
return p
|
|
}
|
|
|
|
function gvk_mesh_free(m: Mesh) -> void {
|
|
if m == null { return }
|
|
if m.vbufs != null { for i in 0 .. m.n_vbufs { if m.vbufs[i] > 0 { gvk_buf_release(m.vbufs[i]) } } }
|
|
m.n_vbufs = 0
|
|
m.vbo = 0
|
|
if m.ebo > 0 { gvk_buf_release(m.ebo); m.ebo = 0 }
|
|
}
|
|
|
|
function gvk_scissor(x: int, y: int, w: int, h: int) -> void {
|
|
if gvk_sc == null { return }
|
|
gvk_sc[0] = 1; gvk_sc[1] = x; gvk_sc[2] = y; gvk_sc[3] = w; gvk_sc[4] = h
|
|
}
|
|
function gvk_scissor_off() -> void { if gvk_sc != null { gvk_sc[0] = 0 } }
|
|
function gvk_viewport(x: int, y: int, w: int, h: int) -> void {
|
|
if gvk_vp == null { return }
|
|
gvk_vp[0] = x; gvk_vp[1] = y; gvk_vp[2] = w; gvk_vp[3] = h
|
|
}
|
|
function gvk_clear_color(r: float, g: float, b: float, a: float) -> void {
|
|
if gvk_clear_rgba == null { return }
|
|
gvk_clear_rgba[0] = r; gvk_clear_rgba[1] = g; gvk_clear_rgba[2] = b; gvk_clear_rgba[3] = a
|
|
}
|
|
function gvk_fb_colors(fb: int, n: int) -> void { if gvk_fb_ncolor != null and fb >= 0 and fb < 4096 { gvk_fb_ncolor[fb] = n } }
|
|
# The framebuffer changes: a clear still waiting for this one's pass runs now, on this target,
|
|
# rather than becoming the load op of whichever pass begins next.
|
|
function gvk_rebind(fb: int) -> void {
|
|
if fb == gvk_fb_cur { return }
|
|
if not gvk_in_pass and gvk_clear_bits != 0 { gvk_pass_begin(gpu_fb, gpu_fb_at(gvk_fb_cur)) }
|
|
gvk_pass_end()
|
|
gvk_fb_cur = fb
|
|
}
|
|
function gvk_fb_forget(fb: int) -> void {
|
|
if fb == gvk_fb_cur { gvk_pass_end() }
|
|
gvk_fb_colors(fb, 1)
|
|
}
|
|
|
|
# the render state gpu.ludic has cached, with OpenGL's defaults where nothing was set yet
|
|
function gvk_state_now() -> GvkState {
|
|
if gvk_state == null { gvk_state = new GvkState }
|
|
let st = gvk_state
|
|
st.depth_test = 0; if gpu_s_depth_test == 1 { st.depth_test = 1 }
|
|
st.depth_write = 1; if gpu_s_depth_write == 0 { st.depth_write = 0 }
|
|
st.depth_func = GL_LESS; if gpu_s_depth_func > 0 { st.depth_func = gpu_s_depth_func }
|
|
st.blend = 0; if gpu_s_blend == 1 { st.blend = 1 }
|
|
st.blend_src = GL_ONE; if gpu_s_blend_src >= 0 { st.blend_src = gpu_s_blend_src }
|
|
st.blend_dst = GL_ZERO; if gpu_s_blend_dst >= 0 { st.blend_dst = gpu_s_blend_dst }
|
|
st.cull = 0; if gpu_s_cull == 1 { st.cull = 1 }
|
|
st.cull_face = GL_BACK; if gpu_s_cull_face > 0 { st.cull_face = gpu_s_cull_face }
|
|
st.color_write = 1; if gpu_s_color_write == 0 { st.color_write = 0 }
|
|
st.a2c = 0; if gpu_s_a2c == 1 { st.a2c = 1 }
|
|
st.bias = 0; if gpu_s_bias == 1 { st.bias = 1 }
|
|
st.bias_factor = float_bits(gpu_s_bias_f)
|
|
st.bias_units = float_bits(gpu_s_bias_u)
|
|
st.wireframe = gvk_wireframe
|
|
return st
|
|
}
|
|
# An indirect draw goes through gvk_draw like any other - the same pipeline, set and buffers -
|
|
# and only its last call differs, so the record is handed over in these for that one draw.
|
|
var gvk_ind_buf: int = 0
|
|
var gvk_ind_off: int = 0
|
|
var gvk_ind_n: int = 0
|
|
var gvk_ind_cbuf: int = 0
|
|
var gvk_ind_coff: int = 0
|
|
# x * y * z mesh-shader invocations with the current program (a *.mesh one) and state
|
|
var gvk_mesh_x: int = 0
|
|
var gvk_mesh_y: int = 0
|
|
var gvk_mesh_z: int = 0
|
|
function gvk_draw_mesh_tasks_now(x: int, y: int, z: int) -> void {
|
|
if not gvk_has_mesh or gvk_mesh_fn == null or x <= 0 or y <= 0 or z <= 0 { return }
|
|
gvk_mesh_x = x; gvk_mesh_y = y; gvk_mesh_z = z
|
|
gvk_draw_now(null, 0, 0, 1)
|
|
gvk_mesh_x = 0
|
|
}
|
|
function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void {
|
|
if m == null or m.ebo == 0 or cmds <= 0 or n <= 0 { return }
|
|
gvk_ind_buf = cmds; gvk_ind_off = offset; gvk_ind_n = n; gvk_ind_cbuf = count_buf; gvk_ind_coff = count_off
|
|
gvk_draw_now(m, 0, 0, 1)
|
|
gvk_ind_buf = 0; gvk_ind_cbuf = 0
|
|
}
|
|
|
|
function gvk_draw_now(m: Mesh, first: int, count: int, instances: int) -> void {
|
|
if gvk_prof() { gvk_n_asked += 1 }
|
|
gvk_draw(gpu_prog_cur, m, gvk_state_now(), first, count, instances, gpu_tx, GPU_TX_W, gpu_tx_cap, gpu_fb, gpu_fb_at(gvk_fb_cur))
|
|
}
|
|
|
|
# the colour (and / or depth) of the read framebuffer into the draw framebuffer, same size
|
|
function gvk_fb_att(fb: int, depth: bool) -> int {
|
|
if fb == 0 { if depth { return gvk_screen_depth }; return gvk_screen_color }
|
|
let o = gpu_fb_at(fb)
|
|
if o < 0 { return 0 }
|
|
if depth { return gpu_fb[o + 2] }
|
|
return gpu_fb[o]
|
|
}
|
|
function gvk_tex_samples_of(tex: int) -> int {
|
|
if tex <= 0 or gvk_tex_samples == null or tex >= len(gvk_tex_samples) or gvk_tex_samples[tex] < 1 { return 1 }
|
|
return gvk_tex_samples[tex]
|
|
}
|
|
# A multisampled image into a single-sampled one: a pass that draws nothing and resolves on its end -
|
|
# colour averaged, depth from sample zero. vkCmdResolveImage cannot resolve depth; a pass can.
|
|
function gvk_resolve(cb: pointer, src: int, dst: int, depth: bool, w: int, h: int) -> void {
|
|
var layout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL
|
|
var mode = VK_RESOLVE_MODE_AVERAGE_BIT
|
|
if depth { layout = VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL; mode = VK_RESOLVE_MODE_SAMPLE_ZERO_BIT }
|
|
gvk_att_barrier(cb, src, 0, depth, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, layout)
|
|
gvk_att_barrier(cb, dst, 0, depth, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, layout)
|
|
let aw = VkRenderingAttachmentInfo_sizeof
|
|
let att = bytes(aw)
|
|
Vk.zero(att, aw)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO)
|
|
Vk.put_i64(att, VkRenderingAttachmentInfo_imageView, gvk_view_of(src, 0))
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_imageLayout, layout)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_resolveMode, mode)
|
|
Vk.put_i64(att, VkRenderingAttachmentInfo_resolveImageView, gvk_view_of(dst, 0))
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_resolveImageLayout, layout)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE)
|
|
let ri = bytes(VkRenderingInfo_sizeof)
|
|
Vk.zero(ri, VkRenderingInfo_sizeof)
|
|
Vk.put_i32(ri, VkRenderingInfo_sType, VK_STRUCTURE_TYPE_RENDERING_INFO)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_width, w)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_height, h)
|
|
Vk.put_i32(ri, VkRenderingInfo_layerCount, 1)
|
|
if depth { Vk.put_ptr(ri, VkRenderingInfo_pDepthAttachment, att) }
|
|
else { Vk.put_i32(ri, VkRenderingInfo_colorAttachmentCount, 1); Vk.put_ptr(ri, VkRenderingInfo_pColorAttachments, att) }
|
|
Vk.cmd_begin_rendering(cb, ri)
|
|
Vk.cmd_end_rendering(cb)
|
|
gvk_att_barrier(cb, src, 0, depth, layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
gvk_att_barrier(cb, dst, 0, depth, layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
function gvk_copy(cb: pointer, src: int, dst: int, depth: bool, w: int, h: int) -> void {
|
|
if src <= 0 or dst <= 0 or gvk_tex_image[src] == 0 or gvk_tex_image[dst] == 0 { return }
|
|
if gvk_tex_samples_of(src) > 1 and gvk_tex_samples_of(dst) == 1 { gvk_resolve(cb, src, dst, depth, w, h); return }
|
|
gvk_barrier(cb, gvk_tex_image[src], depth, 0, 1, gvk_tex_layers[src], VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
|
|
gvk_barrier(cb, gvk_tex_image[dst], depth, 0, 1, gvk_tex_layers[dst], VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
|
let ic = bytes(VkImageCopy_sizeof)
|
|
Vk.zero(ic, VkImageCopy_sizeof)
|
|
var aspect = VK_IMAGE_ASPECT_COLOR_BIT
|
|
if depth { aspect = VK_IMAGE_ASPECT_DEPTH_BIT }
|
|
Vk.put_i32(ic, VkImageCopy_srcSubresource + VkImageSubresourceLayers_aspectMask, aspect)
|
|
Vk.put_i32(ic, VkImageCopy_srcSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
Vk.put_i32(ic, VkImageCopy_dstSubresource + VkImageSubresourceLayers_aspectMask, aspect)
|
|
Vk.put_i32(ic, VkImageCopy_dstSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_width, w)
|
|
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_height, h)
|
|
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_depth, 1)
|
|
Vk.cmd_copy_image(cb, gvk_tex_image[src], VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, gvk_tex_image[dst], VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ic)
|
|
gvk_barrier(cb, gvk_tex_image[src], depth, 0, 1, gvk_tex_layers[src], VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
gvk_barrier(cb, gvk_tex_image[dst], depth, 0, 1, gvk_tex_layers[dst], VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
function gvk_blit(w: int, h: int, mask: int) -> void {
|
|
gvk_pass_end()
|
|
let cb = gvk_frame_cb()
|
|
if (mask & GL_COLOR_BUFFER_BIT) != 0 { gvk_copy(cb, gvk_fb_att(gvk_fb_read, false), gvk_fb_att(gvk_fb_draw, false), false, w, h) }
|
|
if (mask & GL_DEPTH_BUFFER_BIT) != 0 { gvk_copy(cb, gvk_fb_att(gvk_fb_read, true), gvk_fb_att(gvk_fb_draw, true), true, w, h) }
|
|
}
|
|
|
|
# the screen as RGB8, bottom row first, as glReadPixels hands it back (a photograph)
|
|
function gvk_read_screen(w: int, h: int, out: pointer) -> void {
|
|
gvk_present()
|
|
let px = bytes(gvk_screen_w * gvk_screen_h * 4)
|
|
if not gvk_tex_read(gvk_screen_color, GL_RGBA8, gvk_screen_w, gvk_screen_h, GL_RGBA, GL_UNSIGNED_BYTE, px) { return }
|
|
let dst: pointer = out
|
|
for y in 0 .. h {
|
|
for x in 0 .. w {
|
|
let o = (y * gvk_screen_w + x) * 4
|
|
let q = (y * w + x) * 3
|
|
dst[q] = px[o]; dst[q + 1] = px[o + 1]; dst[q + 2] = px[o + 2]
|
|
}
|
|
}
|
|
}
|
|
|
|
# ---- the window: surface and swapchain ----------------------------------------------------------
|
|
# Windows only for now (the Win32 surface); gpu_select keeps a window elsewhere on OpenGL.
|
|
extern function win_native_window() -> pointer = "win_native_window"
|
|
extern function win_native_instance() -> pointer = "win_native_instance"
|
|
|
|
var gvk_surface: long = 0
|
|
var gvk_swap: long = 0
|
|
var gvk_swap_n: int = 0
|
|
var gvk_swap_images: bytes = null
|
|
var gvk_swap_fmt: int = 0
|
|
var gvk_swap_w: int = 0
|
|
var gvk_swap_h: int = 0
|
|
var gvk_vsync: bool = true
|
|
var gvk_acq_fence: bytes = null
|
|
var gvk_swap_stale: bool = false
|
|
|
|
function gvk_surface_make() -> bool {
|
|
if gvk_surface != 0 { return true }
|
|
let hwnd = win_native_window()
|
|
if hwnd == null { gvk_why = "no window to present to"; return false }
|
|
let sci = bytes(VkWin32SurfaceCreateInfoKHR_sizeof)
|
|
Vk.zero(sci, VkWin32SurfaceCreateInfoKHR_sizeof)
|
|
Vk.put_i32(sci, VkWin32SurfaceCreateInfoKHR_sType, VK_STRUCTURE_TYPE_WIN32_SURFACE_CREATE_INFO_KHR)
|
|
Vk.put_ptr(sci, VkWin32SurfaceCreateInfoKHR_hinstance, win_native_instance())
|
|
Vk.put_ptr(sci, VkWin32SurfaceCreateInfoKHR_hwnd, hwnd)
|
|
let out = bytes(8)
|
|
let r = Vk.create_win32_surface_khr(gvk_inst, sci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateWin32SurfaceKHR", r) }
|
|
gvk_surface = gvk_handle(out)
|
|
let ok = bytes(4)
|
|
Vk.put_i32(ok, 0, 0)
|
|
Vk.get_physical_device_surface_support_khr(gvk_pd, gvk_family, gvk_surface, ok)
|
|
if Vk.get_i32(ok, 0) != 1 { gvk_why = "the graphics queue cannot present to the window"; return false }
|
|
let fci = bytes(VkFenceCreateInfo_sizeof)
|
|
Vk.zero(fci, VkFenceCreateInfo_sizeof)
|
|
Vk.put_i32(fci, VkFenceCreateInfo_sType, VK_STRUCTURE_TYPE_FENCE_CREATE_INFO)
|
|
gvk_acq_fence = bytes(8)
|
|
return Vk.create_fence(gvk_dev, fci, null, gvk_acq_fence) == VK_SUCCESS
|
|
}
|
|
|
|
# (Re)make the swapchain for a w x h client area. The old one is handed over and then destroyed.
|
|
function gvk_swap_make(w: int, h: int) -> bool {
|
|
if not gvk_surface_make() { return false }
|
|
Vk.device_wait_idle(gvk_dev)
|
|
let caps = bytes(VkSurfaceCapabilitiesKHR_sizeof)
|
|
Vk.get_physical_device_surface_capabilities_khr(gvk_pd, gvk_surface, caps)
|
|
var ew = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_currentExtent + VkExtent2D_width)
|
|
var eh = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_currentExtent + VkExtent2D_height)
|
|
if ew == -1 or ew <= 0 { ew = w; eh = h }
|
|
if ew <= 0 or eh <= 0 { return false } # minimised: keep the old chain until it has a size
|
|
let cnt = bytes(4)
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_physical_device_surface_formats_khr(gvk_pd, gvk_surface, cnt, null)
|
|
let nf = Vk.get_i32(cnt, 0)
|
|
let fmts = bytes(nf * VkSurfaceFormatKHR_sizeof + 8)
|
|
Vk.get_physical_device_surface_formats_khr(gvk_pd, gvk_surface, cnt, fmts)
|
|
# the screen image holds display-ready 8-bit colour, so the swapchain takes it unconverted
|
|
var fmt = Vk.get_i32(fmts, VkSurfaceFormatKHR_format)
|
|
var cs = Vk.get_i32(fmts, VkSurfaceFormatKHR_colorSpace)
|
|
for i in 0 .. nf {
|
|
let f = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_format)
|
|
let c = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_colorSpace)
|
|
if (f == VK_FORMAT_B8G8R8A8_UNORM or f == VK_FORMAT_R8G8B8A8_UNORM) and c == VK_COLOR_SPACE_SRGB_NONLINEAR_KHR { fmt = f; cs = c }
|
|
}
|
|
var hdr = false
|
|
if gvk_hdr_want and gvk_has_colorspace {
|
|
for i in 0 .. nf {
|
|
let f = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_format)
|
|
let c = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_colorSpace)
|
|
if f == VK_FORMAT_A2B10G10R10_UNORM_PACK32 and c == VK_COLOR_SPACE_HDR10_ST2084_EXT { fmt = f; cs = c; hdr = true }
|
|
}
|
|
if not hdr { print("r3d: vulkan: HDR output asked for, but this display offers no HDR10 swapchain") }
|
|
}
|
|
# the screen image carries the swapchain's depth of colour: remade when HDR comes or goes
|
|
if hdr != gvk_hdr_on { gvk_hdr_on = hdr; gvk_screen_make(gvk_screen_w, gvk_screen_h) }
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_physical_device_surface_present_modes_khr(gvk_pd, gvk_surface, cnt, null)
|
|
let nm = Vk.get_i32(cnt, 0)
|
|
let modes = bytes(nm * 4 + 8)
|
|
Vk.get_physical_device_surface_present_modes_khr(gvk_pd, gvk_surface, cnt, modes)
|
|
# vsync: FIFO, which every device has. Off: mailbox where offered (no tearing), else immediate.
|
|
var mode = VK_PRESENT_MODE_FIFO_KHR
|
|
if not gvk_vsync {
|
|
for i in 0 .. nm { if Vk.get_i32(modes, i * 4) == VK_PRESENT_MODE_IMMEDIATE_KHR { mode = VK_PRESENT_MODE_IMMEDIATE_KHR } }
|
|
for i in 0 .. nm { if Vk.get_i32(modes, i * 4) == VK_PRESENT_MODE_MAILBOX_KHR { mode = VK_PRESENT_MODE_MAILBOX_KHR } }
|
|
}
|
|
var n = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_minImageCount) + 1
|
|
let mx = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_maxImageCount)
|
|
if mx > 0 and n > mx { n = mx }
|
|
let sci = bytes(VkSwapchainCreateInfoKHR_sizeof)
|
|
Vk.zero(sci, VkSwapchainCreateInfoKHR_sizeof)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_sType, VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR)
|
|
Vk.put_i64(sci, VkSwapchainCreateInfoKHR_surface, gvk_surface)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_minImageCount, n)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageFormat, fmt)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageColorSpace, cs)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageExtent + VkExtent2D_width, ew)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageExtent + VkExtent2D_height, eh)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageArrayLayers, 1)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageUsage, VK_IMAGE_USAGE_TRANSFER_DST_BIT)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageSharingMode, VK_SHARING_MODE_EXCLUSIVE)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_preTransform, Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_currentTransform))
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_compositeAlpha, VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_presentMode, mode)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_clipped, 1)
|
|
Vk.put_i64(sci, VkSwapchainCreateInfoKHR_oldSwapchain, gvk_swap)
|
|
let out = bytes(8)
|
|
let r = Vk.create_swapchain_khr(gvk_dev, sci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateSwapchainKHR", r) }
|
|
if gvk_swap != 0 { Vk.destroy_swapchain_khr(gvk_dev, gvk_swap, null) }
|
|
gvk_swap = gvk_handle(out)
|
|
if gvk_hdr_on and gvk_has_hdr_meta { gvk_hdr_metadata() }
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_swapchain_images_khr(gvk_dev, gvk_swap, cnt, null)
|
|
gvk_swap_n = Vk.get_i32(cnt, 0)
|
|
gvk_swap_images = bytes(gvk_swap_n * 8 + 8)
|
|
Vk.get_swapchain_images_khr(gvk_dev, gvk_swap, cnt, gvk_swap_images)
|
|
gvk_swap_fmt = fmt
|
|
gvk_swap_w = ew
|
|
gvk_swap_h = eh
|
|
gvk_swap_stale = false
|
|
var mname = "fifo"
|
|
if mode == VK_PRESENT_MODE_MAILBOX_KHR { mname = "mailbox" }
|
|
if mode == VK_PRESENT_MODE_IMMEDIATE_KHR { mname = "immediate" }
|
|
var cname = "SDR"
|
|
if gvk_hdr_on { cname = "HDR10" }
|
|
print(`r3d: vulkan swapchain {ew}x{eh}, {gvk_swap_n} images, {mname}, {cname}`)
|
|
return true
|
|
}
|
|
|
|
function gvk_present_window() -> void {
|
|
let cb = gvk_frame_cb()
|
|
gvk_pass_end()
|
|
if gvk_swap_stale { gvk_once_end(cb); gvk_cb = null; gvk_swap_make(gvk_swap_w, gvk_swap_h); return }
|
|
let idx = bytes(4)
|
|
Vk.reset_fences(gvk_dev, 1, gvk_acq_fence)
|
|
let forever: long = -1
|
|
let zero: long = 0
|
|
var r = Vk.acquire_next_image_khr(gvk_dev, gvk_swap, forever, zero, Vk.get_i64(gvk_acq_fence, 0), idx)
|
|
if r == VK_ERROR_OUT_OF_DATE_KHR { gvk_once_end(cb); gvk_cb = null; gvk_swap_make(gvk_swap_w, gvk_swap_h); return }
|
|
Vk.wait_for_fences(gvk_dev, 1, gvk_acq_fence, 1, forever)
|
|
let i = Vk.get_i32(idx, 0)
|
|
let dst = Vk.get_i64(gvk_swap_images, i * 8)
|
|
let src = gvk_tex_image[gvk_screen_color]
|
|
gvk_barrier(cb, src, false, 0, 1, 1, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
|
|
gvk_barrier(cb, dst, false, 0, 1, 1, VK_IMAGE_LAYOUT_UNDEFINED, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
|
let blit = bytes(VkImageBlit_sizeof)
|
|
Vk.zero(blit, VkImageBlit_sizeof)
|
|
Vk.put_i32(blit, VkImageBlit_srcSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
|
Vk.put_i32(blit, VkImageBlit_srcSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
# the screen image's row 0 is the bottom of the picture: read it from the top edge down
|
|
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_y, gvk_screen_h)
|
|
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_x, gvk_screen_w)
|
|
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_z, 1)
|
|
Vk.put_i32(blit, VkImageBlit_dstSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
|
Vk.put_i32(blit, VkImageBlit_dstSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_x, gvk_swap_w)
|
|
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_y, gvk_swap_h)
|
|
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_z, 1)
|
|
Vk.cmd_blit_image(cb, src, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dst, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, blit, VK_FILTER_LINEAR)
|
|
gvk_barrier(cb, src, false, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
gvk_barrier(cb, dst, false, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR)
|
|
gvk_once_end(cb)
|
|
gvk_cb = null
|
|
let pi = bytes(VkPresentInfoKHR_sizeof)
|
|
Vk.zero(pi, VkPresentInfoKHR_sizeof)
|
|
Vk.put_i32(pi, VkPresentInfoKHR_sType, VK_STRUCTURE_TYPE_PRESENT_INFO_KHR)
|
|
let chains = bytes(8)
|
|
Vk.put_i64(chains, 0, gvk_swap)
|
|
Vk.put_i32(pi, VkPresentInfoKHR_swapchainCount, 1)
|
|
Vk.put_ptr(pi, VkPresentInfoKHR_pSwapchains, chains)
|
|
Vk.put_ptr(pi, VkPresentInfoKHR_pImageIndices, idx)
|
|
gsl_before_present()
|
|
r = Vk.queue_present_khr(gvk_queue, pi)
|
|
if r == VK_ERROR_OUT_OF_DATE_KHR or r == VK_SUBOPTIMAL_KHR { gvk_swap_stale = true }
|
|
gsl_after_present()
|
|
}
|
|
|
|
# a window's client area changed: the screen images and the swapchain follow it
|
|
var gvk_size_buf: words = null
|
|
function gvk_resize_check() -> bool {
|
|
if gvk_swap == 0 { return false }
|
|
if gvk_size_buf == null { gvk_size_buf = words(4) }
|
|
win_gl_drawable(gvk_size_buf)
|
|
let w = gvk_size_buf[0]
|
|
let h = gvk_size_buf[1]
|
|
if w <= 0 or h <= 0 { return false }
|
|
if w == gl_w and h == gl_h and not gvk_swap_stale { return false }
|
|
gvk_flush()
|
|
gl_w = w
|
|
gl_h = h
|
|
gvk_screen_make(w, h)
|
|
gvk_swap_make(w, h)
|
|
return true
|
|
}
|
|
|
|
# Mipmaps for a texture the renderer asks for mid-frame (the exposure measure reads the HDR scene's
|
|
# smallest level every frame): recorded into the open frame after its pass, so they cost no submit.
|
|
# A chain that has to grow first still goes through its one-shot path.
|
|
function gvk_mips_now(tex: int, w: int, h: int) -> void {
|
|
if gvk_cb == null or gvk_tex_levels[tex] <= 1 { gvk_flush(); gvk_tex_mips(tex, w, h); return }
|
|
gvk_pass_end()
|
|
gvk_tex_mips_into(gvk_cb, tex, w, h)
|
|
}
|
|
|
|
# ---- R3D_VK_PROF: where a Vulkan frame's CPU time goes ---------------------------------------------
|
|
# Every 120 frames: draws and flushes per frame, and milliseconds per frame spent finding pipelines,
|
|
# filling descriptor sets, and inside draws altogether. Off, it costs one flag test a draw.
|
|
var gvk_prof_state: int = -1
|
|
var gvk_us_pipe: long = 0
|
|
var gvk_us_set: long = 0
|
|
var gvk_us_draw: long = 0
|
|
var gvk_n_draws: int = 0
|
|
var gvk_n_asked: int = 0 # draws that reached gvk_draw_now: every one the renderer asked for
|
|
var gvk_n_flush: int = 0
|
|
var gvk_prof_frames: int = 0
|
|
var gvk_n_pipe_new: int = 0 # pipelines made this profile window
|
|
function gvk_prof() -> bool {
|
|
if gvk_prof_state < 0 { gvk_prof_state = 0; if r3d_env_has("R3D_VK_PROF") { gvk_prof_state = 1 } }
|
|
return gvk_prof_state == 1
|
|
}
|
|
function gvk_prof_frame() -> void {
|
|
if not gvk_prof() { return }
|
|
gvk_prof_frames += 1
|
|
if gvk_prof_frames < 120 { return }
|
|
let f = gvk_prof_frames
|
|
let pipe = Text.to_int(string(gvk_us_pipe)) / f
|
|
let set = Text.to_int(string(gvk_us_set)) / f
|
|
let draw = Text.to_int(string(gvk_us_draw)) / f
|
|
print(`r3d: vulkan per frame: {gvk_n_draws / f} draws, {gvk_n_flush / f} flushes; pipelines {pipe / 1000}.{(pipe / 100) % 10} ms, sets {set / 1000}.{(set / 100) % 10} ms, inside draws {draw / 1000}.{(draw / 100) % 10} ms; {gvk_n_pipe_new} pipelines made`)
|
|
# every draw asked for should have been made: a gap is a layer missing from the frame, which is how
|
|
# MoltenVK's refused skinned pipelines went unseen (the drawstats session found it by counting)
|
|
if gvk_n_asked != gvk_n_draws {
|
|
print(`r3d: vulkan: {(gvk_n_asked - gvk_n_draws) / f} draws a frame were asked for and not made ({gvk_n_asked / f} asked, {gvk_n_draws / f} made)`)
|
|
}
|
|
let zero: long = 0
|
|
gvk_us_pipe = zero; gvk_us_set = zero; gvk_us_draw = zero
|
|
gvk_n_draws = 0; gvk_n_asked = 0; gvk_n_flush = 0; gvk_prof_frames = 0; gvk_n_pipe_new = 0
|
|
}
|
|
|
|
# ---- the pipeline cache, by integers -----------------------------------------------------------
|
|
# gvk_pipeline builds and keys pipelines by a string; a draw only needs to find one it already has.
|
|
# A mesh carries an interned layout id (its recorded layout, buffers named by order), the render
|
|
# state packs into one int, and the pass's formats into another; the program's last hit is tried
|
|
# first, then the entries it owns.
|
|
var gvk_layout_keys: []string = null
|
|
var gvk_pc_prog: []int = null
|
|
var gvk_pc_layout: []int = null
|
|
var gvk_pc_state: []int = null
|
|
var gvk_pc_bias: []int = null
|
|
var gvk_pc_pass: []int = null
|
|
var gvk_pc_pipe: []long = null
|
|
var gvk_pc_last: words = null
|
|
|
|
function gvk_layout_id(m: Mesh) -> int {
|
|
if m == null or m.attrs == null { return 1 }
|
|
if m.vk_layout > 0 { return m.vk_layout }
|
|
let key = gvk_layout_key(m)
|
|
if gvk_layout_keys == null { gvk_layout_keys = new []string }
|
|
var id = 0
|
|
for i in 0 .. len(gvk_layout_keys) { if id == 0 and gvk_layout_keys[i] == key { id = i + 2 } }
|
|
if id == 0 { push(gvk_layout_keys, key); id = len(gvk_layout_keys) + 1 }
|
|
m.vk_layout = id
|
|
return id
|
|
}
|
|
function gvk_blend_index(f: int) -> int {
|
|
if f == GL_ZERO { return 0 }
|
|
if f == GL_ONE { return 1 }
|
|
if f == GL_SRC_ALPHA { return 2 }
|
|
if f == GL_ONE_MINUS_SRC_ALPHA { return 3 }
|
|
if f == GL_DST_ALPHA { return 4 }
|
|
if f == GL_ONE_MINUS_DST_ALPHA { return 5 }
|
|
if f == GL_SRC_COLOR { return 6 }
|
|
if f == GL_ONE_MINUS_SRC_COLOR { return 7 }
|
|
return 8
|
|
}
|
|
function gvk_pipeline_fast(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: int, depth_fmt: int, samples: int) -> long {
|
|
let layout = gvk_layout_id(m)
|
|
var state = st.depth_test | (st.depth_write << 1) | (st.blend << 2) | (st.cull << 3) | (st.color_write << 4) | (st.a2c << 5) | (st.bias << 6) | (st.wireframe << 7)
|
|
state = state | ((st.depth_func & 15) << 8) | ((st.cull_face & 15) << 12) | (gvk_blend_index(st.blend_src) << 16) | (gvk_blend_index(st.blend_dst) << 20)
|
|
let bias = st.bias_factor * 31 + st.bias_units
|
|
let pass = ((n_color * 1000 + color_fmt) * 1000 + depth_fmt) * 64 + samples
|
|
if gvk_pc_prog == null {
|
|
gvk_pc_prog = new []int; gvk_pc_layout = new []int; gvk_pc_state = new []int; gvk_pc_bias = new []int
|
|
gvk_pc_pass = new []int; gvk_pc_pipe = new []long
|
|
gvk_pc_last = words(4096)
|
|
for i in 0 .. 4096 { gvk_pc_last[i] = -1 }
|
|
}
|
|
if p < 4096 {
|
|
let k = gvk_pc_last[p]
|
|
if k >= 0 and gvk_pc_layout[k] == layout and gvk_pc_state[k] == state and gvk_pc_pass[k] == pass and gvk_pc_bias[k] == bias { return gvk_pc_pipe[k] }
|
|
}
|
|
for k in 0 .. len(gvk_pc_prog) {
|
|
if gvk_pc_prog[k] == p and gvk_pc_layout[k] == layout and gvk_pc_state[k] == state and gvk_pc_pass[k] == pass and gvk_pc_bias[k] == bias {
|
|
if p < 4096 { gvk_pc_last[p] = k }
|
|
return gvk_pc_pipe[k]
|
|
}
|
|
}
|
|
let pipe = gvk_pipeline(p, m, st, n_color, color_fmt, depth_fmt, samples)
|
|
if pipe == 0 { return pipe }
|
|
push(gvk_pc_prog, p); push(gvk_pc_layout, layout); push(gvk_pc_state, state)
|
|
push(gvk_pc_bias, bias); push(gvk_pc_pass, pass); push(gvk_pc_pipe, pipe)
|
|
if p < 4096 { gvk_pc_last[p] = len(gvk_pc_prog) - 1 }
|
|
return pipe
|
|
}
|