r3d_hdr_calibrate(peak, paper, black) feeds the tonemap's and the overlay's HDR10 variants and the display's HDR metadata; ov_hdr_nits draws a calibration patch at a number of nits. gpu_caps_probe asks the running Vulkan renderer's instance instead of making and destroying a second one under Streamline's interposer, which left the next swapchain rebuild calling address 0. The overlay gets an HDR10 variant, and the tonemap brightens HDR highlights by one factor rather than per channel. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
1914 lines
98 KiB
Text
1914 lines
98 KiB
Text
# gpu_vk_draw.ludic — the Vulkan backend's drawing: programs as manifest variants, pipelines
|
|
# built from what OpenGL decides at the draw, uniform blocks and descriptor sets per draw, and the
|
|
# frame's passes with dynamic rendering, through to the present and the screenshot.
|
|
#
|
|
# It reads render3d's own records - the SPIR-V manifest (gpu_manifest.ludic), a Mesh's recorded
|
|
# vertex layout, gpu.ludic's texture and framebuffer records - so it is compiled only inside
|
|
# render3d, after gpu_vk.ludic and gpu_vk_res.ludic.
|
|
|
|
# ---- programs -----------------------------------------------------------------------------
|
|
# A program handle is a manifest variant on Vulkan: its two SPIR-V modules, a descriptor set
|
|
# layout (binding 0 the vertex stage's uniform block, 1 the fragment stage's, the samplers at
|
|
# the manifest's bindings from 2) and the pipeline layout over it. Programs are never compiled
|
|
# here; one whose variant is not in the manifest cannot draw, and says so once.
|
|
var gvk_prog_var: []GpuVariant = null
|
|
var gvk_prog_vs: []long = null
|
|
var gvk_prog_fs: []long = null
|
|
var gvk_prog_dsl: []long = null
|
|
var gvk_prog_layout: []long = null
|
|
var gvk_spv_dir: string = ""
|
|
|
|
function gvk_read_spv(path: string) -> bytes {
|
|
let f = file_open(path, "rb")
|
|
if f == null { return null }
|
|
file_seek(f, 0, 2)
|
|
let n = file_tell(f)
|
|
file_seek(f, 0, 0)
|
|
let b = bytes(n + 4)
|
|
file_read(f, b, n)
|
|
file_close(f)
|
|
gvk_spv_len = n
|
|
return b
|
|
}
|
|
var gvk_spv_len: int = 0
|
|
|
|
function gvk_module(path: string) -> long {
|
|
let zero: long = 0
|
|
let spv = gvk_read_spv(path)
|
|
if spv == null { print(`r3d: vulkan: no SPIR-V at {path}`); return zero }
|
|
let smci = bytes(VkShaderModuleCreateInfo_sizeof)
|
|
Vk.zero(smci, VkShaderModuleCreateInfo_sizeof)
|
|
Vk.put_i32(smci, VkShaderModuleCreateInfo_sType, VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO)
|
|
let code_size: long = gvk_spv_len
|
|
Vk.put_i64(smci, VkShaderModuleCreateInfo_codeSize, code_size)
|
|
Vk.put_ptr(smci, VkShaderModuleCreateInfo_pCode, spv)
|
|
let out = bytes(8)
|
|
let r = Vk.create_shader_module(gvk_dev, smci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateShaderModule {path}`, r); return zero }
|
|
return gvk_handle(out)
|
|
}
|
|
|
|
# The Vulkan side of a program handle the renderer already made (gpu_program): the handle's
|
|
# manifest key finds the variant. Returns false when there is no such variant.
|
|
# a program whose first stage is a mesh shader (its first file is *.mesh): its pipeline has no vertex
|
|
# input, and its first stage's bindings are the mesh stage's
|
|
var gvk_prog_mesh: []int = null
|
|
function gvk_stage_first(p: int) -> int {
|
|
if gvk_prog_mesh != null and p < len(gvk_prog_mesh) and gvk_prog_mesh[p] == 1 { return VK_SHADER_STAGE_MESH_BIT_EXT }
|
|
return VK_SHADER_STAGE_VERTEX_BIT
|
|
}
|
|
function gvk_program(p: int, key: string, spv_dir: string) -> bool {
|
|
let zero: long = 0
|
|
if gvk_prog_var == null {
|
|
gvk_prog_var = new []GpuVariant; gvk_prog_vs = new []long; gvk_prog_fs = new []long
|
|
gvk_prog_dsl = new []long; gvk_prog_layout = new []long; gvk_prog_mesh = new []int
|
|
}
|
|
while len(gvk_prog_var) <= p {
|
|
push(gvk_prog_var, null); push(gvk_prog_vs, zero); push(gvk_prog_fs, zero); push(gvk_prog_dsl, zero); push(gvk_prog_layout, zero); push(gvk_prog_mesh, 0)
|
|
}
|
|
let parts = Text.split(key, "|")
|
|
var defs = ""
|
|
if len(parts) > 2 { defs = parts[2] }
|
|
let v = gpu_variant_find_key(parts[0], parts[1], defs)
|
|
if v == null { print(`r3d: vulkan: no SPIR-V variant for {key}`); return false }
|
|
let vs = gvk_module(`{spv_dir}/{v.id}.vert.spv`)
|
|
if Text.ends_with(v.vs, ".mesh") { gvk_prog_mesh[p] = 1 } else { gvk_prog_mesh[p] = 0 }
|
|
let fs = gvk_module(`{spv_dir}/{v.id}.frag.spv`)
|
|
if vs == 0 or fs == 0 { return false }
|
|
|
|
let nt = len(v.t_name)
|
|
var nb = nt
|
|
if v.vblock >= 0 { nb += 1 }
|
|
if v.fblock >= 0 { nb += 1 }
|
|
let bw = VkDescriptorSetLayoutBinding_sizeof
|
|
let binds = bytes(bw * (nb + 1))
|
|
Vk.zero(binds, bw * (nb + 1))
|
|
var k = 0
|
|
if v.vblock >= 0 {
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, 0)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p))
|
|
k += 1
|
|
}
|
|
if v.fblock >= 0 {
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_FRAGMENT_BIT)
|
|
k += 1
|
|
}
|
|
for t in 0 .. nt {
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, v.t_bind[t])
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, gvk_stage_first(p) | VK_SHADER_STAGE_FRAGMENT_BIT)
|
|
k += 1
|
|
}
|
|
let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.zero(dslci, VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_bindingCount, nb)
|
|
Vk.put_ptr(dslci, VkDescriptorSetLayoutCreateInfo_pBindings, binds)
|
|
let dsl = bytes(8)
|
|
var r = Vk.create_descriptor_set_layout(gvk_dev, dslci, null, dsl)
|
|
if r != VK_SUCCESS { return gvk_fail(`vkCreateDescriptorSetLayout for {key}`, r) }
|
|
let plci = bytes(VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.zero(plci, VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_setLayoutCount, 1)
|
|
Vk.put_ptr(plci, VkPipelineLayoutCreateInfo_pSetLayouts, dsl)
|
|
let out = bytes(8)
|
|
r = Vk.create_pipeline_layout(gvk_dev, plci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail(`vkCreatePipelineLayout for {key}`, r) }
|
|
gvk_prog_var[p] = v; gvk_prog_vs[p] = vs; gvk_prog_fs[p] = fs
|
|
gvk_prog_dsl[p] = Vk.get_i64(dsl, 0); gvk_prog_layout[p] = gvk_handle(out)
|
|
return true
|
|
}
|
|
|
|
# ---- compute ------------------------------------------------------------------------------
|
|
# A compute program is <name>.comp.spv beside the manifest, with bindings fixed by convention:
|
|
# 0 the parameter block (a uniform buffer, copied into the frame's ring at the dispatch) and
|
|
# 1 .. n storage buffers. Handles start at 1.
|
|
var gvk_cp_pipe: []long = null
|
|
var gvk_cp_layout: []long = null
|
|
var gvk_cp_dsl: []long = null
|
|
var gvk_cp_nbuf: []int = null
|
|
|
|
function gvk_compute_new(name: string, n_bufs: int) -> int {
|
|
let zero: long = 0
|
|
if gvk_cp_pipe == null {
|
|
gvk_cp_pipe = new []long; gvk_cp_layout = new []long; gvk_cp_dsl = new []long; gvk_cp_nbuf = new []int
|
|
push(gvk_cp_pipe, zero); push(gvk_cp_layout, zero); push(gvk_cp_dsl, zero); push(gvk_cp_nbuf, 0)
|
|
}
|
|
let module = gvk_module(`{gvk_spv_dir}/{name}.comp.spv`)
|
|
if module == 0 { return 0 }
|
|
let nb = n_bufs + 1
|
|
let bw = VkDescriptorSetLayoutBinding_sizeof
|
|
let binds = bytes(bw * nb)
|
|
Vk.zero(binds, bw * nb)
|
|
for k in 0 .. nb {
|
|
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
|
|
if k == 0 { kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_binding, k)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorType, kind)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_descriptorCount, 1)
|
|
Vk.put_i32(binds, k * bw + VkDescriptorSetLayoutBinding_stageFlags, VK_SHADER_STAGE_COMPUTE_BIT)
|
|
}
|
|
let dslci = bytes(VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.zero(dslci, VkDescriptorSetLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(dslci, VkDescriptorSetLayoutCreateInfo_bindingCount, nb)
|
|
Vk.put_ptr(dslci, VkDescriptorSetLayoutCreateInfo_pBindings, binds)
|
|
let dsl = bytes(8)
|
|
var r = Vk.create_descriptor_set_layout(gvk_dev, dslci, null, dsl)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateDescriptorSetLayout for compute {name}`, r); return 0 }
|
|
let plci = bytes(VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.zero(plci, VkPipelineLayoutCreateInfo_sizeof)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO)
|
|
Vk.put_i32(plci, VkPipelineLayoutCreateInfo_setLayoutCount, 1)
|
|
Vk.put_ptr(plci, VkPipelineLayoutCreateInfo_pSetLayouts, dsl)
|
|
let out = bytes(8)
|
|
r = Vk.create_pipeline_layout(gvk_dev, plci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreatePipelineLayout for compute {name}`, r); return 0 }
|
|
let layout = gvk_handle(out)
|
|
let cpci = bytes(VkComputePipelineCreateInfo_sizeof)
|
|
Vk.zero(cpci, VkComputePipelineCreateInfo_sizeof)
|
|
Vk.put_i32(cpci, VkComputePipelineCreateInfo_sType, VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO)
|
|
let so = VkComputePipelineCreateInfo_stage
|
|
Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
|
|
Vk.put_i32(cpci, so + VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_COMPUTE_BIT)
|
|
Vk.put_i64(cpci, so + VkPipelineShaderStageCreateInfo_module, module)
|
|
Vk.put_ptr(cpci, so + VkPipelineShaderStageCreateInfo_pName, "main")
|
|
Vk.put_i64(cpci, VkComputePipelineCreateInfo_layout, layout)
|
|
r = Vk.create_compute_pipelines(gvk_dev, zero, 1, cpci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateComputePipelines {name}`, r); return 0 }
|
|
push(gvk_cp_pipe, gvk_handle(out)); push(gvk_cp_layout, layout); push(gvk_cp_dsl, Vk.get_i64(dsl, 0)); push(gvk_cp_nbuf, n_bufs)
|
|
return len(gvk_cp_pipe) - 1
|
|
}
|
|
|
|
# Record a dispatch of compute program c into the frame, between passes: params (n_params
|
|
# bytes, std140) as binding 0 and bufs[0 .. n) as bindings 1 .. n.
|
|
function gvk_dispatch(c: int, params: pointer, n_params: int, bufs: words, gx: int, gy: int, gz: int) -> void {
|
|
let zero: long = 0
|
|
if gvk_cp_pipe == null or c <= 0 or c >= len(gvk_cp_pipe) { return }
|
|
gvk_pass_end()
|
|
let cb = gvk_frame_cb()
|
|
let nb = gvk_cp_nbuf[c]
|
|
let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof)
|
|
Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO)
|
|
Vk.put_i64(dsai, VkDescriptorSetAllocateInfo_descriptorPool, gvk_dpool)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_descriptorSetCount, 1)
|
|
let layouts = bytes(8)
|
|
Vk.put_i64(layouts, 0, gvk_cp_dsl[c])
|
|
Vk.put_ptr(dsai, VkDescriptorSetAllocateInfo_pSetLayouts, layouts)
|
|
let sets = bytes(8)
|
|
let r = Vk.allocate_descriptor_sets(gvk_dev, dsai, sets)
|
|
if r != VK_SUCCESS { gvk_fail("vkAllocateDescriptorSets (compute)", r); return }
|
|
let set = Vk.get_i64(sets, 0)
|
|
let at = gvk_ring_put(params, n_params)
|
|
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return }
|
|
let ww = VkWriteDescriptorSet_sizeof
|
|
let bw = VkDescriptorBufferInfo_sizeof
|
|
let writes = bytes(ww * (nb + 1))
|
|
Vk.zero(writes, ww * (nb + 1))
|
|
let infos = bytes(bw * (nb + 1))
|
|
Vk.zero(infos, bw * (nb + 1))
|
|
for k in 0 .. nb + 1 {
|
|
var buf = gvk_ring_buf
|
|
var off: long = 0
|
|
var range: long = 0
|
|
var kind = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER
|
|
if k == 0 { off = at; range = n_params; kind = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER }
|
|
else { buf = bufs[k - 1]; range = gvk_buf_size[buf]; gvk_buf_used[buf] = gvk_frame_no }
|
|
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_buffer, gvk_buf[buf])
|
|
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_offset, off)
|
|
Vk.put_i64(infos, k * bw + VkDescriptorBufferInfo_range, range)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
|
|
Vk.put_i64(writes, k * ww + VkWriteDescriptorSet_dstSet, set)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_dstBinding, k)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorCount, 1)
|
|
Vk.put_i32(writes, k * ww + VkWriteDescriptorSet_descriptorType, kind)
|
|
Vk.put_ptr(writes, k * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(infos, k * bw))
|
|
}
|
|
Vk.update_descriptor_sets(gvk_dev, nb + 1, writes, 0, null)
|
|
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_pipe[c])
|
|
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_COMPUTE, gvk_cp_layout[c], 0, 1, sets, 0, null)
|
|
Vk.cmd_dispatch(cb, gx, gy, gz)
|
|
}
|
|
|
|
# R3D_VK_PROBE=1: one dispatch over a GPU-owned buffer, read back - the compute path end to end
|
|
function gvk_compute_probe() -> void {
|
|
let c = gvk_compute_new("probe", 1)
|
|
if c == 0 { print("r3d: vulkan compute probe: FAILED (no program)"); return }
|
|
let n = 1000
|
|
let b = gvk_buf_new()
|
|
let data = bytes(n * 4)
|
|
for i in 0 .. n { Vk.put_i32(data, i * 4, fi(i)) }
|
|
gvk_buf_upload(b, n * 4, data)
|
|
gvk_buf_gpu_owned(b)
|
|
let pr = bytes(8)
|
|
Vk.put_i32(pr, 0, n)
|
|
Vk.put_i32(pr, 4, fi(3))
|
|
let bufs = words(1)
|
|
bufs[0] = b
|
|
gvk_dispatch(c, pr, 8, bufs, (n + 63) / 64, 1, 1)
|
|
gvk_flush()
|
|
var bad = 0
|
|
let mp = gvk_buf_map[b]
|
|
for i in 0 .. n { if Vk.get_i32(mp, i * 4) != fi(i * 3) { bad += 1 } }
|
|
if bad == 0 { print(`r3d: vulkan compute probe OK ({n} values)`) } else { print(`r3d: vulkan compute probe: FAILED ({bad} of {n} wrong)`) }
|
|
}
|
|
|
|
# ---- pipelines ----------------------------------------------------------------------------
|
|
# A pipeline is everything OpenGL decides at the draw: the program, the vertex layout the mesh
|
|
# recorded, the render state the renderer set, and the formats and sample count of the pass it
|
|
# draws into. Each distinct combination is built once, the first time it is drawn.
|
|
property GvkState {
|
|
depth_test: int = 0,
|
|
depth_write: int = 1,
|
|
depth_func: int = 0x0201, # GL_LESS
|
|
blend: int = 0,
|
|
blend_src: int = 1, # GL_ONE
|
|
blend_dst: int = 0, # GL_ZERO
|
|
cull: int = 0,
|
|
cull_face: int = 0x0405, # GL_BACK
|
|
color_write: int = 1,
|
|
a2c: int = 0,
|
|
bias: int = 0,
|
|
bias_factor: int = 0, # float bits
|
|
bias_units: int = 0, # float bits
|
|
wireframe: int = 0
|
|
}
|
|
var gvk_pipe_keys: []string = null
|
|
var gvk_pipe: []long = null
|
|
|
|
# does the mesh leave shader input `loc` unfed (no attribute recorded there)?
|
|
function gvk_input_unfed(m: Mesh, loc: int) -> bool {
|
|
if m == null or m.attrs == null or loc < 0 or loc >= m.n_attrs { return true }
|
|
return m.attrs[loc * GPU_ATTR_W + 1] == 0
|
|
}
|
|
# the float format a zero-fed shader input reads, from its manifest type
|
|
function gvk_input_format(t: string) -> int {
|
|
if t == "float" { return VK_FORMAT_R32_SFLOAT }
|
|
if t == "vec2" { return VK_FORMAT_R32G32_SFLOAT }
|
|
if t == "vec3" { return VK_FORMAT_R32G32B32_SFLOAT }
|
|
return VK_FORMAT_R32G32B32A32_SFLOAT
|
|
}
|
|
# 64 KB of zeros, the buffer every unfed input reads (one vec4 per instance, up to 4096)
|
|
var gvk_zero_vbuf: int = 0
|
|
var gvk_skip_said: words = null # per program: its skipped draws have been reported
|
|
function gvk_zero_vbuf_get() -> int {
|
|
if gvk_zero_vbuf == 0 {
|
|
gvk_zero_vbuf = gvk_buf_new()
|
|
let z = bytes(65536)
|
|
Vk.zero(z, 65536)
|
|
gvk_buf_upload(gvk_zero_vbuf, 65536, z)
|
|
}
|
|
return gvk_zero_vbuf
|
|
}
|
|
|
|
function gvk_attr_format(comps: int, type: int, normalized: int) -> int {
|
|
if type == GPU_U8 {
|
|
if normalized == 1 { if comps == 4 { return VK_FORMAT_R8G8B8A8_UNORM }; if comps == 3 { return VK_FORMAT_R8G8B8_UNORM }; if comps == 2 { return VK_FORMAT_R8G8_UNORM }; return VK_FORMAT_R8_UNORM }
|
|
# not normalised, read by a float input (a_joints is a vec4): OpenGL converts the integer to a
|
|
# float, and Vulkan's form of that is USCALED - a UINT format against a float input is invalid
|
|
if comps == 4 { return VK_FORMAT_R8G8B8A8_USCALED }; if comps == 2 { return VK_FORMAT_R8G8_USCALED }; return VK_FORMAT_R8_USCALED
|
|
}
|
|
if type == GPU_U16 {
|
|
if normalized == 1 { if comps == 4 { return VK_FORMAT_R16G16B16A16_UNORM }; if comps == 2 { return VK_FORMAT_R16G16_UNORM }; return VK_FORMAT_R16_UNORM }
|
|
if comps == 4 { return VK_FORMAT_R16G16B16A16_USCALED }; if comps == 2 { return VK_FORMAT_R16G16_USCALED }; return VK_FORMAT_R16_USCALED
|
|
}
|
|
if comps == 4 { return VK_FORMAT_R32G32B32A32_SFLOAT }
|
|
if comps == 3 { return VK_FORMAT_R32G32B32_SFLOAT }
|
|
if comps == 2 { return VK_FORMAT_R32G32_SFLOAT }
|
|
return VK_FORMAT_R32_SFLOAT
|
|
}
|
|
function gvk_blend_factor(f: int) -> int {
|
|
if f == GL_ONE { return VK_BLEND_FACTOR_ONE }
|
|
if f == GL_SRC_ALPHA { return VK_BLEND_FACTOR_SRC_ALPHA }
|
|
if f == GL_ONE_MINUS_SRC_ALPHA { return VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA }
|
|
if f == GL_DST_ALPHA { return VK_BLEND_FACTOR_DST_ALPHA }
|
|
if f == GL_ONE_MINUS_DST_ALPHA { return VK_BLEND_FACTOR_ONE_MINUS_DST_ALPHA }
|
|
if f == GL_SRC_COLOR { return VK_BLEND_FACTOR_SRC_COLOR }
|
|
if f == GL_ONE_MINUS_SRC_COLOR { return VK_BLEND_FACTOR_ONE_MINUS_SRC_COLOR }
|
|
return VK_BLEND_FACTOR_ZERO
|
|
}
|
|
function gvk_depth_op(f: int) -> int {
|
|
if f == GL_LEQUAL { return VK_COMPARE_OP_LESS_OR_EQUAL }
|
|
if f == GL_EQUAL { return VK_COMPARE_OP_EQUAL }
|
|
if f == GL_ALWAYS { return VK_COMPARE_OP_ALWAYS }
|
|
if f == GL_GREATER { return VK_COMPARE_OP_GREATER }
|
|
if f == GL_GEQUAL { return VK_COMPARE_OP_GREATER_OR_EQUAL }
|
|
if f == GL_NEVER { return VK_COMPARE_OP_NEVER }
|
|
if f == GL_NOTEQUAL { return VK_COMPARE_OP_NOT_EQUAL }
|
|
return VK_COMPARE_OP_LESS
|
|
}
|
|
# the vertex layout a mesh recorded (gpu.ludic's attrs), as part of a pipeline key
|
|
# Buffers are named by the order they are first read in, not by handle: a scatter mesh re-pointed
|
|
# at another instance buffer keeps its layout, and so its pipeline.
|
|
function gvk_layout_key(m: Mesh) -> string {
|
|
if m == null or m.attrs == null { return "none" }
|
|
var k = ""
|
|
let seen = words(GPU_MAX_ATTRS)
|
|
var ns = 0
|
|
for i in 0 .. m.n_attrs {
|
|
let o = i * GPU_ATTR_W
|
|
if m.attrs[o + 1] == 0 { continue }
|
|
var bi = -1
|
|
for q in 0 .. ns { if seen[q] == m.attrs[o] and bi < 0 { bi = q } }
|
|
if bi < 0 { bi = ns; seen[ns] = m.attrs[o]; ns += 1 }
|
|
k = k + `{i}:{bi}:{m.attrs[o + 1]}:{m.attrs[o + 2]}:{m.attrs[o + 3]}:{m.attrs[o + 4]}:{m.attrs[o + 5]}:{m.attrs[o + 6]};`
|
|
}
|
|
return k
|
|
}
|
|
|
|
# The pipeline for program p drawing mesh m (null for a draw without vertex input, like the
|
|
# full-screen triangle) with state st into a pass of n_color colour attachments of format
|
|
# color_fmt, a depth attachment of depth_fmt (VK_FORMAT_UNDEFINED for none) at `samples`.
|
|
function gvk_pipeline(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: int, depth_fmt: int, samples: int) -> long {
|
|
let zero: long = 0
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return zero }
|
|
let key = `{p}|{gvk_layout_key(m)}|{st.depth_test},{st.depth_write},{st.depth_func},{st.blend},{st.blend_src},{st.blend_dst},{st.cull},{st.cull_face},{st.color_write},{st.a2c},{st.bias},{st.bias_factor},{st.bias_units},{st.wireframe}|{n_color},{color_fmt},{depth_fmt},{samples}`
|
|
if gvk_pipe_keys == null { gvk_pipe_keys = new []string; gvk_pipe = new []long }
|
|
for i in 0 .. len(gvk_pipe_keys) { if gvk_pipe_keys[i] == key { return gvk_pipe[i] } }
|
|
|
|
let ss = VkPipelineShaderStageCreateInfo_sizeof
|
|
let stages = bytes(ss * 2)
|
|
Vk.zero(stages, ss * 2)
|
|
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
|
|
Vk.put_i32(stages, VkPipelineShaderStageCreateInfo_stage, gvk_stage_first(p))
|
|
Vk.put_i64(stages, VkPipelineShaderStageCreateInfo_module, gvk_prog_vs[p])
|
|
Vk.put_ptr(stages, VkPipelineShaderStageCreateInfo_pName, "main")
|
|
Vk.put_i32(stages, ss + VkPipelineShaderStageCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO)
|
|
Vk.put_i32(stages, ss + VkPipelineShaderStageCreateInfo_stage, VK_SHADER_STAGE_FRAGMENT_BIT)
|
|
Vk.put_i64(stages, ss + VkPipelineShaderStageCreateInfo_module, gvk_prog_fs[p])
|
|
Vk.put_ptr(stages, ss + VkPipelineShaderStageCreateInfo_pName, "main")
|
|
|
|
# vertex input: one binding per distinct buffer the mesh's attributes read, in the order met;
|
|
# an attribute the shader does not read is left out
|
|
let aw = VkVertexInputAttributeDescription_sizeof
|
|
let bdw = VkVertexInputBindingDescription_sizeof
|
|
let attrs = bytes(aw * (GPU_MAX_ATTRS + 1))
|
|
let bnds = bytes(bdw * (GPU_MAX_VBUFS + 1))
|
|
Vk.zero(attrs, aw * (GPU_MAX_ATTRS + 1))
|
|
Vk.zero(bnds, bdw * (GPU_MAX_VBUFS + 1))
|
|
var na = 0
|
|
var nbd = 0
|
|
if m != null and m.attrs != null {
|
|
let bufs = words(GPU_MAX_VBUFS)
|
|
for i in 0 .. m.n_attrs {
|
|
let o = i * GPU_ATTR_W
|
|
if m.attrs[o + 1] == 0 { continue }
|
|
var wanted = false
|
|
for q in 0 .. len(v.i_loc) { if v.i_loc[q] == i { wanted = true } }
|
|
if not wanted { continue }
|
|
var bi = -1
|
|
for q in 0 .. nbd { if bufs[q] == m.attrs[o] and bi < 0 { bi = q } }
|
|
if bi < 0 and nbd < GPU_MAX_VBUFS {
|
|
bi = nbd
|
|
bufs[nbd] = m.attrs[o]
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_binding, nbd)
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_stride, m.attrs[o + 3])
|
|
if m.attrs[o + 6] == 1 { Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_inputRate, VK_VERTEX_INPUT_RATE_INSTANCE) }
|
|
nbd += 1
|
|
}
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_location, i)
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_binding, bi)
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_format, gvk_attr_format(m.attrs[o + 1], m.attrs[o + 2], m.attrs[o + 5]))
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_offset, m.attrs[o + 4])
|
|
na += 1
|
|
}
|
|
}
|
|
# A shader input the mesh does not provide (a_joints on a mesh without a skin) reads zeros from a
|
|
# shared buffer, as OpenGL's disabled attribute reads its default. Vulkan requires every input the
|
|
# vertex stage declares to be fed (VUID-VkGraphicsPipelineCreateInfo-Input-07904).
|
|
var need_zero = false
|
|
for q in 0 .. len(v.i_loc) {
|
|
if not gvk_input_unfed(m, v.i_loc[q]) or na >= GPU_MAX_ATTRS { continue }
|
|
if not need_zero {
|
|
need_zero = true
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_binding, nbd)
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_stride, 16)
|
|
Vk.put_i32(bnds, nbd * bdw + VkVertexInputBindingDescription_inputRate, VK_VERTEX_INPUT_RATE_INSTANCE)
|
|
}
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_location, v.i_loc[q])
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_binding, nbd)
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_format, gvk_input_format(v.i_type[q]))
|
|
Vk.put_i32(attrs, na * aw + VkVertexInputAttributeDescription_offset, 0)
|
|
na += 1
|
|
}
|
|
if need_zero { nbd += 1 }
|
|
let vin = bytes(VkPipelineVertexInputStateCreateInfo_sizeof)
|
|
Vk.zero(vin, VkPipelineVertexInputStateCreateInfo_sizeof)
|
|
Vk.put_i32(vin, VkPipelineVertexInputStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO)
|
|
Vk.put_i32(vin, VkPipelineVertexInputStateCreateInfo_vertexBindingDescriptionCount, nbd)
|
|
Vk.put_ptr(vin, VkPipelineVertexInputStateCreateInfo_pVertexBindingDescriptions, bnds)
|
|
Vk.put_i32(vin, VkPipelineVertexInputStateCreateInfo_vertexAttributeDescriptionCount, na)
|
|
Vk.put_ptr(vin, VkPipelineVertexInputStateCreateInfo_pVertexAttributeDescriptions, attrs)
|
|
|
|
let ias = bytes(VkPipelineInputAssemblyStateCreateInfo_sizeof)
|
|
Vk.zero(ias, VkPipelineInputAssemblyStateCreateInfo_sizeof)
|
|
Vk.put_i32(ias, VkPipelineInputAssemblyStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO)
|
|
Vk.put_i32(ias, VkPipelineInputAssemblyStateCreateInfo_topology, VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST)
|
|
let vps = bytes(VkPipelineViewportStateCreateInfo_sizeof)
|
|
Vk.zero(vps, VkPipelineViewportStateCreateInfo_sizeof)
|
|
Vk.put_i32(vps, VkPipelineViewportStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO)
|
|
Vk.put_i32(vps, VkPipelineViewportStateCreateInfo_viewportCount, 1)
|
|
Vk.put_i32(vps, VkPipelineViewportStateCreateInfo_scissorCount, 1)
|
|
|
|
let rs = bytes(VkPipelineRasterizationStateCreateInfo_sizeof)
|
|
Vk.zero(rs, VkPipelineRasterizationStateCreateInfo_sizeof)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO)
|
|
if st.wireframe == 1 { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_polygonMode, VK_POLYGON_MODE_LINE) } else { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_polygonMode, VK_POLYGON_MODE_FILL) }
|
|
if st.cull == 1 {
|
|
if st.cull_face == GL_FRONT { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_cullMode, VK_CULL_MODE_FRONT_BIT) } else { Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_cullMode, VK_CULL_MODE_BACK_BIT) }
|
|
}
|
|
# no y flip between the APIs' clip spaces and their framebuffers' rows, so a triangle that is
|
|
# counter-clockwise on OpenGL's bottom-up window is clockwise in Vulkan's top-down one
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_frontFace, VK_FRONT_FACE_CLOCKWISE)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_lineWidth, 0x3F800000)
|
|
if st.bias == 1 {
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_depthBiasEnable, 1)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_depthBiasSlopeFactor, st.bias_factor)
|
|
Vk.put_i32(rs, VkPipelineRasterizationStateCreateInfo_depthBiasConstantFactor, st.bias_units)
|
|
}
|
|
let ms = bytes(VkPipelineMultisampleStateCreateInfo_sizeof)
|
|
Vk.zero(ms, VkPipelineMultisampleStateCreateInfo_sizeof)
|
|
Vk.put_i32(ms, VkPipelineMultisampleStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO)
|
|
Vk.put_i32(ms, VkPipelineMultisampleStateCreateInfo_rasterizationSamples, samples)
|
|
# OpenGL ignores alpha to coverage without a multisampled target; Vulkan with one sample would
|
|
# instead drop every fragment under half alpha, which erased the meadow's flowers
|
|
if samples > 1 { Vk.put_i32(ms, VkPipelineMultisampleStateCreateInfo_alphaToCoverageEnable, st.a2c) }
|
|
let ds = bytes(VkPipelineDepthStencilStateCreateInfo_sizeof)
|
|
Vk.zero(ds, VkPipelineDepthStencilStateCreateInfo_sizeof)
|
|
Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO)
|
|
if depth_fmt != VK_FORMAT_UNDEFINED {
|
|
Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_depthTestEnable, st.depth_test)
|
|
# OpenGL writes no depth with the test off, whatever the mask says
|
|
if st.depth_test == 1 { Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_depthWriteEnable, st.depth_write) }
|
|
Vk.put_i32(ds, VkPipelineDepthStencilStateCreateInfo_depthCompareOp, gvk_depth_op(st.depth_func))
|
|
}
|
|
let cbw = VkPipelineColorBlendAttachmentState_sizeof
|
|
let cba = bytes(cbw * (n_color + 1))
|
|
Vk.zero(cba, cbw * (n_color + 1))
|
|
for c in 0 .. n_color {
|
|
if st.color_write == 1 { Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_colorWriteMask, 15) }
|
|
if st.blend == 1 {
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_blendEnable, 1)
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_srcColorBlendFactor, gvk_blend_factor(st.blend_src))
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_dstColorBlendFactor, gvk_blend_factor(st.blend_dst))
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_srcAlphaBlendFactor, gvk_blend_factor(st.blend_src))
|
|
Vk.put_i32(cba, c * cbw + VkPipelineColorBlendAttachmentState_dstAlphaBlendFactor, gvk_blend_factor(st.blend_dst))
|
|
}
|
|
}
|
|
let cbs = bytes(VkPipelineColorBlendStateCreateInfo_sizeof)
|
|
Vk.zero(cbs, VkPipelineColorBlendStateCreateInfo_sizeof)
|
|
Vk.put_i32(cbs, VkPipelineColorBlendStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO)
|
|
Vk.put_i32(cbs, VkPipelineColorBlendStateCreateInfo_attachmentCount, n_color)
|
|
Vk.put_ptr(cbs, VkPipelineColorBlendStateCreateInfo_pAttachments, cba)
|
|
let dyn_states = bytes(8)
|
|
Vk.put_i32(dyn_states, 0, VK_DYNAMIC_STATE_VIEWPORT)
|
|
Vk.put_i32(dyn_states, 4, VK_DYNAMIC_STATE_SCISSOR)
|
|
let dys = bytes(VkPipelineDynamicStateCreateInfo_sizeof)
|
|
Vk.zero(dys, VkPipelineDynamicStateCreateInfo_sizeof)
|
|
Vk.put_i32(dys, VkPipelineDynamicStateCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO)
|
|
Vk.put_i32(dys, VkPipelineDynamicStateCreateInfo_dynamicStateCount, 2)
|
|
Vk.put_ptr(dys, VkPipelineDynamicStateCreateInfo_pDynamicStates, dyn_states)
|
|
let formats = bytes(4 * (n_color + 1))
|
|
for c in 0 .. n_color { Vk.put_i32(formats, c * 4, color_fmt) }
|
|
let prci = bytes(VkPipelineRenderingCreateInfo_sizeof)
|
|
Vk.zero(prci, VkPipelineRenderingCreateInfo_sizeof)
|
|
Vk.put_i32(prci, VkPipelineRenderingCreateInfo_sType, VK_STRUCTURE_TYPE_PIPELINE_RENDERING_CREATE_INFO)
|
|
Vk.put_i32(prci, VkPipelineRenderingCreateInfo_colorAttachmentCount, n_color)
|
|
Vk.put_ptr(prci, VkPipelineRenderingCreateInfo_pColorAttachmentFormats, formats)
|
|
Vk.put_i32(prci, VkPipelineRenderingCreateInfo_depthAttachmentFormat, depth_fmt)
|
|
|
|
let gpci = bytes(VkGraphicsPipelineCreateInfo_sizeof)
|
|
Vk.zero(gpci, VkGraphicsPipelineCreateInfo_sizeof)
|
|
Vk.put_i32(gpci, VkGraphicsPipelineCreateInfo_sType, VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pNext, prci)
|
|
Vk.put_i32(gpci, VkGraphicsPipelineCreateInfo_stageCount, 2)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pStages, stages)
|
|
# a mesh-shader pipeline has neither: the mesh stage makes its own vertices
|
|
if gvk_stage_first(p) == VK_SHADER_STAGE_VERTEX_BIT {
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pVertexInputState, vin)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pInputAssemblyState, ias)
|
|
}
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pViewportState, vps)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pRasterizationState, rs)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pMultisampleState, ms)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pDepthStencilState, ds)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pColorBlendState, cbs)
|
|
Vk.put_ptr(gpci, VkGraphicsPipelineCreateInfo_pDynamicState, dys)
|
|
Vk.put_i64(gpci, VkGraphicsPipelineCreateInfo_layout, gvk_prog_layout[p])
|
|
let out = bytes(8)
|
|
let r = Vk.create_graphics_pipelines(gvk_dev, zero, 1, gpci, null, out)
|
|
if r != VK_SUCCESS { gvk_fail(`vkCreateGraphicsPipelines for {v.vs} + {v.fs}`, r); return zero }
|
|
let pipe = gvk_handle(out)
|
|
# R3D_VK_PROF names each pipeline as it is made: one made during play is a stall a warm-up missed
|
|
if gvk_prof() { gvk_n_pipe_new += 1; print(`r3d: vulkan pipeline {len(gvk_pipe) + 1}: {v.vs} + {v.fs}, {n_color} colour format {color_fmt}, depth {depth_fmt}, {samples}x`) }
|
|
push(gvk_pipe_keys, key)
|
|
push(gvk_pipe, pipe)
|
|
return pipe
|
|
}
|
|
|
|
# ---- uniforms -----------------------------------------------------------------------------
|
|
# On Vulkan a loose uniform is a place in its program's uniform block: glslang's relaxed mode
|
|
# gathered each stage's loose uniforms into one block and the manifest says where each one sits.
|
|
# A uniform "location" is program * 4096 + the index of its first manifest entry, so -1 still
|
|
# means "not in this program". Writing one writes every stage that declares it; the blocks go to
|
|
# the GPU at the draw.
|
|
var gvk_ublk_v: []bytes = null # per program: the vertex stage's block as last set
|
|
var gvk_ublk_f: []bytes = null
|
|
var gvk_prog_tex: []words = null # per program: the texture bound to each manifest sampler
|
|
var gvk_prog_unit: []words = null # per program: the unit + 1 each sampler was pointed at with u_i
|
|
var gvk_prog_dirty: []int = null # per program: its blocks or textures changed since its last set
|
|
var gvk_set_last: []long = null # per program: the descriptor set its last draw used
|
|
var gvk_set_frame: []int = null # per program: the frame that set belongs to
|
|
var gvk_set_tex: []words = null # per program: the textures that set was filled with
|
|
|
|
function gvk_same(a: pointer, b: pointer, n: int) -> bool {
|
|
var i = 0
|
|
while i < n { if Vk.get_i32(a, i) != Vk.get_i32(b, i) { return false }; i += 4 }
|
|
return true
|
|
}
|
|
|
|
function gvk_uniform_blocks(p: int) -> void {
|
|
if gvk_ublk_v == null {
|
|
gvk_ublk_v = new []bytes; gvk_ublk_f = new []bytes; gvk_prog_tex = new []words; gvk_prog_unit = new []words
|
|
gvk_prog_dirty = new []int; gvk_set_last = new []long; gvk_set_frame = new []int; gvk_set_tex = new []words
|
|
}
|
|
let zero: long = 0
|
|
while len(gvk_ublk_v) <= p {
|
|
push(gvk_ublk_v, null); push(gvk_ublk_f, null); push(gvk_prog_tex, null); push(gvk_prog_unit, null)
|
|
push(gvk_prog_dirty, 1); push(gvk_set_last, zero); push(gvk_set_frame, 0); push(gvk_set_tex, null)
|
|
}
|
|
let v = gvk_prog_var[p]
|
|
if v == null or gvk_ublk_v[p] != null or gvk_prog_tex[p] != null { return }
|
|
if v.vblock > 0 { let b = bytes(v.vblock); Vk.zero(b, v.vblock); gvk_ublk_v[p] = b }
|
|
if v.fblock > 0 { let b = bytes(v.fblock); Vk.zero(b, v.fblock); gvk_ublk_f[p] = b }
|
|
let nt = len(v.t_name) + 1
|
|
let t = words(nt)
|
|
let u = words(nt)
|
|
for i in 0 .. nt { t[i] = 0; u[i] = 0 }
|
|
gvk_prog_tex[p] = t
|
|
gvk_prog_unit[p] = u
|
|
let st = words(nt)
|
|
for i in 0 .. nt { st[i] = -1 }
|
|
gvk_set_tex[p] = st
|
|
}
|
|
|
|
function gvk_uniform(p: int, name: string) -> int {
|
|
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) { return -1 }
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return -1 }
|
|
for i in 0 .. len(v.u_name) { if v.u_name[i] == name { return p * 4096 + i } }
|
|
# a sampler: OpenGL code points it at a texture unit once with u_i and then only binds units
|
|
# (the actors do), so its location is kept apart and u_i on it records the unit
|
|
for i in 0 .. len(v.t_name) { if v.t_name[i] == name { return p * 4096 + 2048 + i } }
|
|
return -1
|
|
}
|
|
|
|
# n elements of `size` bytes each from src into every block that declares the uniform at loc
|
|
function gvk_u_set(loc: int, src: pointer, size: int, n: int) -> void {
|
|
if loc < 0 { return }
|
|
let p = loc / 4096
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return }
|
|
gvk_uniform_blocks(p)
|
|
if loc % 4096 >= 2048 {
|
|
let si = loc % 4096 - 2048
|
|
if si < len(v.t_name) and gvk_prog_unit[p][si] != Vk.get_i32(src, 0) + 1 { gvk_prog_unit[p][si] = Vk.get_i32(src, 0) + 1; gvk_prog_dirty[p] = 1 }
|
|
return
|
|
}
|
|
let name = v.u_name[loc % 4096]
|
|
for j in 0 .. len(v.u_name) {
|
|
if v.u_name[j] != name { continue }
|
|
var blk = gvk_ublk_v[p]
|
|
var cap = v.vblock
|
|
if v.u_stage[j] == 1 { blk = gvk_ublk_f[p]; cap = v.fblock }
|
|
if blk == null { continue }
|
|
var stride = v.u_stride[j]
|
|
if stride == 0 { stride = size }
|
|
var count = n
|
|
if count > v.u_count[j] { count = v.u_count[j] }
|
|
for k in 0 .. count {
|
|
let at = v.u_off[j] + k * stride
|
|
# the same value again (most per-draw uniforms repeat) leaves the program's set reusable
|
|
if at + size <= cap and not gvk_same(mem_off(blk, at), mem_off(src, k * size), size) {
|
|
mem_copy(mem_off(blk, at), mem_off(src, k * size), size)
|
|
gvk_prog_dirty[p] = 1
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
# a sampler uniform: the texture for the manifest binding with that name
|
|
function gvk_bind_texture(p: int, name: string, tex: int) -> void {
|
|
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) { return }
|
|
let v = gvk_prog_var[p]
|
|
if v == null { return }
|
|
gvk_uniform_blocks(p)
|
|
for i in 0 .. len(v.t_name) { if v.t_name[i] == name and gvk_prog_tex[p][i] != tex { gvk_prog_tex[p][i] = tex; gvk_prog_dirty[p] = 1 } }
|
|
}
|
|
|
|
# ---- per-frame uniform ring and descriptor sets -------------------------------------------
|
|
# Each draw's blocks are copied into one host-visible ring buffer at the device's alignment and
|
|
# its descriptor set comes from a pool that is reset with the frame. Both are rewound at
|
|
# gvk_frame_reset.
|
|
const GVK_RING_BYTES: int = 64 * 1024 * 1024
|
|
var gvk_ring_buf: int = 0 # a gvk_buf handle
|
|
var gvk_ring_off: int = 0
|
|
var gvk_ring_align: int = 256
|
|
var gvk_dpool: long = 0
|
|
var gvk_white: int = 0 # a 1x1 white texture for a sampler nothing was bound to
|
|
var gvk_msaa_max: int = 0 # samples the scene may use (gvk_frame_init reads the device's limits)
|
|
|
|
function gvk_frame_init() -> bool {
|
|
let props = bytes(VkPhysicalDeviceProperties_sizeof)
|
|
Vk.get_physical_device_properties(gvk_pd, props)
|
|
let al = Vk.get_i64(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_minUniformBufferOffsetAlignment)
|
|
gvk_ring_align = Text.to_int(string(al))
|
|
if gvk_ring_align < 16 { gvk_ring_align = 16 }
|
|
# MSAA: the most samples (up to the 4 the scene asks for) both a colour and a depth target can take
|
|
let lim = VkPhysicalDeviceProperties_limits
|
|
let counts = Vk.get_i32(props, lim + VkPhysicalDeviceLimits_framebufferColorSampleCounts) & Vk.get_i32(props, lim + VkPhysicalDeviceLimits_framebufferDepthSampleCounts)
|
|
gvk_msaa_max = 1
|
|
if (counts & VK_SAMPLE_COUNT_2_BIT) != 0 { gvk_msaa_max = 2 }
|
|
if (counts & VK_SAMPLE_COUNT_4_BIT) != 0 { gvk_msaa_max = 4 }
|
|
gvk_ring_buf = gvk_buf_new()
|
|
if not gvk_buf_reserve(gvk_ring_buf, GVK_RING_BYTES) { return false }
|
|
let sizes = bytes(VkDescriptorPoolSize_sizeof * 3)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_descriptorCount, 32768)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_descriptorCount, 131072)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_STORAGE_BUFFER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof * 2 + VkDescriptorPoolSize_descriptorCount, 16384)
|
|
let dpci = bytes(VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.zero(dpci, VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_maxSets, 16384)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 3)
|
|
Vk.put_ptr(dpci, VkDescriptorPoolCreateInfo_pPoolSizes, sizes)
|
|
let out = bytes(8)
|
|
let r = Vk.create_descriptor_pool(gvk_dev, dpci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateDescriptorPool", r) }
|
|
gvk_dpool = gvk_handle(out)
|
|
if not gvk_kpool_make() { return false }
|
|
gvk_white = gvk_tex_new()
|
|
let px = bytes(4)
|
|
px[0] = 255; px[1] = 255; px[2] = 255; px[3] = 255
|
|
return gvk_tex_storage(gvk_white, false, GL_RGBA8, 1, 1, 1, false) and gvk_tex_upload(gvk_white, GL_RGBA8, 1, 1, 1, GL_RGBA, GL_UNSIGNED_BYTE, px)
|
|
}
|
|
|
|
function gvk_frame_reset() -> void {
|
|
gvk_ring_off = 0
|
|
Vk.reset_descriptor_pool(gvk_dev, gvk_dpool, 0)
|
|
}
|
|
|
|
# a block into the ring; its offset, or -1 when the frame has used the whole ring
|
|
function gvk_ring_put(blk: pointer, n: int) -> int {
|
|
let at = (gvk_ring_off + gvk_ring_align - 1) / gvk_ring_align * gvk_ring_align
|
|
if at + n > GVK_RING_BYTES { return -1 }
|
|
mem_copy(mem_off(gvk_buf_map[gvk_ring_buf], at), blk, n)
|
|
gvk_ring_off = at + n
|
|
return at
|
|
}
|
|
|
|
# ---- descriptor sets that survive the frame -------------------------------------------------
|
|
# A draw's set carries its textures; its uniform blocks are dynamic uniform buffers into the
|
|
# frame's ring, bound with this draw's offsets. So one set serves every draw of a program with the
|
|
# same textures, this frame and the frames after it, and a draw that changes only uniforms allocates
|
|
# and writes no set at all - which was about half of a Vulkan frame's CPU time (8.5 ms at the camp).
|
|
# The sets live in a pool of their own that is only reset when it fills. The key holds each
|
|
# texture's handle and generation (bumped when the image behind a handle is replaced) and its
|
|
# sampler, so a set is never matched against a texture it no longer describes.
|
|
const GVK_KPOOL_SETS: int = 8192
|
|
var gvk_kpool: long = 0
|
|
var gvk_sc_prog: []int = null # per cached set: its program
|
|
var gvk_sc_koff: []int = null # per cached set: where its key starts in gvk_sc_keys
|
|
var gvk_sc_set: []long = null
|
|
var gvk_sc_next: []int = null # the program's next cached set, -1 at the end
|
|
var gvk_sc_keys: []long = null # per texture of a key: handle * 65536 + generation, sampler
|
|
var gvk_sc_head: words = null # per program (< 4096): its most recently made set, -1 if none
|
|
var gvk_sc_tmp: []long = null # this draw's key while it is looked up
|
|
var gvk_set_offs: bytes = null # this draw's dynamic offsets, in binding order
|
|
var gvk_set_ndyn: int = 0
|
|
var gvk_ub_frame: []int = null # per program: the frame its blocks were last copied into the ring
|
|
var gvk_ub_offv: []int = null
|
|
var gvk_ub_offf: []int = null
|
|
var gvk_sc_made: int = 0 # sets made since start (R3D_VK_PROF)
|
|
var gvk_default_smp: long = 0 # linear, clamped: for a texture slot nothing was bound to
|
|
|
|
function gvk_kpool_make() -> bool {
|
|
let sizes = bytes(VkDescriptorPoolSize_sizeof * 2)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_descriptorCount, GVK_KPOOL_SETS * 2)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_type, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_i32(sizes, VkDescriptorPoolSize_sizeof + VkDescriptorPoolSize_descriptorCount, GVK_KPOOL_SETS * 8)
|
|
let dpci = bytes(VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.zero(dpci, VkDescriptorPoolCreateInfo_sizeof)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_maxSets, GVK_KPOOL_SETS)
|
|
Vk.put_i32(dpci, VkDescriptorPoolCreateInfo_poolSizeCount, 2)
|
|
Vk.put_ptr(dpci, VkDescriptorPoolCreateInfo_pPoolSizes, sizes)
|
|
let out = bytes(8)
|
|
let r = Vk.create_descriptor_pool(gvk_dev, dpci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateDescriptorPool (kept sets)", r) }
|
|
gvk_kpool = gvk_handle(out)
|
|
gvk_sc_clear()
|
|
return true
|
|
}
|
|
|
|
function gvk_sc_clear() -> void {
|
|
gvk_sc_prog = new []int; gvk_sc_koff = new []int; gvk_sc_set = new []long; gvk_sc_next = new []int
|
|
gvk_sc_keys = new []long
|
|
if gvk_sc_head == null { gvk_sc_head = words(4096) }
|
|
for i in 0 .. 4096 { gvk_sc_head[i] = -1 }
|
|
}
|
|
|
|
# the pool is full: finish the frame recorded so far (it may use sets from it), then start again
|
|
function gvk_kpool_reset() -> void {
|
|
gvk_flush()
|
|
Vk.reset_descriptor_pool(gvk_dev, gvk_kpool, 0)
|
|
gvk_sc_clear()
|
|
}
|
|
|
|
function gvk_draw_set(p: int, tx: words, tx_w: int, tx_cap: int) -> long {
|
|
let zero: long = 0
|
|
let v = gvk_prog_var[p]
|
|
gvk_uniform_blocks(p)
|
|
if gvk_sc_tmp == null {
|
|
gvk_sc_tmp = new []long
|
|
gvk_set_offs = bytes(16)
|
|
gvk_ub_frame = new []int; gvk_ub_offv = new []int; gvk_ub_offf = new []int
|
|
}
|
|
while len(gvk_ub_frame) <= p { push(gvk_ub_frame, 0); push(gvk_ub_offv, 0); push(gvk_ub_offf, 0) }
|
|
let nt = len(v.t_name)
|
|
# the key: every texture as this draw resolves it, and its sampler
|
|
while len(gvk_sc_tmp) < nt * 2 + 2 { push(gvk_sc_tmp, zero) }
|
|
for t in 0 .. nt {
|
|
var tex = gvk_prog_tex[p][t]
|
|
# nothing bound by name: the texture on the unit the sampler was pointed at, as OpenGL reads it
|
|
let unit1 = gvk_prog_unit[p][t]
|
|
if tex <= 0 and unit1 > 0 and unit1 <= 32 and gpu_unit_2d != null { tex = gpu_unit_2d[unit1 - 1] }
|
|
if tex <= 0 or tex >= len(gvk_tex_image) or gvk_tex_image[tex] == 0 { tex = gvk_white }
|
|
var smp: long = 0
|
|
let o = tex * tx_w
|
|
if tex != gvk_white and tex < tx_cap {
|
|
smp = gvk_tex_sampler(tex, tx[o + 5], tx[o + 6], tx[o + 7], tx[o + 8], tx[o + 9], tx[o + 11])
|
|
} else {
|
|
# the sampler for a slot nothing was bound to, made once: looking it up by its string key on
|
|
# every draw was the costly part of the key for the shadow and depth variants
|
|
if gvk_default_smp == 0 { gvk_default_smp = gvk_sampler(GL_LINEAR, GL_LINEAR, GL_CLAMP_TO_EDGE, GL_CLAMP_TO_EDGE, 0, 0) }
|
|
smp = gvk_default_smp
|
|
}
|
|
let tg: long = tex * 65536 + gvk_tex_gen[tex]
|
|
gvk_sc_tmp[t * 2] = tg
|
|
gvk_sc_tmp[t * 2 + 1] = smp
|
|
}
|
|
var set: long = 0
|
|
var e = -1
|
|
if p < 4096 { e = gvk_sc_head[p] }
|
|
while e >= 0 and set == 0 {
|
|
let ko = gvk_sc_koff[e]
|
|
var same = true
|
|
var t = 0
|
|
while same and t < nt * 2 { if gvk_sc_keys[ko + t] != gvk_sc_tmp[t] { same = false }; t += 1 }
|
|
if same { set = gvk_sc_set[e] } else { e = gvk_sc_next[e] }
|
|
}
|
|
if set == 0 {
|
|
set = gvk_sc_make(p, nt)
|
|
if set == 0 { return zero }
|
|
}
|
|
# the uniform blocks: copied into the ring once per frame, and again only when a value changed
|
|
if gvk_prog_dirty[p] == 1 or gvk_ub_frame[p] != gvk_frame_no {
|
|
if gvk_ublk_v[p] != null and v.vblock > 0 {
|
|
let at = gvk_ring_put(gvk_ublk_v[p], v.vblock)
|
|
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return zero }
|
|
gvk_ub_offv[p] = at
|
|
}
|
|
if gvk_ublk_f[p] != null and v.fblock > 0 {
|
|
let at = gvk_ring_put(gvk_ublk_f[p], v.fblock)
|
|
if at < 0 { print("r3d: vulkan: the frame's uniform ring is full"); return zero }
|
|
gvk_ub_offf[p] = at
|
|
}
|
|
gvk_ub_frame[p] = gvk_frame_no
|
|
gvk_prog_dirty[p] = 0
|
|
}
|
|
gvk_set_ndyn = 0
|
|
if v.vblock >= 0 { Vk.put_i32(gvk_set_offs, gvk_set_ndyn * 4, gvk_ub_offv[p]); gvk_set_ndyn += 1 }
|
|
if v.fblock >= 0 { Vk.put_i32(gvk_set_offs, gvk_set_ndyn * 4, gvk_ub_offf[p]); gvk_set_ndyn += 1 }
|
|
gvk_buf_used[gvk_ring_buf] = gvk_frame_no
|
|
return set
|
|
}
|
|
|
|
# a new kept set for program p with the textures in gvk_sc_tmp, written and cached
|
|
function gvk_sc_make(p: int, nt: int) -> long {
|
|
let zero: long = 0
|
|
let v = gvk_prog_var[p]
|
|
let dsai = bytes(VkDescriptorSetAllocateInfo_sizeof)
|
|
let layouts = bytes(8)
|
|
let sets = bytes(8)
|
|
var r = 0
|
|
var tries = 0
|
|
while tries < 2 {
|
|
Vk.zero(dsai, VkDescriptorSetAllocateInfo_sizeof)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_sType, VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO)
|
|
Vk.put_i64(dsai, VkDescriptorSetAllocateInfo_descriptorPool, gvk_kpool)
|
|
Vk.put_i32(dsai, VkDescriptorSetAllocateInfo_descriptorSetCount, 1)
|
|
Vk.put_i64(layouts, 0, gvk_prog_dsl[p])
|
|
Vk.put_ptr(dsai, VkDescriptorSetAllocateInfo_pSetLayouts, layouts)
|
|
r = Vk.allocate_descriptor_sets(gvk_dev, dsai, sets)
|
|
if r == VK_SUCCESS { tries = 2 } else { gvk_kpool_reset(); tries += 1 }
|
|
}
|
|
if r != VK_SUCCESS { gvk_fail("vkAllocateDescriptorSets (kept sets)", r); return zero }
|
|
let set = Vk.get_i64(sets, 0)
|
|
let ww = VkWriteDescriptorSet_sizeof
|
|
let writes = bytes(ww * (nt + 3))
|
|
Vk.zero(writes, ww * (nt + 3))
|
|
let bis = bytes(VkDescriptorBufferInfo_sizeof * 2)
|
|
let iis = bytes(VkDescriptorImageInfo_sizeof * (nt + 1))
|
|
var nw = 0
|
|
var bi = 0
|
|
for stage in 0 .. 2 {
|
|
var size = v.vblock
|
|
if stage == 1 { size = v.fblock }
|
|
if size < 0 { continue }
|
|
let at = bi * VkDescriptorBufferInfo_sizeof
|
|
let off0: long = 0
|
|
var range: long = size
|
|
if size == 0 { range = 16 }
|
|
Vk.put_i64(bis, at + VkDescriptorBufferInfo_buffer, gvk_buf[gvk_ring_buf])
|
|
Vk.put_i64(bis, at + VkDescriptorBufferInfo_offset, off0)
|
|
Vk.put_i64(bis, at + VkDescriptorBufferInfo_range, range)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
|
|
Vk.put_i64(writes, nw * ww + VkWriteDescriptorSet_dstSet, set)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_dstBinding, stage)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorCount, 1)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorType, VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC)
|
|
Vk.put_ptr(writes, nw * ww + VkWriteDescriptorSet_pBufferInfo, mem_off(bis, at))
|
|
nw += 1
|
|
bi += 1
|
|
}
|
|
let iw = VkDescriptorImageInfo_sizeof
|
|
for t in 0 .. nt {
|
|
let tex = Text.to_int(string(gvk_sc_tmp[t * 2] / 65536))
|
|
Vk.put_i64(iis, t * iw + VkDescriptorImageInfo_sampler, gvk_sc_tmp[t * 2 + 1])
|
|
Vk.put_i64(iis, t * iw + VkDescriptorImageInfo_imageView, gvk_tex_view[tex])
|
|
Vk.put_i32(iis, t * iw + VkDescriptorImageInfo_imageLayout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_sType, VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET)
|
|
Vk.put_i64(writes, nw * ww + VkWriteDescriptorSet_dstSet, set)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_dstBinding, v.t_bind[t])
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorCount, 1)
|
|
Vk.put_i32(writes, nw * ww + VkWriteDescriptorSet_descriptorType, VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER)
|
|
Vk.put_ptr(writes, nw * ww + VkWriteDescriptorSet_pImageInfo, mem_off(iis, t * iw))
|
|
nw += 1
|
|
}
|
|
if nw > 0 { Vk.update_descriptor_sets(gvk_dev, nw, writes, 0, null) }
|
|
let e = len(gvk_sc_set)
|
|
push(gvk_sc_prog, p); push(gvk_sc_koff, len(gvk_sc_keys)); push(gvk_sc_set, set)
|
|
for t in 0 .. nt * 2 { push(gvk_sc_keys, gvk_sc_tmp[t]) }
|
|
var head = -1
|
|
if p < 4096 { head = gvk_sc_head[p]; gvk_sc_head[p] = e }
|
|
push(gvk_sc_next, head)
|
|
gvk_sc_made += 1
|
|
return set
|
|
}
|
|
|
|
# ---- the frame and its passes -------------------------------------------------------------
|
|
# One command buffer records the whole frame. Binding a framebuffer ends the pass in progress;
|
|
# the next one begins at its first clear or draw, so a clear that comes first becomes the pass's
|
|
# load op. For the length of a pass its attachments sit in the attachment layouts, and between
|
|
# passes every image is back in SHADER_READ_ONLY where samplers expect it. The present submits
|
|
# and waits: the frame is finished before the next one starts, as it is on OpenGL.
|
|
#
|
|
# Framebuffer 0 is the screen: a colour and a depth image made at gvk_screen_make. Headless that
|
|
# is all the screen there is; with a window the present copies it to the swapchain.
|
|
var gvk_cb: pointer = null
|
|
var gvk_in_pass: bool = false
|
|
var gvk_fb_cur: int = 0
|
|
var gvk_fb_read: int = 0
|
|
var gvk_fb_draw: int = 0
|
|
var gvk_pass_w: int = 0
|
|
var gvk_pass_h: int = 0
|
|
var gvk_pass_ncolor: int = 0
|
|
var gvk_pass_cfmt: int = 0
|
|
var gvk_pass_dfmt: int = 0
|
|
var gvk_pass_samples: int = 1 # the attachments' sample count, which every pipeline in the pass must match
|
|
var gvk_pass_col: words = null # the pass's colour attachments: texture, layer + 1
|
|
var gvk_pass_dep: words = null # its depth attachment: texture, layer + 1
|
|
var gvk_clear_bits: int = 0 # GL_COLOR_BUFFER_BIT / GL_DEPTH_BUFFER_BIT waiting for the pass
|
|
var gvk_clear_rgba: words = null # float bits
|
|
var gvk_vp: words = null # x, y, w, h
|
|
var gvk_sc: words = null # scissor on, x, y (OpenGL rows), w, h
|
|
var gvk_screen_color: int = 0
|
|
var gvk_screen_depth: int = 0
|
|
var gvk_screen_w: int = 0
|
|
var gvk_screen_h: int = 0
|
|
var gvk_layer_views: []string = null # "tex:layer" -> index into gvk_layer_view
|
|
var gvk_layer_view: []long = null
|
|
var gvk_fb_ncolor: words = null # per framebuffer: colour slots drawn (draw buffers; 0 = none)
|
|
|
|
# ---- HDR output ------------------------------------------------------------------------------
|
|
# HDR10 (ST 2084 over BT.2020) when the player asks for it and the surface offers it. The screen
|
|
# image and the tonemap's LDR image go 10-bit with it; the tonemap and the overlay switch to their
|
|
# HDR10 variants (gpu_hdr_active). Headless there is no surface, so never.
|
|
var gvk_hdr_want: bool = false # r3d_hdr: the setting
|
|
var gvk_hdr_on: bool = false # the swapchain is HDR10 now
|
|
var gvk_has_colorspace: bool = false # the instance took VK_EXT_swapchain_colorspace
|
|
var gvk_has_hdr_meta: bool = false # the device took VK_EXT_hdr_metadata
|
|
# R3D_HDR=1 / 0 overrides the setting, for a desktop test
|
|
function r3d_hdr(on: bool) -> void {
|
|
var want = on
|
|
if Os.has_env("R3D_HDR") { want = Text.to_int(Os.env("R3D_HDR")) != 0 }
|
|
if want == gvk_hdr_want { return }
|
|
gvk_hdr_want = want
|
|
if gvk_swap != 0 { gvk_swap_stale = true }
|
|
}
|
|
function gpu_hdr_active() -> bool { return gpu_kind == GPU_VK and gvk_hdr_on }
|
|
function gvk_screen_fmt() -> int { if gvk_hdr_on { return GL_RGB10_A2 }; return GL_RGBA8 }
|
|
|
|
# The player's calibration of their display, in nits (float bits): the brightest it shows, where the
|
|
# picture's and the interface's white sit, and how far the darkest shade is lifted. Displays differ by
|
|
# an order of magnitude - a 400-nit monitor and a 2000-nit television - and one fixed curve either
|
|
# clips the first's highlights flat or leaves the second dim.
|
|
var r3d_hdr_peak: int = 0 # 0 until r3d_hdr_calibrate: 1000
|
|
var r3d_hdr_paper: int = 0 # 200
|
|
var r3d_hdr_black: int = 0 # 0
|
|
function r3d_hdr_peak_nits() -> int { if r3d_hdr_peak == 0 { return fi(1000) }; return r3d_hdr_peak }
|
|
function r3d_hdr_paper_nits() -> int { if r3d_hdr_paper == 0 { return fi(200) }; return r3d_hdr_paper }
|
|
function r3d_hdr_black_nits() -> int { return r3d_hdr_black }
|
|
function r3d_hdr_calibrate(peak: int, paper: int, black: int) -> void {
|
|
var pk = f_clamp(peak, fi(100), fi(10000))
|
|
let pp = f_clamp(paper, fi(80), fi(1000))
|
|
if f_ls(pk, pp) { pk = pp }
|
|
let bl = f_clamp(black, F_ZERO, fi(5))
|
|
if pk == r3d_hdr_peak and pp == r3d_hdr_paper and bl == r3d_hdr_black { return }
|
|
r3d_hdr_peak = pk; r3d_hdr_paper = pp; r3d_hdr_black = bl
|
|
# the display is told what the picture now reaches
|
|
if gvk_hdr_on and gvk_has_hdr_meta and gvk_swap != 0 { gvk_hdr_metadata() }
|
|
}
|
|
|
|
# what the picture is: graded in BT.709 around D65, highlights to the calibrated peak, paper white average
|
|
function gvk_hdr_metadata() -> void {
|
|
let md = bytes(VkHdrMetadataEXT_sizeof)
|
|
Vk.zero(md, VkHdrMetadataEXT_sizeof)
|
|
Vk.put_i32(md, VkHdrMetadataEXT_sType, VK_STRUCTURE_TYPE_HDR_METADATA_EXT)
|
|
Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryRed + VkXYColorEXT_x, fl(0.64)); Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryRed + VkXYColorEXT_y, fl(0.33))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryGreen + VkXYColorEXT_x, fl(0.30)); Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryGreen + VkXYColorEXT_y, fl(0.60))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryBlue + VkXYColorEXT_x, fl(0.15)); Vk.put_i32(md, VkHdrMetadataEXT_displayPrimaryBlue + VkXYColorEXT_y, fl(0.06))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_whitePoint + VkXYColorEXT_x, fl(0.3127)); Vk.put_i32(md, VkHdrMetadataEXT_whitePoint + VkXYColorEXT_y, fl(0.3290))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_maxLuminance, r3d_hdr_peak_nits())
|
|
Vk.put_i32(md, VkHdrMetadataEXT_minLuminance, fl(0.001))
|
|
Vk.put_i32(md, VkHdrMetadataEXT_maxContentLightLevel, r3d_hdr_peak_nits())
|
|
Vk.put_i32(md, VkHdrMetadataEXT_maxFrameAverageLightLevel, r3d_hdr_paper_nits())
|
|
let chains = bytes(8)
|
|
Vk.put_i64(chains, 0, gvk_swap)
|
|
Vk.set_hdr_metadata_ext(gvk_dev, 1, chains, md)
|
|
}
|
|
|
|
function gvk_screen_make(w: int, h: int) -> bool {
|
|
if gvk_vp == null {
|
|
gvk_vp = words(4); gvk_sc = words(5); gvk_clear_rgba = words(4)
|
|
gvk_pass_col = words(4); gvk_pass_dep = words(2)
|
|
gvk_fb_ncolor = words(4096)
|
|
for i in 0 .. 4096 { gvk_fb_ncolor[i] = 1 }
|
|
for i in 0 .. 5 { gvk_sc[i] = 0 }
|
|
}
|
|
gvk_screen_w = w
|
|
gvk_screen_h = h
|
|
if gvk_screen_color == 0 { gvk_screen_color = gvk_tex_new(); gvk_screen_depth = gvk_tex_new() }
|
|
gvk_vp[0] = 0; gvk_vp[1] = 0; gvk_vp[2] = w; gvk_vp[3] = h
|
|
return gvk_tex_storage(gvk_screen_color, false, gvk_screen_fmt(), w, h, 1, false) and gvk_tex_storage(gvk_screen_depth, false, GL_DEPTH_COMPONENT32F, w, h, 1, false)
|
|
}
|
|
|
|
function gvk_frame_cb() -> pointer {
|
|
if gvk_cb == null {
|
|
gvk_frame_reset()
|
|
gvk_cb = gvk_once_begin()
|
|
}
|
|
return gvk_cb
|
|
}
|
|
|
|
# The view a pass draws into: level 0 only (an attachment view has exactly one level, and a target
|
|
# the exposure measure mipmaps has several), and one layer of an array image for a cascade drawn
|
|
# on its own. Keyed by the image's generation, so a replaced image never reuses a stale view.
|
|
function gvk_view_of(tex: int, layer1: int) -> long {
|
|
if layer1 == 0 and gvk_tex_levels[tex] <= 1 and gvk_tex_layers[tex] <= 1 { return gvk_tex_view[tex] }
|
|
let key = `{tex}:{gvk_tex_gen[tex]}:{layer1}`
|
|
if gvk_layer_views == null { gvk_layer_views = new []string; gvk_layer_view = new []long }
|
|
for i in 0 .. len(gvk_layer_views) { if gvk_layer_views[i] == key { return gvk_layer_view[i] } }
|
|
let depth = gvk_tex_vkfmt[tex] == VK_FORMAT_D32_SFLOAT
|
|
let vci = bytes(VkImageViewCreateInfo_sizeof)
|
|
Vk.zero(vci, VkImageViewCreateInfo_sizeof)
|
|
Vk.put_i32(vci, VkImageViewCreateInfo_sType, VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO)
|
|
Vk.put_i64(vci, VkImageViewCreateInfo_image, gvk_tex_image[tex])
|
|
Vk.put_i32(vci, VkImageViewCreateInfo_viewType, VK_IMAGE_VIEW_TYPE_2D)
|
|
Vk.put_i32(vci, VkImageViewCreateInfo_format, gvk_tex_vkfmt[tex])
|
|
let sr = VkImageViewCreateInfo_subresourceRange
|
|
if depth { Vk.put_i32(vci, sr + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT) } else { Vk.put_i32(vci, sr + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT) }
|
|
Vk.put_i32(vci, sr + VkImageSubresourceRange_levelCount, 1)
|
|
if layer1 > 0 { Vk.put_i32(vci, sr + VkImageSubresourceRange_baseArrayLayer, layer1 - 1) }
|
|
Vk.put_i32(vci, sr + VkImageSubresourceRange_layerCount, 1)
|
|
let out = bytes(8)
|
|
let zero: long = 0
|
|
if Vk.create_image_view(gvk_dev, vci, null, out) != VK_SUCCESS { return zero }
|
|
push(gvk_layer_views, key)
|
|
push(gvk_layer_view, gvk_handle(out))
|
|
return gvk_handle(out)
|
|
}
|
|
# the layer range a barrier for an attachment covers: the whole image unless one layer is drawn
|
|
function gvk_att_barrier(cb: pointer, tex: int, layer1: int, depth: bool, old_layout: int, new_layout: int) -> void {
|
|
# an attachment whose image was never made (no memory for it) has nothing to transition
|
|
if tex <= 0 or tex >= len(gvk_tex_image) or gvk_tex_image[tex] == 0 { return }
|
|
let b = bytes(VkImageMemoryBarrier_sizeof)
|
|
Vk.zero(b, VkImageMemoryBarrier_sizeof)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_sType, VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_srcAccessMask, gvk_layout_access(old_layout))
|
|
Vk.put_i32(b, VkImageMemoryBarrier_dstAccessMask, gvk_layout_access(new_layout))
|
|
Vk.put_i32(b, VkImageMemoryBarrier_oldLayout, old_layout)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_newLayout, new_layout)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_srcQueueFamilyIndex, VK_QUEUE_FAMILY_IGNORED)
|
|
Vk.put_i32(b, VkImageMemoryBarrier_dstQueueFamilyIndex, VK_QUEUE_FAMILY_IGNORED)
|
|
Vk.put_i64(b, VkImageMemoryBarrier_image, gvk_tex_image[tex])
|
|
let r = VkImageMemoryBarrier_subresourceRange
|
|
if depth { Vk.put_i32(b, r + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT) } else { Vk.put_i32(b, r + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT) }
|
|
Vk.put_i32(b, r + VkImageSubresourceRange_levelCount, 1)
|
|
if layer1 == 0 { Vk.put_i32(b, r + VkImageSubresourceRange_layerCount, gvk_tex_layers[tex]) }
|
|
else { Vk.put_i32(b, r + VkImageSubresourceRange_baseArrayLayer, layer1 - 1); Vk.put_i32(b, r + VkImageSubresourceRange_layerCount, 1) }
|
|
Vk.cmd_pipeline_barrier(cb, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 0, null, 0, null, 1, b)
|
|
}
|
|
|
|
# fb's attachments from gpu.ludic's record (colour 0, colour 1, depth texture, depth layer + 1,
|
|
# colour rb, depth rb, samples, colour layer + 1), framebuffer 0 being the screen
|
|
function gvk_pass_collect(fb: int, rec: words, rec_o: int) -> void {
|
|
for i in 0 .. 4 { gvk_pass_col[i] = 0 }
|
|
gvk_pass_dep[0] = 0; gvk_pass_dep[1] = 0
|
|
gvk_pass_ncolor = 0
|
|
if fb == 0 {
|
|
gvk_pass_col[0] = gvk_screen_color; gvk_pass_ncolor = 1
|
|
gvk_pass_dep[0] = gvk_screen_depth
|
|
return
|
|
}
|
|
if rec_o < 0 { return }
|
|
var want = 1
|
|
if fb < 4096 { want = gvk_fb_ncolor[fb] }
|
|
for slot in 0 .. 2 {
|
|
if slot < want and rec[rec_o + slot] > 0 {
|
|
gvk_pass_col[gvk_pass_ncolor * 2] = rec[rec_o + slot]
|
|
if slot == 0 { gvk_pass_col[gvk_pass_ncolor * 2 + 1] = rec[rec_o + 7] }
|
|
gvk_pass_ncolor += 1
|
|
}
|
|
}
|
|
gvk_pass_dep[0] = rec[rec_o + 2]
|
|
gvk_pass_dep[1] = rec[rec_o + 3]
|
|
}
|
|
|
|
function gvk_pass_begin(rec: words, rec_o: int) -> void {
|
|
if gvk_in_pass { return }
|
|
let cb = gvk_frame_cb()
|
|
gvk_pass_collect(gvk_fb_cur, rec, rec_o)
|
|
let aw = VkRenderingAttachmentInfo_sizeof
|
|
let catt = bytes(aw * 3)
|
|
Vk.zero(catt, aw * 3)
|
|
gvk_pass_w = 0
|
|
gvk_pass_h = 0
|
|
gvk_pass_cfmt = VK_FORMAT_UNDEFINED
|
|
gvk_pass_dfmt = VK_FORMAT_UNDEFINED
|
|
gvk_pass_samples = 1
|
|
for c in 0 .. gvk_pass_ncolor {
|
|
let tex = gvk_pass_col[c * 2]
|
|
let layer1 = gvk_pass_col[c * 2 + 1]
|
|
gvk_att_barrier(cb, tex, layer1, false, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO)
|
|
Vk.put_i64(catt, c * aw + VkRenderingAttachmentInfo_imageView, gvk_view_of(tex, layer1))
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_imageLayout, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE)
|
|
if (gvk_clear_bits & GL_COLOR_BUFFER_BIT) != 0 {
|
|
Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_CLEAR)
|
|
for k in 0 .. 4 { Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_clearValue + k * 4, gvk_clear_rgba[k]) }
|
|
} else { Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) }
|
|
gvk_pass_cfmt = gvk_tex_vkfmt[tex]
|
|
gvk_pass_samples = gvk_tex_samples_of(tex)
|
|
if gvk_pass_w == 0 { gvk_pass_w = gvk_tex_w(tex); gvk_pass_h = gvk_tex_h(tex) }
|
|
}
|
|
let datt = bytes(aw)
|
|
Vk.zero(datt, aw)
|
|
let dtex = gvk_pass_dep[0]
|
|
if dtex > 0 {
|
|
gvk_att_barrier(cb, dtex, gvk_pass_dep[1], true, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO)
|
|
Vk.put_i64(datt, VkRenderingAttachmentInfo_imageView, gvk_view_of(dtex, gvk_pass_dep[1]))
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_imageLayout, VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL)
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE)
|
|
if (gvk_clear_bits & GL_DEPTH_BUFFER_BIT) != 0 {
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_CLEAR)
|
|
Vk.put_i32(datt, VkRenderingAttachmentInfo_clearValue, 0x3F800000)
|
|
} else { Vk.put_i32(datt, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) }
|
|
gvk_pass_dfmt = VK_FORMAT_D32_SFLOAT
|
|
gvk_pass_samples = gvk_tex_samples_of(dtex)
|
|
if gvk_pass_w == 0 { gvk_pass_w = gvk_tex_w(dtex); gvk_pass_h = gvk_tex_h(dtex) }
|
|
}
|
|
gvk_clear_bits = 0
|
|
let ri = bytes(VkRenderingInfo_sizeof)
|
|
Vk.zero(ri, VkRenderingInfo_sizeof)
|
|
Vk.put_i32(ri, VkRenderingInfo_sType, VK_STRUCTURE_TYPE_RENDERING_INFO)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_width, gvk_pass_w)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_height, gvk_pass_h)
|
|
Vk.put_i32(ri, VkRenderingInfo_layerCount, 1)
|
|
Vk.put_i32(ri, VkRenderingInfo_colorAttachmentCount, gvk_pass_ncolor)
|
|
Vk.put_ptr(ri, VkRenderingInfo_pColorAttachments, catt)
|
|
if dtex > 0 { Vk.put_ptr(ri, VkRenderingInfo_pDepthAttachment, datt) }
|
|
Vk.cmd_begin_rendering(cb, ri)
|
|
gvk_in_pass = true
|
|
}
|
|
|
|
function gvk_pass_end() -> void {
|
|
if not gvk_in_pass { return }
|
|
let cb = gvk_cb
|
|
Vk.cmd_end_rendering(cb)
|
|
for c in 0 .. gvk_pass_ncolor {
|
|
gvk_att_barrier(cb, gvk_pass_col[c * 2], gvk_pass_col[c * 2 + 1], false, VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
if gvk_pass_dep[0] > 0 {
|
|
gvk_att_barrier(cb, gvk_pass_dep[0], gvk_pass_dep[1], true, VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
gvk_in_pass = false
|
|
}
|
|
|
|
# a clear: the load op of a pass that has not begun, an explicit clear inside one that has
|
|
function gvk_clear(mask: int, rec: words, rec_o: int) -> void {
|
|
if not gvk_in_pass { gvk_clear_bits = gvk_clear_bits | mask; return }
|
|
let cb = gvk_cb
|
|
let caw = VkClearAttachment_sizeof
|
|
let atts = bytes(caw * 4)
|
|
Vk.zero(atts, caw * 4)
|
|
var n = 0
|
|
if (mask & GL_COLOR_BUFFER_BIT) != 0 {
|
|
for c in 0 .. gvk_pass_ncolor {
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_colorAttachment, c)
|
|
for k in 0 .. 4 { Vk.put_i32(atts, n * caw + VkClearAttachment_clearValue + k * 4, gvk_clear_rgba[k]) }
|
|
n += 1
|
|
}
|
|
}
|
|
if (mask & GL_DEPTH_BUFFER_BIT) != 0 and gvk_pass_dep[0] > 0 {
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT)
|
|
Vk.put_i32(atts, n * caw + VkClearAttachment_clearValue, 0x3F800000)
|
|
n += 1
|
|
}
|
|
if n == 0 { return }
|
|
let rect = bytes(VkClearRect_sizeof)
|
|
Vk.zero(rect, VkClearRect_sizeof)
|
|
Vk.put_i32(rect, VkClearRect_rect + VkRect2D_extent + VkExtent2D_width, gvk_pass_w)
|
|
Vk.put_i32(rect, VkClearRect_rect + VkRect2D_extent + VkExtent2D_height, gvk_pass_h)
|
|
Vk.put_i32(rect, VkClearRect_layerCount, 1)
|
|
Vk.cmd_clear_attachments(cb, n, atts, 1, rect)
|
|
}
|
|
|
|
function gvk_tex_w(tex: int) -> int { return gvk_tex_dims_w[tex] }
|
|
function gvk_tex_h(tex: int) -> int { return gvk_tex_dims_h[tex] }
|
|
|
|
# viewport and scissor for the draw; OpenGL's rows count from the bottom and so do a Vulkan
|
|
# target's here (no y flip), so both pass straight through
|
|
function gvk_set_view(cb: pointer) -> void {
|
|
let vp = bytes(VkViewport_sizeof)
|
|
Vk.zero(vp, VkViewport_sizeof)
|
|
Vk.put_i32(vp, VkViewport_x, fi(gvk_vp[0]))
|
|
Vk.put_i32(vp, VkViewport_y, fi(gvk_vp[1]))
|
|
Vk.put_i32(vp, VkViewport_width, fi(gvk_vp[2]))
|
|
Vk.put_i32(vp, VkViewport_height, fi(gvk_vp[3]))
|
|
Vk.put_i32(vp, VkViewport_maxDepth, 0x3F800000)
|
|
Vk.cmd_set_viewport(cb, 0, 1, vp)
|
|
let sc = bytes(VkRect2D_sizeof)
|
|
Vk.zero(sc, VkRect2D_sizeof)
|
|
if gvk_sc[0] == 1 {
|
|
Vk.put_i32(sc, VkRect2D_offset + VkOffset2D_x, gvk_sc[1])
|
|
Vk.put_i32(sc, VkRect2D_offset + VkOffset2D_y, gvk_sc[2])
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_width, gvk_sc[3])
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_height, gvk_sc[4])
|
|
} else {
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_width, gvk_pass_w)
|
|
Vk.put_i32(sc, VkRect2D_extent + VkExtent2D_height, gvk_pass_h)
|
|
}
|
|
Vk.cmd_set_scissor(cb, 0, 1, sc)
|
|
}
|
|
|
|
# A draw of mesh m with program p and state st: the pipeline, the view, the set, the buffers.
|
|
# first / count select vertices or indices; count 0 means the mesh's own count. instances >= 1.
|
|
function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instances: int, tx: words, tx_w: int, tx_cap: int, rec: words, rec_o: int) -> void {
|
|
if gvk_prog_var == null or p <= 0 or p >= len(gvk_prog_var) or gvk_prog_var[p] == null { return }
|
|
let prof = gvk_prof()
|
|
var t0: long = 0
|
|
if prof { t0 = gl_now_us() }
|
|
gvk_pass_begin(rec, rec_o)
|
|
let cb = gvk_cb
|
|
let samples = gvk_pass_samples
|
|
let pipe = gvk_pipeline_fast(p, m, st, gvk_pass_ncolor, gvk_pass_cfmt, gvk_pass_dfmt, samples)
|
|
if pipe == 0 {
|
|
# Say so, once per program: a driver that refuses a pipeline other drivers accept (MoltenVK
|
|
# rejected every skinned one) otherwise leaves a whole layer missing from the frame in silence.
|
|
if gvk_skip_said == null { gvk_skip_said = words(4096); for i in 0 .. 4096 { gvk_skip_said[i] = 0 } }
|
|
if p < 4096 and gvk_skip_said[p] == 0 {
|
|
gvk_skip_said[p] = 1
|
|
print(`r3d: vulkan: no pipeline for {gpu_program_key(p)}; its draws are skipped`)
|
|
}
|
|
return
|
|
}
|
|
var t1: long = 0
|
|
if prof { t1 = gl_now_us(); gvk_us_pipe = gvk_us_pipe + (t1 - t0) }
|
|
Vk.cmd_bind_pipeline(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, pipe)
|
|
gvk_set_view(cb)
|
|
let set = gvk_draw_set(p, tx, tx_w, tx_cap)
|
|
if set == 0 { return }
|
|
if prof { gvk_us_set = gvk_us_set + (gl_now_us() - t1) }
|
|
let sets = bytes(8)
|
|
Vk.put_i64(sets, 0, set)
|
|
Vk.cmd_bind_descriptor_sets(cb, VK_PIPELINE_BIND_POINT_GRAPHICS, gvk_prog_layout[p], 0, 1, sets, gvk_set_ndyn, gvk_set_offs)
|
|
let v = gvk_prog_var[p]
|
|
if m != null and m.attrs != null {
|
|
# the same bindings, in the same order, as gvk_pipeline gave the layout
|
|
let bufs = bytes(8 * (GPU_MAX_VBUFS + 1))
|
|
let offs = bytes(8 * (GPU_MAX_VBUFS + 1))
|
|
Vk.zero(offs, 8 * (GPU_MAX_VBUFS + 1))
|
|
let seen = words(GPU_MAX_VBUFS)
|
|
var nbd = 0
|
|
for i in 0 .. m.n_attrs {
|
|
let o = i * GPU_ATTR_W
|
|
if m.attrs[o + 1] == 0 { continue }
|
|
var wanted = false
|
|
for q in 0 .. len(v.i_loc) { if v.i_loc[q] == i { wanted = true } }
|
|
if not wanted { continue }
|
|
var known = false
|
|
for q in 0 .. nbd { if seen[q] == m.attrs[o] { known = true } }
|
|
if not known and nbd < GPU_MAX_VBUFS {
|
|
seen[nbd] = m.attrs[o]
|
|
Vk.put_i64(bufs, nbd * 8, gvk_buf[m.attrs[o]])
|
|
gvk_buf_used[m.attrs[o]] = gvk_frame_no
|
|
nbd += 1
|
|
}
|
|
}
|
|
var unfed = false
|
|
for q in 0 .. len(v.i_loc) { if gvk_input_unfed(m, v.i_loc[q]) { unfed = true } }
|
|
if unfed and nbd <= GPU_MAX_VBUFS {
|
|
let zb = gvk_zero_vbuf_get()
|
|
Vk.put_i64(bufs, nbd * 8, gvk_buf[zb])
|
|
gvk_buf_used[zb] = gvk_frame_no
|
|
nbd += 1
|
|
}
|
|
if nbd > 0 { Vk.cmd_bind_vertex_buffers(cb, 0, nbd, bufs, offs) }
|
|
}
|
|
var n = count
|
|
if n == 0 and m != null { n = m.count }
|
|
if gvk_mesh_x > 0 {
|
|
# a mesh-shader dispatch (gpu_draw_mesh_tasks): the device's command, through a pointer
|
|
Vk.sl_call_piii(gvk_mesh_fn, cb, gvk_mesh_x, gvk_mesh_y, gvk_mesh_z)
|
|
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
|
|
return
|
|
}
|
|
if m != null and m.ebo != 0 {
|
|
let zero: long = 0
|
|
var itype = VK_INDEX_TYPE_UINT32
|
|
if m.itype == GL_UNSIGNED_SHORT { itype = VK_INDEX_TYPE_UINT16 }
|
|
Vk.cmd_bind_index_buffer(cb, gvk_buf[m.ebo], zero, itype)
|
|
gvk_buf_used[m.ebo] = gvk_frame_no
|
|
if gvk_ind_buf > 0 {
|
|
# the draws are records in a buffer (gvk_draw_indirect_now); the GPU may have written them
|
|
let ioff: long = gvk_ind_off
|
|
gvk_buf_used[gvk_ind_buf] = gvk_frame_no
|
|
if gvk_ind_cbuf > 0 and gvk_has_dic {
|
|
let coff: long = gvk_ind_coff
|
|
gvk_buf_used[gvk_ind_cbuf] = gvk_frame_no
|
|
Vk.cmd_draw_indexed_indirect_count(cb, gvk_buf[gvk_ind_buf], ioff, gvk_buf[gvk_ind_cbuf], coff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof)
|
|
} else {
|
|
Vk.cmd_draw_indexed_indirect(cb, gvk_buf[gvk_ind_buf], ioff, gvk_ind_n, VkDrawIndexedIndirectCommand_sizeof)
|
|
}
|
|
} else {
|
|
Vk.cmd_draw_indexed(cb, n, instances, first, 0, 0)
|
|
}
|
|
} else {
|
|
Vk.cmd_draw(cb, n, instances, first, 0)
|
|
}
|
|
if prof { gvk_n_draws += 1; gvk_us_draw = gvk_us_draw + (gl_now_us() - t0) }
|
|
}
|
|
|
|
# The frame so far, submitted and waited for, so work that submits on its own - an upload, a
|
|
# read-back, a new or freed image or buffer - happens after the draws recorded before it, in the
|
|
# order OpenGL would have done them. Costs a submit per such call while the backend comes up.
|
|
function gvk_flush() -> void {
|
|
if gvk_cb == null { return }
|
|
gvk_n_flush += 1
|
|
gvk_pass_end()
|
|
gvk_once_end(gvk_cb)
|
|
gvk_cb = null
|
|
gvk_frame_no += 1
|
|
gvk_retire_flush()
|
|
}
|
|
|
|
# The finished frame: submitted and waited for. With a window, the screen image is blitted into
|
|
# the swapchain's next image first - flipped, since the screen image keeps OpenGL's bottom-up rows
|
|
# - and that image is presented.
|
|
function gvk_present() -> void {
|
|
gvk_prof_frame()
|
|
if gvk_swap != 0 { gvk_present_window(); gvk_frame_no += 1; gvk_retire_flush(); return }
|
|
if gvk_cb == null { return }
|
|
gvk_pass_end()
|
|
gvk_once_end(gvk_cb)
|
|
gvk_cb = null
|
|
gvk_frame_no += 1
|
|
gvk_retire_flush()
|
|
}
|
|
|
|
# The screen as a binary PPM, top row first. The frame so far is finished first, then read back;
|
|
# row 0 of the image is OpenGL's bottom row, so rows are written last to first, as gl_screenshot does.
|
|
function gvk_screenshot(path: string) -> bool {
|
|
# the screen image is PQ-encoded 10-bit while HDR is on, which an 8-bit PPM would misread
|
|
if gvk_hdr_on { print("r3d: vulkan: screenshots are SDR only - turn HDR output off to take one"); return false }
|
|
gvk_present()
|
|
let w = gvk_screen_w
|
|
let h = gvk_screen_h
|
|
let px = bytes(w * h * 4)
|
|
if not gvk_tex_read(gvk_screen_color, GL_RGBA8, w, h, GL_RGBA, GL_UNSIGNED_BYTE, px) { return false }
|
|
let f = file_open(path, "wb")
|
|
if f == null { return false }
|
|
let hdr = `P6\n{w} {h}\n255\n`
|
|
file_write(f, hdr, len(hdr))
|
|
let row = bytes(w * 3)
|
|
var y = h - 1
|
|
while y >= 0 {
|
|
for x in 0 .. w {
|
|
let o = (y * w + x) * 4
|
|
row[x * 3] = px[o]; row[x * 3 + 1] = px[o + 1]; row[x * 3 + 2] = px[o + 2]
|
|
}
|
|
file_write(f, row, w * 3)
|
|
y -= 1
|
|
}
|
|
file_close(f)
|
|
return true
|
|
}
|
|
|
|
# ---- what gpu.ludic's Vulkan branches call -----------------------------------------------------
|
|
var gvk_fb_counter: int = 0
|
|
var gvk_prog_counter: int = 0
|
|
var gvk_wireframe: int = 0
|
|
var gvk_state: GvkState = null
|
|
|
|
# the manifest the programs are looked up in, from the renderer's own shader directory
|
|
function gvk_manifest() -> bool {
|
|
r3d_find_root()
|
|
gvk_spv_dir = `{r3d_root}/shaders/spv`
|
|
return gpu_manifest_load(`{gvk_spv_dir}/manifest.txt`) > 0 and len(gpu_variants) > 0
|
|
}
|
|
|
|
# The screen: a colour and a depth image. Headless they are the asked-for size; with a window
|
|
# they are its client area in pixels, and the swapchain is made on it.
|
|
function gvk_open(w: int, h: int, title: string) -> bool {
|
|
gl_w = w
|
|
gl_h = h
|
|
if is_windowed() {
|
|
win_open(w, h, 1, title)
|
|
win_gl_resize(w, h)
|
|
if gvk_size_buf == null { gvk_size_buf = words(4) }
|
|
win_gl_drawable(gvk_size_buf)
|
|
if gvk_size_buf[0] > 0 and gvk_size_buf[1] > 0 { gl_w = gvk_size_buf[0]; gl_h = gvk_size_buf[1] }
|
|
gl_scale = win_gl_scale()
|
|
}
|
|
if not gvk_frame_init() { return false }
|
|
if not gvk_screen_make(gl_w, gl_h) { return false }
|
|
if Os.has_env("R3D_VK_PROBE") { gvk_compute_probe() }
|
|
if is_windowed() { return gvk_swap_make(gl_w, gl_h) }
|
|
return true
|
|
}
|
|
|
|
# a program handle for a variant; the key is gpu_program's, so gpu_program_key works on both
|
|
function gvk_program_new(vs: string, fs: string, defines: string) -> int {
|
|
gvk_prog_counter += 1
|
|
let p = gvk_prog_counter
|
|
let key = `{vs}|{fs}|{Text.replace(defines, "\n", ";")}`
|
|
if gpu_prog_ids == null { gpu_prog_ids = new []int; gpu_prog_keys = new []string }
|
|
push(gpu_prog_ids, p)
|
|
push(gpu_prog_keys, key)
|
|
if not gvk_program(p, key, gvk_spv_dir) { return 0 }
|
|
return p
|
|
}
|
|
|
|
function gvk_mesh_free(m: Mesh) -> void {
|
|
if m == null { return }
|
|
if m.vbufs != null { for i in 0 .. m.n_vbufs { if m.vbufs[i] > 0 { gvk_buf_release(m.vbufs[i]) } } }
|
|
m.n_vbufs = 0
|
|
m.vbo = 0
|
|
if m.ebo > 0 { gvk_buf_release(m.ebo); m.ebo = 0 }
|
|
}
|
|
|
|
function gvk_scissor(x: int, y: int, w: int, h: int) -> void {
|
|
if gvk_sc == null { return }
|
|
gvk_sc[0] = 1; gvk_sc[1] = x; gvk_sc[2] = y; gvk_sc[3] = w; gvk_sc[4] = h
|
|
}
|
|
function gvk_scissor_off() -> void { if gvk_sc != null { gvk_sc[0] = 0 } }
|
|
function gvk_viewport(x: int, y: int, w: int, h: int) -> void {
|
|
if gvk_vp == null { return }
|
|
gvk_vp[0] = x; gvk_vp[1] = y; gvk_vp[2] = w; gvk_vp[3] = h
|
|
}
|
|
function gvk_clear_color(r: int, g: int, b: int, a: int) -> void {
|
|
if gvk_clear_rgba == null { return }
|
|
gvk_clear_rgba[0] = r; gvk_clear_rgba[1] = g; gvk_clear_rgba[2] = b; gvk_clear_rgba[3] = a
|
|
}
|
|
function gvk_fb_colors(fb: int, n: int) -> void { if gvk_fb_ncolor != null and fb >= 0 and fb < 4096 { gvk_fb_ncolor[fb] = n } }
|
|
# The framebuffer changes: a clear still waiting for this one's pass runs now, on this target,
|
|
# rather than becoming the load op of whichever pass begins next.
|
|
function gvk_rebind(fb: int) -> void {
|
|
if fb == gvk_fb_cur { return }
|
|
if not gvk_in_pass and gvk_clear_bits != 0 { gvk_pass_begin(gpu_fb, gpu_fb_at(gvk_fb_cur)) }
|
|
gvk_pass_end()
|
|
gvk_fb_cur = fb
|
|
}
|
|
function gvk_fb_forget(fb: int) -> void {
|
|
if fb == gvk_fb_cur { gvk_pass_end() }
|
|
gvk_fb_colors(fb, 1)
|
|
}
|
|
|
|
# the render state gpu.ludic has cached, with OpenGL's defaults where nothing was set yet
|
|
function gvk_state_now() -> GvkState {
|
|
if gvk_state == null { gvk_state = new GvkState }
|
|
let st = gvk_state
|
|
st.depth_test = 0; if gpu_s_depth_test == 1 { st.depth_test = 1 }
|
|
st.depth_write = 1; if gpu_s_depth_write == 0 { st.depth_write = 0 }
|
|
st.depth_func = GL_LESS; if gpu_s_depth_func > 0 { st.depth_func = gpu_s_depth_func }
|
|
st.blend = 0; if gpu_s_blend == 1 { st.blend = 1 }
|
|
st.blend_src = GL_ONE; if gpu_s_blend_src >= 0 { st.blend_src = gpu_s_blend_src }
|
|
st.blend_dst = GL_ZERO; if gpu_s_blend_dst >= 0 { st.blend_dst = gpu_s_blend_dst }
|
|
st.cull = 0; if gpu_s_cull == 1 { st.cull = 1 }
|
|
st.cull_face = GL_BACK; if gpu_s_cull_face > 0 { st.cull_face = gpu_s_cull_face }
|
|
st.color_write = 1; if gpu_s_color_write == 0 { st.color_write = 0 }
|
|
st.a2c = 0; if gpu_s_a2c == 1 { st.a2c = 1 }
|
|
st.bias = 0; if gpu_s_bias == 1 { st.bias = 1 }
|
|
st.bias_factor = gpu_s_bias_f
|
|
st.bias_units = gpu_s_bias_u
|
|
st.wireframe = gvk_wireframe
|
|
return st
|
|
}
|
|
# An indirect draw goes through gvk_draw like any other - the same pipeline, set and buffers -
|
|
# and only its last call differs, so the record is handed over in these for that one draw.
|
|
var gvk_ind_buf: int = 0
|
|
var gvk_ind_off: int = 0
|
|
var gvk_ind_n: int = 0
|
|
var gvk_ind_cbuf: int = 0
|
|
var gvk_ind_coff: int = 0
|
|
# x * y * z mesh-shader invocations with the current program (a *.mesh one) and state
|
|
var gvk_mesh_x: int = 0
|
|
var gvk_mesh_y: int = 0
|
|
var gvk_mesh_z: int = 0
|
|
function gvk_draw_mesh_tasks_now(x: int, y: int, z: int) -> void {
|
|
if not gvk_has_mesh or gvk_mesh_fn == null or x <= 0 or y <= 0 or z <= 0 { return }
|
|
gvk_mesh_x = x; gvk_mesh_y = y; gvk_mesh_z = z
|
|
gvk_draw_now(null, 0, 0, 1)
|
|
gvk_mesh_x = 0
|
|
}
|
|
function gvk_draw_indirect_now(m: Mesh, cmds: int, offset: int, n: int, count_buf: int, count_off: int) -> void {
|
|
if m == null or m.ebo == 0 or cmds <= 0 or n <= 0 { return }
|
|
gvk_ind_buf = cmds; gvk_ind_off = offset; gvk_ind_n = n; gvk_ind_cbuf = count_buf; gvk_ind_coff = count_off
|
|
gvk_draw_now(m, 0, 0, 1)
|
|
gvk_ind_buf = 0; gvk_ind_cbuf = 0
|
|
}
|
|
|
|
function gvk_draw_now(m: Mesh, first: int, count: int, instances: int) -> void {
|
|
if gvk_prof() { gvk_n_asked += 1 }
|
|
gvk_draw(gpu_prog_cur, m, gvk_state_now(), first, count, instances, gpu_tx, GPU_TX_W, gpu_tx_cap, gpu_fb, gpu_fb_at(gvk_fb_cur))
|
|
}
|
|
|
|
# the colour (and / or depth) of the read framebuffer into the draw framebuffer, same size
|
|
function gvk_fb_att(fb: int, depth: bool) -> int {
|
|
if fb == 0 { if depth { return gvk_screen_depth }; return gvk_screen_color }
|
|
let o = gpu_fb_at(fb)
|
|
if o < 0 { return 0 }
|
|
if depth { return gpu_fb[o + 2] }
|
|
return gpu_fb[o]
|
|
}
|
|
function gvk_tex_samples_of(tex: int) -> int {
|
|
if tex <= 0 or gvk_tex_samples == null or tex >= len(gvk_tex_samples) or gvk_tex_samples[tex] < 1 { return 1 }
|
|
return gvk_tex_samples[tex]
|
|
}
|
|
# A multisampled image into a single-sampled one: a pass that draws nothing and resolves on its end -
|
|
# colour averaged, depth from sample zero. vkCmdResolveImage cannot resolve depth; a pass can.
|
|
function gvk_resolve(cb: pointer, src: int, dst: int, depth: bool, w: int, h: int) -> void {
|
|
var layout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL
|
|
var mode = VK_RESOLVE_MODE_AVERAGE_BIT
|
|
if depth { layout = VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL; mode = VK_RESOLVE_MODE_SAMPLE_ZERO_BIT }
|
|
gvk_att_barrier(cb, src, 0, depth, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, layout)
|
|
gvk_att_barrier(cb, dst, 0, depth, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, layout)
|
|
let aw = VkRenderingAttachmentInfo_sizeof
|
|
let att = bytes(aw)
|
|
Vk.zero(att, aw)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO)
|
|
Vk.put_i64(att, VkRenderingAttachmentInfo_imageView, gvk_view_of(src, 0))
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_imageLayout, layout)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_resolveMode, mode)
|
|
Vk.put_i64(att, VkRenderingAttachmentInfo_resolveImageView, gvk_view_of(dst, 0))
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_resolveImageLayout, layout)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD)
|
|
Vk.put_i32(att, VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE)
|
|
let ri = bytes(VkRenderingInfo_sizeof)
|
|
Vk.zero(ri, VkRenderingInfo_sizeof)
|
|
Vk.put_i32(ri, VkRenderingInfo_sType, VK_STRUCTURE_TYPE_RENDERING_INFO)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_width, w)
|
|
Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_height, h)
|
|
Vk.put_i32(ri, VkRenderingInfo_layerCount, 1)
|
|
if depth { Vk.put_ptr(ri, VkRenderingInfo_pDepthAttachment, att) }
|
|
else { Vk.put_i32(ri, VkRenderingInfo_colorAttachmentCount, 1); Vk.put_ptr(ri, VkRenderingInfo_pColorAttachments, att) }
|
|
Vk.cmd_begin_rendering(cb, ri)
|
|
Vk.cmd_end_rendering(cb)
|
|
gvk_att_barrier(cb, src, 0, depth, layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
gvk_att_barrier(cb, dst, 0, depth, layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
function gvk_copy(cb: pointer, src: int, dst: int, depth: bool, w: int, h: int) -> void {
|
|
if src <= 0 or dst <= 0 or gvk_tex_image[src] == 0 or gvk_tex_image[dst] == 0 { return }
|
|
if gvk_tex_samples_of(src) > 1 and gvk_tex_samples_of(dst) == 1 { gvk_resolve(cb, src, dst, depth, w, h); return }
|
|
gvk_barrier(cb, gvk_tex_image[src], depth, 0, 1, gvk_tex_layers[src], VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
|
|
gvk_barrier(cb, gvk_tex_image[dst], depth, 0, 1, gvk_tex_layers[dst], VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
|
let ic = bytes(VkImageCopy_sizeof)
|
|
Vk.zero(ic, VkImageCopy_sizeof)
|
|
var aspect = VK_IMAGE_ASPECT_COLOR_BIT
|
|
if depth { aspect = VK_IMAGE_ASPECT_DEPTH_BIT }
|
|
Vk.put_i32(ic, VkImageCopy_srcSubresource + VkImageSubresourceLayers_aspectMask, aspect)
|
|
Vk.put_i32(ic, VkImageCopy_srcSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
Vk.put_i32(ic, VkImageCopy_dstSubresource + VkImageSubresourceLayers_aspectMask, aspect)
|
|
Vk.put_i32(ic, VkImageCopy_dstSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_width, w)
|
|
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_height, h)
|
|
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_depth, 1)
|
|
Vk.cmd_copy_image(cb, gvk_tex_image[src], VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, gvk_tex_image[dst], VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ic)
|
|
gvk_barrier(cb, gvk_tex_image[src], depth, 0, 1, gvk_tex_layers[src], VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
gvk_barrier(cb, gvk_tex_image[dst], depth, 0, 1, gvk_tex_layers[dst], VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
}
|
|
function gvk_blit(w: int, h: int, mask: int) -> void {
|
|
gvk_pass_end()
|
|
let cb = gvk_frame_cb()
|
|
if (mask & GL_COLOR_BUFFER_BIT) != 0 { gvk_copy(cb, gvk_fb_att(gvk_fb_read, false), gvk_fb_att(gvk_fb_draw, false), false, w, h) }
|
|
if (mask & GL_DEPTH_BUFFER_BIT) != 0 { gvk_copy(cb, gvk_fb_att(gvk_fb_read, true), gvk_fb_att(gvk_fb_draw, true), true, w, h) }
|
|
}
|
|
|
|
# the screen as RGB8, bottom row first, as glReadPixels hands it back (a photograph)
|
|
function gvk_read_screen(w: int, h: int, out: pointer) -> void {
|
|
gvk_present()
|
|
let px = bytes(gvk_screen_w * gvk_screen_h * 4)
|
|
if not gvk_tex_read(gvk_screen_color, GL_RGBA8, gvk_screen_w, gvk_screen_h, GL_RGBA, GL_UNSIGNED_BYTE, px) { return }
|
|
let dst: pointer = out
|
|
for y in 0 .. h {
|
|
for x in 0 .. w {
|
|
let o = (y * gvk_screen_w + x) * 4
|
|
let q = (y * w + x) * 3
|
|
dst[q] = px[o]; dst[q + 1] = px[o + 1]; dst[q + 2] = px[o + 2]
|
|
}
|
|
}
|
|
}
|
|
|
|
# ---- the window: surface and swapchain ----------------------------------------------------------
|
|
# Windows only for now (the Win32 surface); gpu_select keeps a window elsewhere on OpenGL.
|
|
extern function win_native_window() -> pointer = "win_native_window"
|
|
extern function win_native_instance() -> pointer = "win_native_instance"
|
|
|
|
var gvk_surface: long = 0
|
|
var gvk_swap: long = 0
|
|
var gvk_swap_n: int = 0
|
|
var gvk_swap_images: bytes = null
|
|
var gvk_swap_fmt: int = 0
|
|
var gvk_swap_w: int = 0
|
|
var gvk_swap_h: int = 0
|
|
var gvk_vsync: bool = true
|
|
var gvk_acq_fence: bytes = null
|
|
var gvk_swap_stale: bool = false
|
|
|
|
function gvk_surface_make() -> bool {
|
|
if gvk_surface != 0 { return true }
|
|
let hwnd = win_native_window()
|
|
if hwnd == null { gvk_why = "no window to present to"; return false }
|
|
let sci = bytes(VkWin32SurfaceCreateInfoKHR_sizeof)
|
|
Vk.zero(sci, VkWin32SurfaceCreateInfoKHR_sizeof)
|
|
Vk.put_i32(sci, VkWin32SurfaceCreateInfoKHR_sType, VK_STRUCTURE_TYPE_WIN32_SURFACE_CREATE_INFO_KHR)
|
|
Vk.put_ptr(sci, VkWin32SurfaceCreateInfoKHR_hinstance, win_native_instance())
|
|
Vk.put_ptr(sci, VkWin32SurfaceCreateInfoKHR_hwnd, hwnd)
|
|
let out = bytes(8)
|
|
let r = Vk.create_win32_surface_khr(gvk_inst, sci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateWin32SurfaceKHR", r) }
|
|
gvk_surface = gvk_handle(out)
|
|
let ok = bytes(4)
|
|
Vk.put_i32(ok, 0, 0)
|
|
Vk.get_physical_device_surface_support_khr(gvk_pd, gvk_family, gvk_surface, ok)
|
|
if Vk.get_i32(ok, 0) != 1 { gvk_why = "the graphics queue cannot present to the window"; return false }
|
|
let fci = bytes(VkFenceCreateInfo_sizeof)
|
|
Vk.zero(fci, VkFenceCreateInfo_sizeof)
|
|
Vk.put_i32(fci, VkFenceCreateInfo_sType, VK_STRUCTURE_TYPE_FENCE_CREATE_INFO)
|
|
gvk_acq_fence = bytes(8)
|
|
return Vk.create_fence(gvk_dev, fci, null, gvk_acq_fence) == VK_SUCCESS
|
|
}
|
|
|
|
# (Re)make the swapchain for a w x h client area. The old one is handed over and then destroyed.
|
|
function gvk_swap_make(w: int, h: int) -> bool {
|
|
if not gvk_surface_make() { return false }
|
|
Vk.device_wait_idle(gvk_dev)
|
|
let caps = bytes(VkSurfaceCapabilitiesKHR_sizeof)
|
|
Vk.get_physical_device_surface_capabilities_khr(gvk_pd, gvk_surface, caps)
|
|
var ew = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_currentExtent + VkExtent2D_width)
|
|
var eh = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_currentExtent + VkExtent2D_height)
|
|
if ew == -1 or ew <= 0 { ew = w; eh = h }
|
|
if ew <= 0 or eh <= 0 { return false } # minimised: keep the old chain until it has a size
|
|
let cnt = bytes(4)
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_physical_device_surface_formats_khr(gvk_pd, gvk_surface, cnt, null)
|
|
let nf = Vk.get_i32(cnt, 0)
|
|
let fmts = bytes(nf * VkSurfaceFormatKHR_sizeof + 8)
|
|
Vk.get_physical_device_surface_formats_khr(gvk_pd, gvk_surface, cnt, fmts)
|
|
# the screen image holds display-ready 8-bit colour, so the swapchain takes it unconverted
|
|
var fmt = Vk.get_i32(fmts, VkSurfaceFormatKHR_format)
|
|
var cs = Vk.get_i32(fmts, VkSurfaceFormatKHR_colorSpace)
|
|
for i in 0 .. nf {
|
|
let f = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_format)
|
|
let c = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_colorSpace)
|
|
if (f == VK_FORMAT_B8G8R8A8_UNORM or f == VK_FORMAT_R8G8B8A8_UNORM) and c == VK_COLOR_SPACE_SRGB_NONLINEAR_KHR { fmt = f; cs = c }
|
|
}
|
|
var hdr = false
|
|
if gvk_hdr_want and gvk_has_colorspace {
|
|
for i in 0 .. nf {
|
|
let f = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_format)
|
|
let c = Vk.get_i32(fmts, i * VkSurfaceFormatKHR_sizeof + VkSurfaceFormatKHR_colorSpace)
|
|
if f == VK_FORMAT_A2B10G10R10_UNORM_PACK32 and c == VK_COLOR_SPACE_HDR10_ST2084_EXT { fmt = f; cs = c; hdr = true }
|
|
}
|
|
if not hdr { print("r3d: vulkan: HDR output asked for, but this display offers no HDR10 swapchain") }
|
|
}
|
|
# the screen image carries the swapchain's depth of colour: remade when HDR comes or goes
|
|
if hdr != gvk_hdr_on { gvk_hdr_on = hdr; gvk_screen_make(gvk_screen_w, gvk_screen_h) }
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_physical_device_surface_present_modes_khr(gvk_pd, gvk_surface, cnt, null)
|
|
let nm = Vk.get_i32(cnt, 0)
|
|
let modes = bytes(nm * 4 + 8)
|
|
Vk.get_physical_device_surface_present_modes_khr(gvk_pd, gvk_surface, cnt, modes)
|
|
# vsync: FIFO, which every device has. Off: mailbox where offered (no tearing), else immediate.
|
|
var mode = VK_PRESENT_MODE_FIFO_KHR
|
|
if not gvk_vsync {
|
|
for i in 0 .. nm { if Vk.get_i32(modes, i * 4) == VK_PRESENT_MODE_IMMEDIATE_KHR { mode = VK_PRESENT_MODE_IMMEDIATE_KHR } }
|
|
for i in 0 .. nm { if Vk.get_i32(modes, i * 4) == VK_PRESENT_MODE_MAILBOX_KHR { mode = VK_PRESENT_MODE_MAILBOX_KHR } }
|
|
}
|
|
var n = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_minImageCount) + 1
|
|
let mx = Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_maxImageCount)
|
|
if mx > 0 and n > mx { n = mx }
|
|
let sci = bytes(VkSwapchainCreateInfoKHR_sizeof)
|
|
Vk.zero(sci, VkSwapchainCreateInfoKHR_sizeof)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_sType, VK_STRUCTURE_TYPE_SWAPCHAIN_CREATE_INFO_KHR)
|
|
Vk.put_i64(sci, VkSwapchainCreateInfoKHR_surface, gvk_surface)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_minImageCount, n)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageFormat, fmt)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageColorSpace, cs)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageExtent + VkExtent2D_width, ew)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageExtent + VkExtent2D_height, eh)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageArrayLayers, 1)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageUsage, VK_IMAGE_USAGE_TRANSFER_DST_BIT)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_imageSharingMode, VK_SHARING_MODE_EXCLUSIVE)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_preTransform, Vk.get_i32(caps, VkSurfaceCapabilitiesKHR_currentTransform))
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_compositeAlpha, VK_COMPOSITE_ALPHA_OPAQUE_BIT_KHR)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_presentMode, mode)
|
|
Vk.put_i32(sci, VkSwapchainCreateInfoKHR_clipped, 1)
|
|
Vk.put_i64(sci, VkSwapchainCreateInfoKHR_oldSwapchain, gvk_swap)
|
|
let out = bytes(8)
|
|
let r = Vk.create_swapchain_khr(gvk_dev, sci, null, out)
|
|
if r != VK_SUCCESS { return gvk_fail("vkCreateSwapchainKHR", r) }
|
|
if gvk_swap != 0 { Vk.destroy_swapchain_khr(gvk_dev, gvk_swap, null) }
|
|
gvk_swap = gvk_handle(out)
|
|
if gvk_hdr_on and gvk_has_hdr_meta { gvk_hdr_metadata() }
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_swapchain_images_khr(gvk_dev, gvk_swap, cnt, null)
|
|
gvk_swap_n = Vk.get_i32(cnt, 0)
|
|
gvk_swap_images = bytes(gvk_swap_n * 8 + 8)
|
|
Vk.get_swapchain_images_khr(gvk_dev, gvk_swap, cnt, gvk_swap_images)
|
|
gvk_swap_fmt = fmt
|
|
gvk_swap_w = ew
|
|
gvk_swap_h = eh
|
|
gvk_swap_stale = false
|
|
var mname = "fifo"
|
|
if mode == VK_PRESENT_MODE_MAILBOX_KHR { mname = "mailbox" }
|
|
if mode == VK_PRESENT_MODE_IMMEDIATE_KHR { mname = "immediate" }
|
|
var cname = "SDR"
|
|
if gvk_hdr_on { cname = "HDR10" }
|
|
print(`r3d: vulkan swapchain {ew}x{eh}, {gvk_swap_n} images, {mname}, {cname}`)
|
|
return true
|
|
}
|
|
|
|
function gvk_present_window() -> void {
|
|
let cb = gvk_frame_cb()
|
|
gvk_pass_end()
|
|
if gvk_swap_stale { gvk_once_end(cb); gvk_cb = null; gvk_swap_make(gvk_swap_w, gvk_swap_h); return }
|
|
let idx = bytes(4)
|
|
Vk.reset_fences(gvk_dev, 1, gvk_acq_fence)
|
|
let forever: long = -1
|
|
let zero: long = 0
|
|
var r = Vk.acquire_next_image_khr(gvk_dev, gvk_swap, forever, zero, Vk.get_i64(gvk_acq_fence, 0), idx)
|
|
if r == VK_ERROR_OUT_OF_DATE_KHR { gvk_once_end(cb); gvk_cb = null; gvk_swap_make(gvk_swap_w, gvk_swap_h); return }
|
|
Vk.wait_for_fences(gvk_dev, 1, gvk_acq_fence, 1, forever)
|
|
let i = Vk.get_i32(idx, 0)
|
|
let dst = Vk.get_i64(gvk_swap_images, i * 8)
|
|
let src = gvk_tex_image[gvk_screen_color]
|
|
gvk_barrier(cb, src, false, 0, 1, 1, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
|
|
gvk_barrier(cb, dst, false, 0, 1, 1, VK_IMAGE_LAYOUT_UNDEFINED, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
|
let blit = bytes(VkImageBlit_sizeof)
|
|
Vk.zero(blit, VkImageBlit_sizeof)
|
|
Vk.put_i32(blit, VkImageBlit_srcSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
|
Vk.put_i32(blit, VkImageBlit_srcSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
# the screen image's row 0 is the bottom of the picture: read it from the top edge down
|
|
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_y, gvk_screen_h)
|
|
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_x, gvk_screen_w)
|
|
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_z, 1)
|
|
Vk.put_i32(blit, VkImageBlit_dstSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
|
Vk.put_i32(blit, VkImageBlit_dstSubresource + VkImageSubresourceLayers_layerCount, 1)
|
|
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_x, gvk_swap_w)
|
|
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_y, gvk_swap_h)
|
|
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_z, 1)
|
|
Vk.cmd_blit_image(cb, src, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dst, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, blit, VK_FILTER_LINEAR)
|
|
gvk_barrier(cb, src, false, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
|
gvk_barrier(cb, dst, false, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_PRESENT_SRC_KHR)
|
|
gvk_once_end(cb)
|
|
gvk_cb = null
|
|
let pi = bytes(VkPresentInfoKHR_sizeof)
|
|
Vk.zero(pi, VkPresentInfoKHR_sizeof)
|
|
Vk.put_i32(pi, VkPresentInfoKHR_sType, VK_STRUCTURE_TYPE_PRESENT_INFO_KHR)
|
|
let chains = bytes(8)
|
|
Vk.put_i64(chains, 0, gvk_swap)
|
|
Vk.put_i32(pi, VkPresentInfoKHR_swapchainCount, 1)
|
|
Vk.put_ptr(pi, VkPresentInfoKHR_pSwapchains, chains)
|
|
Vk.put_ptr(pi, VkPresentInfoKHR_pImageIndices, idx)
|
|
gsl_before_present()
|
|
r = Vk.queue_present_khr(gvk_queue, pi)
|
|
if r == VK_ERROR_OUT_OF_DATE_KHR or r == VK_SUBOPTIMAL_KHR { gvk_swap_stale = true }
|
|
gsl_after_present()
|
|
}
|
|
|
|
# a window's client area changed: the screen images and the swapchain follow it
|
|
var gvk_size_buf: words = null
|
|
function gvk_resize_check() -> bool {
|
|
if gvk_swap == 0 { return false }
|
|
if gvk_size_buf == null { gvk_size_buf = words(4) }
|
|
win_gl_drawable(gvk_size_buf)
|
|
let w = gvk_size_buf[0]
|
|
let h = gvk_size_buf[1]
|
|
if w <= 0 or h <= 0 { return false }
|
|
if w == gl_w and h == gl_h and not gvk_swap_stale { return false }
|
|
gvk_flush()
|
|
gl_w = w
|
|
gl_h = h
|
|
gvk_screen_make(w, h)
|
|
gvk_swap_make(w, h)
|
|
return true
|
|
}
|
|
|
|
# Mipmaps for a texture the renderer asks for mid-frame (the exposure measure reads the HDR scene's
|
|
# smallest level every frame): recorded into the open frame after its pass, so they cost no submit.
|
|
# A chain that has to grow first still goes through its one-shot path.
|
|
function gvk_mips_now(tex: int, w: int, h: int) -> void {
|
|
if gvk_cb == null or gvk_tex_levels[tex] <= 1 { gvk_flush(); gvk_tex_mips(tex, w, h); return }
|
|
gvk_pass_end()
|
|
gvk_tex_mips_into(gvk_cb, tex, w, h)
|
|
}
|
|
|
|
# ---- R3D_VK_PROF: where a Vulkan frame's CPU time goes ---------------------------------------------
|
|
# Every 120 frames: draws and flushes per frame, and milliseconds per frame spent finding pipelines,
|
|
# filling descriptor sets, and inside draws altogether. Off, it costs one flag test a draw.
|
|
var gvk_prof_state: int = -1
|
|
var gvk_us_pipe: long = 0
|
|
var gvk_us_set: long = 0
|
|
var gvk_us_draw: long = 0
|
|
var gvk_n_draws: int = 0
|
|
var gvk_n_asked: int = 0 # draws that reached gvk_draw_now: every one the renderer asked for
|
|
var gvk_n_flush: int = 0
|
|
var gvk_prof_frames: int = 0
|
|
var gvk_n_pipe_new: int = 0 # pipelines made this profile window
|
|
function gvk_prof() -> bool {
|
|
if gvk_prof_state < 0 { gvk_prof_state = 0; if Os.has_env("R3D_VK_PROF") { gvk_prof_state = 1 } }
|
|
return gvk_prof_state == 1
|
|
}
|
|
function gvk_prof_frame() -> void {
|
|
if not gvk_prof() { return }
|
|
gvk_prof_frames += 1
|
|
if gvk_prof_frames < 120 { return }
|
|
let f = gvk_prof_frames
|
|
let pipe = Text.to_int(string(gvk_us_pipe)) / f
|
|
let set = Text.to_int(string(gvk_us_set)) / f
|
|
let draw = Text.to_int(string(gvk_us_draw)) / f
|
|
print(`r3d: vulkan per frame: {gvk_n_draws / f} draws, {gvk_n_flush / f} flushes; pipelines {pipe / 1000}.{(pipe / 100) % 10} ms, sets {set / 1000}.{(set / 100) % 10} ms, inside draws {draw / 1000}.{(draw / 100) % 10} ms; {gvk_n_pipe_new} pipelines made`)
|
|
# every draw asked for should have been made: a gap is a layer missing from the frame, which is how
|
|
# MoltenVK's refused skinned pipelines went unseen (the drawstats session found it by counting)
|
|
if gvk_n_asked != gvk_n_draws {
|
|
print(`r3d: vulkan: {(gvk_n_asked - gvk_n_draws) / f} draws a frame were asked for and not made ({gvk_n_asked / f} asked, {gvk_n_draws / f} made)`)
|
|
}
|
|
let zero: long = 0
|
|
gvk_us_pipe = zero; gvk_us_set = zero; gvk_us_draw = zero
|
|
gvk_n_draws = 0; gvk_n_asked = 0; gvk_n_flush = 0; gvk_prof_frames = 0; gvk_n_pipe_new = 0
|
|
}
|
|
|
|
# ---- the pipeline cache, by integers -----------------------------------------------------------
|
|
# gvk_pipeline builds and keys pipelines by a string; a draw only needs to find one it already has.
|
|
# A mesh carries an interned layout id (its recorded layout, buffers named by order), the render
|
|
# state packs into one int, and the pass's formats into another; the program's last hit is tried
|
|
# first, then the entries it owns.
|
|
var gvk_layout_keys: []string = null
|
|
var gvk_pc_prog: []int = null
|
|
var gvk_pc_layout: []int = null
|
|
var gvk_pc_state: []int = null
|
|
var gvk_pc_bias: []int = null
|
|
var gvk_pc_pass: []int = null
|
|
var gvk_pc_pipe: []long = null
|
|
var gvk_pc_last: words = null
|
|
|
|
function gvk_layout_id(m: Mesh) -> int {
|
|
if m == null or m.attrs == null { return 1 }
|
|
if m.vk_layout > 0 { return m.vk_layout }
|
|
let key = gvk_layout_key(m)
|
|
if gvk_layout_keys == null { gvk_layout_keys = new []string }
|
|
var id = 0
|
|
for i in 0 .. len(gvk_layout_keys) { if id == 0 and gvk_layout_keys[i] == key { id = i + 2 } }
|
|
if id == 0 { push(gvk_layout_keys, key); id = len(gvk_layout_keys) + 1 }
|
|
m.vk_layout = id
|
|
return id
|
|
}
|
|
function gvk_blend_index(f: int) -> int {
|
|
if f == GL_ZERO { return 0 }
|
|
if f == GL_ONE { return 1 }
|
|
if f == GL_SRC_ALPHA { return 2 }
|
|
if f == GL_ONE_MINUS_SRC_ALPHA { return 3 }
|
|
if f == GL_DST_ALPHA { return 4 }
|
|
if f == GL_ONE_MINUS_DST_ALPHA { return 5 }
|
|
if f == GL_SRC_COLOR { return 6 }
|
|
if f == GL_ONE_MINUS_SRC_COLOR { return 7 }
|
|
return 8
|
|
}
|
|
function gvk_pipeline_fast(p: int, m: Mesh, st: GvkState, n_color: int, color_fmt: int, depth_fmt: int, samples: int) -> long {
|
|
let layout = gvk_layout_id(m)
|
|
var state = st.depth_test | (st.depth_write << 1) | (st.blend << 2) | (st.cull << 3) | (st.color_write << 4) | (st.a2c << 5) | (st.bias << 6) | (st.wireframe << 7)
|
|
state = state | ((st.depth_func & 15) << 8) | ((st.cull_face & 15) << 12) | (gvk_blend_index(st.blend_src) << 16) | (gvk_blend_index(st.blend_dst) << 20)
|
|
let bias = st.bias_factor * 31 + st.bias_units
|
|
let pass = ((n_color * 1000 + color_fmt) * 1000 + depth_fmt) * 64 + samples
|
|
if gvk_pc_prog == null {
|
|
gvk_pc_prog = new []int; gvk_pc_layout = new []int; gvk_pc_state = new []int; gvk_pc_bias = new []int
|
|
gvk_pc_pass = new []int; gvk_pc_pipe = new []long
|
|
gvk_pc_last = words(4096)
|
|
for i in 0 .. 4096 { gvk_pc_last[i] = -1 }
|
|
}
|
|
if p < 4096 {
|
|
let k = gvk_pc_last[p]
|
|
if k >= 0 and gvk_pc_layout[k] == layout and gvk_pc_state[k] == state and gvk_pc_pass[k] == pass and gvk_pc_bias[k] == bias { return gvk_pc_pipe[k] }
|
|
}
|
|
for k in 0 .. len(gvk_pc_prog) {
|
|
if gvk_pc_prog[k] == p and gvk_pc_layout[k] == layout and gvk_pc_state[k] == state and gvk_pc_pass[k] == pass and gvk_pc_bias[k] == bias {
|
|
if p < 4096 { gvk_pc_last[p] = k }
|
|
return gvk_pc_pipe[k]
|
|
}
|
|
}
|
|
let pipe = gvk_pipeline(p, m, st, n_color, color_fmt, depth_fmt, samples)
|
|
if pipe == 0 { return pipe }
|
|
push(gvk_pc_prog, p); push(gvk_pc_layout, layout); push(gvk_pc_state, state)
|
|
push(gvk_pc_bias, bias); push(gvk_pc_pass, pass); push(gvk_pc_pipe, pipe)
|
|
if p < 4096 { gvk_pc_last[p] = len(gvk_pc_prog) - 1 }
|
|
return pipe
|
|
}
|