The renderer starts itself from the first frame (gpu_select), so the device, its tables, the manifest and programs, grass, shadows, sky, terrain textures and the actor pool read as frame allocations: each is @alloc_ok as start-up, with gvk_fail (a failure) and the actor census and texture dump (debug switches). ov_nine's four corner/uv arrays are one floats(16) made in gvk_startup_state; ac_in_light's four points are made there too. Compiled (steady, and ludic deps over main 4316ff97). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
774 lines
45 KiB
Text
774 lines
45 KiB
Text
# gpu_vk.ludic — the Vulkan side of gpu.ludic: the device every other part of the backend
|
|
# draws with, the memory it allocates from and the one-shot command buffers that uploads,
|
|
# bakes and read-backs go through.
|
|
#
|
|
# gpu.ludic keeps the renderer's handles (a texture, a buffer, a framebuffer, a program as an
|
|
# int) and branches on the backend; on Vulkan those ints index tables of Vulkan objects kept
|
|
# here. Nothing in this file runs unless the Vulkan backend was chosen and came up.
|
|
|
|
# ---- the device ---------------------------------------------------------------------------
|
|
|
|
@alloc_ok("a failure, said once: the message is built only when something is already wrong")
|
|
function gvk_fail(render3d_st: mut Render3dState, what: string, r: int) -> bool {
|
|
render3d_st.gvk_why = `{what} failed (VkResult {r})`
|
|
gvk_note(render3d_st, `r3d: vulkan: {render3d_st.gvk_why}`)
|
|
return false
|
|
}
|
|
|
|
# A Vulkan failure is printed, and with R3D_VK_ERRLOG=<file> also appended to that file, opened and
|
|
# closed for each line: stdout is buffered, and a crash straight after a failure (an image that was
|
|
# never made, then used) used to take the one line that explained it with it.
|
|
function gvk_note(render3d_st: mut Render3dState, msg: string) -> void {
|
|
print(msg)
|
|
if not render3d_st.gvk_errlog_read {
|
|
render3d_st.gvk_errlog_read = true
|
|
if r3d_env_has(render3d_st, "R3D_VK_ERRLOG") { render3d_st.gvk_errlog = r3d_env(render3d_st, "R3D_VK_ERRLOG") }
|
|
}
|
|
if render3d_st.gvk_errlog == null { return }
|
|
let f = file_open(render3d_st.gvk_errlog, "ab")
|
|
if f == null { return }
|
|
let line = msg + "\n"
|
|
file_write(f, line, len(line))
|
|
file_close(f)
|
|
}
|
|
function gvk_handle(out: bytes) -> long { return Vk.get_i64(out, 0) }
|
|
|
|
function gvk_ext_in(props: bytes, n: int, want: string) -> bool {
|
|
for i in 0 .. n {
|
|
if string(Vk.at(props, i * VkExtensionProperties_sizeof + VkExtensionProperties_extensionName)) == want { return true }
|
|
}
|
|
return false
|
|
}
|
|
|
|
# Instance, the first discrete GPU (else the first listed), its first graphics queue, and a
|
|
# device with the Tier 1 floor switched on. False, with gvk_why set, if any of it is missing:
|
|
# the caller stays on OpenGL.
|
|
|
|
function gvk_surface_ext() -> string {
|
|
if Os.platform() == "macos" { return VK_EXT_METAL_SURFACE_EXTENSION_NAME }
|
|
return VK_KHR_WIN32_SURFACE_EXTENSION_NAME
|
|
}
|
|
|
|
# MoltenVK's command pooling keeps every command object a frame ever recorded and never gives one
|
|
# back, so the heap grew each time a frame drew more than any before (~650 bytes a draw, for as long
|
|
# as the game ran). Off, the objects are made and freed with their command buffer, at no measured
|
|
# cost (3.7 ms either way over 520 actors). Read when the library loads, so before any Vk call.
|
|
function gvk_no_command_pooling() -> void {
|
|
if Os.platform() != "macos" or Os.has_env("MVK_CONFIG_USE_COMMAND_POOLING") { return }
|
|
Os.set_env("MVK_CONFIG_USE_COMMAND_POOLING", "0")
|
|
}
|
|
|
|
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
|
|
function gvk_init(render3d_st: mut Render3dState) -> bool {
|
|
if render3d_st.gvk_ready { return true }
|
|
gvk_no_command_pooling()
|
|
# MoltenVK's host allocations counted for the memory fence (plan 25.1c); every create and destroy
|
|
# is given the same callbacks, as Vulkan asks, so it is decided once, before the instance
|
|
if r3d_env_has(render3d_st, "R3D_ALLOC_VK") { render3d_st.gvk_ac = Vk.alloc_callbacks() }
|
|
gsl_boot(render3d_st)
|
|
if Vk.open() == 0 { render3d_st.gvk_why = "no Vulkan loader"; return false }
|
|
gsl_init(render3d_st)
|
|
|
|
let cnt = bytes(4)
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.enumerate_instance_extension_properties(null, cnt, null)
|
|
let nie = Vk.get_i32(cnt, 0)
|
|
let iexts = bytes(nie * VkExtensionProperties_sizeof + 8)
|
|
Vk.enumerate_instance_extension_properties(null, cnt, iexts)
|
|
# MoltenVK is a portability driver, and is only listed to a program that says it knows
|
|
let portability = gvk_ext_in(iexts, nie, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME)
|
|
let app = bytes(VkApplicationInfo_sizeof)
|
|
Vk.zero(app, VkApplicationInfo_sizeof)
|
|
Vk.put_i32(app, VkApplicationInfo_sType, VK_STRUCTURE_TYPE_APPLICATION_INFO)
|
|
Vk.put_ptr(app, VkApplicationInfo_pApplicationName, "ludic.render3d")
|
|
Vk.put_i32(app, VkApplicationInfo_apiVersion, (1 << 22) | (3 << 12))
|
|
let iext_names = bytes(32)
|
|
var n_iext = 0
|
|
let ici = bytes(VkInstanceCreateInfo_sizeof)
|
|
Vk.zero(ici, VkInstanceCreateInfo_sizeof)
|
|
Vk.put_i32(ici, VkInstanceCreateInfo_sType, VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO)
|
|
Vk.put_ptr(ici, VkInstanceCreateInfo_pApplicationInfo, app)
|
|
if portability {
|
|
Vk.put_ptr(iext_names, n_iext * 8, VK_KHR_PORTABILITY_ENUMERATION_EXTENSION_NAME); n_iext += 1
|
|
Vk.put_i32(ici, VkInstanceCreateInfo_flags, VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR)
|
|
}
|
|
# a window's surface: Win32 on Windows, Metal (MoltenVK) on macOS
|
|
let surf_ext = gvk_surface_ext()
|
|
render3d_st.gvk_has_surface = render3d_st.gvk_want_surface and gvk_ext_in(iexts, nie, VK_KHR_SURFACE_EXTENSION_NAME) and gvk_ext_in(iexts, nie, surf_ext)
|
|
if render3d_st.gvk_want_surface and not render3d_st.gvk_has_surface { render3d_st.gvk_why = `no {surf_ext} in this Vulkan instance`; return false }
|
|
if render3d_st.gvk_has_surface {
|
|
Vk.put_ptr(iext_names, n_iext * 8, VK_KHR_SURFACE_EXTENSION_NAME); n_iext += 1
|
|
Vk.put_ptr(iext_names, n_iext * 8, surf_ext); n_iext += 1
|
|
}
|
|
# R3D_VK_LABELS: name each pass for a GPU capture (MoltenVK makes them Metal debug groups)
|
|
render3d_st.gvk_labels = r3d_env_has(render3d_st, "R3D_VK_LABELS") and gvk_ext_in(iexts, nie, VK_EXT_DEBUG_UTILS_EXTENSION_NAME) and Vk.has("vkCmdBeginDebugUtilsLabelEXT") == 1
|
|
if render3d_st.gvk_labels { Vk.put_ptr(iext_names, n_iext * 8, VK_EXT_DEBUG_UTILS_EXTENSION_NAME); n_iext += 1 }
|
|
# HDR output: an instance only lists the HDR colour spaces when it asks for them
|
|
render3d_st.gvk_has_colorspace = render3d_st.gvk_has_surface and gvk_ext_in(iexts, nie, VK_EXT_SWAPCHAIN_COLOR_SPACE_EXTENSION_NAME)
|
|
if render3d_st.gvk_has_colorspace { Vk.put_ptr(iext_names, n_iext * 8, VK_EXT_SWAPCHAIN_COLOR_SPACE_EXTENSION_NAME); n_iext += 1 }
|
|
if n_iext > 0 {
|
|
Vk.put_i32(ici, VkInstanceCreateInfo_enabledExtensionCount, n_iext)
|
|
Vk.put_ptr(ici, VkInstanceCreateInfo_ppEnabledExtensionNames, iext_names)
|
|
}
|
|
let out = bytes(8)
|
|
var r = Vk.create_instance(ici, render3d_st.gvk_ac, out)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateInstance", r) }
|
|
render3d_st.gvk_inst = Vk.get_ptr(out, 0)
|
|
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.enumerate_physical_devices(render3d_st.gvk_inst, cnt, null)
|
|
let nd = Vk.get_i32(cnt, 0)
|
|
if nd == 0 { render3d_st.gvk_why = "no Vulkan device"; return false }
|
|
let devs = bytes(nd * 8 + 8)
|
|
Vk.enumerate_physical_devices(render3d_st.gvk_inst, cnt, devs)
|
|
let props = bytes(VkPhysicalDeviceProperties_sizeof)
|
|
var pick = 0
|
|
for d in 0 .. nd {
|
|
Vk.get_physical_device_properties(Vk.get_ptr(devs, d * 8), props)
|
|
if Vk.get_i32(props, VkPhysicalDeviceProperties_deviceType) == VK_PHYSICAL_DEVICE_TYPE_DISCRETE_GPU { pick = d; break }
|
|
}
|
|
render3d_st.gvk_pd = Vk.get_ptr(devs, pick * 8)
|
|
Vk.get_physical_device_properties(render3d_st.gvk_pd, props)
|
|
render3d_st.gvk_device_name = string(Vk.at(props, VkPhysicalDeviceProperties_deviceName))
|
|
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.get_physical_device_queue_family_properties(render3d_st.gvk_pd, cnt, null)
|
|
let nq = Vk.get_i32(cnt, 0)
|
|
let qprops = bytes(nq * VkQueueFamilyProperties_sizeof + 8)
|
|
Vk.get_physical_device_queue_family_properties(render3d_st.gvk_pd, cnt, qprops)
|
|
for q in 0 .. nq {
|
|
let flags = Vk.get_i32(qprops, q * VkQueueFamilyProperties_sizeof + VkQueueFamilyProperties_queueFlags)
|
|
if render3d_st.gvk_family < 0 and (flags & VK_QUEUE_GRAPHICS_BIT) != 0 { render3d_st.gvk_family = q }
|
|
}
|
|
if render3d_st.gvk_family < 0 { render3d_st.gvk_why = `{render3d_st.gvk_device_name} has no graphics queue`; return false }
|
|
|
|
# what the device offers, then the same structs handed back asking for the floor
|
|
let f13 = bytes(VkPhysicalDeviceVulkan13Features_sizeof)
|
|
Vk.zero(f13, VkPhysicalDeviceVulkan13Features_sizeof)
|
|
Vk.put_i32(f13, VkPhysicalDeviceVulkan13Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES)
|
|
let f12 = bytes(VkPhysicalDeviceVulkan12Features_sizeof)
|
|
Vk.zero(f12, VkPhysicalDeviceVulkan12Features_sizeof)
|
|
Vk.put_i32(f12, VkPhysicalDeviceVulkan12Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES)
|
|
Vk.put_ptr(f12, VkPhysicalDeviceVulkan12Features_pNext, f13)
|
|
let f2 = bytes(VkPhysicalDeviceFeatures2_sizeof)
|
|
Vk.zero(f2, VkPhysicalDeviceFeatures2_sizeof)
|
|
Vk.put_i32(f2, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
|
|
Vk.put_ptr(f2, VkPhysicalDeviceFeatures2_pNext, f12)
|
|
Vk.get_physical_device_features2(render3d_st.gvk_pd, f2)
|
|
var missing = ""
|
|
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_dynamicRendering) != 1 { missing = missing + " dynamicRendering" }
|
|
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_synchronization2) != 1 { missing = missing + " synchronization2" }
|
|
if Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_descriptorIndexing) != 1 { missing = missing + " descriptorIndexing" }
|
|
if Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_timelineSemaphore) != 1 { missing = missing + " timelineSemaphore" }
|
|
if len(missing) > 0 { render3d_st.gvk_why = `{render3d_st.gvk_device_name} lacks{missing}`; return false }
|
|
# anisotropic filtering is a 1.0 feature every desktop GPU has; ask for it when it is there
|
|
let aniso = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_samplerAnisotropy) == 1
|
|
if aniso { render3d_st.gvk_max_aniso = Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_maxSamplerAnisotropy) }
|
|
let want13 = bytes(VkPhysicalDeviceVulkan13Features_sizeof)
|
|
Vk.zero(want13, VkPhysicalDeviceVulkan13Features_sizeof)
|
|
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES)
|
|
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_dynamicRendering, 1)
|
|
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_synchronization2, 1)
|
|
# maintenance4: glslang's mesh stages declare their work group size with LocalSizeId, which needs it
|
|
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_maintenance4) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_maintenance4, 1) }
|
|
# Streamline's hooks keep their own data on our objects (private data slots), and ask the device
|
|
# to have the feature rather than turning it on themselves
|
|
if render3d_st.gsl_on and Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_privateData) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_privateData, 1) }
|
|
let want12 = bytes(VkPhysicalDeviceVulkan12Features_sizeof)
|
|
Vk.zero(want12, VkPhysicalDeviceVulkan12Features_sizeof)
|
|
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES)
|
|
Vk.put_ptr(want12, VkPhysicalDeviceVulkan12Features_pNext, want13)
|
|
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_descriptorIndexing, 1)
|
|
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_timelineSemaphore, 1)
|
|
let want2 = bytes(VkPhysicalDeviceFeatures2_sizeof)
|
|
Vk.zero(want2, VkPhysicalDeviceFeatures2_sizeof)
|
|
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
|
|
Vk.put_ptr(want2, VkPhysicalDeviceFeatures2_pNext, want12)
|
|
if aniso { Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_samplerAnisotropy, 1) }
|
|
# GPU-driven drawing: one indirect buffer holds a layer's draws, and a compute pass may write
|
|
# how many of them there are. Asked for where the device has them; the renderer checks
|
|
# gvk_has_mdi / gvk_has_dic before it takes that path.
|
|
render3d_st.gvk_has_mdi = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect) == 1 and Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance) == 1
|
|
# BC-compressed textures (BC4/5/7): every desktop GPU, Apple silicon through MoltenVK too; a
|
|
# texture with a .dds beside its .png is uploaded compressed when this is on (texture.ludic)
|
|
render3d_st.gvk_has_bc = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_textureCompressionBC) == 1
|
|
if render3d_st.gvk_has_bc { Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_textureCompressionBC, 1) }
|
|
if render3d_st.gvk_has_mdi {
|
|
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect, 1)
|
|
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance, 1)
|
|
}
|
|
render3d_st.gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1
|
|
if render3d_st.gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) }
|
|
# R3D_PROF: per-pass GPU time from timestamp queries (gvk_query_*). The slots are reused every few
|
|
# frames, so they are reset from the host - which is a feature to ask for.
|
|
render3d_st.gvk_has_hqr = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_hostQueryReset) == 1 and Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampComputeAndGraphics) == 1
|
|
if render3d_st.gvk_has_hqr {
|
|
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_hostQueryReset, 1)
|
|
render3d_st.gvk_ts_period = float_from_bits(Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampPeriod))
|
|
}
|
|
|
|
Vk.put_i32(cnt, 0, 0)
|
|
Vk.enumerate_device_extension_properties(render3d_st.gvk_pd, null, cnt, null)
|
|
let nde = Vk.get_i32(cnt, 0)
|
|
let dexts = bytes(nde * VkExtensionProperties_sizeof + 8)
|
|
Vk.enumerate_device_extension_properties(render3d_st.gvk_pd, null, cnt, dexts)
|
|
# mesh-shader grass: the extension, and its meshShader feature asked for where the device has it.
|
|
# R3D_NO_MESH=1 leaves it off.
|
|
render3d_st.gvk_has_mesh = false
|
|
if gvk_ext_in(dexts, nde, VK_EXT_MESH_SHADER_EXTENSION_NAME) and not r3d_env_has(render3d_st, "R3D_NO_MESH") {
|
|
let fm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
|
Vk.zero(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
|
Vk.put_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT)
|
|
let qm = bytes(VkPhysicalDeviceFeatures2_sizeof)
|
|
Vk.zero(qm, VkPhysicalDeviceFeatures2_sizeof)
|
|
Vk.put_i32(qm, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
|
|
Vk.put_ptr(qm, VkPhysicalDeviceFeatures2_pNext, fm)
|
|
Vk.get_physical_device_features2(render3d_st.gvk_pd, qm)
|
|
if Vk.get_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader) == 1 {
|
|
let wm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
|
Vk.zero(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
|
Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT)
|
|
Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader, 1)
|
|
Vk.put_ptr(want13, VkPhysicalDeviceVulkan13Features_pNext, wm)
|
|
render3d_st.gvk_has_mesh = true
|
|
}
|
|
}
|
|
let prio = bytes(4)
|
|
Vk.put_i32(prio, 0, 0x3F800000)
|
|
let qci = bytes(VkDeviceQueueCreateInfo_sizeof)
|
|
Vk.zero(qci, VkDeviceQueueCreateInfo_sizeof)
|
|
Vk.put_i32(qci, VkDeviceQueueCreateInfo_sType, VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO)
|
|
Vk.put_i32(qci, VkDeviceQueueCreateInfo_queueFamilyIndex, render3d_st.gvk_family)
|
|
Vk.put_i32(qci, VkDeviceQueueCreateInfo_queueCount, 1)
|
|
Vk.put_ptr(qci, VkDeviceQueueCreateInfo_pQueuePriorities, prio)
|
|
let dext_names = bytes(48)
|
|
var n_dext = 0
|
|
let dci = bytes(VkDeviceCreateInfo_sizeof)
|
|
Vk.zero(dci, VkDeviceCreateInfo_sizeof)
|
|
Vk.put_i32(dci, VkDeviceCreateInfo_sType, VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO)
|
|
Vk.put_ptr(dci, VkDeviceCreateInfo_pNext, want2)
|
|
Vk.put_i32(dci, VkDeviceCreateInfo_queueCreateInfoCount, 1)
|
|
Vk.put_ptr(dci, VkDeviceCreateInfo_pQueueCreateInfos, qci)
|
|
if gvk_ext_in(dexts, nde, "VK_KHR_portability_subset") { Vk.put_ptr(dext_names, n_dext * 8, "VK_KHR_portability_subset"); n_dext += 1 }
|
|
if render3d_st.gvk_has_surface {
|
|
if not gvk_ext_in(dexts, nde, VK_KHR_SWAPCHAIN_EXTENSION_NAME) { render3d_st.gvk_why = `{render3d_st.gvk_device_name} has no swapchain`; return false }
|
|
Vk.put_ptr(dext_names, n_dext * 8, VK_KHR_SWAPCHAIN_EXTENSION_NAME); n_dext += 1
|
|
# Reflex: Streamline adds VK_NV_low_latency2 to this device, which needs present ids it does not add
|
|
if render3d_st.gsl_on and gvk_ext_in(dexts, nde, "VK_KHR_present_id") { Vk.put_ptr(dext_names, n_dext * 8, "VK_KHR_present_id"); n_dext += 1 }
|
|
# HDR output tells the display what the picture holds
|
|
# and only where the loader has the command: Streamline's interposer exports no vkSetHdrMetadataEXT,
|
|
# and calling its thunk there crashed the game the moment the swapchain came up HDR10
|
|
render3d_st.gvk_has_hdr_meta = gvk_ext_in(dexts, nde, VK_EXT_HDR_METADATA_EXTENSION_NAME) and Vk.has("vkSetHdrMetadataEXT") == 1
|
|
if render3d_st.gvk_has_hdr_meta { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_HDR_METADATA_EXTENSION_NAME); n_dext += 1 }
|
|
}
|
|
if render3d_st.gvk_has_mesh { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_MESH_SHADER_EXTENSION_NAME); n_dext += 1 }
|
|
if n_dext > 0 {
|
|
Vk.put_i32(dci, VkDeviceCreateInfo_enabledExtensionCount, n_dext)
|
|
Vk.put_ptr(dci, VkDeviceCreateInfo_ppEnabledExtensionNames, dext_names)
|
|
}
|
|
r = Vk.create_device(render3d_st.gvk_pd, dci, render3d_st.gvk_ac, out)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateDevice", r) }
|
|
render3d_st.gvk_dev = Vk.get_ptr(out, 0)
|
|
Vk.get_device_queue(render3d_st.gvk_dev, render3d_st.gvk_family, 0, out)
|
|
render3d_st.gvk_queue = Vk.get_ptr(out, 0)
|
|
gsl_probe_device(render3d_st, render3d_st.gvk_pd)
|
|
if render3d_st.gvk_has_mesh {
|
|
render3d_st.gvk_mesh_fn = Vk.get_device_proc_addr(render3d_st.gvk_dev, "vkCmdDrawMeshTasksEXT")
|
|
if render3d_st.gvk_mesh_fn == null { render3d_st.gvk_has_mesh = false }
|
|
}
|
|
render3d_st.gvk_mp = bytes(VkPhysicalDeviceMemoryProperties_sizeof)
|
|
Vk.get_physical_device_memory_properties(render3d_st.gvk_pd, render3d_st.gvk_mp)
|
|
|
|
if not gvk_cmd_init(render3d_st) { return false }
|
|
render3d_st.gvk_ready = true
|
|
gvk_startup_state(render3d_st)
|
|
print(`r3d: vulkan on {render3d_st.gvk_device_name}, anisotropy up to {gvk_aniso_x(render3d_st.gvk_max_aniso)}x, multi-draw indirect {render3d_st.gvk_has_mdi}, indirect count {render3d_st.gvk_has_dic}, mesh shaders {render3d_st.gvk_has_mesh}`)
|
|
return true
|
|
}
|
|
|
|
# a positive float, as its bits, to a whole number (16.0 -> 16) for a message
|
|
function gvk_aniso_x(bits: int) -> int {
|
|
let e = ((bits >> 23) & 255) - 127
|
|
if e < 0 { return 0 }
|
|
return ((bits & 0x7FFFFF) | 0x800000) >> (23 - e)
|
|
}
|
|
|
|
# ---- memory -------------------------------------------------------------------------------
|
|
# The first memory type a resource allows with every property wanted; -1 if there is none.
|
|
function gvk_mem_type(render3d_st: Render3dState, allowed: int, want: int) -> int {
|
|
for t in 0 .. Vk.get_i32(render3d_st.gvk_mp, VkPhysicalDeviceMemoryProperties_memoryTypeCount) {
|
|
let pf = Vk.get_i32(render3d_st.gvk_mp, VkPhysicalDeviceMemoryProperties_memoryTypes + t * VkMemoryType_sizeof + VkMemoryType_propertyFlags)
|
|
if ((allowed >> t) & 1) == 1 and (pf & want) == want { return t }
|
|
}
|
|
return -1
|
|
}
|
|
# Memory for one resource, from its requirements (a VkMemoryRequirements). One allocation per
|
|
# resource while the backend comes up; the block allocator with sub-allocation replaces it
|
|
# before the forest and the streams are on this backend, which allocate thousands.
|
|
function gvk_alloc(render3d_st: mut Render3dState, req: bytes, want: int) -> long {
|
|
let allowed = Vk.get_i32(req, VkMemoryRequirements_memoryTypeBits)
|
|
var t = gvk_mem_type(render3d_st, allowed, want)
|
|
# device-local is a preference; host-visible is a need
|
|
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(render3d_st, allowed, 0) }
|
|
let zero: long = 0
|
|
if t < 0 { print(`r3d: vulkan: no memory type for properties {want}`); return zero }
|
|
let mai = gvk_tmp(render3d_st, VkMemoryAllocateInfo_sizeof)
|
|
Vk.zero(mai, VkMemoryAllocateInfo_sizeof)
|
|
Vk.put_i32(mai, VkMemoryAllocateInfo_sType, VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO)
|
|
Vk.put_i64(mai, VkMemoryAllocateInfo_allocationSize, Vk.get_i64(req, VkMemoryRequirements_size))
|
|
Vk.put_i32(mai, VkMemoryAllocateInfo_memoryTypeIndex, t)
|
|
let out = gvk_tmp(render3d_st, 8)
|
|
render3d_st.gvk_mk_mem += 1
|
|
let r = Vk.allocate_memory(render3d_st.gvk_dev, mai, render3d_st.gvk_ac, out)
|
|
if r != VK_SUCCESS { print(`r3d: vulkan: vkAllocateMemory failed (VkResult {r})`); return zero }
|
|
render3d_st.gvk_n_allocs += 1
|
|
return gvk_handle(out)
|
|
}
|
|
|
|
# ---- sub-allocation ----------------------------------------------------------------------
|
|
# Device memory in blocks, resources carved out of them: one vkAllocateMemory per resource met the
|
|
# driver's limit in the self-tests' second world (8735 live allocations and a shadow map refused).
|
|
# Blocks are 64 MB of device-local memory or of host-visible, kept apart for images and for
|
|
# buffers (so the driver's granularity between the two never matters), first fit with alignment,
|
|
# and a freed range merges with its neighbours. A host-visible block is mapped once; a buffer's
|
|
# pointer is the block's plus its offset. Anything over 16 MB still gets an allocation of its own:
|
|
# bigger blocks held back video memory a second world then could not get (a 4096^2 shadow map and
|
|
# then its 2048^2 fallback were refused with 71 allocations live on an 8 GB card).
|
|
# An allocation is an id (0 is none), kept where a memory handle used to be.
|
|
const GVK_BLOCK_DEVICE: int = 67108864
|
|
const GVK_BLOCK_HOST: int = 67108864
|
|
const GVK_OWN_OVER: int = 16777216
|
|
|
|
@alloc_ok("made once per resource and kept for its life (a texture, program, sampler, view, layout or memory block is created when first asked for)")
|
|
function gvk_mem_init(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gvk_al_blk != null { return }
|
|
let zero: long = 0
|
|
render3d_st.gvk_blk_mem = new []long; render3d_st.gvk_blk_kind = new []int; render3d_st.gvk_blk_size = new []int; render3d_st.gvk_blk_map = new []pointer
|
|
render3d_st.gvk_fr_blk = new []int; render3d_st.gvk_fr_off = new []int; render3d_st.gvk_fr_len = new []int
|
|
render3d_st.gvk_al_blk = new []int; render3d_st.gvk_al_mem = new []long; render3d_st.gvk_al_off = new []int; render3d_st.gvk_al_len = new []int
|
|
render3d_st.gvk_al_map = new []pointer; render3d_st.gvk_al_spare = new []int
|
|
push(render3d_st.gvk_al_blk, -1); push(render3d_st.gvk_al_mem, zero); push(render3d_st.gvk_al_off, 0); push(render3d_st.gvk_al_len, 0); push(render3d_st.gvk_al_map, null)
|
|
}
|
|
|
|
# one vkAllocateMemory of `size` bytes of memory type t, mapped when host-visible; 0 on failure
|
|
function gvk_mem_raw(render3d_st: mut Render3dState, t: int, size: int, host: bool, out_map: []pointer) -> long {
|
|
let zero: long = 0
|
|
let mai = gvk_tmp(render3d_st, VkMemoryAllocateInfo_sizeof)
|
|
Vk.zero(mai, VkMemoryAllocateInfo_sizeof)
|
|
Vk.put_i32(mai, VkMemoryAllocateInfo_sType, VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO)
|
|
let size_l: long = size
|
|
Vk.put_i64(mai, VkMemoryAllocateInfo_allocationSize, size_l)
|
|
Vk.put_i32(mai, VkMemoryAllocateInfo_memoryTypeIndex, t)
|
|
let out = gvk_tmp(render3d_st, 8)
|
|
render3d_st.gvk_mk_mem += 1
|
|
let r = Vk.allocate_memory(render3d_st.gvk_dev, mai, render3d_st.gvk_ac, out)
|
|
if r != VK_SUCCESS { gvk_note(render3d_st, `r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {render3d_st.gvk_n_allocs} allocations live)`); return zero }
|
|
render3d_st.gvk_n_allocs += 1
|
|
let mem = gvk_handle(out)
|
|
out_map[0] = null
|
|
if host {
|
|
let pp = gvk_tmp(render3d_st, 8)
|
|
if Vk.map_memory(render3d_st.gvk_dev, mem, zero, size_l, 0, pp) != VK_SUCCESS { gvk_note(render3d_st, "r3d: vulkan: vkMapMemory failed"); return zero }
|
|
out_map[0] = Vk.get_ptr(pp, 0)
|
|
}
|
|
return mem
|
|
}
|
|
|
|
function gvk_fr_put(render3d_st: mut Render3dState, b: int, off: int, len_: int) -> void {
|
|
if len_ <= 0 { return }
|
|
for i in 0 .. len(render3d_st.gvk_fr_blk) { if render3d_st.gvk_fr_len[i] == 0 { render3d_st.gvk_fr_blk[i] = b; render3d_st.gvk_fr_off[i] = off; render3d_st.gvk_fr_len[i] = len_; return } }
|
|
push(render3d_st.gvk_fr_blk, b); push(render3d_st.gvk_fr_off, off); push(render3d_st.gvk_fr_len, len_)
|
|
}
|
|
|
|
# Memory for a resource from its requirements (a VkMemoryRequirements): an allocation id, 0 if none.
|
|
@alloc_ok("made once per resource and kept for its life (a texture, program, sampler, view, layout or memory block is created when first asked for)")
|
|
function gvk_mem_new(render3d_st: mut Render3dState, req: bytes, want: int, image: bool) -> int {
|
|
gvk_mem_init(render3d_st)
|
|
let allowed = Vk.get_i32(req, VkMemoryRequirements_memoryTypeBits)
|
|
var t = gvk_mem_type(render3d_st, allowed, want)
|
|
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(render3d_st, allowed, 0) }
|
|
if t < 0 { gvk_note(render3d_st, `r3d: vulkan: no memory type for properties {want}`); return 0 }
|
|
let host = (want & VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT) != 0
|
|
let size = int(Vk.get_i64(req, VkMemoryRequirements_size))
|
|
var align = int(Vk.get_i64(req, VkMemoryRequirements_alignment))
|
|
if align < 1 { align = 1 }
|
|
var blk = -1
|
|
var off = 0
|
|
let zero: long = 0
|
|
# the mapped pointer comes back through a one-slot list made once, not one per allocation
|
|
if render3d_st.gvk_map_slot == null {
|
|
render3d_st.gvk_map_slot = new []pointer
|
|
push(render3d_st.gvk_map_slot, null)
|
|
}
|
|
let mp = render3d_st.gvk_map_slot
|
|
mp[0] = null
|
|
if size <= GVK_OWN_OVER {
|
|
var kind = t * 2
|
|
if image { kind += 1 }
|
|
var i = 0
|
|
while i < len(render3d_st.gvk_fr_blk) and blk < 0 {
|
|
let n = render3d_st.gvk_fr_len[i]
|
|
if n > 0 and render3d_st.gvk_blk_kind[render3d_st.gvk_fr_blk[i]] == kind {
|
|
let start = (render3d_st.gvk_fr_off[i] + align - 1) / align * align
|
|
let end = render3d_st.gvk_fr_off[i] + n
|
|
if start + size <= end {
|
|
blk = render3d_st.gvk_fr_blk[i]; off = start
|
|
let front = start - render3d_st.gvk_fr_off[i]
|
|
let back = end - (start + size)
|
|
if front > 0 { render3d_st.gvk_fr_len[i] = front } else { render3d_st.gvk_fr_len[i] = 0 }
|
|
gvk_fr_put(render3d_st, blk, start + size, back)
|
|
}
|
|
}
|
|
i += 1
|
|
}
|
|
if blk < 0 {
|
|
var bsize = GVK_BLOCK_DEVICE
|
|
if host { bsize = GVK_BLOCK_HOST }
|
|
let mem = gvk_mem_raw(render3d_st, t, bsize, host, mp)
|
|
if mem != 0 {
|
|
# into the record of a block given back, when there is one: a buffer made again every few
|
|
# frames empties a block and makes the next, and each made a new record for good
|
|
blk = -1
|
|
for e in 0 .. len(render3d_st.gvk_blk_kind) { if blk < 0 and render3d_st.gvk_blk_kind[e] == -1 and render3d_st.gvk_blk_mem[e] == 0 { blk = e } }
|
|
if blk < 0 {
|
|
push(render3d_st.gvk_blk_mem, mem); push(render3d_st.gvk_blk_kind, kind); push(render3d_st.gvk_blk_size, bsize); push(render3d_st.gvk_blk_map, mp[0])
|
|
blk = len(render3d_st.gvk_blk_mem) - 1
|
|
} else {
|
|
render3d_st.gvk_blk_mem[blk] = mem; render3d_st.gvk_blk_kind[blk] = kind; render3d_st.gvk_blk_size[blk] = bsize; render3d_st.gvk_blk_map[blk] = mp[0]
|
|
}
|
|
off = 0
|
|
gvk_fr_put(render3d_st, blk, size, bsize - size)
|
|
}
|
|
}
|
|
}
|
|
var own: long = 0
|
|
var map: pointer = null
|
|
if blk < 0 {
|
|
own = gvk_mem_raw(render3d_st, t, size, host, mp)
|
|
if own == 0 { return 0 }
|
|
map = mp[0]
|
|
} else if render3d_st.gvk_blk_map[blk] != null {
|
|
map = mem_off(render3d_st.gvk_blk_map[blk], off)
|
|
}
|
|
var a = 0
|
|
if len(render3d_st.gvk_al_spare) > 0 {
|
|
let spare = render3d_st.gvk_al_spare
|
|
a = spare[len(spare) - 1]
|
|
List.pop(spare)
|
|
} else {
|
|
push(render3d_st.gvk_al_blk, -1); push(render3d_st.gvk_al_mem, zero); push(render3d_st.gvk_al_off, 0); push(render3d_st.gvk_al_len, 0); push(render3d_st.gvk_al_map, null)
|
|
a = len(render3d_st.gvk_al_blk) - 1
|
|
}
|
|
render3d_st.gvk_al_blk[a] = blk; render3d_st.gvk_al_mem[a] = own; render3d_st.gvk_al_off[a] = off; render3d_st.gvk_al_len[a] = size; render3d_st.gvk_al_map[a] = map
|
|
return a
|
|
}
|
|
|
|
function gvk_mem_handle(render3d_st: Render3dState, a: int) -> long {
|
|
if render3d_st.gvk_al_blk[a] < 0 { return render3d_st.gvk_al_mem[a] }
|
|
return render3d_st.gvk_blk_mem[render3d_st.gvk_al_blk[a]]
|
|
}
|
|
function gvk_mem_offset(render3d_st: Render3dState, a: int) -> long {
|
|
let o: long = render3d_st.gvk_al_off[a]
|
|
return o
|
|
}
|
|
function gvk_mem_ptr(render3d_st: Render3dState, a: int) -> pointer { return render3d_st.gvk_al_map[a] }
|
|
|
|
# give an allocation back: its own memory is freed, a range returns to its block and merges
|
|
function gvk_mem_free(render3d_st: mut Render3dState, a: int) -> void {
|
|
if render3d_st.gvk_al_blk == null or a <= 0 or a >= len(render3d_st.gvk_al_blk) or render3d_st.gvk_al_len[a] == 0 { return }
|
|
let b = render3d_st.gvk_al_blk[a]
|
|
let zero: long = 0
|
|
if b < 0 {
|
|
if render3d_st.gvk_al_map[a] != null { Vk.unmap_memory(render3d_st.gvk_dev, render3d_st.gvk_al_mem[a]) }
|
|
render3d_st.gvk_mk_x_mem += 1
|
|
Vk.free_memory(render3d_st.gvk_dev, render3d_st.gvk_al_mem[a], render3d_st.gvk_ac)
|
|
render3d_st.gvk_n_allocs -= 1
|
|
} else {
|
|
var off = render3d_st.gvk_al_off[a]
|
|
var n = render3d_st.gvk_al_len[a]
|
|
# merge with a free range that ends where this starts, and one that starts where this ends
|
|
for i in 0 .. len(render3d_st.gvk_fr_blk) {
|
|
if render3d_st.gvk_fr_len[i] > 0 and render3d_st.gvk_fr_blk[i] == b and render3d_st.gvk_fr_off[i] + render3d_st.gvk_fr_len[i] == off {
|
|
off = render3d_st.gvk_fr_off[i]; n += render3d_st.gvk_fr_len[i]; render3d_st.gvk_fr_len[i] = 0
|
|
}
|
|
}
|
|
for i in 0 .. len(render3d_st.gvk_fr_blk) {
|
|
if render3d_st.gvk_fr_len[i] > 0 and render3d_st.gvk_fr_blk[i] == b and render3d_st.gvk_fr_off[i] == off + n {
|
|
n += render3d_st.gvk_fr_len[i]; render3d_st.gvk_fr_len[i] = 0
|
|
}
|
|
}
|
|
if off == 0 and n == render3d_st.gvk_blk_size[b] {
|
|
# the block is empty again: give it back, or a world swapped out keeps its memory for good
|
|
if render3d_st.gvk_blk_map[b] != null { Vk.unmap_memory(render3d_st.gvk_dev, render3d_st.gvk_blk_mem[b]) }
|
|
render3d_st.gvk_mk_x_mem += 1
|
|
Vk.free_memory(render3d_st.gvk_dev, render3d_st.gvk_blk_mem[b], render3d_st.gvk_ac)
|
|
render3d_st.gvk_n_allocs -= 1
|
|
render3d_st.gvk_blk_mem[b] = zero; render3d_st.gvk_blk_map[b] = null; render3d_st.gvk_blk_kind[b] = -1; render3d_st.gvk_blk_size[b] = 0
|
|
} else {
|
|
gvk_fr_put(render3d_st, b, off, n)
|
|
}
|
|
}
|
|
render3d_st.gvk_al_len[a] = 0; render3d_st.gvk_al_mem[a] = zero; render3d_st.gvk_al_map[a] = null; render3d_st.gvk_al_blk[a] = -1
|
|
push(render3d_st.gvk_al_spare, a)
|
|
}
|
|
function gvk_mem_id(x: long) -> int { return int(x) }
|
|
|
|
# ---- one-shot commands --------------------------------------------------------------------
|
|
# Uploads, bakes and read-backs record into a command buffer, submit it and wait. The frame
|
|
# itself does not go through here.
|
|
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
|
|
function gvk_cmd_init(render3d_st: mut Render3dState) -> bool {
|
|
let cpi = bytes(VkCommandPoolCreateInfo_sizeof)
|
|
Vk.zero(cpi, VkCommandPoolCreateInfo_sizeof)
|
|
Vk.put_i32(cpi, VkCommandPoolCreateInfo_sType, VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO)
|
|
Vk.put_i32(cpi, VkCommandPoolCreateInfo_flags, VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT)
|
|
Vk.put_i32(cpi, VkCommandPoolCreateInfo_queueFamilyIndex, render3d_st.gvk_family)
|
|
let out = bytes(8)
|
|
var r = Vk.create_command_pool(render3d_st.gvk_dev, cpi, render3d_st.gvk_ac, out)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateCommandPool", r) }
|
|
render3d_st.gvk_pool = gvk_handle(out)
|
|
let fci = bytes(VkFenceCreateInfo_sizeof)
|
|
Vk.zero(fci, VkFenceCreateInfo_sizeof)
|
|
Vk.put_i32(fci, VkFenceCreateInfo_sType, VK_STRUCTURE_TYPE_FENCE_CREATE_INFO)
|
|
render3d_st.gvk_fence = bytes(8)
|
|
r = Vk.create_fence(render3d_st.gvk_dev, fci, render3d_st.gvk_ac, render3d_st.gvk_fence)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateFence", r) }
|
|
render3d_st.gvk_frame_fence = bytes(8)
|
|
r = Vk.create_fence(render3d_st.gvk_dev, fci, render3d_st.gvk_ac, render3d_st.gvk_frame_fence)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateFence", r) }
|
|
return true
|
|
}
|
|
# a command buffer, begun
|
|
# ---- scratch -------------------------------------------------------------------------------
|
|
# The structs a Vulkan call reads are copied before it returns, so the ones built for a draw, a
|
|
# pass, a barrier or a submit need not outlive it. They were each a bytes() - a malloc nothing
|
|
# freed, several per draw: about 0.5 MB a frame, and one windowed run grew past 150 GB. They
|
|
# come from this ring instead: one block, handed out in order and wrapped when it is used up.
|
|
const GVK_TMP_BYTES: int = 1048576
|
|
|
|
function gvk_tmp(render3d_st: mut Render3dState, n: int) -> pointer {
|
|
let sz = (n + 15) / 16 * 16
|
|
if sz > GVK_TMP_BYTES { print(`r3d: vulkan: {n} bytes of scratch asked for at once`); return null }
|
|
if render3d_st.gvk_tmp_off + sz > GVK_TMP_BYTES { render3d_st.gvk_tmp_off = 0 }
|
|
let p = mem_off(render3d_st.gvk_tmp_buf, render3d_st.gvk_tmp_off)
|
|
render3d_st.gvk_tmp_off += sz
|
|
return p
|
|
}
|
|
|
|
function gvk_once_begin(render3d_st: mut Render3dState) -> pointer {
|
|
let cbai = gvk_tmp(render3d_st, VkCommandBufferAllocateInfo_sizeof)
|
|
Vk.zero(cbai, VkCommandBufferAllocateInfo_sizeof)
|
|
Vk.put_i32(cbai, VkCommandBufferAllocateInfo_sType, VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO)
|
|
Vk.put_i64(cbai, VkCommandBufferAllocateInfo_commandPool, render3d_st.gvk_pool)
|
|
Vk.put_i32(cbai, VkCommandBufferAllocateInfo_level, VK_COMMAND_BUFFER_LEVEL_PRIMARY)
|
|
Vk.put_i32(cbai, VkCommandBufferAllocateInfo_commandBufferCount, 1)
|
|
let cbs = gvk_tmp(render3d_st, 8)
|
|
render3d_st.gvk_mk_cmd += 1
|
|
if Vk.allocate_command_buffers(render3d_st.gvk_dev, cbai, cbs) != VK_SUCCESS { return null }
|
|
let cb = Vk.get_ptr(cbs, 0)
|
|
let cbbi = gvk_tmp(render3d_st, VkCommandBufferBeginInfo_sizeof)
|
|
Vk.zero(cbbi, VkCommandBufferBeginInfo_sizeof)
|
|
Vk.put_i32(cbbi, VkCommandBufferBeginInfo_sType, VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO)
|
|
Vk.put_i32(cbbi, VkCommandBufferBeginInfo_flags, VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT)
|
|
Vk.begin_command_buffer(cb, cbbi)
|
|
return cb
|
|
}
|
|
# end it, submit it, wait for it, free it
|
|
function gvk_once_end(render3d_st: mut Render3dState, cb: pointer) -> bool {
|
|
gvk_frame_wait(render3d_st)
|
|
var r = Vk.end_command_buffer(cb)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkEndCommandBuffer", r) }
|
|
let cbs = gvk_tmp(render3d_st, 8)
|
|
Vk.put_ptr(cbs, 0, cb)
|
|
Vk.reset_fences(render3d_st.gvk_dev, 1, render3d_st.gvk_fence)
|
|
let si = gvk_tmp(render3d_st, VkSubmitInfo_sizeof)
|
|
Vk.zero(si, VkSubmitInfo_sizeof)
|
|
Vk.put_i32(si, VkSubmitInfo_sType, VK_STRUCTURE_TYPE_SUBMIT_INFO)
|
|
Vk.put_i32(si, VkSubmitInfo_commandBufferCount, 1)
|
|
Vk.put_ptr(si, VkSubmitInfo_pCommandBuffers, cbs)
|
|
r = Vk.queue_submit(render3d_st.gvk_queue, 1, si, Vk.get_i64(render3d_st.gvk_fence, 0))
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkQueueSubmit", r) }
|
|
let forever: long = -1
|
|
r = Vk.wait_for_fences(render3d_st.gvk_dev, 1, render3d_st.gvk_fence, 1, forever)
|
|
Vk.free_command_buffers(render3d_st.gvk_dev, render3d_st.gvk_pool, 1, cbs)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkWaitForFences", r) }
|
|
return true
|
|
}
|
|
|
|
# ---- one frame in flight -------------------------------------------------------------------
|
|
# A window's frame is submitted and presented without waiting, so the game's next tick runs on the
|
|
# CPU while the GPU draws: waiting at the present serialised the two, which cost MoltenVK a third
|
|
# of its frame rate against OpenGL. Anything that would reuse what that frame reads - the next
|
|
# frame's command buffer and ring, a one-off submit, a descriptor pool reset, a swapchain rebuild -
|
|
# waits for it first (gvk_frame_wait), and a buffer it reads is busy (gvk_buf_busy). The frame's
|
|
# uniform ring and descriptor pool are double-buffered by frame_no & 1, so the next frame records
|
|
# while this one draws: waiting at the next frame's start left the GPU idle for the whole of its
|
|
# recording and MoltenVK's encoding (a third of a frame). On by default on macOS; R3D_VK_INFLIGHT=0/1
|
|
# says otherwise.
|
|
function gvk_inflight_on(render3d_st: mut Render3dState) -> bool {
|
|
if render3d_st.gvk_inflight < 0 {
|
|
render3d_st.gvk_inflight = 0
|
|
if Os.platform() == "macos" { render3d_st.gvk_inflight = 1 }
|
|
render3d_st.gvk_nopool = r3d_env_has(render3d_st, "R3D_VK_NOPOOL")
|
|
if r3d_env_has(render3d_st, "R3D_VK_INFLIGHT") { render3d_st.gvk_inflight = Text.to_int(r3d_env(render3d_st, "R3D_VK_INFLIGHT")) }
|
|
}
|
|
return render3d_st.gvk_inflight == 1
|
|
}
|
|
|
|
function gvk_frame_submit(render3d_st: mut Render3dState, cb: pointer) -> bool {
|
|
gvk_frame_wait(render3d_st)
|
|
var r = Vk.end_command_buffer(cb)
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkEndCommandBuffer", r) }
|
|
let cbs = gvk_tmp(render3d_st, 8)
|
|
Vk.put_ptr(cbs, 0, cb)
|
|
Vk.reset_fences(render3d_st.gvk_dev, 1, render3d_st.gvk_frame_fence)
|
|
let si = gvk_tmp(render3d_st, VkSubmitInfo_sizeof)
|
|
Vk.zero(si, VkSubmitInfo_sizeof)
|
|
Vk.put_i32(si, VkSubmitInfo_sType, VK_STRUCTURE_TYPE_SUBMIT_INFO)
|
|
Vk.put_i32(si, VkSubmitInfo_commandBufferCount, 1)
|
|
Vk.put_ptr(si, VkSubmitInfo_pCommandBuffers, cbs)
|
|
r = Vk.queue_submit(render3d_st.gvk_queue, 1, si, Vk.get_i64(render3d_st.gvk_frame_fence, 0))
|
|
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkQueueSubmit", r) }
|
|
render3d_st.gvk_frame_pending = true
|
|
render3d_st.gvk_frame_pending_cb = cb
|
|
render3d_st.gvk_frame_pending_no = render3d_st.gvk_frame_no
|
|
return true
|
|
}
|
|
|
|
function gvk_frame_wait(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gvk_frame_pending { return }
|
|
let forever: long = -1
|
|
Vk.wait_for_fences(render3d_st.gvk_dev, 1, render3d_st.gvk_frame_fence, 1, forever)
|
|
let cbs = gvk_tmp(render3d_st, 8)
|
|
Vk.put_ptr(cbs, 0, render3d_st.gvk_frame_pending_cb)
|
|
Vk.free_command_buffers(render3d_st.gvk_dev, render3d_st.gvk_pool, 1, cbs)
|
|
render3d_st.gvk_frame_pending = false
|
|
render3d_st.gvk_frame_pending_cb = null
|
|
gvk_retire_upto(render3d_st, render3d_st.gvk_frame_pending_no)
|
|
}
|
|
|
|
# ---- teardown -----------------------------------------------------------------------------
|
|
function gvk_shutdown(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gvk_ready { return }
|
|
gvk_frame_wait(render3d_st)
|
|
Vk.device_wait_idle(render3d_st.gvk_dev)
|
|
gsl_shutdown(render3d_st)
|
|
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_fence, 0), render3d_st.gvk_ac)
|
|
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_frame_fence, 0), render3d_st.gvk_ac)
|
|
Vk.destroy_command_pool(render3d_st.gvk_dev, render3d_st.gvk_pool, render3d_st.gvk_ac)
|
|
if render3d_st.gvk_prime_dst != 0 { Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_prime_dst, render3d_st.gvk_ac); render3d_st.gvk_prime_dst = 0 }
|
|
Vk.destroy_device(render3d_st.gvk_dev, render3d_st.gvk_ac)
|
|
Vk.destroy_instance(render3d_st.gvk_inst, render3d_st.gvk_ac)
|
|
render3d_st.gvk_ready = false
|
|
}
|
|
|
|
# ---- GPU timestamps (R3D_PROF) ------------------------------------------------------------
|
|
# prof.ludic's query slots, each a pair of timestamps: 2 * id where the pass starts, 2 * id + 1 where it
|
|
# ends. A slot is read back PROF_RING frames after it was written and reset just before it is written
|
|
# again. On a device without host query reset or timestamps, every read says "not yet" and the report
|
|
# stays empty.
|
|
function gvk_query_new(render3d_st: mut Render3dState, n: int, ids: words) -> void {
|
|
for i in 0 .. n { ids[i] = i }
|
|
if not render3d_st.gvk_has_hqr or render3d_st.gvk_dev == null { return }
|
|
let qci = gvk_tmp(render3d_st, VkQueryPoolCreateInfo_sizeof)
|
|
Vk.zero(qci, VkQueryPoolCreateInfo_sizeof)
|
|
Vk.put_i32(qci, VkQueryPoolCreateInfo_sType, VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO)
|
|
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryType, VK_QUERY_TYPE_TIMESTAMP)
|
|
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryCount, n * 2)
|
|
let out = gvk_tmp(render3d_st, 8)
|
|
if Vk.create_query_pool(render3d_st.gvk_dev, qci, render3d_st.gvk_ac, out) != VK_SUCCESS { return }
|
|
render3d_st.gvk_qpool = gvk_handle(out)
|
|
Vk.reset_query_pool(render3d_st.gvk_dev, render3d_st.gvk_qpool, 0, n * 2)
|
|
}
|
|
function gvk_query_begin(render3d_st: mut Render3dState, id: int) -> void {
|
|
if render3d_st.gvk_qpool == 0 { return }
|
|
Vk.reset_query_pool(render3d_st.gvk_dev, render3d_st.gvk_qpool, id * 2, 2)
|
|
Vk.cmd_write_timestamp(gvk_frame_cb(render3d_st), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, render3d_st.gvk_qpool, id * 2)
|
|
render3d_st.gvk_q_active = id
|
|
}
|
|
function gvk_query_end(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gvk_qpool == 0 or render3d_st.gvk_q_active < 0 { return }
|
|
Vk.cmd_write_timestamp(gvk_frame_cb(render3d_st), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, render3d_st.gvk_qpool, render3d_st.gvk_q_active * 2 + 1)
|
|
render3d_st.gvk_q_active = -1
|
|
}
|
|
function gvk_query_result(render3d_st: mut Render3dState, id: int, out: words) -> bool {
|
|
if render3d_st.gvk_qpool == 0 { return false }
|
|
let data = gvk_tmp(render3d_st, 16)
|
|
let size: long = 16
|
|
let stride: long = 8
|
|
if Vk.get_query_pool_results(render3d_st.gvk_dev, render3d_st.gvk_qpool, id * 2, 2, size, data, stride, VK_QUERY_RESULT_64_BIT) != VK_SUCCESS { return false }
|
|
let ticks = int(Vk.get_i64(data, 8) - Vk.get_i64(data, 0))
|
|
if ticks < 0 { return false }
|
|
out[0] = int(float(ticks) * render3d_st.gvk_ts_period)
|
|
return true
|
|
}
|
|
|
|
# ---- debug labels (R3D_VK_LABELS) ------------------------------------------------------------
|
|
function gvk_label_begin(render3d_st: mut Render3dState, name: pointer) -> void {
|
|
# only inside a frame: a pass the load profiles comes before the frame's pools exist
|
|
if render3d_st.gpu_kind != GPU_VK or render3d_st.gvk_cb == null { return }
|
|
let cb = render3d_st.gvk_cb
|
|
let li = gvk_tmp(render3d_st, VkDebugUtilsLabelEXT_sizeof)
|
|
Vk.zero(li, VkDebugUtilsLabelEXT_sizeof)
|
|
Vk.put_i32(li, VkDebugUtilsLabelEXT_sType, VK_STRUCTURE_TYPE_DEBUG_UTILS_LABEL_EXT)
|
|
Vk.put_ptr(li, VkDebugUtilsLabelEXT_pLabelName, name)
|
|
Vk.cmd_begin_debug_utils_label_ext(cb, li)
|
|
render3d_st.gvk_label_depth += 1
|
|
}
|
|
function gvk_label_end(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gpu_kind != GPU_VK or render3d_st.gvk_label_depth <= 0 or render3d_st.gvk_cb == null { return }
|
|
Vk.cmd_end_debug_utils_label_ext(render3d_st.gvk_cb)
|
|
render3d_st.gvk_label_depth -= 1
|
|
}
|
|
# the frame's command buffer ends with every label it opened closed (called where it ends: a
|
|
# pointer compares by what it points at, so once_end cannot ask which buffer it was handed)
|
|
function gvk_labels_close(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gvk_cb == null { render3d_st.gvk_label_depth = 0; return }
|
|
while render3d_st.gvk_label_depth > 0 {
|
|
Vk.cmd_end_debug_utils_label_ext(render3d_st.gvk_cb)
|
|
render3d_st.gvk_label_depth -= 1
|
|
}
|
|
}
|
|
|
|
# Everything a frame's draws use as scratch or as a table, made once the device is up: each was made
|
|
# on the first draw that needed it, which put an allocation in whichever frame that was (plan 25.2)
|
|
const GVK_PROGS_MAX: int = 4096
|
|
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
|
|
function gvk_startup_state(render3d_st: mut Render3dState) -> void {
|
|
let zero: long = 0
|
|
render3d_st.gvk_tmp_buf = bytes(GVK_TMP_BYTES)
|
|
render3d_st.gvk_skip_said = words(4096)
|
|
for i in 0 .. 4096 { render3d_st.gvk_skip_said[i] = 0 }
|
|
render3d_st.gvk_seen = words(GPU_MAX_VBUFS)
|
|
if render3d_st.gvk_state == null { render3d_st.gvk_state = new GvkState }
|
|
render3d_st.gvk_size_buf = words(4)
|
|
if render3d_st.gvk_tex_spare == null { render3d_st.gvk_tex_spare = new []int }
|
|
if render3d_st.gvk_buf_spare == null { render3d_st.gvk_buf_spare = new []int }
|
|
if render3d_st.gvk_retired_frame == null { render3d_st.gvk_retired_frame = new []int }
|
|
if render3d_st.gvk_buf_gpu == null { render3d_st.gvk_buf_gpu = new []int }
|
|
render3d_st.gvk_sc_tmp = new []long
|
|
let sct = render3d_st.gvk_sc_tmp
|
|
for i in 0 .. 130 { push(sct, zero) }
|
|
render3d_st.gvk_set_offs = bytes(16)
|
|
render3d_st.gvk_ub_frame = new []int; render3d_st.gvk_ub_offv = new []int; render3d_st.gvk_ub_offf = new []int
|
|
let ubf = render3d_st.gvk_ub_frame
|
|
let ubv = render3d_st.gvk_ub_offv
|
|
let ubo = render3d_st.gvk_ub_offf
|
|
for i in 0 .. GVK_PROGS_MAX {
|
|
push(ubf, 0); push(ubv, 0); push(ubo, 0)
|
|
}
|
|
render3d_st.gvk_pc_prog = new []int; render3d_st.gvk_pc_layout = new []int; render3d_st.gvk_pc_state = new []int; render3d_st.gvk_pc_bias = new []int
|
|
render3d_st.gvk_pc_pass = new []int; render3d_st.gvk_pc_pipe = new []long
|
|
render3d_st.gvk_pc_last = words(4096)
|
|
for i in 0 .. 4096 { render3d_st.gvk_pc_last[i] = -1 }
|
|
render3d_st.gg_rb = words(4); render3d_st.gg_cb = words(1); render3d_st.gg_bufs = words(3)
|
|
render3d_st.gg_texs = words(3); render3d_st.gg_pr = words(48)
|
|
if render3d_st.gpu_u_tmp == null { render3d_st.gpu_u_tmp = words(4) }
|
|
if render3d_st.gpu_unit_2d == null {
|
|
render3d_st.gpu_unit_2d = words(32)
|
|
for i in 0 .. 32 { render3d_st.gpu_unit_2d[i] = -1 }
|
|
}
|
|
if render3d_st.cam_planes == null { render3d_st.cam_planes = floats(16) }
|
|
if render3d_st.ov_nine_buf == null { render3d_st.ov_nine_buf = floats(16) }
|
|
if render3d_st.ac_lp0 == null { render3d_st.ac_lp0 = floats(3); render3d_st.ac_lpx = floats(3); render3d_st.ac_lpy = floats(3); render3d_st.ac_lpz = floats(3) }
|
|
}
|