diff --git a/packages/ludic.render3d/camera.ludic b/packages/ludic.render3d/camera.ludic index 69827155..9246258e 100644 --- a/packages/ludic.render3d/camera.ludic +++ b/packages/ludic.render3d/camera.ludic @@ -36,6 +36,7 @@ function cam_init(aspect: int) -> void { cam_update() } function cam_begin_frame(n: int, w: int, h: int) -> void { + gsl_jitter_frame() cam_update() } function cam_set(x: int, y: int, z: int, yaw_deg: int, pitch_deg: int) -> void { @@ -56,6 +57,12 @@ function cam_update() -> void { m4_perspective(cam_proj, cam_fov, cam_aspect, cam_near, cam_far) m4_mul(cam_vp_clean, cam_proj, cam_view) m4_inverse(cam_inv_vp_clean, cam_vp_clean) + # DLSS's sub-pixel jitter (streamline.ludic), in the projection the scene draws with only: + # culling, depth reconstruction and last frame's matrix keep the clean one + if gsl_jitter_x != 0 or gsl_jitter_y != 0 { + cam_proj[8] = f_sub(cam_proj[8], gsl_jitter_x) + cam_proj[9] = f_sub(cam_proj[9], gsl_jitter_y) + } m4_mul(cam_vp, cam_proj, cam_view) m4_inverse(cam_inv_vp, cam_vp) m4_inverse(cam_inv_proj, cam_proj) diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index da690a3a..3e907d2e 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -386,9 +386,10 @@ function gpu_caps_probe() -> void { } # Whether the renderer actually draws a feature yet. Until a feature lands, choosing it is saved and -# shown, and says it takes effect later. The Vulkan renderer itself draws the whole game now (phase -# 37-38); ray tracing, DLSS, Reflex, HDR output and mesh-shader grass do not yet. -function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN } +# shown, and says it takes effect later. The Vulkan renderer draws the whole game (phase 37-38), and +# DLSS super resolution and Reflex run through NVIDIA Streamline (streamline.ludic); whether this +# machine can use one is the caps' question, not this one. +function gpu_feature_implemented(f: int) -> bool { return f == GF_VULKAN or f == GF_DLSS or f == GF_REFLEX } # ---- vertex data -------------------------------------------------------------------- # A Mesh is built through these and records what it is made of - which buffer feeds which diff --git a/packages/ludic.render3d/gpu_vk.ludic b/packages/ludic.render3d/gpu_vk.ludic index 265c7cd7..01cad858 100644 --- a/packages/ludic.render3d/gpu_vk.ludic +++ b/packages/ludic.render3d/gpu_vk.ludic @@ -61,7 +61,9 @@ var gvk_has_dic: bool = false # drawIndirectCount function gvk_init() -> bool { if gvk_ready { return true } + gsl_boot() if Vk.open() == 0 { gvk_why = "no Vulkan loader"; return false } + gsl_init() let cnt = bytes(4) Vk.put_i32(cnt, 0, 0) @@ -156,6 +158,9 @@ function gvk_init() -> bool { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES) Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_dynamicRendering, 1) Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_synchronization2, 1) + # Streamline's hooks keep their own data on our objects (private data slots), and ask the device + # to have the feature rather than turning it on themselves + if gsl_on and Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_privateData) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_privateData, 1) } let want12 = bytes(VkPhysicalDeviceVulkan12Features_sizeof) Vk.zero(want12, VkPhysicalDeviceVulkan12Features_sizeof) Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES) @@ -210,6 +215,8 @@ function gvk_init() -> bool { if gvk_has_surface { if not gvk_ext_in(dexts, nde, VK_KHR_SWAPCHAIN_EXTENSION_NAME) { gvk_why = `{gvk_device_name} has no swapchain`; return false } Vk.put_ptr(dext_names, n_dext * 8, VK_KHR_SWAPCHAIN_EXTENSION_NAME); n_dext += 1 + # Reflex: Streamline adds VK_NV_low_latency2 to this device, which needs present ids it does not add + if gsl_on and gvk_ext_in(dexts, nde, "VK_KHR_present_id") { Vk.put_ptr(dext_names, n_dext * 8, "VK_KHR_present_id"); n_dext += 1 } } if n_dext > 0 { Vk.put_i32(dci, VkDeviceCreateInfo_enabledExtensionCount, n_dext) @@ -220,6 +227,7 @@ function gvk_init() -> bool { gvk_dev = Vk.get_ptr(out, 0) Vk.get_device_queue(gvk_dev, gvk_family, 0, out) gvk_queue = Vk.get_ptr(out, 0) + gsl_probe_device(gvk_pd) gvk_mp = bytes(VkPhysicalDeviceMemoryProperties_sizeof) Vk.get_physical_device_memory_properties(gvk_pd, gvk_mp) @@ -521,6 +529,7 @@ function gvk_once_end(cb: pointer) -> bool { function gvk_shutdown() -> void { if not gvk_ready { return } Vk.device_wait_idle(gvk_dev) + gsl_shutdown() Vk.destroy_fence(gvk_dev, Vk.get_i64(gvk_fence, 0), null) Vk.destroy_command_pool(gvk_dev, gvk_pool, null) Vk.destroy_device(gvk_dev, null) diff --git a/packages/ludic.render3d/gpu_vk_draw.ludic b/packages/ludic.render3d/gpu_vk_draw.ludic index 7d53e9d3..b927d0b3 100644 --- a/packages/ludic.render3d/gpu_vk_draw.ludic +++ b/packages/ludic.render3d/gpu_vk_draw.ludic @@ -1678,8 +1678,10 @@ function gvk_present_window() -> void { Vk.put_i32(pi, VkPresentInfoKHR_swapchainCount, 1) Vk.put_ptr(pi, VkPresentInfoKHR_pSwapchains, chains) Vk.put_ptr(pi, VkPresentInfoKHR_pImageIndices, idx) + gsl_before_present() r = Vk.queue_present_khr(gvk_queue, pi) if r == VK_ERROR_OUT_OF_DATE_KHR or r == VK_SUBOPTIMAL_KHR { gvk_swap_stale = true } + gsl_after_present() } # a window's client area changed: the screen images and the swapchain follow it diff --git a/packages/ludic.render3d/gpu_vk_res.ludic b/packages/ludic.render3d/gpu_vk_res.ludic index c50e169a..8250c33a 100644 --- a/packages/ludic.render3d/gpu_vk_res.ludic +++ b/packages/ludic.render3d/gpu_vk_res.ludic @@ -171,6 +171,9 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer if samples > 1 { levels = 1 } var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT if depth { usage = usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } else { usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT } + # DLSS reads and writes the frame from compute: a float colour target is storage too while + # Streamline is running (8-bit sRGB formats cannot be, so only the float ones) + if gsl_on and samples == 1 and (vkfmt == VK_FORMAT_R16G16B16A16_SFLOAT or vkfmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT } let ici = bytes(VkImageCreateInfo_sizeof) Vk.zero(ici, VkImageCreateInfo_sizeof) Vk.put_i32(ici, VkImageCreateInfo_sType, VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO) diff --git a/packages/ludic.render3d/post.ludic b/packages/ludic.render3d/post.ludic index 3093a81f..e5ea51f2 100644 --- a/packages/ludic.render3d/post.ludic +++ b/packages/ludic.render3d/post.ludic @@ -42,6 +42,8 @@ var post_depth_copy: Target = null var post_prev: Target = null # last frame's scene colour, for the SSGI bounce only var post_scene: Target = null # this frame's scene colour before the water, for refraction var post_frame: int = 0 +var post_color_w: int = 0 # its size: the display's when DLSS upscaled it +var post_color_h: int = 0 var post_color: int = 0 # the HDR colour the rest of post reads # the resolved depth, copied so passes can read it while drawing into the frame var post_p_sharp: int = 0 var post_sharpen: int = 0 @@ -258,7 +260,7 @@ function post_bloom_pass() -> void { gpu_depth_test(false) gpu_blend(false) var src = post_color - var sw = post_w; var sh = post_h + var sw = post_color_w; var sh = post_color_h gpu_use_program(post_p_down) for i in 0 .. BLOOM_LEVELS { let t = post_bloom[i] @@ -315,7 +317,7 @@ function post_tonemap(color_tex: int) -> void { gpu_viewport(0, 0, gl_w, gl_h) gpu_use_program(post_p_sharp) r3d_bind_2d(post_p_sharp, "u_src", 0, post_ldr.color) - u_f2(gpu_uniform(post_p_sharp, "u_texel"), fr(1, post_w), fr(1, post_h)) + u_f2(gpu_uniform(post_p_sharp, "u_texel"), fr(1, post_ldr.w), fr(1, post_ldr.h)) u_f(gpu_uniform(post_p_sharp, "u_amount"), post_sharpen) u_f(gpu_uniform(post_p_sharp, "u_grain"), post_grain) u_f(gpu_uniform(post_p_sharp, "u_time"), r3d_time) diff --git a/packages/ludic.render3d/r3d.ludic b/packages/ludic.render3d/r3d.ludic index 82c5b680..104bc5a2 100644 --- a/packages/ludic.render3d/r3d.ludic +++ b/packages/ludic.render3d/r3d.ludic @@ -29,4 +29,5 @@ import "actor.ludic" import "stream.ludic" import "grass.ludic" import "water.ludic" +import "streamline.ludic" import "render.ludic" diff --git a/packages/ludic.render3d/render.ludic b/packages/ludic.render3d/render.ludic index 618d88f5..eb198125 100644 --- a/packages/ludic.render3d/render.ludic +++ b/packages/ludic.render3d/render.ludic @@ -179,6 +179,7 @@ function r3d_frame(time: int) -> void { r3d_resize() } r3d_time = time + gsl_frame_start() # the frame's counters close here, before any of its own work: the window each of them # covers is exactly one frame, from this point to the same point next time prof_gen_frame() @@ -246,7 +247,9 @@ function r3d_frame(time: int) -> void { water_draw(post_depth_copy.depth) prof_end() } - post_color = post_hdr.color + post_color = post_hdr.color; post_color_w = post_w; post_color_h = post_h + # DLSS super resolution: the lit frame up to the display's size, before anything reads it + if gsl_dlss_live() { prof_begin("DLSS"); post_color = gsl_dlss_eval(); prof_end() } if not post_no_gi { prof_begin("SSAO/GI"); post_ssao_pass(); prof_end() } if r3d_debug_max { tex_max(post_hdr.color, post_hdr.w, post_hdr.h, "hdr") } prof_begin("bloom") diff --git a/packages/ludic.render3d/streamline.ludic b/packages/ludic.render3d/streamline.ludic new file mode 100644 index 00000000..d2ec902a --- /dev/null +++ b/packages/ludic.render3d/streamline.ludic @@ -0,0 +1,480 @@ +# ============================================================================ +# streamline.ludic — NVIDIA Streamline (SDK 2.14.1) on the Vulkan renderer: DLSS +# super resolution, Reflex and its latency markers. +# +# On Windows the loader opens sl.interposer.dll instead of vulkan-1.dll when it is +# beside the executable (runtime/native/vk_win.ll). slInit has to run before the +# instance is made - Streamline's own vkCreateInstance / vkCreateDevice add what its +# plugins need - so gvk_init calls gsl_boot() before Vk.open() and gsl_init() straight +# after. Nothing here runs on macOS, on OpenGL, or where the DLLs are missing: every +# entry point checks gsl_on and the renderer draws exactly as it did. +# +# Every sl structure starts with next, a GUID and a size_t version (32 bytes); the +# offsets below are the SDK headers' x64 layout. GUIDs are written as 16-bit halves so +# no literal needs more than 31 bits. +# ============================================================================ + +const GSL_DLSS: int = 0 +const GSL_REFLEX: int = 3 +const GSL_PCL: int = 4 +const GSL_DLSS_G: int = 1000 +const GSL_DLSS_RR: int = 1001 + +# sl::PCLMarker +const GSL_SIM_START: int = 0 +const GSL_SIM_END: int = 1 +const GSL_SUBMIT_START: int = 2 +const GSL_SUBMIT_END: int = 3 +const GSL_PRESENT_START: int = 4 +const GSL_PRESENT_END: int = 5 + +const GSL_LAYOUT_READ: int = 5 # VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL + +var gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in +var gsl_dlss_ok: bool = false # what this adapter can run, from slIsFeatureSupported +var gsl_rr_ok: bool = false +var gsl_fg_ok: bool = false +var gsl_reflex_ok: bool = false +var gsl_pcl_ok: bool = false + +var gsl_feats: bytes = null # kept alive: Preferences points at it +var gsl_logdir: bytes = null # and at this +var gsl_vp: bytes = null # sl::ViewportHandle 0, the one view +var gsl_tok_buf: bytes = null +var gsl_idx_buf: bytes = null +var gsl_token: pointer = null # this frame's sl::FrameToken, owned by Streamline +var gsl_frame_n: int = 0 +var gsl_said: bool = false # one failure message, not one a frame +var gsl_fresh: bool = false # a token was taken since the last frame started (a present took it) + +# ---- structure headers ------------------------------------------------------------------ +function gsl_le16(v: int) -> int { return ((v >> 8) & 255) | ((v & 255) << 8) } +function gsl_header(p: pointer, a: int, b: int, c: int, d: int, e: int, f: int, g: int, h: int, version: int) -> void { + Vk.put_i32(p, 8, (a << 16) | b) + Vk.put_i32(p, 12, c | (d << 16)) + Vk.put_i32(p, 16, gsl_le16(e) | (gsl_le16(f) << 16)) + Vk.put_i32(p, 20, gsl_le16(g) | (gsl_le16(h) << 16)) + let v: long = version + Vk.put_i64(p, 24, v) +} +function gsl_struct(size: int) -> bytes { + let p = bytes(size) + Vk.zero(p, size) + return p +} + +# a feature's own function (slDLSSSetOptions, slReflexSleep, ...), or null +function gsl_fn(feature: int, name: string) -> pointer { + let out = bytes(8) + let zero: long = 0 + Vk.put_i64(out, 0, zero) + if Vk.sl_get_feature_function(feature, name, out) != 0 { return null } + return Vk.get_ptr(out, 0) +} + +# ---- start and stop --------------------------------------------------------------------- +# Before Vk.open(): ask the loader for the interposer. R3D_NO_STREAMLINE=1 keeps plain Vulkan. +function gsl_boot() -> void { + if Os.platform() != "windows" { return } + if Os.has_env("R3D_NO_STREAMLINE") { return } + Vk.sl_prefer(1) +} + +# After Vk.open(), before vkCreateInstance. +function gsl_init() -> void { + if gsl_on or Vk.sl_active() == 0 { return } + gsl_feats = gsl_struct(20) + Vk.put_i32(gsl_feats, 0, GSL_DLSS); Vk.put_i32(gsl_feats, 4, GSL_REFLEX); Vk.put_i32(gsl_feats, 8, GSL_PCL) + Vk.put_i32(gsl_feats, 12, GSL_DLSS_RR); Vk.put_i32(gsl_feats, 16, GSL_DLSS_G) + # sl::Preferences + let pref = gsl_struct(144) + gsl_header(pref, 0x1ca1, 0x0965, 0xbf8e, 0x432b, 0x8da1, 0x6716, 0xd879, 0xfb14, 1) + # logLevel eOff; R3D_SL_LOG= writes Streamline's own verbose log (sl.log) there + var level = 0 + if Os.has_env("R3D_SL_LOG") { + level = 2 + let dir: pointer = Os.env("R3D_SL_LOG") + let n = Text.length(Os.env("R3D_SL_LOG")) + gsl_logdir = gsl_struct(n * 2 + 8) # pathToLogsAndData is a wide string + for i in 0 .. n { Vk.put_i32(gsl_logdir, i * 2, dir[i] & 255) } + Vk.put_ptr(pref, 56, gsl_logdir) + } + Vk.put_i32(pref, 36, level) + # eDisableCLStateTracking | eAllowOTA | eLoadDownloadedPlugins | eUseFrameBasedResourceTagging + let flags: long = 1 | 8 | 64 | 128 + Vk.put_i64(pref, 88, flags) + Vk.put_ptr(pref, 96, gsl_feats) + Vk.put_i32(pref, 104, 5) + Vk.put_i32(pref, 112, 0) # engine eCustom + Vk.put_ptr(pref, 120, "ludic render3d") + Vk.put_ptr(pref, 128, "a0f57b54-1daf-4934-90ae-c4035c19df04") + Vk.put_i32(pref, 136, 2) # renderAPI eVulkan + let r = Vk.sl_init(pref, gsl_sdk_version()) + if r != 0 { print(`r3d: streamline: slInit failed ({r}); DLSS and Reflex are off`); return } + gsl_on = true + gsl_vp = gsl_struct(40) + gsl_header(gsl_vp, 0x171b, 0x6435, 0x9b3c, 0x4fc8, 0x9994, 0xfbe5, 0x2569, 0xaaa4, 1) + gsl_tok_buf = bytes(8) + gsl_idx_buf = bytes(4) +} + +# kSDKVersion, built from longs: (2 << 48) on ints is computed in 32 bits and arrives as garbage, +# which slInit answers with eErrorInvalidParameter before it has even opened a log +function gsl_sdk_version() -> long { + let major: long = 2 + let minor: long = 14 + let patch: long = 1 + let magic: long = 0xfedc + return (major << 48) | (minor << 32) | (patch << 16) | magic +} + +# After vkCreateDevice: what this adapter supports. +function gsl_probe_device(pd: pointer) -> void { + if not gsl_on { return } + let ai = gsl_struct(56) # sl::AdapterInfo + gsl_header(ai, 0x0677, 0x315f, 0xa746, 0x4492, 0x9f42, 0xcb61, 0x42c9, 0xc3d4, 1) + Vk.put_ptr(ai, 48, pd) + gsl_dlss_ok = Vk.sl_is_feature_supported(GSL_DLSS, ai) == 0 + gsl_rr_ok = Vk.sl_is_feature_supported(GSL_DLSS_RR, ai) == 0 + gsl_fg_ok = Vk.sl_is_feature_supported(GSL_DLSS_G, ai) == 0 + gsl_reflex_ok = Vk.sl_is_feature_supported(GSL_REFLEX, ai) == 0 + gsl_pcl_ok = Vk.sl_is_feature_supported(GSL_PCL, ai) == 0 + print(`r3d: streamline: DLSS {gsl_dlss_ok}, ray reconstruction {gsl_rr_ok}, frame generation {gsl_fg_ok}, Reflex {gsl_reflex_ok}`) +} + +# Before the device goes. +function gsl_shutdown() -> void { + if not gsl_on { return } + Vk.sl_shutdown() + gsl_on = false +} + +# ---- frames, Reflex and the latency markers --------------------------------------------- +# A frame's token is taken straight after the previous present, where Reflex sleeps: the +# wait lands before the game reads input and simulates, which is the latency it removes. +var gsl_reflex_mode: int = 0 # 0 off, 1 low latency, 2 low latency + boost +var gsl_reflex_applied: int = -1 +var gsl_f_sleep: pointer = null +var gsl_f_marker: pointer = null + +function r3d_reflex(mode: int) -> void { + gsl_reflex_mode = mode + if Os.has_env("R3D_REFLEX") { gsl_reflex_mode = Text.to_int(Os.env("R3D_REFLEX")) } +} +function r3d_reflex_live() -> bool { return gsl_on and gsl_reflex_ok and gsl_reflex_mode > 0 } + +function gsl_marker(m: int) -> void { + if not gsl_pcl_ok or gsl_token == null { return } + if gsl_f_marker == null { gsl_f_marker = gsl_fn(GSL_PCL, "slPCLSetMarker") } + if gsl_f_marker != null { Vk.sl_call_ip(gsl_f_marker, m, gsl_token) } +} + +function gsl_reflex_apply() -> void { + if not gsl_reflex_ok or gsl_reflex_applied == gsl_reflex_mode { return } + let f = gsl_fn(GSL_REFLEX, "slReflexSetOptions") + if f == null { return } + let o = gsl_struct(48) # sl::ReflexOptions + gsl_header(o, 0xf03a, 0xf81a, 0x6d0b, 0x4902, 0xa651, 0xc496, 0x5e21, 0x5434, 1) + Vk.put_i32(o, 32, gsl_reflex_mode) + if gsl_pcl_ok { Vk.put_i32(o, 40, 1) } # useMarkersToOptimize + if Vk.sl_call_p(f, o) == 0 { gsl_reflex_applied = gsl_reflex_mode } +} + +function gsl_new_frame() -> void { + Vk.put_i32(gsl_idx_buf, 0, gsl_frame_n) + gsl_frame_n += 1 + gsl_token = null + gsl_fresh = true + if Vk.sl_get_new_frame_token(gsl_tok_buf, gsl_idx_buf) == 0 { gsl_token = Vk.get_ptr(gsl_tok_buf, 0) } + if gsl_token == null { return } + gsl_reflex_apply() + if r3d_reflex_live() { + if gsl_f_sleep == null { gsl_f_sleep = gsl_fn(GSL_REFLEX, "slReflexSleep") } + if gsl_f_sleep != null { Vk.sl_call_p(gsl_f_sleep, gsl_token) } + } + gsl_marker(GSL_SIM_START) +} + +# r3d_frame's first line: the game has simulated, the renderer starts recording +function gsl_frame_start() -> void { + if not gsl_on or gpu_kind != GPU_VK { return } + # a headless run never presents: each frame takes its own token here + if not gsl_fresh { gsl_new_frame() } + gsl_fresh = false + gsl_marker(GSL_SIM_END) + gsl_marker(GSL_SUBMIT_START) +} +function gsl_before_present() -> void { + if not gsl_on { return } + gsl_marker(GSL_SUBMIT_END) + gsl_marker(GSL_PRESENT_START) +} +function gsl_after_present() -> void { + if not gsl_on { return } + gsl_marker(GSL_PRESENT_END) + gsl_new_frame() +} + +# ---- DLSS super resolution ---------------------------------------------------------------- +# The lit HDR frame (after the water, before occlusion, bloom and the tonemap) is upscaled to +# the display's size; everything after it reads the result by UV, so only the LDR image and the +# sharpen pass change size. The projection is jittered on a Halton (2, 3) cycle while DLSS is on. +# There are no per-object motion vectors yet: a zero target is tagged and Streamline adds the +# camera's own motion from depth and clipToPrevClip. +var gsl_dlss_mode: int = 0 # 0 off, 1 DLAA, 2 quality, 3 balanced, 4 performance +var gsl_opts: bytes = null +var gsl_opt_mode: int = -1 +var gsl_opt_w: int = 0 +var gsl_opt_h: int = 0 +var gsl_rw: int = 0 # the render size DLSS asked for +var gsl_rh: int = 0 +var gsl_set_mode: int = -1 +var gsl_set_w: int = 0 +var gsl_set_h: int = 0 +var gsl_mv: Target = null +var gsl_out: Target = null +var gsl_consts: bytes = null +var gsl_tags: bytes = null +var gsl_res: bytes = null +var gsl_inputs: bytes = null +var gsl_prev_vp: words = null +var gsl_reset: bool = true +var gsl_jitter_x: int = 0 # float bits, NDC offsets the projection carries this frame +var gsl_jitter_y: int = 0 +var gsl_jpx: int = 0 # float bits, the same in pixels +var gsl_jpy: int = 0 + +# R3D_DLSS=0..4 overrides the setting, for a headless take +function r3d_dlss(mode: int) -> void { + var m = mode + if Os.has_env("R3D_DLSS") { m = Text.to_int(Os.env("R3D_DLSS")) } + if m != gsl_dlss_mode { gsl_reset = true } + gsl_dlss_mode = m +} +function gsl_dlss_live() -> bool { return gsl_on and gsl_dlss_ok and gsl_dlss_mode > 0 and gpu_kind == GPU_VK and gsl_token != null } +function r3d_dlss_live() -> bool { return gsl_on and gsl_dlss_ok and gsl_dlss_mode > 0 and gpu_kind == GPU_VK } + +# sl::DLSSMode from the setting +function gsl_sl_mode(m: int) -> int { + if m == 1 { return 6 } # eDLAA + if m == 2 { return 3 } # eMaxQuality + if m == 3 { return 2 } # eBalanced + if m == 4 { return 1 } # eMaxPerformance + return 0 +} + +function gsl_fill_options(w: int, h: int) -> void { + if gsl_opts == null { gsl_opts = gsl_struct(88) } + Vk.zero(gsl_opts, 88) + gsl_header(gsl_opts, 0x6ac8, 0x26e4, 0x4c61, 0x4101, 0xa92d, 0x638d, 0x4210, 0x57b8, 3) + Vk.put_i32(gsl_opts, 32, gsl_sl_mode(gsl_dlss_mode)) + Vk.put_i32(gsl_opts, 36, w) + Vk.put_i32(gsl_opts, 40, h) + Vk.put_i32(gsl_opts, 48, F_ONE) # preExposure + Vk.put_i32(gsl_opts, 52, F_ONE) # exposureScale + Vk.put_i32(gsl_opts, 56, 1) # colorBuffersHDR eTrue +} + +# the render size for the display's size and the mode, asked once per change +function gsl_optimal() -> void { + if gsl_opt_mode == gsl_dlss_mode and gsl_opt_w == gl_w and gsl_opt_h == gl_h { return } + gsl_opt_mode = gsl_dlss_mode; gsl_opt_w = gl_w; gsl_opt_h = gl_h + gsl_rw = gl_w; gsl_rh = gl_h + let f = gsl_fn(GSL_DLSS, "slDLSSGetOptimalSettings") + if f == null { return } + gsl_fill_options(gl_w, gl_h) + let os = gsl_struct(64) # sl::DLSSOptimalSettings + gsl_header(os, 0xef1d, 0x0957, 0xfd58, 0x4df7, 0xb504, 0x8b69, 0xd8aa, 0x6b76, 1) + if Vk.sl_call_pp(f, gsl_opts, os) == 0 { + let w = Vk.get_i32(os, 32) + let h = Vk.get_i32(os, 36) + if w > 0 and h > 0 { gsl_rw = w; gsl_rh = h } + } +} +function r3d_dlss_render_w() -> int { if not r3d_dlss_live() { return gl_w }; gsl_optimal(); return gsl_rw } +function r3d_dlss_render_h() -> int { if not r3d_dlss_live() { return gl_h }; gsl_optimal(); return gsl_rh } + +# a radical-inverse sample in [0, 1), float bits +function gsl_halton(i: int, b: int) -> int { + var f = F_ONE + var r = F_ZERO + var k = i + let fb = fi(b) + while k > 0 { + f = f_div(f, fb) + r = f_add(r, f_mul(f, fi(k % b))) + k = k / b + } + return r +} + +# cam_begin_frame: this frame's sub-pixel offset, before the camera builds its matrices +function gsl_jitter_frame() -> void { + gsl_jitter_x = F_ZERO; gsl_jitter_y = F_ZERO; gsl_jpx = F_ZERO; gsl_jpy = F_ZERO + if not r3d_dlss_live() or post_w <= 0 or post_h <= 0 { return } + # DLSS wants at least 8 x (display / render)^2 phases; 32 covers performance mode + let i = (gsl_frame_n % 32) + 1 + gsl_jpx = f_sub(gsl_halton(i, 2), F_HALF) + gsl_jpy = f_sub(gsl_halton(i, 3), F_HALF) + gsl_jitter_x = f_div(f_mul(F_TWO, gsl_jpx), fi(post_w)) + gsl_jitter_y = f_div(f_mul(F_TWO, gsl_jpy), fi(post_h)) +} + +# sl::Resource for one of the renderer's textures, in the layout every pass leaves them in +function gsl_resource(at: int, tex: int) -> void { + let p = Vk.at(gsl_res, at) + Vk.zero(p, 112) + gsl_header(p, 0x3a9d, 0x70cf, 0x2418, 0x4b72, 0x8391, 0x13f8, 0x721c, 0x7261, 1) + Vk.put_i64(p, 40, gvk_tex_image[tex]) + Vk.put_i64(p, 48, gvk_mem_handle(gvk_mem_id(gvk_tex_mem[tex]))) + Vk.put_i64(p, 56, gvk_tex_view[tex]) + Vk.put_i32(p, 64, GSL_LAYOUT_READ) + Vk.put_i32(p, 68, gvk_tex_dims_w[tex]) + Vk.put_i32(p, 72, gvk_tex_dims_h[tex]) + Vk.put_i32(p, 76, gvk_tex_vkfmt[tex]) + Vk.put_i32(p, 80, gvk_tex_levels[tex]) + Vk.put_i32(p, 84, gvk_tex_layers[tex]) + Vk.put_i32(p, 100, gvk_tex_usage(tex)) +} +# the usage gvk_tex_storage gave the image +function gvk_tex_usage(tex: int) -> int { + let fmt = gvk_tex_vkfmt[tex] + var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT + if fmt == VK_FORMAT_D32_SFLOAT { return usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } + usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT + if gvk_tex_samples[tex] <= 1 and (fmt == VK_FORMAT_R16G16B16A16_SFLOAT or fmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT } + return usage +} + +# sl::ResourceTag i, pointing at resource i +function gsl_tag(i: int, buffer: int, w: int, h: int) -> void { + let p = Vk.at(gsl_tags, i * 64) + Vk.zero(p, 64) + gsl_header(p, 0x4c6a, 0x5aad, 0xb445, 0x496c, 0x87ff, 0x1af3, 0x845b, 0xe653, 1) + Vk.put_ptr(p, 32, Vk.at(gsl_res, i * 112)) + Vk.put_i32(p, 40, buffer) + Vk.put_i32(p, 44, 2) # eValidUntilEvaluate + Vk.put_i32(p, 56, w) + Vk.put_i32(p, 60, h) +} + +# a column-major matrix into a row-major sl::float4x4 +function gsl_put_m4(p: pointer, at: int, m: words) -> void { + for r in 0 .. 4 { for c in 0 .. 4 { Vk.put_i32(p, at + (r * 4 + c) * 4, m[c * 4 + r]) } } +} +function gsl_put_v3(p: pointer, at: int, v: words) -> void { + Vk.put_i32(p, at, v[0]); Vk.put_i32(p, at + 4, v[1]); Vk.put_i32(p, at + 8, v[2]) +} + +# The renderer's clip space is OpenGL's; what Vulkan stores is depth remapped to [0, 1] and +# row 0 at NDC y = -1. Streamline reads images with row 0 at the top, so the matrices it is +# given carry both: y flipped, z' = (z + w) / 2. +function gsl_clip_fix(m: words) -> void { + m4_identity(m) + m[5] = f_neg(F_ONE) + m[10] = F_HALF + m[14] = F_HALF +} + +function gsl_constants() -> void { + if gsl_consts == null { gsl_consts = gsl_struct(456) } + let k = gsl_consts + Vk.zero(k, 456) + gsl_header(k, 0xdcd3, 0x5ad7, 0x4e4a, 0x4bad, 0xa90c, 0xe0c4, 0x9eb2, 0x3afe, 2) + let fix = m4_new(); let proj = m4_new(); let v2c = m4_new(); let c2v = m4_new() + let cur = m4_new(); let prev = m4_new(); let inv_cur = m4_new(); let c2p = m4_new(); let p2c = m4_new() + gsl_clip_fix(fix) + m4_perspective(proj, cam_fov, cam_aspect, cam_near, cam_far) + m4_mul(v2c, fix, proj) + m4_inverse(c2v, v2c) + gsl_put_m4(k, 32, v2c) # cameraViewToClip (no jitter) + gsl_put_m4(k, 96, c2v) # clipToCameraView + let ident = m4_new() + m4_identity(ident) + gsl_put_m4(k, 160, ident) # clipToLensClip + if gsl_prev_vp == null { gsl_prev_vp = m4_new(); for i in 0 .. 16 { gsl_prev_vp[i] = cam_vp_clean[i] } } + m4_mul(cur, fix, cam_vp_clean) + m4_mul(prev, fix, gsl_prev_vp) + m4_inverse(inv_cur, cur) + m4_mul(c2p, prev, inv_cur) + m4_inverse(p2c, c2p) + gsl_put_m4(k, 224, c2p) # clipToPrevClip + gsl_put_m4(k, 288, p2c) # prevClipToClip + # the sample's offset from the pixel centre, in the flipped image + Vk.put_i32(k, 352, f_neg(gsl_jpx)) + Vk.put_i32(k, 356, gsl_jpy) + Vk.put_i32(k, 360, F_ONE) # mvecScale + Vk.put_i32(k, 364, F_ONE) + gsl_put_v3(k, 376, cam_pos) + let up = words(3) + v3_cross(up, cam_right, cam_fwd) + gsl_put_v3(k, 388, up) + gsl_put_v3(k, 400, cam_right) + gsl_put_v3(k, 412, cam_fwd) + Vk.put_i32(k, 424, cam_near) + Vk.put_i32(k, 428, cam_far) + Vk.put_i32(k, 432, cam_fov) + Vk.put_i32(k, 436, cam_aspect) + Vk.put_i32(k, 440, F_ZERO) # motionVectorsInvalidValue + # depthInverted, cameraMotionIncluded, motionVectors3D false; reset on a cut; not orthographic, + # not dilated, not jittered + if gsl_reset { Vk.put_i32(k, 444, 256 * 256 * 256) } + Vk.put_i32(k, 452, fi(40)) # minRelativeLinearDepthObjectSeparation + free(fix); free(proj); free(v2c); free(c2v); free(cur); free(prev); free(inv_cur); free(c2p); free(p2c); free(ident); free(up) +} + +# make the DLSS targets at this frame's sizes +function gsl_targets() -> void { + if gsl_mv == null or gsl_mv.w != post_w or gsl_mv.h != post_h { + if gsl_mv != null { target_free(gsl_mv) } + gsl_mv = target_new(post_w, post_h, GL_RG16F, GL_RG, GL_HALF_FLOAT, false, GL_NEAREST) + gsl_reset = true + } + if gsl_out == null or gsl_out.w != gl_w or gsl_out.h != gl_h { + if gsl_out != null { target_free(gsl_out) } + gsl_out = target_new(gl_w, gl_h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR) + gsl_reset = true + } + # the LDR image the tonemap writes follows the upscaled size + if post_ldr.w != gl_w or post_ldr.h != gl_h { + target_free(post_ldr) + post_ldr = target_new(gl_w, gl_h, GL_RGBA8, GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR) + } + if gsl_res == null { gsl_res = gsl_struct(4 * 112); gsl_tags = gsl_struct(4 * 64); gsl_inputs = bytes(8) } +} + +# Upscale post_hdr into gsl_out; the colour the rest of post reads (post_hdr's own if it failed). +function gsl_dlss_eval() -> int { + gsl_targets() + # no motion of its own: the camera's comes from depth + target_bind(gsl_mv) + gpu_clear_color(0.0, 0.0, 0.0, 0.0) + gpu_clear(GL_COLOR_BUFFER_BIT) + gvk_pass_end() + let cb = gvk_frame_cb() + if gsl_set_mode != gsl_dlss_mode or gsl_set_w != gl_w or gsl_set_h != gl_h { + let f = gsl_fn(GSL_DLSS, "slDLSSSetOptions") + gsl_fill_options(gl_w, gl_h) + if f != null and Vk.sl_call_pp(f, gsl_vp, gsl_opts) == 0 { gsl_set_mode = gsl_dlss_mode; gsl_set_w = gl_w; gsl_set_h = gl_h } + } + gsl_constants() + Vk.sl_set_constants(gsl_consts, gsl_token, gsl_vp) + gsl_resource(0, post_hdr.depth); gsl_tag(0, 0, post_w, post_h) # kBufferTypeDepth + gsl_resource(112, gsl_mv.color); gsl_tag(1, 1, post_w, post_h) # kBufferTypeMotionVectors + gsl_resource(224, post_hdr.color); gsl_tag(2, 3, post_w, post_h) # kBufferTypeScalingInputColor + gsl_resource(336, gsl_out.color); gsl_tag(3, 4, gl_w, gl_h) # kBufferTypeScalingOutputColor + Vk.sl_set_tag_for_frame(gsl_token, gsl_vp, gsl_tags, 4, cb) + Vk.put_ptr(gsl_inputs, 0, gsl_vp) + let r = Vk.sl_evaluate_feature(GSL_DLSS, gsl_token, gsl_inputs, 1, cb) + for i in 0 .. 16 { gsl_prev_vp[i] = cam_vp_clean[i] } + gsl_reset = false + # Streamline records its own pipeline and descriptors into the command buffer; nothing needs + # forgetting, because every gvk_draw binds its pipeline, view and set afresh + if r != 0 { + if not gsl_said { print(`r3d: streamline: DLSS evaluate failed ({r}); drawing without it`); gsl_said = true } + post_color_w = post_w; post_color_h = post_h + return post_hdr.color + } + post_color_w = gl_w; post_color_h = gl_h + return gsl_out.color +} diff --git a/tools/ludic-cli/bundle.ludic b/tools/ludic-cli/bundle.ludic index 4746d4db..92fcd4ed 100644 --- a/tools/ludic-cli/bundle.ludic +++ b/tools/ludic-cli/bundle.ludic @@ -310,6 +310,21 @@ function cmd_bundle_windows() -> int { packed = true } + # 2b. `app native ""`: native libraries loaded at run time (NVIDIA Streamline's DLLs and + # their licence texts) go beside the executable, where LoadLibrary finds them. Never packed: + # a DLL has to be a real file. + let native = manifest_app(m, "native") + if native != "" { + if not Fs.exists(native) { + err(`ludic bundle: app native "{native}" does not exist\n`) + return 1 + } + if not shq(`cp -R "{native}"/. "{root}/"`) { + err("ludic bundle: could not place the native libraries\n") + return 1 + } + } + # 3. what to mount and where the game writes var idx = "" if packed { idx = idx + "pack game.lpak" + nl() }