# ============================================================================ # streamline.ludic — NVIDIA Streamline (SDK 2.14.1) on the Vulkan renderer: DLSS # super resolution, Reflex and its latency markers. # # On Windows the loader opens sl.interposer.dll instead of vulkan-1.dll when it is # beside the executable (runtime/native/vk_win.ll). slInit has to run before the # instance is made - Streamline's own vkCreateInstance / vkCreateDevice add what its # plugins need - so gvk_init calls gsl_boot() before Vk.open() and gsl_init() straight # after. Nothing here runs on macOS, on OpenGL, or where the DLLs are missing: every # entry point checks gsl_on and the renderer draws exactly as it did. # # Every sl structure starts with next, a GUID and a size_t version (32 bytes); the # offsets below are the SDK headers' x64 layout. GUIDs are written as 16-bit halves so # no literal needs more than 31 bits. # ============================================================================ const GSL_DLSS: int = 0 const GSL_REFLEX: int = 3 const GSL_PCL: int = 4 const GSL_DLSS_G: int = 1000 const GSL_DLSS_RR: int = 1001 # sl::PCLMarker const GSL_SIM_START: int = 0 const GSL_SIM_END: int = 1 const GSL_SUBMIT_START: int = 2 const GSL_SUBMIT_END: int = 3 const GSL_PRESENT_START: int = 4 const GSL_PRESENT_END: int = 5 const GSL_LAYOUT_READ: int = 5 # VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL # ---- structure headers ------------------------------------------------------------------ function gsl_le16(v: int) -> int { return ((v >> 8) & 255) | ((v & 255) << 8) } function gsl_header(p: pointer, a: int, b: int, c: int, d: int, e: int, f: int, g: int, h: int, version: int) -> void { Vk.put_i32(p, 8, (a << 16) | b) Vk.put_i32(p, 12, c | (d << 16)) Vk.put_i32(p, 16, gsl_le16(e) | (gsl_le16(f) << 16)) Vk.put_i32(p, 20, gsl_le16(g) | (gsl_le16(h) << 16)) let v: long = version Vk.put_i64(p, 24, v) } @alloc_ok("DLSS set-up: its structs and entry points, made once") function gsl_struct(size: int) -> bytes { let p = bytes(size) Vk.zero(p, size) return p } # a feature's own function (slDLSSSetOptions, slReflexSleep, ...), or null @alloc_ok("DLSS set-up: its structs and entry points, made once") function gsl_fn(feature: int, name: string) -> pointer { let out = bytes(8) let zero: long = 0 Vk.put_i64(out, 0, zero) if Vk.sl_get_feature_function(feature, name, out) != 0 { return null } return Vk.get_ptr(out, 0) } # ---- start and stop --------------------------------------------------------------------- # Before Vk.open(): ask the loader for the interposer. R3D_NO_STREAMLINE=1 keeps plain Vulkan. function gsl_boot(render3d_st: mut Render3dState) -> void { if Os.platform() != "windows" { return } if r3d_env_has(render3d_st, "R3D_NO_STREAMLINE") { return } Vk.sl_prefer(1) } # After Vk.open(), before vkCreateInstance. @alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play") function gsl_init(render3d_st: mut Render3dState) -> void { # the evaluate call's structs, made once with DLSS rather than by the first frame it drew if render3d_st.gsl_res == null { render3d_st.gsl_res = gsl_struct(4 * 112); render3d_st.gsl_tags = gsl_struct(4 * 64); render3d_st.gsl_inputs = bytes(8) } if render3d_st.gsl_on or Vk.sl_active() == 0 { return } render3d_st.gsl_feats = gsl_struct(20) Vk.put_i32(render3d_st.gsl_feats, 0, GSL_DLSS); Vk.put_i32(render3d_st.gsl_feats, 4, GSL_REFLEX); Vk.put_i32(render3d_st.gsl_feats, 8, GSL_PCL) Vk.put_i32(render3d_st.gsl_feats, 12, GSL_DLSS_RR); Vk.put_i32(render3d_st.gsl_feats, 16, GSL_DLSS_G) # sl::Preferences let pref = gsl_struct(144) gsl_header(pref, 0x1ca1, 0x0965, 0xbf8e, 0x432b, 0x8da1, 0x6716, 0xd879, 0xfb14, 1) # logLevel eOff; R3D_SL_LOG= writes Streamline's own verbose log (sl.log) there var level = 0 if r3d_env_has(render3d_st, "R3D_SL_LOG") { level = 2 let dir: pointer = r3d_env(render3d_st, "R3D_SL_LOG") let n = Text.length(r3d_env(render3d_st, "R3D_SL_LOG")) render3d_st.gsl_logdir = gsl_struct(n * 2 + 8) # pathToLogsAndData is a wide string for i in 0 .. n { Vk.put_i32(render3d_st.gsl_logdir, i * 2, dir[i] & 255) } Vk.put_ptr(pref, 56, render3d_st.gsl_logdir) } Vk.put_i32(pref, 36, level) # eDisableCLStateTracking | eAllowOTA | eLoadDownloadedPlugins | eUseFrameBasedResourceTagging let flags: long = 1 | 8 | 64 | 128 Vk.put_i64(pref, 88, flags) Vk.put_ptr(pref, 96, render3d_st.gsl_feats) Vk.put_i32(pref, 104, 5) Vk.put_i32(pref, 112, 0) # engine eCustom Vk.put_ptr(pref, 120, "ludic render3d") Vk.put_ptr(pref, 128, "a0f57b54-1daf-4934-90ae-c4035c19df04") Vk.put_i32(pref, 136, 2) # renderAPI eVulkan let r = Vk.sl_init(pref, gsl_sdk_version()) if r != 0 { print(`r3d: streamline: slInit failed ({r}); DLSS and Reflex are off`); return } render3d_st.gsl_on = true render3d_st.gsl_vp = gsl_struct(40) gsl_header(render3d_st.gsl_vp, 0x171b, 0x6435, 0x9b3c, 0x4fc8, 0x9994, 0xfbe5, 0x2569, 0xaaa4, 1) render3d_st.gsl_tok_buf = bytes(8) render3d_st.gsl_idx_buf = bytes(4) } # kSDKVersion, built from longs: (2 << 48) on ints is computed in 32 bits and arrives as garbage, # which slInit answers with eErrorInvalidParameter before it has even opened a log function gsl_sdk_version() -> long { let major: long = 2 let minor: long = 14 let patch: long = 1 let magic: long = 0xfedc return (major << 48) | (minor << 32) | (patch << 16) | magic } # After vkCreateDevice: what this adapter supports. function gsl_probe_device(render3d_st: mut Render3dState, pd: pointer) -> void { if not render3d_st.gsl_on { return } let ai = gsl_struct(56) # sl::AdapterInfo gsl_header(ai, 0x0677, 0x315f, 0xa746, 0x4492, 0x9f42, 0xcb61, 0x42c9, 0xc3d4, 1) Vk.put_ptr(ai, 48, pd) render3d_st.gsl_dlss_ok = Vk.sl_is_feature_supported(GSL_DLSS, ai) == 0 render3d_st.gsl_rr_ok = Vk.sl_is_feature_supported(GSL_DLSS_RR, ai) == 0 render3d_st.gsl_fg_ok = Vk.sl_is_feature_supported(GSL_DLSS_G, ai) == 0 render3d_st.gsl_reflex_ok = Vk.sl_is_feature_supported(GSL_REFLEX, ai) == 0 render3d_st.gsl_pcl_ok = Vk.sl_is_feature_supported(GSL_PCL, ai) == 0 print(`r3d: streamline: DLSS {render3d_st.gsl_dlss_ok}, ray reconstruction {render3d_st.gsl_rr_ok}, frame generation {render3d_st.gsl_fg_ok}, Reflex {render3d_st.gsl_reflex_ok}`) } # Before the device goes. function gsl_shutdown(render3d_st: mut Render3dState) -> void { if not render3d_st.gsl_on { return } Vk.sl_shutdown() render3d_st.gsl_on = false } # ---- frames, Reflex and the latency markers --------------------------------------------- # A frame's token is taken straight after the previous present, where Reflex sleeps: the # wait lands before the game reads input and simulates, which is the latency it removes. function r3d_reflex(render3d_st: mut Render3dState, mode: int) -> void { render3d_st.gsl_reflex_mode = mode if r3d_env_has(render3d_st, "R3D_REFLEX") { render3d_st.gsl_reflex_mode = Text.to_int(r3d_env(render3d_st, "R3D_REFLEX")) } } function r3d_reflex_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_reflex_ok and render3d_st.gsl_reflex_mode > 0 } function gsl_marker(render3d_st: mut Render3dState, m: int) -> void { if not render3d_st.gsl_pcl_ok or render3d_st.gsl_token == null { return } if render3d_st.gsl_f_marker == null { render3d_st.gsl_f_marker = gsl_fn(GSL_PCL, "slPCLSetMarker") } if render3d_st.gsl_f_marker != null { Vk.sl_call_ip(render3d_st.gsl_f_marker, m, render3d_st.gsl_token) } } function gsl_reflex_apply(render3d_st: mut Render3dState) -> void { if not render3d_st.gsl_reflex_ok or render3d_st.gsl_reflex_applied == render3d_st.gsl_reflex_mode { return } let f = gsl_fn(GSL_REFLEX, "slReflexSetOptions") if f == null { return } let o = gsl_struct(48) # sl::ReflexOptions gsl_header(o, 0xf03a, 0xf81a, 0x6d0b, 0x4902, 0xa651, 0xc496, 0x5e21, 0x5434, 1) Vk.put_i32(o, 32, render3d_st.gsl_reflex_mode) if render3d_st.gsl_pcl_ok { Vk.put_i32(o, 40, 1) } # useMarkersToOptimize if Vk.sl_call_p(f, o) == 0 { render3d_st.gsl_reflex_applied = render3d_st.gsl_reflex_mode } } function gsl_new_frame(render3d_st: mut Render3dState) -> void { Vk.put_i32(render3d_st.gsl_idx_buf, 0, render3d_st.gsl_frame_n) render3d_st.gsl_frame_n += 1 render3d_st.gsl_token = null render3d_st.gsl_fresh = true if Vk.sl_get_new_frame_token(render3d_st.gsl_tok_buf, render3d_st.gsl_idx_buf) == 0 { render3d_st.gsl_token = Vk.get_ptr(render3d_st.gsl_tok_buf, 0) } if render3d_st.gsl_token == null { return } gsl_reflex_apply(render3d_st) if r3d_reflex_live(render3d_st) { if render3d_st.gsl_f_sleep == null { render3d_st.gsl_f_sleep = gsl_fn(GSL_REFLEX, "slReflexSleep") } if render3d_st.gsl_f_sleep != null { Vk.sl_call_p(render3d_st.gsl_f_sleep, render3d_st.gsl_token) } } gsl_marker(render3d_st, GSL_SIM_START) } # r3d_frame's first line: the game has simulated, the renderer starts recording function gsl_frame_start(render3d_st: mut Render3dState) -> void { if not render3d_st.gsl_on or render3d_st.gpu_kind != GPU_VK { return } # a headless run never presents: each frame takes its own token here if not render3d_st.gsl_fresh { gsl_new_frame(render3d_st) } render3d_st.gsl_fresh = false gsl_marker(render3d_st, GSL_SIM_END) gsl_marker(render3d_st, GSL_SUBMIT_START) } function gsl_before_present(render3d_st: mut Render3dState) -> void { if not render3d_st.gsl_on { return } gsl_marker(render3d_st, GSL_SUBMIT_END) gsl_marker(render3d_st, GSL_PRESENT_START) } function gsl_after_present(render3d_st: mut Render3dState) -> void { if not render3d_st.gsl_on { return } gsl_marker(render3d_st, GSL_PRESENT_END) gsl_new_frame(render3d_st) } # ---- DLSS super resolution ---------------------------------------------------------------- # The lit HDR frame (after the water, before occlusion, bloom and the tonemap) is upscaled to # the display's size; everything after it reads the result by UV, so only the LDR image and the # sharpen pass change size. The projection is jittered on a Halton (2, 3) cycle while DLSS is on. # There are no per-object motion vectors yet: a zero target is tagged and Streamline adds the # camera's own motion from depth and clipToPrevClip. # R3D_DLSS=0..4 overrides the setting, for a headless take function r3d_dlss(render3d_st: mut Render3dState, mode: int) -> void { var m = mode if r3d_env_has(render3d_st, "R3D_DLSS") { m = Text.to_int(r3d_env(render3d_st, "R3D_DLSS")) } if m != render3d_st.gsl_dlss_mode { render3d_st.gsl_reset = true } render3d_st.gsl_dlss_mode = m } function gsl_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK and render3d_st.gsl_token != null } function r3d_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK } # sl::DLSSMode from the setting function gsl_sl_mode(m: int) -> int { if m == 1 { return 6 } # eDLAA if m == 2 { return 3 } # eMaxQuality if m == 3 { return 2 } # eBalanced if m == 4 { return 1 } # eMaxPerformance return 0 } function gsl_fill_options(render3d_st: mut Render3dState, w: int, h: int) -> void { if render3d_st.gsl_opts == null { render3d_st.gsl_opts = gsl_struct(88) } Vk.zero(render3d_st.gsl_opts, 88) gsl_header(render3d_st.gsl_opts, 0x6ac8, 0x26e4, 0x4c61, 0x4101, 0xa92d, 0x638d, 0x4210, 0x57b8, 3) Vk.put_i32(render3d_st.gsl_opts, 32, gsl_sl_mode(render3d_st.gsl_dlss_mode)) Vk.put_i32(render3d_st.gsl_opts, 36, w) Vk.put_i32(render3d_st.gsl_opts, 40, h) Vk.put_i32(render3d_st.gsl_opts, 48, float_bits(1.0)) # preExposure Vk.put_i32(render3d_st.gsl_opts, 52, float_bits(1.0)) # exposureScale Vk.put_i32(render3d_st.gsl_opts, 56, 1) # colorBuffersHDR eTrue # The model: preset K (the transformer NVIDIA calls its best image quality) in every mode. The # defaults put Performance on preset M, which on an RTX 3070 Ti at 4K evaluated in 18 ms against # K's 2.8 - slower than no DLSS at all (33 fps against 41; with K, 60). R3D_DLSS_PRESET= # (sl::DLSSPreset: 11 K, 12 L, 13 M) sets every mode's, to measure them against each other. var preset = 11 if r3d_env_has(render3d_st, "R3D_DLSS_PRESET") { preset = Text.to_int(r3d_env(render3d_st, "R3D_DLSS_PRESET")) } if preset > 0 { for f in 0 .. 6 { Vk.put_i32(render3d_st.gsl_opts, 60 + f * 4, preset) } } # dlaa, quality, balanced, performance, ultra performance, ultra quality } # the render size for the display's size and the mode, asked once per change function gsl_optimal(render3d_st: mut Render3dState) -> void { if render3d_st.gsl_opt_mode == render3d_st.gsl_dlss_mode and render3d_st.gsl_opt_w == gl_width() and render3d_st.gsl_opt_h == gl_height() { return } render3d_st.gsl_opt_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_opt_w = gl_width(); render3d_st.gsl_opt_h = gl_height() render3d_st.gsl_rw = gl_width(); render3d_st.gsl_rh = gl_height() let f = gsl_fn(GSL_DLSS, "slDLSSGetOptimalSettings") if f == null { return } gsl_fill_options(render3d_st, gl_width(), gl_height()) let os = gsl_struct(64) # sl::DLSSOptimalSettings gsl_header(os, 0xef1d, 0x0957, 0xfd58, 0x4df7, 0xb504, 0x8b69, 0xd8aa, 0x6b76, 1) if Vk.sl_call_pp(f, render3d_st.gsl_opts, os) == 0 { let w = Vk.get_i32(os, 32) let h = Vk.get_i32(os, 36) if w > 0 and h > 0 { render3d_st.gsl_rw = w; render3d_st.gsl_rh = h } } } function r3d_dlss_render_w(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_width() }; gsl_optimal(render3d_st); return render3d_st.gsl_rw } function r3d_dlss_render_h(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_height() }; gsl_optimal(render3d_st); return render3d_st.gsl_rh } # a radical-inverse sample in [0, 1), float bits function gsl_halton(i: int, b: int) -> float { var f = 1.0 var r = 0.0 var k = i let fb = float(b) while k > 0 { f = f / fb r = r + f * float(k % b) k = k / b } return r } # cam_begin_frame: this frame's sub-pixel offset, before the camera builds its matrices function gsl_jitter_frame(render3d_st: mut Render3dState) -> void { render3d_st.gsl_jitter_x = 0.0; render3d_st.gsl_jitter_y = 0.0; render3d_st.gsl_jpx = 0.0; render3d_st.gsl_jpy = 0.0 # a jittered frame nobody resolves shakes on screen however still the camera is: jitter only while # this frame holds a DLSS token and the last evaluate worked. (Not gsl_fresh: gsl_frame_start has # already taken the token and cleared it by the time the camera asks, which turned jitter off.) if not r3d_dlss_live(render3d_st) or not render3d_st.gsl_eval_ok or render3d_st.gsl_token == null or render3d_st.post_w <= 0 or render3d_st.post_h <= 0 { return } # DLSS wants at least 8 x (display / render)^2 phases; 32 covers performance mode let i = (render3d_st.gsl_frame_n % 32) + 1 render3d_st.gsl_jpx = gsl_halton(i, 2) - 0.5 render3d_st.gsl_jpy = gsl_halton(i, 3) - 0.5 render3d_st.gsl_jitter_x = 2.0 * render3d_st.gsl_jpx / float(render3d_st.post_w) render3d_st.gsl_jitter_y = 2.0 * render3d_st.gsl_jpy / float(render3d_st.post_h) } # sl::Resource for one of the renderer's textures, in the layout every pass leaves them in function gsl_resource(render3d_st: Render3dState, at: int, tex: int) -> void { let p = Vk.at(render3d_st.gsl_res, at) Vk.zero(p, 112) gsl_header(p, 0x3a9d, 0x70cf, 0x2418, 0x4b72, 0x8391, 0x13f8, 0x721c, 0x7261, 1) Vk.put_i64(p, 40, render3d_st.gvk_tex_image[tex]) Vk.put_i64(p, 48, gvk_mem_handle(render3d_st, gvk_mem_id(render3d_st.gvk_tex_mem[tex]))) Vk.put_i64(p, 56, render3d_st.gvk_tex_view[tex]) Vk.put_i32(p, 64, GSL_LAYOUT_READ) Vk.put_i32(p, 68, render3d_st.gvk_tex_dims_w[tex]) Vk.put_i32(p, 72, render3d_st.gvk_tex_dims_h[tex]) Vk.put_i32(p, 76, render3d_st.gvk_tex_vkfmt[tex]) Vk.put_i32(p, 80, render3d_st.gvk_tex_levels[tex]) Vk.put_i32(p, 84, render3d_st.gvk_tex_layers[tex]) Vk.put_i32(p, 100, gvk_tex_usage(render3d_st, tex)) } # the usage gvk_tex_storage gave the image function gvk_tex_usage(render3d_st: Render3dState, tex: int) -> int { let fmt = render3d_st.gvk_tex_vkfmt[tex] var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT if fmt == VK_FORMAT_D32_SFLOAT { return usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT if render3d_st.gvk_tex_samples[tex] <= 1 and (fmt == VK_FORMAT_R16G16B16A16_SFLOAT or fmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT } return usage } # sl::ResourceTag i, pointing at resource i function gsl_tag(render3d_st: Render3dState, i: int, buffer: int, w: int, h: int) -> void { let p = Vk.at(render3d_st.gsl_tags, i * 64) Vk.zero(p, 64) gsl_header(p, 0x4c6a, 0x5aad, 0xb445, 0x496c, 0x87ff, 0x1af3, 0x845b, 0xe653, 1) Vk.put_ptr(p, 32, Vk.at(render3d_st.gsl_res, i * 112)) Vk.put_i32(p, 40, buffer) Vk.put_i32(p, 44, 2) # eValidUntilEvaluate Vk.put_i32(p, 56, w) Vk.put_i32(p, 60, h) } # a column-major matrix into a row-major sl::float4x4 function gsl_put_m4(p: pointer, at: int, m: floats) -> void { for r in 0 .. 4 { for c in 0 .. 4 { Vk.put_i32(p, at + (r * 4 + c) * 4, float_bits(m[c * 4 + r])) } } } function gsl_put_v3(p: pointer, at: int, v: floats) -> void { Vk.put_i32(p, at, float_bits(v[0])); Vk.put_i32(p, at + 4, float_bits(v[1])); Vk.put_i32(p, at + 8, float_bits(v[2])) } # The renderer's clip space is OpenGL's; what Vulkan stores is depth remapped to [0, 1] and # row 0 at NDC y = -1. Streamline reads images with row 0 at the top, so the matrices it is # given carry both: y flipped, z' = (z + w) / 2. function gsl_clip_fix(m: floats) -> void { m4_identity(m) m[5] = -1.0 m[10] = 0.5 m[14] = 0.5 } function gsl_constants(render3d_st: mut Render3dState) -> void { if render3d_st.gsl_consts == null { render3d_st.gsl_consts = gsl_struct(456) } let k = render3d_st.gsl_consts Vk.zero(k, 456) gsl_header(k, 0xdcd3, 0x5ad7, 0x4e4a, 0x4bad, 0xa90c, 0xe0c4, 0x9eb2, 0x3afe, 2) let fix = m4_new(); let proj = m4_new(); let v2c = m4_new(); let c2v = m4_new() let cur = m4_new(); let prev = m4_new(); let inv_cur = m4_new(); let c2p = m4_new(); let p2c = m4_new() gsl_clip_fix(fix) m4_perspective(proj, render3d_st.cam_fov, render3d_st.cam_aspect, render3d_st.cam_near, render3d_st.cam_far) m4_mul(v2c, fix, proj) m4_inverse(c2v, v2c) gsl_put_m4(k, 32, v2c) # cameraViewToClip (no jitter) gsl_put_m4(k, 96, c2v) # clipToCameraView let ident = m4_new() m4_identity(ident) gsl_put_m4(k, 160, ident) # clipToLensClip if render3d_st.gsl_prev_vp == null { render3d_st.gsl_prev_vp = m4_new(); for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] } } m4_mul(cur, fix, render3d_st.cam_vp_clean) m4_mul(prev, fix, render3d_st.gsl_prev_vp) m4_inverse(inv_cur, cur) m4_mul(c2p, prev, inv_cur) m4_inverse(p2c, c2p) gsl_put_m4(k, 224, c2p) # clipToPrevClip gsl_put_m4(k, 288, p2c) # prevClipToClip # the sample's offset from the pixel centre, in the image Streamline sees: row 0 at the top, so the # vertical offset flips with it. With jy unflipped DLSS resolved the ground into concentric # rings; with jx flipped too, thin stems doubled sideways (PC shots, 2026-09-15). var jx = -render3d_st.gsl_jpx var jy = -render3d_st.gsl_jpy # R3D_DLSS_JX / R3D_DLSS_JY = -1 flip a sign, to check the convention against the picture if r3d_env_has(render3d_st, "R3D_DLSS_JX") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JX")) < 0 { jx = -jx } if r3d_env_has(render3d_st, "R3D_DLSS_JY") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JY")) < 0 { jy = -jy } Vk.put_i32(k, 352, float_bits(jx)) Vk.put_i32(k, 356, float_bits(jy)) Vk.put_i32(k, 360, float_bits(1.0)) # mvecScale Vk.put_i32(k, 364, float_bits(1.0)) gsl_put_v3(k, 376, render3d_st.cam_pos) let up = render3d_st.gsl_up # made with the state v3_cross(up, render3d_st.cam_right, render3d_st.cam_fwd) gsl_put_v3(k, 388, up) gsl_put_v3(k, 400, render3d_st.cam_right) gsl_put_v3(k, 412, render3d_st.cam_fwd) Vk.put_i32(k, 424, float_bits(render3d_st.cam_near)) Vk.put_i32(k, 428, float_bits(render3d_st.cam_far)) Vk.put_i32(k, 432, float_bits(render3d_st.cam_fov)) Vk.put_i32(k, 436, float_bits(render3d_st.cam_aspect)) Vk.put_i32(k, 440, float_bits(0.0)) # motionVectorsInvalidValue # depthInverted, cameraMotionIncluded, motionVectors3D false; reset on a cut; not orthographic, # not dilated, not jittered if render3d_st.gsl_reset { Vk.put_i32(k, 444, 256 * 256 * 256) } Vk.put_i32(k, 452, float_bits(40.0)) # minRelativeLinearDepthObjectSeparation free(fix); free(proj); free(v2c); free(c2v); free(cur); free(prev); free(inv_cur); free(c2p); free(p2c); free(ident) } # make the DLSS targets at this frame's sizes function gsl_targets(render3d_st: mut Render3dState) -> void { if render3d_st.gsl_mv == null or render3d_st.gsl_mv.w != render3d_st.post_w or render3d_st.gsl_mv.h != render3d_st.post_h { if render3d_st.gsl_mv != null { target_free(render3d_st, render3d_st.gsl_mv) } render3d_st.gsl_mv = target_new(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_RG16F, GL_RG, GL_HALF_FLOAT, false, GL_NEAREST) render3d_st.gsl_reset = true } if render3d_st.gsl_out == null or render3d_st.gsl_out.w != gl_width() or render3d_st.gsl_out.h != gl_height() { if render3d_st.gsl_out != null { target_free(render3d_st, render3d_st.gsl_out) } render3d_st.gsl_out = target_new(render3d_st, gl_width(), gl_height(), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR) render3d_st.gsl_reset = true } # the LDR image the tonemap writes follows the upscaled size if render3d_st.post_ldr.w != gl_width() or render3d_st.post_ldr.h != gl_height() { target_free(render3d_st, render3d_st.post_ldr) render3d_st.post_ldr = target_new(render3d_st, gl_width(), gl_height(), post_ldr_fmt(render3d_st), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR) render3d_st.post_ldr_hdr = gpu_hdr_active(render3d_st) } } # Upscale post_hdr into gsl_out; the colour the rest of post reads (post_hdr's own if it failed). function gsl_dlss_eval(render3d_st: mut Render3dState) -> int { gsl_targets(render3d_st) # no motion of its own: the camera's comes from depth target_bind(render3d_st, render3d_st.gsl_mv) gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0) gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT) gvk_pass_end(render3d_st) let cb = gvk_frame_cb(render3d_st) if render3d_st.gsl_set_mode != render3d_st.gsl_dlss_mode or render3d_st.gsl_set_w != gl_width() or render3d_st.gsl_set_h != gl_height() { let f = gsl_fn(GSL_DLSS, "slDLSSSetOptions") gsl_fill_options(render3d_st, gl_width(), gl_height()) if f != null and Vk.sl_call_pp(f, render3d_st.gsl_vp, render3d_st.gsl_opts) == 0 { render3d_st.gsl_set_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_set_w = gl_width(); render3d_st.gsl_set_h = gl_height() } } gsl_constants(render3d_st) Vk.sl_set_constants(render3d_st.gsl_consts, render3d_st.gsl_token, render3d_st.gsl_vp) gsl_resource(render3d_st, 0, render3d_st.post_hdr.depth); gsl_tag(render3d_st, 0, 0, render3d_st.post_w, render3d_st.post_h) # kBufferTypeDepth gsl_resource(render3d_st, 112, render3d_st.gsl_mv.color); gsl_tag(render3d_st, 1, 1, render3d_st.post_w, render3d_st.post_h) # kBufferTypeMotionVectors gsl_resource(render3d_st, 224, render3d_st.post_hdr.color); gsl_tag(render3d_st, 2, 3, render3d_st.post_w, render3d_st.post_h) # kBufferTypeScalingInputColor gsl_resource(render3d_st, 336, render3d_st.gsl_out.color); gsl_tag(render3d_st, 3, 4, gl_width(), gl_height()) # kBufferTypeScalingOutputColor Vk.sl_set_tag_for_frame(render3d_st.gsl_token, render3d_st.gsl_vp, render3d_st.gsl_tags, 4, cb) Vk.put_ptr(render3d_st.gsl_inputs, 0, render3d_st.gsl_vp) let r = Vk.sl_evaluate_feature(GSL_DLSS, render3d_st.gsl_token, render3d_st.gsl_inputs, 1, cb) for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] } render3d_st.gsl_reset = false # Streamline records its own pipeline and descriptors into the command buffer; nothing needs # forgetting, because every gvk_draw binds its pipeline, view and set afresh render3d_st.gsl_eval_ok = r == 0 if r != 0 { if not render3d_st.gsl_said { gsl_say_eval_failed(r); render3d_st.gsl_said = true } render3d_st.post_color_w = render3d_st.post_w; render3d_st.post_color_h = render3d_st.post_h return render3d_st.post_hdr.color } render3d_st.post_color_w = gl_width(); render3d_st.post_color_h = gl_height() return render3d_st.gsl_out.color } # messages, each built in a function of its own so the path that says it holds no allocation @alloc_ok("a message, built only when it is said: a failure, a warning or a debug switch") function gsl_say_eval_failed(r: int) -> void { print(`r3d: streamline: DLSS evaluate failed ({r}); drawing without it`) }