ludic/packages/ludic.render3d/streamline.ludic
Orkuncakilkaya b9f9cb8495 render3d: r3d_fog_wall(dist) - fog that closes to the sky's horizon colour from 0.3 dist to dist, and nothing drawn past dist + 8 m
Every lit program blends toward the horizon colour (the prefiltered sky with the sun's inscatter,
so it carries the day's grade); the sky's horizon band is pulled to the same colour, wider the
closer the wall. The far plane, scatter and stream cull reach and the shadow cascades are clamped
to the wall; the reflection pass shares the shaders and the cull. 0 is off: every branch is gated
on u_fog_wall > 0 and r3d_reach hands back the far it was given. R3D_FOG_WALL=<m> for a shot.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-29 15:07:40 +03:00

466 lines
25 KiB
Text

# ============================================================================
# streamline.ludic — NVIDIA Streamline (SDK 2.14.1) on the Vulkan renderer: DLSS
# super resolution, Reflex and its latency markers.
#
# On Windows the loader opens sl.interposer.dll instead of vulkan-1.dll when it is
# beside the executable (runtime/native/vk_win.ll). slInit has to run before the
# instance is made - Streamline's own vkCreateInstance / vkCreateDevice add what its
# plugins need - so gvk_init calls gsl_boot() before Vk.open() and gsl_init() straight
# after. Nothing here runs on macOS, on OpenGL, or where the DLLs are missing: every
# entry point checks gsl_on and the renderer draws exactly as it did.
#
# Every sl structure starts with next, a GUID and a size_t version (32 bytes); the
# offsets below are the SDK headers' x64 layout. GUIDs are written as 16-bit halves so
# no literal needs more than 31 bits.
# ============================================================================
const GSL_DLSS: int = 0
const GSL_REFLEX: int = 3
const GSL_PCL: int = 4
const GSL_DLSS_G: int = 1000
const GSL_DLSS_RR: int = 1001
# sl::PCLMarker
const GSL_SIM_START: int = 0
const GSL_SIM_END: int = 1
const GSL_SUBMIT_START: int = 2
const GSL_SUBMIT_END: int = 3
const GSL_PRESENT_START: int = 4
const GSL_PRESENT_END: int = 5
const GSL_LAYOUT_READ: int = 5 # VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL
# ---- structure headers ------------------------------------------------------------------
function gsl_le16(v: int) -> int { return ((v >> 8) & 255) | ((v & 255) << 8) }
function gsl_header(p: pointer, a: int, b: int, c: int, d: int, e: int, f: int, g: int, h: int, version: int) -> void {
Vk.put_i32(p, 8, (a << 16) | b)
Vk.put_i32(p, 12, c | (d << 16))
Vk.put_i32(p, 16, gsl_le16(e) | (gsl_le16(f) << 16))
Vk.put_i32(p, 20, gsl_le16(g) | (gsl_le16(h) << 16))
let v: long = version
Vk.put_i64(p, 24, v)
}
@alloc_ok("DLSS set-up: its structs and entry points, made once")
function gsl_struct(size: int) -> bytes {
let p = bytes(size)
Vk.zero(p, size)
return p
}
# a feature's own function (slDLSSSetOptions, slReflexSleep, ...), or null
@alloc_ok("DLSS set-up: its structs and entry points, made once")
function gsl_fn(feature: int, name: string) -> pointer {
let out = bytes(8)
let zero: long = 0
Vk.put_i64(out, 0, zero)
if Vk.sl_get_feature_function(feature, name, out) != 0 { return null }
return Vk.get_ptr(out, 0)
}
# ---- start and stop ---------------------------------------------------------------------
# Before Vk.open(): ask the loader for the interposer. R3D_NO_STREAMLINE=1 keeps plain Vulkan.
function gsl_boot(render3d_st: mut Render3dState) -> void {
if Os.platform() != "windows" { return }
if r3d_env_has(render3d_st, "R3D_NO_STREAMLINE") { return }
Vk.sl_prefer(1)
}
# After Vk.open(), before vkCreateInstance.
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
function gsl_init(render3d_st: mut Render3dState) -> void {
# the evaluate call's structs, made once with DLSS rather than by the first frame it drew
if render3d_st.gsl_res == null { render3d_st.gsl_res = gsl_struct(4 * 112); render3d_st.gsl_tags = gsl_struct(4 * 64); render3d_st.gsl_inputs = bytes(8) }
if render3d_st.gsl_on or Vk.sl_active() == 0 { return }
render3d_st.gsl_feats = gsl_struct(20)
Vk.put_i32(render3d_st.gsl_feats, 0, GSL_DLSS); Vk.put_i32(render3d_st.gsl_feats, 4, GSL_REFLEX); Vk.put_i32(render3d_st.gsl_feats, 8, GSL_PCL)
Vk.put_i32(render3d_st.gsl_feats, 12, GSL_DLSS_RR); Vk.put_i32(render3d_st.gsl_feats, 16, GSL_DLSS_G)
# sl::Preferences
let pref = gsl_struct(144)
gsl_header(pref, 0x1ca1, 0x0965, 0xbf8e, 0x432b, 0x8da1, 0x6716, 0xd879, 0xfb14, 1)
# logLevel eOff; R3D_SL_LOG=<directory> writes Streamline's own verbose log (sl.log) there
var level = 0
if r3d_env_has(render3d_st, "R3D_SL_LOG") {
level = 2
let dir: pointer = r3d_env(render3d_st, "R3D_SL_LOG")
let n = Text.length(r3d_env(render3d_st, "R3D_SL_LOG"))
render3d_st.gsl_logdir = gsl_struct(n * 2 + 8) # pathToLogsAndData is a wide string
for i in 0 .. n { Vk.put_i32(render3d_st.gsl_logdir, i * 2, dir[i] & 255) }
Vk.put_ptr(pref, 56, render3d_st.gsl_logdir)
}
Vk.put_i32(pref, 36, level)
# eDisableCLStateTracking | eAllowOTA | eLoadDownloadedPlugins | eUseFrameBasedResourceTagging
let flags: long = 1 | 8 | 64 | 128
Vk.put_i64(pref, 88, flags)
Vk.put_ptr(pref, 96, render3d_st.gsl_feats)
Vk.put_i32(pref, 104, 5)
Vk.put_i32(pref, 112, 0) # engine eCustom
Vk.put_ptr(pref, 120, "ludic render3d")
Vk.put_ptr(pref, 128, "a0f57b54-1daf-4934-90ae-c4035c19df04")
Vk.put_i32(pref, 136, 2) # renderAPI eVulkan
let r = Vk.sl_init(pref, gsl_sdk_version())
if r != 0 { print(`r3d: streamline: slInit failed ({r}); DLSS and Reflex are off`); return }
render3d_st.gsl_on = true
render3d_st.gsl_vp = gsl_struct(40)
gsl_header(render3d_st.gsl_vp, 0x171b, 0x6435, 0x9b3c, 0x4fc8, 0x9994, 0xfbe5, 0x2569, 0xaaa4, 1)
render3d_st.gsl_tok_buf = bytes(8)
render3d_st.gsl_idx_buf = bytes(4)
}
# kSDKVersion, built from longs: (2 << 48) on ints is computed in 32 bits and arrives as garbage,
# which slInit answers with eErrorInvalidParameter before it has even opened a log
function gsl_sdk_version() -> long {
let major: long = 2
let minor: long = 14
let patch: long = 1
let magic: long = 0xfedc
return (major << 48) | (minor << 32) | (patch << 16) | magic
}
# After vkCreateDevice: what this adapter supports.
function gsl_probe_device(render3d_st: mut Render3dState, pd: pointer) -> void {
if not render3d_st.gsl_on { return }
let ai = gsl_struct(56) # sl::AdapterInfo
gsl_header(ai, 0x0677, 0x315f, 0xa746, 0x4492, 0x9f42, 0xcb61, 0x42c9, 0xc3d4, 1)
Vk.put_ptr(ai, 48, pd)
render3d_st.gsl_dlss_ok = Vk.sl_is_feature_supported(GSL_DLSS, ai) == 0
render3d_st.gsl_rr_ok = Vk.sl_is_feature_supported(GSL_DLSS_RR, ai) == 0
render3d_st.gsl_fg_ok = Vk.sl_is_feature_supported(GSL_DLSS_G, ai) == 0
render3d_st.gsl_reflex_ok = Vk.sl_is_feature_supported(GSL_REFLEX, ai) == 0
render3d_st.gsl_pcl_ok = Vk.sl_is_feature_supported(GSL_PCL, ai) == 0
print(`r3d: streamline: DLSS {render3d_st.gsl_dlss_ok}, ray reconstruction {render3d_st.gsl_rr_ok}, frame generation {render3d_st.gsl_fg_ok}, Reflex {render3d_st.gsl_reflex_ok}`)
}
# Before the device goes.
function gsl_shutdown(render3d_st: mut Render3dState) -> void {
if not render3d_st.gsl_on { return }
Vk.sl_shutdown()
render3d_st.gsl_on = false
}
# ---- frames, Reflex and the latency markers ---------------------------------------------
# A frame's token is taken straight after the previous present, where Reflex sleeps: the
# wait lands before the game reads input and simulates, which is the latency it removes.
function r3d_reflex(render3d_st: mut Render3dState, mode: int) -> void {
render3d_st.gsl_reflex_mode = mode
if r3d_env_has(render3d_st, "R3D_REFLEX") { render3d_st.gsl_reflex_mode = Text.to_int(r3d_env(render3d_st, "R3D_REFLEX")) }
}
function r3d_reflex_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_reflex_ok and render3d_st.gsl_reflex_mode > 0 }
function gsl_marker(render3d_st: mut Render3dState, m: int) -> void {
if not render3d_st.gsl_pcl_ok or render3d_st.gsl_token == null { return }
if render3d_st.gsl_f_marker == null { render3d_st.gsl_f_marker = gsl_fn(GSL_PCL, "slPCLSetMarker") }
if render3d_st.gsl_f_marker != null { Vk.sl_call_ip(render3d_st.gsl_f_marker, m, render3d_st.gsl_token) }
}
function gsl_reflex_apply(render3d_st: mut Render3dState) -> void {
if not render3d_st.gsl_reflex_ok or render3d_st.gsl_reflex_applied == render3d_st.gsl_reflex_mode { return }
let f = gsl_fn(GSL_REFLEX, "slReflexSetOptions")
if f == null { return }
let o = gsl_struct(48) # sl::ReflexOptions
gsl_header(o, 0xf03a, 0xf81a, 0x6d0b, 0x4902, 0xa651, 0xc496, 0x5e21, 0x5434, 1)
Vk.put_i32(o, 32, render3d_st.gsl_reflex_mode)
if render3d_st.gsl_pcl_ok { Vk.put_i32(o, 40, 1) } # useMarkersToOptimize
if Vk.sl_call_p(f, o) == 0 { render3d_st.gsl_reflex_applied = render3d_st.gsl_reflex_mode }
}
function gsl_new_frame(render3d_st: mut Render3dState) -> void {
Vk.put_i32(render3d_st.gsl_idx_buf, 0, render3d_st.gsl_frame_n)
render3d_st.gsl_frame_n += 1
render3d_st.gsl_token = null
render3d_st.gsl_fresh = true
if Vk.sl_get_new_frame_token(render3d_st.gsl_tok_buf, render3d_st.gsl_idx_buf) == 0 { render3d_st.gsl_token = Vk.get_ptr(render3d_st.gsl_tok_buf, 0) }
if render3d_st.gsl_token == null { return }
gsl_reflex_apply(render3d_st)
if r3d_reflex_live(render3d_st) {
if render3d_st.gsl_f_sleep == null { render3d_st.gsl_f_sleep = gsl_fn(GSL_REFLEX, "slReflexSleep") }
if render3d_st.gsl_f_sleep != null { Vk.sl_call_p(render3d_st.gsl_f_sleep, render3d_st.gsl_token) }
}
gsl_marker(render3d_st, GSL_SIM_START)
}
# r3d_frame's first line: the game has simulated, the renderer starts recording
function gsl_frame_start(render3d_st: mut Render3dState) -> void {
if not render3d_st.gsl_on or render3d_st.gpu_kind != GPU_VK { return }
# a headless run never presents: each frame takes its own token here
if not render3d_st.gsl_fresh { gsl_new_frame(render3d_st) }
render3d_st.gsl_fresh = false
gsl_marker(render3d_st, GSL_SIM_END)
gsl_marker(render3d_st, GSL_SUBMIT_START)
}
function gsl_before_present(render3d_st: mut Render3dState) -> void {
if not render3d_st.gsl_on { return }
gsl_marker(render3d_st, GSL_SUBMIT_END)
gsl_marker(render3d_st, GSL_PRESENT_START)
}
function gsl_after_present(render3d_st: mut Render3dState) -> void {
if not render3d_st.gsl_on { return }
gsl_marker(render3d_st, GSL_PRESENT_END)
gsl_new_frame(render3d_st)
}
# ---- DLSS super resolution ----------------------------------------------------------------
# The lit HDR frame (after the water, before occlusion, bloom and the tonemap) is upscaled to
# the display's size; everything after it reads the result by UV, so only the LDR image and the
# sharpen pass change size. The projection is jittered on a Halton (2, 3) cycle while DLSS is on.
# There are no per-object motion vectors yet: a zero target is tagged and Streamline adds the
# camera's own motion from depth and clipToPrevClip.
# R3D_DLSS=0..4 overrides the setting, for a headless take
function r3d_dlss(render3d_st: mut Render3dState, mode: int) -> void {
var m = mode
if r3d_env_has(render3d_st, "R3D_DLSS") { m = Text.to_int(r3d_env(render3d_st, "R3D_DLSS")) }
if m != render3d_st.gsl_dlss_mode { render3d_st.gsl_reset = true }
render3d_st.gsl_dlss_mode = m
}
function gsl_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK and render3d_st.gsl_token != null }
function r3d_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK }
# sl::DLSSMode from the setting
function gsl_sl_mode(m: int) -> int {
if m == 1 { return 6 } # eDLAA
if m == 2 { return 3 } # eMaxQuality
if m == 3 { return 2 } # eBalanced
if m == 4 { return 1 } # eMaxPerformance
return 0
}
function gsl_fill_options(render3d_st: mut Render3dState, w: int, h: int) -> void {
if render3d_st.gsl_opts == null { render3d_st.gsl_opts = gsl_struct(88) }
Vk.zero(render3d_st.gsl_opts, 88)
gsl_header(render3d_st.gsl_opts, 0x6ac8, 0x26e4, 0x4c61, 0x4101, 0xa92d, 0x638d, 0x4210, 0x57b8, 3)
Vk.put_i32(render3d_st.gsl_opts, 32, gsl_sl_mode(render3d_st.gsl_dlss_mode))
Vk.put_i32(render3d_st.gsl_opts, 36, w)
Vk.put_i32(render3d_st.gsl_opts, 40, h)
Vk.put_i32(render3d_st.gsl_opts, 48, float_bits(1.0)) # preExposure
Vk.put_i32(render3d_st.gsl_opts, 52, float_bits(1.0)) # exposureScale
Vk.put_i32(render3d_st.gsl_opts, 56, 1) # colorBuffersHDR eTrue
# The model: preset K (the transformer NVIDIA calls its best image quality) in every mode. The
# defaults put Performance on preset M, which on an RTX 3070 Ti at 4K evaluated in 18 ms against
# K's 2.8 - slower than no DLSS at all (33 fps against 41; with K, 60). R3D_DLSS_PRESET=<n>
# (sl::DLSSPreset: 11 K, 12 L, 13 M) sets every mode's, to measure them against each other.
var preset = 11
if r3d_env_has(render3d_st, "R3D_DLSS_PRESET") { preset = Text.to_int(r3d_env(render3d_st, "R3D_DLSS_PRESET")) }
if preset > 0 { for f in 0 .. 6 { Vk.put_i32(render3d_st.gsl_opts, 60 + f * 4, preset) } } # dlaa, quality, balanced, performance, ultra performance, ultra quality
}
# the render size for the display's size and the mode, asked once per change
function gsl_optimal(render3d_st: mut Render3dState) -> void {
if render3d_st.gsl_opt_mode == render3d_st.gsl_dlss_mode and render3d_st.gsl_opt_w == gl_width() and render3d_st.gsl_opt_h == gl_height() { return }
render3d_st.gsl_opt_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_opt_w = gl_width(); render3d_st.gsl_opt_h = gl_height()
render3d_st.gsl_rw = gl_width(); render3d_st.gsl_rh = gl_height()
let f = gsl_fn(GSL_DLSS, "slDLSSGetOptimalSettings")
if f == null { return }
gsl_fill_options(render3d_st, gl_width(), gl_height())
let os = gsl_struct(64) # sl::DLSSOptimalSettings
gsl_header(os, 0xef1d, 0x0957, 0xfd58, 0x4df7, 0xb504, 0x8b69, 0xd8aa, 0x6b76, 1)
if Vk.sl_call_pp(f, render3d_st.gsl_opts, os) == 0 {
let w = Vk.get_i32(os, 32)
let h = Vk.get_i32(os, 36)
if w > 0 and h > 0 { render3d_st.gsl_rw = w; render3d_st.gsl_rh = h }
}
}
function r3d_dlss_render_w(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_width() }; gsl_optimal(render3d_st); return render3d_st.gsl_rw }
function r3d_dlss_render_h(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_height() }; gsl_optimal(render3d_st); return render3d_st.gsl_rh }
# a radical-inverse sample in [0, 1), float bits
function gsl_halton(i: int, b: int) -> float {
var f = 1.0
var r = 0.0
var k = i
let fb = float(b)
while k > 0 {
f = f / fb
r = r + f * float(k % b)
k = k / b
}
return r
}
# cam_begin_frame: this frame's sub-pixel offset, before the camera builds its matrices
function gsl_jitter_frame(render3d_st: mut Render3dState) -> void {
render3d_st.gsl_jitter_x = 0.0; render3d_st.gsl_jitter_y = 0.0; render3d_st.gsl_jpx = 0.0; render3d_st.gsl_jpy = 0.0
# a jittered frame nobody resolves shakes on screen however still the camera is: jitter only while
# this frame holds a DLSS token and the last evaluate worked. (Not gsl_fresh: gsl_frame_start has
# already taken the token and cleared it by the time the camera asks, which turned jitter off.)
if not r3d_dlss_live(render3d_st) or not render3d_st.gsl_eval_ok or render3d_st.gsl_token == null or render3d_st.post_w <= 0 or render3d_st.post_h <= 0 { return }
# DLSS wants at least 8 x (display / render)^2 phases; 32 covers performance mode
let i = (render3d_st.gsl_frame_n % 32) + 1
render3d_st.gsl_jpx = gsl_halton(i, 2) - 0.5
render3d_st.gsl_jpy = gsl_halton(i, 3) - 0.5
render3d_st.gsl_jitter_x = 2.0 * render3d_st.gsl_jpx / float(render3d_st.post_w)
render3d_st.gsl_jitter_y = 2.0 * render3d_st.gsl_jpy / float(render3d_st.post_h)
}
# sl::Resource for one of the renderer's textures, in the layout every pass leaves them in
function gsl_resource(render3d_st: Render3dState, at: int, tex: int) -> void {
let p = Vk.at(render3d_st.gsl_res, at)
Vk.zero(p, 112)
gsl_header(p, 0x3a9d, 0x70cf, 0x2418, 0x4b72, 0x8391, 0x13f8, 0x721c, 0x7261, 1)
Vk.put_i64(p, 40, render3d_st.gvk_tex_image[tex])
Vk.put_i64(p, 48, gvk_mem_handle(render3d_st, gvk_mem_id(render3d_st.gvk_tex_mem[tex])))
Vk.put_i64(p, 56, render3d_st.gvk_tex_view[tex])
Vk.put_i32(p, 64, GSL_LAYOUT_READ)
Vk.put_i32(p, 68, render3d_st.gvk_tex_dims_w[tex])
Vk.put_i32(p, 72, render3d_st.gvk_tex_dims_h[tex])
Vk.put_i32(p, 76, render3d_st.gvk_tex_vkfmt[tex])
Vk.put_i32(p, 80, render3d_st.gvk_tex_levels[tex])
Vk.put_i32(p, 84, render3d_st.gvk_tex_layers[tex])
Vk.put_i32(p, 100, gvk_tex_usage(render3d_st, tex))
}
# the usage gvk_tex_storage gave the image
function gvk_tex_usage(render3d_st: Render3dState, tex: int) -> int {
let fmt = render3d_st.gvk_tex_vkfmt[tex]
var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT
if fmt == VK_FORMAT_D32_SFLOAT { return usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT }
usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT
if render3d_st.gvk_tex_samples[tex] <= 1 and (fmt == VK_FORMAT_R16G16B16A16_SFLOAT or fmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }
return usage
}
# sl::ResourceTag i, pointing at resource i
function gsl_tag(render3d_st: Render3dState, i: int, buffer: int, w: int, h: int) -> void {
let p = Vk.at(render3d_st.gsl_tags, i * 64)
Vk.zero(p, 64)
gsl_header(p, 0x4c6a, 0x5aad, 0xb445, 0x496c, 0x87ff, 0x1af3, 0x845b, 0xe653, 1)
Vk.put_ptr(p, 32, Vk.at(render3d_st.gsl_res, i * 112))
Vk.put_i32(p, 40, buffer)
Vk.put_i32(p, 44, 2) # eValidUntilEvaluate
Vk.put_i32(p, 56, w)
Vk.put_i32(p, 60, h)
}
# a column-major matrix into a row-major sl::float4x4
function gsl_put_m4(p: pointer, at: int, m: floats) -> void {
for r in 0 .. 4 { for c in 0 .. 4 { Vk.put_i32(p, at + (r * 4 + c) * 4, float_bits(m[c * 4 + r])) } }
}
function gsl_put_v3(p: pointer, at: int, v: floats) -> void {
Vk.put_i32(p, at, float_bits(v[0])); Vk.put_i32(p, at + 4, float_bits(v[1])); Vk.put_i32(p, at + 8, float_bits(v[2]))
}
# The renderer's clip space is OpenGL's; what Vulkan stores is depth remapped to [0, 1] and
# row 0 at NDC y = -1. Streamline reads images with row 0 at the top, so the matrices it is
# given carry both: y flipped, z' = (z + w) / 2.
function gsl_clip_fix(m: floats) -> void {
m4_identity(m)
m[5] = -1.0
m[10] = 0.5
m[14] = 0.5
}
function gsl_constants(render3d_st: mut Render3dState) -> void {
if render3d_st.gsl_consts == null { render3d_st.gsl_consts = gsl_struct(456) }
let k = render3d_st.gsl_consts
Vk.zero(k, 456)
gsl_header(k, 0xdcd3, 0x5ad7, 0x4e4a, 0x4bad, 0xa90c, 0xe0c4, 0x9eb2, 0x3afe, 2)
let fix = m4_new(); let proj = m4_new(); let v2c = m4_new(); let c2v = m4_new()
let cur = m4_new(); let prev = m4_new(); let inv_cur = m4_new(); let c2p = m4_new(); let p2c = m4_new()
gsl_clip_fix(fix)
m4_perspective(proj, render3d_st.cam_fov, render3d_st.cam_aspect, render3d_st.cam_near, r3d_reach(render3d_st, render3d_st.cam_far))
m4_mul(v2c, fix, proj)
m4_inverse(c2v, v2c)
gsl_put_m4(k, 32, v2c) # cameraViewToClip (no jitter)
gsl_put_m4(k, 96, c2v) # clipToCameraView
let ident = m4_new()
m4_identity(ident)
gsl_put_m4(k, 160, ident) # clipToLensClip
if render3d_st.gsl_prev_vp == null { render3d_st.gsl_prev_vp = m4_new(); for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] } }
m4_mul(cur, fix, render3d_st.cam_vp_clean)
m4_mul(prev, fix, render3d_st.gsl_prev_vp)
m4_inverse(inv_cur, cur)
m4_mul(c2p, prev, inv_cur)
m4_inverse(p2c, c2p)
gsl_put_m4(k, 224, c2p) # clipToPrevClip
gsl_put_m4(k, 288, p2c) # prevClipToClip
# the sample's offset from the pixel centre, in the image Streamline sees: row 0 at the top, so the
# vertical offset flips with it. With jy unflipped DLSS resolved the ground into concentric
# rings; with jx flipped too, thin stems doubled sideways (PC shots, 2026-09-15).
var jx = -render3d_st.gsl_jpx
var jy = -render3d_st.gsl_jpy
# R3D_DLSS_JX / R3D_DLSS_JY = -1 flip a sign, to check the convention against the picture
if r3d_env_has(render3d_st, "R3D_DLSS_JX") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JX")) < 0 { jx = -jx }
if r3d_env_has(render3d_st, "R3D_DLSS_JY") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JY")) < 0 { jy = -jy }
Vk.put_i32(k, 352, float_bits(jx))
Vk.put_i32(k, 356, float_bits(jy))
Vk.put_i32(k, 360, float_bits(1.0)) # mvecScale
Vk.put_i32(k, 364, float_bits(1.0))
gsl_put_v3(k, 376, render3d_st.cam_pos)
let up = render3d_st.gsl_up # made with the state
v3_cross(up, render3d_st.cam_right, render3d_st.cam_fwd)
gsl_put_v3(k, 388, up)
gsl_put_v3(k, 400, render3d_st.cam_right)
gsl_put_v3(k, 412, render3d_st.cam_fwd)
Vk.put_i32(k, 424, float_bits(render3d_st.cam_near))
Vk.put_i32(k, 428, float_bits(r3d_reach(render3d_st, render3d_st.cam_far)))
Vk.put_i32(k, 432, float_bits(render3d_st.cam_fov))
Vk.put_i32(k, 436, float_bits(render3d_st.cam_aspect))
Vk.put_i32(k, 440, float_bits(0.0)) # motionVectorsInvalidValue
# depthInverted, cameraMotionIncluded, motionVectors3D false; reset on a cut; not orthographic,
# not dilated, not jittered
if render3d_st.gsl_reset { Vk.put_i32(k, 444, 256 * 256 * 256) }
Vk.put_i32(k, 452, float_bits(40.0)) # minRelativeLinearDepthObjectSeparation
free(fix); free(proj); free(v2c); free(c2v); free(cur); free(prev); free(inv_cur); free(c2p); free(p2c); free(ident)
}
# make the DLSS targets at this frame's sizes
function gsl_targets(render3d_st: mut Render3dState) -> void {
if render3d_st.gsl_mv == null or render3d_st.gsl_mv.w != render3d_st.post_w or render3d_st.gsl_mv.h != render3d_st.post_h {
if render3d_st.gsl_mv != null { target_free(render3d_st, render3d_st.gsl_mv) }
render3d_st.gsl_mv = target_new(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_RG16F, GL_RG, GL_HALF_FLOAT, false, GL_NEAREST)
render3d_st.gsl_reset = true
}
if render3d_st.gsl_out == null or render3d_st.gsl_out.w != gl_width() or render3d_st.gsl_out.h != gl_height() {
if render3d_st.gsl_out != null { target_free(render3d_st, render3d_st.gsl_out) }
render3d_st.gsl_out = target_new(render3d_st, gl_width(), gl_height(), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
render3d_st.gsl_reset = true
}
# the LDR image the tonemap writes follows the upscaled size
if render3d_st.post_ldr.w != gl_width() or render3d_st.post_ldr.h != gl_height() {
target_free(render3d_st, render3d_st.post_ldr)
render3d_st.post_ldr = target_new(render3d_st, gl_width(), gl_height(), post_ldr_fmt(render3d_st), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
render3d_st.post_ldr_hdr = gpu_hdr_active(render3d_st)
}
}
# Upscale post_hdr into gsl_out; the colour the rest of post reads (post_hdr's own if it failed).
function gsl_dlss_eval(render3d_st: mut Render3dState) -> int {
gsl_targets(render3d_st)
# no motion of its own: the camera's comes from depth
target_bind(render3d_st, render3d_st.gsl_mv)
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT)
gvk_pass_end(render3d_st)
let cb = gvk_frame_cb(render3d_st)
if render3d_st.gsl_set_mode != render3d_st.gsl_dlss_mode or render3d_st.gsl_set_w != gl_width() or render3d_st.gsl_set_h != gl_height() {
let f = gsl_fn(GSL_DLSS, "slDLSSSetOptions")
gsl_fill_options(render3d_st, gl_width(), gl_height())
if f != null and Vk.sl_call_pp(f, render3d_st.gsl_vp, render3d_st.gsl_opts) == 0 { render3d_st.gsl_set_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_set_w = gl_width(); render3d_st.gsl_set_h = gl_height() }
}
gsl_constants(render3d_st)
Vk.sl_set_constants(render3d_st.gsl_consts, render3d_st.gsl_token, render3d_st.gsl_vp)
gsl_resource(render3d_st, 0, render3d_st.post_hdr.depth); gsl_tag(render3d_st, 0, 0, render3d_st.post_w, render3d_st.post_h) # kBufferTypeDepth
gsl_resource(render3d_st, 112, render3d_st.gsl_mv.color); gsl_tag(render3d_st, 1, 1, render3d_st.post_w, render3d_st.post_h) # kBufferTypeMotionVectors
gsl_resource(render3d_st, 224, render3d_st.post_hdr.color); gsl_tag(render3d_st, 2, 3, render3d_st.post_w, render3d_st.post_h) # kBufferTypeScalingInputColor
gsl_resource(render3d_st, 336, render3d_st.gsl_out.color); gsl_tag(render3d_st, 3, 4, gl_width(), gl_height()) # kBufferTypeScalingOutputColor
Vk.sl_set_tag_for_frame(render3d_st.gsl_token, render3d_st.gsl_vp, render3d_st.gsl_tags, 4, cb)
Vk.put_ptr(render3d_st.gsl_inputs, 0, render3d_st.gsl_vp)
let r = Vk.sl_evaluate_feature(GSL_DLSS, render3d_st.gsl_token, render3d_st.gsl_inputs, 1, cb)
for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] }
render3d_st.gsl_reset = false
# Streamline records its own pipeline and descriptors into the command buffer; nothing needs
# forgetting, because every gvk_draw binds its pipeline, view and set afresh
render3d_st.gsl_eval_ok = r == 0
if r != 0 {
if not render3d_st.gsl_said { gsl_say_eval_failed(r); render3d_st.gsl_said = true }
render3d_st.post_color_w = render3d_st.post_w; render3d_st.post_color_h = render3d_st.post_h
return render3d_st.post_hdr.color
}
render3d_st.post_color_w = gl_width(); render3d_st.post_color_h = gl_height()
return render3d_st.gsl_out.color
}
# messages, each built in a function of its own so the path that says it holds no allocation
@alloc_ok("a message, built only when it is said: a failure, a warning or a debug switch")
function gsl_say_eval_failed(r: int) -> void { print(`r3d: streamline: DLSS evaluate failed ({r}); drawing without it`) }