Every lit program blends toward the horizon colour (the prefiltered sky with the sun's inscatter, so it carries the day's grade); the sky's horizon band is pulled to the same colour, wider the closer the wall. The far plane, scatter and stream cull reach and the shadow cascades are clamped to the wall; the reflection pass shares the shaders and the cull. 0 is off: every branch is gated on u_fog_wall > 0 and r3d_reach hands back the far it was given. R3D_FOG_WALL=<m> for a shot. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
466 lines
25 KiB
Text
466 lines
25 KiB
Text
# ============================================================================
|
|
# streamline.ludic — NVIDIA Streamline (SDK 2.14.1) on the Vulkan renderer: DLSS
|
|
# super resolution, Reflex and its latency markers.
|
|
#
|
|
# On Windows the loader opens sl.interposer.dll instead of vulkan-1.dll when it is
|
|
# beside the executable (runtime/native/vk_win.ll). slInit has to run before the
|
|
# instance is made - Streamline's own vkCreateInstance / vkCreateDevice add what its
|
|
# plugins need - so gvk_init calls gsl_boot() before Vk.open() and gsl_init() straight
|
|
# after. Nothing here runs on macOS, on OpenGL, or where the DLLs are missing: every
|
|
# entry point checks gsl_on and the renderer draws exactly as it did.
|
|
#
|
|
# Every sl structure starts with next, a GUID and a size_t version (32 bytes); the
|
|
# offsets below are the SDK headers' x64 layout. GUIDs are written as 16-bit halves so
|
|
# no literal needs more than 31 bits.
|
|
# ============================================================================
|
|
|
|
const GSL_DLSS: int = 0
|
|
const GSL_REFLEX: int = 3
|
|
const GSL_PCL: int = 4
|
|
const GSL_DLSS_G: int = 1000
|
|
const GSL_DLSS_RR: int = 1001
|
|
|
|
# sl::PCLMarker
|
|
const GSL_SIM_START: int = 0
|
|
const GSL_SIM_END: int = 1
|
|
const GSL_SUBMIT_START: int = 2
|
|
const GSL_SUBMIT_END: int = 3
|
|
const GSL_PRESENT_START: int = 4
|
|
const GSL_PRESENT_END: int = 5
|
|
|
|
const GSL_LAYOUT_READ: int = 5 # VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL
|
|
|
|
|
|
|
|
# ---- structure headers ------------------------------------------------------------------
|
|
function gsl_le16(v: int) -> int { return ((v >> 8) & 255) | ((v & 255) << 8) }
|
|
function gsl_header(p: pointer, a: int, b: int, c: int, d: int, e: int, f: int, g: int, h: int, version: int) -> void {
|
|
Vk.put_i32(p, 8, (a << 16) | b)
|
|
Vk.put_i32(p, 12, c | (d << 16))
|
|
Vk.put_i32(p, 16, gsl_le16(e) | (gsl_le16(f) << 16))
|
|
Vk.put_i32(p, 20, gsl_le16(g) | (gsl_le16(h) << 16))
|
|
let v: long = version
|
|
Vk.put_i64(p, 24, v)
|
|
}
|
|
@alloc_ok("DLSS set-up: its structs and entry points, made once")
|
|
function gsl_struct(size: int) -> bytes {
|
|
let p = bytes(size)
|
|
Vk.zero(p, size)
|
|
return p
|
|
}
|
|
|
|
# a feature's own function (slDLSSSetOptions, slReflexSleep, ...), or null
|
|
@alloc_ok("DLSS set-up: its structs and entry points, made once")
|
|
function gsl_fn(feature: int, name: string) -> pointer {
|
|
let out = bytes(8)
|
|
let zero: long = 0
|
|
Vk.put_i64(out, 0, zero)
|
|
if Vk.sl_get_feature_function(feature, name, out) != 0 { return null }
|
|
return Vk.get_ptr(out, 0)
|
|
}
|
|
|
|
# ---- start and stop ---------------------------------------------------------------------
|
|
# Before Vk.open(): ask the loader for the interposer. R3D_NO_STREAMLINE=1 keeps plain Vulkan.
|
|
function gsl_boot(render3d_st: mut Render3dState) -> void {
|
|
if Os.platform() != "windows" { return }
|
|
if r3d_env_has(render3d_st, "R3D_NO_STREAMLINE") { return }
|
|
Vk.sl_prefer(1)
|
|
}
|
|
|
|
# After Vk.open(), before vkCreateInstance.
|
|
@alloc_ok("start-up: the device, its tables, the programs, the passes and the world's first textures are made once, before play")
|
|
function gsl_init(render3d_st: mut Render3dState) -> void {
|
|
# the evaluate call's structs, made once with DLSS rather than by the first frame it drew
|
|
if render3d_st.gsl_res == null { render3d_st.gsl_res = gsl_struct(4 * 112); render3d_st.gsl_tags = gsl_struct(4 * 64); render3d_st.gsl_inputs = bytes(8) }
|
|
if render3d_st.gsl_on or Vk.sl_active() == 0 { return }
|
|
render3d_st.gsl_feats = gsl_struct(20)
|
|
Vk.put_i32(render3d_st.gsl_feats, 0, GSL_DLSS); Vk.put_i32(render3d_st.gsl_feats, 4, GSL_REFLEX); Vk.put_i32(render3d_st.gsl_feats, 8, GSL_PCL)
|
|
Vk.put_i32(render3d_st.gsl_feats, 12, GSL_DLSS_RR); Vk.put_i32(render3d_st.gsl_feats, 16, GSL_DLSS_G)
|
|
# sl::Preferences
|
|
let pref = gsl_struct(144)
|
|
gsl_header(pref, 0x1ca1, 0x0965, 0xbf8e, 0x432b, 0x8da1, 0x6716, 0xd879, 0xfb14, 1)
|
|
# logLevel eOff; R3D_SL_LOG=<directory> writes Streamline's own verbose log (sl.log) there
|
|
var level = 0
|
|
if r3d_env_has(render3d_st, "R3D_SL_LOG") {
|
|
level = 2
|
|
let dir: pointer = r3d_env(render3d_st, "R3D_SL_LOG")
|
|
let n = Text.length(r3d_env(render3d_st, "R3D_SL_LOG"))
|
|
render3d_st.gsl_logdir = gsl_struct(n * 2 + 8) # pathToLogsAndData is a wide string
|
|
for i in 0 .. n { Vk.put_i32(render3d_st.gsl_logdir, i * 2, dir[i] & 255) }
|
|
Vk.put_ptr(pref, 56, render3d_st.gsl_logdir)
|
|
}
|
|
Vk.put_i32(pref, 36, level)
|
|
# eDisableCLStateTracking | eAllowOTA | eLoadDownloadedPlugins | eUseFrameBasedResourceTagging
|
|
let flags: long = 1 | 8 | 64 | 128
|
|
Vk.put_i64(pref, 88, flags)
|
|
Vk.put_ptr(pref, 96, render3d_st.gsl_feats)
|
|
Vk.put_i32(pref, 104, 5)
|
|
Vk.put_i32(pref, 112, 0) # engine eCustom
|
|
Vk.put_ptr(pref, 120, "ludic render3d")
|
|
Vk.put_ptr(pref, 128, "a0f57b54-1daf-4934-90ae-c4035c19df04")
|
|
Vk.put_i32(pref, 136, 2) # renderAPI eVulkan
|
|
let r = Vk.sl_init(pref, gsl_sdk_version())
|
|
if r != 0 { print(`r3d: streamline: slInit failed ({r}); DLSS and Reflex are off`); return }
|
|
render3d_st.gsl_on = true
|
|
render3d_st.gsl_vp = gsl_struct(40)
|
|
gsl_header(render3d_st.gsl_vp, 0x171b, 0x6435, 0x9b3c, 0x4fc8, 0x9994, 0xfbe5, 0x2569, 0xaaa4, 1)
|
|
render3d_st.gsl_tok_buf = bytes(8)
|
|
render3d_st.gsl_idx_buf = bytes(4)
|
|
}
|
|
|
|
# kSDKVersion, built from longs: (2 << 48) on ints is computed in 32 bits and arrives as garbage,
|
|
# which slInit answers with eErrorInvalidParameter before it has even opened a log
|
|
function gsl_sdk_version() -> long {
|
|
let major: long = 2
|
|
let minor: long = 14
|
|
let patch: long = 1
|
|
let magic: long = 0xfedc
|
|
return (major << 48) | (minor << 32) | (patch << 16) | magic
|
|
}
|
|
|
|
# After vkCreateDevice: what this adapter supports.
|
|
function gsl_probe_device(render3d_st: mut Render3dState, pd: pointer) -> void {
|
|
if not render3d_st.gsl_on { return }
|
|
let ai = gsl_struct(56) # sl::AdapterInfo
|
|
gsl_header(ai, 0x0677, 0x315f, 0xa746, 0x4492, 0x9f42, 0xcb61, 0x42c9, 0xc3d4, 1)
|
|
Vk.put_ptr(ai, 48, pd)
|
|
render3d_st.gsl_dlss_ok = Vk.sl_is_feature_supported(GSL_DLSS, ai) == 0
|
|
render3d_st.gsl_rr_ok = Vk.sl_is_feature_supported(GSL_DLSS_RR, ai) == 0
|
|
render3d_st.gsl_fg_ok = Vk.sl_is_feature_supported(GSL_DLSS_G, ai) == 0
|
|
render3d_st.gsl_reflex_ok = Vk.sl_is_feature_supported(GSL_REFLEX, ai) == 0
|
|
render3d_st.gsl_pcl_ok = Vk.sl_is_feature_supported(GSL_PCL, ai) == 0
|
|
print(`r3d: streamline: DLSS {render3d_st.gsl_dlss_ok}, ray reconstruction {render3d_st.gsl_rr_ok}, frame generation {render3d_st.gsl_fg_ok}, Reflex {render3d_st.gsl_reflex_ok}`)
|
|
}
|
|
|
|
# Before the device goes.
|
|
function gsl_shutdown(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gsl_on { return }
|
|
Vk.sl_shutdown()
|
|
render3d_st.gsl_on = false
|
|
}
|
|
|
|
# ---- frames, Reflex and the latency markers ---------------------------------------------
|
|
# A frame's token is taken straight after the previous present, where Reflex sleeps: the
|
|
# wait lands before the game reads input and simulates, which is the latency it removes.
|
|
|
|
function r3d_reflex(render3d_st: mut Render3dState, mode: int) -> void {
|
|
render3d_st.gsl_reflex_mode = mode
|
|
if r3d_env_has(render3d_st, "R3D_REFLEX") { render3d_st.gsl_reflex_mode = Text.to_int(r3d_env(render3d_st, "R3D_REFLEX")) }
|
|
}
|
|
function r3d_reflex_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_reflex_ok and render3d_st.gsl_reflex_mode > 0 }
|
|
|
|
function gsl_marker(render3d_st: mut Render3dState, m: int) -> void {
|
|
if not render3d_st.gsl_pcl_ok or render3d_st.gsl_token == null { return }
|
|
if render3d_st.gsl_f_marker == null { render3d_st.gsl_f_marker = gsl_fn(GSL_PCL, "slPCLSetMarker") }
|
|
if render3d_st.gsl_f_marker != null { Vk.sl_call_ip(render3d_st.gsl_f_marker, m, render3d_st.gsl_token) }
|
|
}
|
|
|
|
function gsl_reflex_apply(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gsl_reflex_ok or render3d_st.gsl_reflex_applied == render3d_st.gsl_reflex_mode { return }
|
|
let f = gsl_fn(GSL_REFLEX, "slReflexSetOptions")
|
|
if f == null { return }
|
|
let o = gsl_struct(48) # sl::ReflexOptions
|
|
gsl_header(o, 0xf03a, 0xf81a, 0x6d0b, 0x4902, 0xa651, 0xc496, 0x5e21, 0x5434, 1)
|
|
Vk.put_i32(o, 32, render3d_st.gsl_reflex_mode)
|
|
if render3d_st.gsl_pcl_ok { Vk.put_i32(o, 40, 1) } # useMarkersToOptimize
|
|
if Vk.sl_call_p(f, o) == 0 { render3d_st.gsl_reflex_applied = render3d_st.gsl_reflex_mode }
|
|
}
|
|
|
|
function gsl_new_frame(render3d_st: mut Render3dState) -> void {
|
|
Vk.put_i32(render3d_st.gsl_idx_buf, 0, render3d_st.gsl_frame_n)
|
|
render3d_st.gsl_frame_n += 1
|
|
render3d_st.gsl_token = null
|
|
render3d_st.gsl_fresh = true
|
|
if Vk.sl_get_new_frame_token(render3d_st.gsl_tok_buf, render3d_st.gsl_idx_buf) == 0 { render3d_st.gsl_token = Vk.get_ptr(render3d_st.gsl_tok_buf, 0) }
|
|
if render3d_st.gsl_token == null { return }
|
|
gsl_reflex_apply(render3d_st)
|
|
if r3d_reflex_live(render3d_st) {
|
|
if render3d_st.gsl_f_sleep == null { render3d_st.gsl_f_sleep = gsl_fn(GSL_REFLEX, "slReflexSleep") }
|
|
if render3d_st.gsl_f_sleep != null { Vk.sl_call_p(render3d_st.gsl_f_sleep, render3d_st.gsl_token) }
|
|
}
|
|
gsl_marker(render3d_st, GSL_SIM_START)
|
|
}
|
|
|
|
# r3d_frame's first line: the game has simulated, the renderer starts recording
|
|
function gsl_frame_start(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gsl_on or render3d_st.gpu_kind != GPU_VK { return }
|
|
# a headless run never presents: each frame takes its own token here
|
|
if not render3d_st.gsl_fresh { gsl_new_frame(render3d_st) }
|
|
render3d_st.gsl_fresh = false
|
|
gsl_marker(render3d_st, GSL_SIM_END)
|
|
gsl_marker(render3d_st, GSL_SUBMIT_START)
|
|
}
|
|
function gsl_before_present(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gsl_on { return }
|
|
gsl_marker(render3d_st, GSL_SUBMIT_END)
|
|
gsl_marker(render3d_st, GSL_PRESENT_START)
|
|
}
|
|
function gsl_after_present(render3d_st: mut Render3dState) -> void {
|
|
if not render3d_st.gsl_on { return }
|
|
gsl_marker(render3d_st, GSL_PRESENT_END)
|
|
gsl_new_frame(render3d_st)
|
|
}
|
|
|
|
# ---- DLSS super resolution ----------------------------------------------------------------
|
|
# The lit HDR frame (after the water, before occlusion, bloom and the tonemap) is upscaled to
|
|
# the display's size; everything after it reads the result by UV, so only the LDR image and the
|
|
# sharpen pass change size. The projection is jittered on a Halton (2, 3) cycle while DLSS is on.
|
|
# There are no per-object motion vectors yet: a zero target is tagged and Streamline adds the
|
|
# camera's own motion from depth and clipToPrevClip.
|
|
|
|
# R3D_DLSS=0..4 overrides the setting, for a headless take
|
|
function r3d_dlss(render3d_st: mut Render3dState, mode: int) -> void {
|
|
var m = mode
|
|
if r3d_env_has(render3d_st, "R3D_DLSS") { m = Text.to_int(r3d_env(render3d_st, "R3D_DLSS")) }
|
|
if m != render3d_st.gsl_dlss_mode { render3d_st.gsl_reset = true }
|
|
render3d_st.gsl_dlss_mode = m
|
|
}
|
|
function gsl_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK and render3d_st.gsl_token != null }
|
|
function r3d_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK }
|
|
|
|
# sl::DLSSMode from the setting
|
|
function gsl_sl_mode(m: int) -> int {
|
|
if m == 1 { return 6 } # eDLAA
|
|
if m == 2 { return 3 } # eMaxQuality
|
|
if m == 3 { return 2 } # eBalanced
|
|
if m == 4 { return 1 } # eMaxPerformance
|
|
return 0
|
|
}
|
|
|
|
function gsl_fill_options(render3d_st: mut Render3dState, w: int, h: int) -> void {
|
|
if render3d_st.gsl_opts == null { render3d_st.gsl_opts = gsl_struct(88) }
|
|
Vk.zero(render3d_st.gsl_opts, 88)
|
|
gsl_header(render3d_st.gsl_opts, 0x6ac8, 0x26e4, 0x4c61, 0x4101, 0xa92d, 0x638d, 0x4210, 0x57b8, 3)
|
|
Vk.put_i32(render3d_st.gsl_opts, 32, gsl_sl_mode(render3d_st.gsl_dlss_mode))
|
|
Vk.put_i32(render3d_st.gsl_opts, 36, w)
|
|
Vk.put_i32(render3d_st.gsl_opts, 40, h)
|
|
Vk.put_i32(render3d_st.gsl_opts, 48, float_bits(1.0)) # preExposure
|
|
Vk.put_i32(render3d_st.gsl_opts, 52, float_bits(1.0)) # exposureScale
|
|
Vk.put_i32(render3d_st.gsl_opts, 56, 1) # colorBuffersHDR eTrue
|
|
# The model: preset K (the transformer NVIDIA calls its best image quality) in every mode. The
|
|
# defaults put Performance on preset M, which on an RTX 3070 Ti at 4K evaluated in 18 ms against
|
|
# K's 2.8 - slower than no DLSS at all (33 fps against 41; with K, 60). R3D_DLSS_PRESET=<n>
|
|
# (sl::DLSSPreset: 11 K, 12 L, 13 M) sets every mode's, to measure them against each other.
|
|
var preset = 11
|
|
if r3d_env_has(render3d_st, "R3D_DLSS_PRESET") { preset = Text.to_int(r3d_env(render3d_st, "R3D_DLSS_PRESET")) }
|
|
if preset > 0 { for f in 0 .. 6 { Vk.put_i32(render3d_st.gsl_opts, 60 + f * 4, preset) } } # dlaa, quality, balanced, performance, ultra performance, ultra quality
|
|
}
|
|
|
|
# the render size for the display's size and the mode, asked once per change
|
|
function gsl_optimal(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gsl_opt_mode == render3d_st.gsl_dlss_mode and render3d_st.gsl_opt_w == gl_width() and render3d_st.gsl_opt_h == gl_height() { return }
|
|
render3d_st.gsl_opt_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_opt_w = gl_width(); render3d_st.gsl_opt_h = gl_height()
|
|
render3d_st.gsl_rw = gl_width(); render3d_st.gsl_rh = gl_height()
|
|
let f = gsl_fn(GSL_DLSS, "slDLSSGetOptimalSettings")
|
|
if f == null { return }
|
|
gsl_fill_options(render3d_st, gl_width(), gl_height())
|
|
let os = gsl_struct(64) # sl::DLSSOptimalSettings
|
|
gsl_header(os, 0xef1d, 0x0957, 0xfd58, 0x4df7, 0xb504, 0x8b69, 0xd8aa, 0x6b76, 1)
|
|
if Vk.sl_call_pp(f, render3d_st.gsl_opts, os) == 0 {
|
|
let w = Vk.get_i32(os, 32)
|
|
let h = Vk.get_i32(os, 36)
|
|
if w > 0 and h > 0 { render3d_st.gsl_rw = w; render3d_st.gsl_rh = h }
|
|
}
|
|
}
|
|
function r3d_dlss_render_w(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_width() }; gsl_optimal(render3d_st); return render3d_st.gsl_rw }
|
|
function r3d_dlss_render_h(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_height() }; gsl_optimal(render3d_st); return render3d_st.gsl_rh }
|
|
|
|
# a radical-inverse sample in [0, 1), float bits
|
|
function gsl_halton(i: int, b: int) -> float {
|
|
var f = 1.0
|
|
var r = 0.0
|
|
var k = i
|
|
let fb = float(b)
|
|
while k > 0 {
|
|
f = f / fb
|
|
r = r + f * float(k % b)
|
|
k = k / b
|
|
}
|
|
return r
|
|
}
|
|
|
|
# cam_begin_frame: this frame's sub-pixel offset, before the camera builds its matrices
|
|
function gsl_jitter_frame(render3d_st: mut Render3dState) -> void {
|
|
render3d_st.gsl_jitter_x = 0.0; render3d_st.gsl_jitter_y = 0.0; render3d_st.gsl_jpx = 0.0; render3d_st.gsl_jpy = 0.0
|
|
# a jittered frame nobody resolves shakes on screen however still the camera is: jitter only while
|
|
# this frame holds a DLSS token and the last evaluate worked. (Not gsl_fresh: gsl_frame_start has
|
|
# already taken the token and cleared it by the time the camera asks, which turned jitter off.)
|
|
if not r3d_dlss_live(render3d_st) or not render3d_st.gsl_eval_ok or render3d_st.gsl_token == null or render3d_st.post_w <= 0 or render3d_st.post_h <= 0 { return }
|
|
# DLSS wants at least 8 x (display / render)^2 phases; 32 covers performance mode
|
|
let i = (render3d_st.gsl_frame_n % 32) + 1
|
|
render3d_st.gsl_jpx = gsl_halton(i, 2) - 0.5
|
|
render3d_st.gsl_jpy = gsl_halton(i, 3) - 0.5
|
|
render3d_st.gsl_jitter_x = 2.0 * render3d_st.gsl_jpx / float(render3d_st.post_w)
|
|
render3d_st.gsl_jitter_y = 2.0 * render3d_st.gsl_jpy / float(render3d_st.post_h)
|
|
}
|
|
|
|
# sl::Resource for one of the renderer's textures, in the layout every pass leaves them in
|
|
function gsl_resource(render3d_st: Render3dState, at: int, tex: int) -> void {
|
|
let p = Vk.at(render3d_st.gsl_res, at)
|
|
Vk.zero(p, 112)
|
|
gsl_header(p, 0x3a9d, 0x70cf, 0x2418, 0x4b72, 0x8391, 0x13f8, 0x721c, 0x7261, 1)
|
|
Vk.put_i64(p, 40, render3d_st.gvk_tex_image[tex])
|
|
Vk.put_i64(p, 48, gvk_mem_handle(render3d_st, gvk_mem_id(render3d_st.gvk_tex_mem[tex])))
|
|
Vk.put_i64(p, 56, render3d_st.gvk_tex_view[tex])
|
|
Vk.put_i32(p, 64, GSL_LAYOUT_READ)
|
|
Vk.put_i32(p, 68, render3d_st.gvk_tex_dims_w[tex])
|
|
Vk.put_i32(p, 72, render3d_st.gvk_tex_dims_h[tex])
|
|
Vk.put_i32(p, 76, render3d_st.gvk_tex_vkfmt[tex])
|
|
Vk.put_i32(p, 80, render3d_st.gvk_tex_levels[tex])
|
|
Vk.put_i32(p, 84, render3d_st.gvk_tex_layers[tex])
|
|
Vk.put_i32(p, 100, gvk_tex_usage(render3d_st, tex))
|
|
}
|
|
# the usage gvk_tex_storage gave the image
|
|
function gvk_tex_usage(render3d_st: Render3dState, tex: int) -> int {
|
|
let fmt = render3d_st.gvk_tex_vkfmt[tex]
|
|
var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT
|
|
if fmt == VK_FORMAT_D32_SFLOAT { return usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT }
|
|
usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT
|
|
if render3d_st.gvk_tex_samples[tex] <= 1 and (fmt == VK_FORMAT_R16G16B16A16_SFLOAT or fmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }
|
|
return usage
|
|
}
|
|
|
|
# sl::ResourceTag i, pointing at resource i
|
|
function gsl_tag(render3d_st: Render3dState, i: int, buffer: int, w: int, h: int) -> void {
|
|
let p = Vk.at(render3d_st.gsl_tags, i * 64)
|
|
Vk.zero(p, 64)
|
|
gsl_header(p, 0x4c6a, 0x5aad, 0xb445, 0x496c, 0x87ff, 0x1af3, 0x845b, 0xe653, 1)
|
|
Vk.put_ptr(p, 32, Vk.at(render3d_st.gsl_res, i * 112))
|
|
Vk.put_i32(p, 40, buffer)
|
|
Vk.put_i32(p, 44, 2) # eValidUntilEvaluate
|
|
Vk.put_i32(p, 56, w)
|
|
Vk.put_i32(p, 60, h)
|
|
}
|
|
|
|
# a column-major matrix into a row-major sl::float4x4
|
|
function gsl_put_m4(p: pointer, at: int, m: floats) -> void {
|
|
for r in 0 .. 4 { for c in 0 .. 4 { Vk.put_i32(p, at + (r * 4 + c) * 4, float_bits(m[c * 4 + r])) } }
|
|
}
|
|
function gsl_put_v3(p: pointer, at: int, v: floats) -> void {
|
|
Vk.put_i32(p, at, float_bits(v[0])); Vk.put_i32(p, at + 4, float_bits(v[1])); Vk.put_i32(p, at + 8, float_bits(v[2]))
|
|
}
|
|
|
|
# The renderer's clip space is OpenGL's; what Vulkan stores is depth remapped to [0, 1] and
|
|
# row 0 at NDC y = -1. Streamline reads images with row 0 at the top, so the matrices it is
|
|
# given carry both: y flipped, z' = (z + w) / 2.
|
|
function gsl_clip_fix(m: floats) -> void {
|
|
m4_identity(m)
|
|
m[5] = -1.0
|
|
m[10] = 0.5
|
|
m[14] = 0.5
|
|
}
|
|
|
|
function gsl_constants(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gsl_consts == null { render3d_st.gsl_consts = gsl_struct(456) }
|
|
let k = render3d_st.gsl_consts
|
|
Vk.zero(k, 456)
|
|
gsl_header(k, 0xdcd3, 0x5ad7, 0x4e4a, 0x4bad, 0xa90c, 0xe0c4, 0x9eb2, 0x3afe, 2)
|
|
let fix = m4_new(); let proj = m4_new(); let v2c = m4_new(); let c2v = m4_new()
|
|
let cur = m4_new(); let prev = m4_new(); let inv_cur = m4_new(); let c2p = m4_new(); let p2c = m4_new()
|
|
gsl_clip_fix(fix)
|
|
m4_perspective(proj, render3d_st.cam_fov, render3d_st.cam_aspect, render3d_st.cam_near, r3d_reach(render3d_st, render3d_st.cam_far))
|
|
m4_mul(v2c, fix, proj)
|
|
m4_inverse(c2v, v2c)
|
|
gsl_put_m4(k, 32, v2c) # cameraViewToClip (no jitter)
|
|
gsl_put_m4(k, 96, c2v) # clipToCameraView
|
|
let ident = m4_new()
|
|
m4_identity(ident)
|
|
gsl_put_m4(k, 160, ident) # clipToLensClip
|
|
if render3d_st.gsl_prev_vp == null { render3d_st.gsl_prev_vp = m4_new(); for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] } }
|
|
m4_mul(cur, fix, render3d_st.cam_vp_clean)
|
|
m4_mul(prev, fix, render3d_st.gsl_prev_vp)
|
|
m4_inverse(inv_cur, cur)
|
|
m4_mul(c2p, prev, inv_cur)
|
|
m4_inverse(p2c, c2p)
|
|
gsl_put_m4(k, 224, c2p) # clipToPrevClip
|
|
gsl_put_m4(k, 288, p2c) # prevClipToClip
|
|
# the sample's offset from the pixel centre, in the image Streamline sees: row 0 at the top, so the
|
|
# vertical offset flips with it. With jy unflipped DLSS resolved the ground into concentric
|
|
# rings; with jx flipped too, thin stems doubled sideways (PC shots, 2026-09-15).
|
|
var jx = -render3d_st.gsl_jpx
|
|
var jy = -render3d_st.gsl_jpy
|
|
# R3D_DLSS_JX / R3D_DLSS_JY = -1 flip a sign, to check the convention against the picture
|
|
if r3d_env_has(render3d_st, "R3D_DLSS_JX") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JX")) < 0 { jx = -jx }
|
|
if r3d_env_has(render3d_st, "R3D_DLSS_JY") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JY")) < 0 { jy = -jy }
|
|
Vk.put_i32(k, 352, float_bits(jx))
|
|
Vk.put_i32(k, 356, float_bits(jy))
|
|
Vk.put_i32(k, 360, float_bits(1.0)) # mvecScale
|
|
Vk.put_i32(k, 364, float_bits(1.0))
|
|
gsl_put_v3(k, 376, render3d_st.cam_pos)
|
|
let up = render3d_st.gsl_up # made with the state
|
|
v3_cross(up, render3d_st.cam_right, render3d_st.cam_fwd)
|
|
gsl_put_v3(k, 388, up)
|
|
gsl_put_v3(k, 400, render3d_st.cam_right)
|
|
gsl_put_v3(k, 412, render3d_st.cam_fwd)
|
|
Vk.put_i32(k, 424, float_bits(render3d_st.cam_near))
|
|
Vk.put_i32(k, 428, float_bits(r3d_reach(render3d_st, render3d_st.cam_far)))
|
|
Vk.put_i32(k, 432, float_bits(render3d_st.cam_fov))
|
|
Vk.put_i32(k, 436, float_bits(render3d_st.cam_aspect))
|
|
Vk.put_i32(k, 440, float_bits(0.0)) # motionVectorsInvalidValue
|
|
# depthInverted, cameraMotionIncluded, motionVectors3D false; reset on a cut; not orthographic,
|
|
# not dilated, not jittered
|
|
if render3d_st.gsl_reset { Vk.put_i32(k, 444, 256 * 256 * 256) }
|
|
Vk.put_i32(k, 452, float_bits(40.0)) # minRelativeLinearDepthObjectSeparation
|
|
free(fix); free(proj); free(v2c); free(c2v); free(cur); free(prev); free(inv_cur); free(c2p); free(p2c); free(ident)
|
|
}
|
|
|
|
# make the DLSS targets at this frame's sizes
|
|
function gsl_targets(render3d_st: mut Render3dState) -> void {
|
|
if render3d_st.gsl_mv == null or render3d_st.gsl_mv.w != render3d_st.post_w or render3d_st.gsl_mv.h != render3d_st.post_h {
|
|
if render3d_st.gsl_mv != null { target_free(render3d_st, render3d_st.gsl_mv) }
|
|
render3d_st.gsl_mv = target_new(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_RG16F, GL_RG, GL_HALF_FLOAT, false, GL_NEAREST)
|
|
render3d_st.gsl_reset = true
|
|
}
|
|
if render3d_st.gsl_out == null or render3d_st.gsl_out.w != gl_width() or render3d_st.gsl_out.h != gl_height() {
|
|
if render3d_st.gsl_out != null { target_free(render3d_st, render3d_st.gsl_out) }
|
|
render3d_st.gsl_out = target_new(render3d_st, gl_width(), gl_height(), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
|
render3d_st.gsl_reset = true
|
|
}
|
|
# the LDR image the tonemap writes follows the upscaled size
|
|
if render3d_st.post_ldr.w != gl_width() or render3d_st.post_ldr.h != gl_height() {
|
|
target_free(render3d_st, render3d_st.post_ldr)
|
|
render3d_st.post_ldr = target_new(render3d_st, gl_width(), gl_height(), post_ldr_fmt(render3d_st), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
|
render3d_st.post_ldr_hdr = gpu_hdr_active(render3d_st)
|
|
}
|
|
}
|
|
|
|
# Upscale post_hdr into gsl_out; the colour the rest of post reads (post_hdr's own if it failed).
|
|
function gsl_dlss_eval(render3d_st: mut Render3dState) -> int {
|
|
gsl_targets(render3d_st)
|
|
# no motion of its own: the camera's comes from depth
|
|
target_bind(render3d_st, render3d_st.gsl_mv)
|
|
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
|
|
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT)
|
|
gvk_pass_end(render3d_st)
|
|
let cb = gvk_frame_cb(render3d_st)
|
|
if render3d_st.gsl_set_mode != render3d_st.gsl_dlss_mode or render3d_st.gsl_set_w != gl_width() or render3d_st.gsl_set_h != gl_height() {
|
|
let f = gsl_fn(GSL_DLSS, "slDLSSSetOptions")
|
|
gsl_fill_options(render3d_st, gl_width(), gl_height())
|
|
if f != null and Vk.sl_call_pp(f, render3d_st.gsl_vp, render3d_st.gsl_opts) == 0 { render3d_st.gsl_set_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_set_w = gl_width(); render3d_st.gsl_set_h = gl_height() }
|
|
}
|
|
gsl_constants(render3d_st)
|
|
Vk.sl_set_constants(render3d_st.gsl_consts, render3d_st.gsl_token, render3d_st.gsl_vp)
|
|
gsl_resource(render3d_st, 0, render3d_st.post_hdr.depth); gsl_tag(render3d_st, 0, 0, render3d_st.post_w, render3d_st.post_h) # kBufferTypeDepth
|
|
gsl_resource(render3d_st, 112, render3d_st.gsl_mv.color); gsl_tag(render3d_st, 1, 1, render3d_st.post_w, render3d_st.post_h) # kBufferTypeMotionVectors
|
|
gsl_resource(render3d_st, 224, render3d_st.post_hdr.color); gsl_tag(render3d_st, 2, 3, render3d_st.post_w, render3d_st.post_h) # kBufferTypeScalingInputColor
|
|
gsl_resource(render3d_st, 336, render3d_st.gsl_out.color); gsl_tag(render3d_st, 3, 4, gl_width(), gl_height()) # kBufferTypeScalingOutputColor
|
|
Vk.sl_set_tag_for_frame(render3d_st.gsl_token, render3d_st.gsl_vp, render3d_st.gsl_tags, 4, cb)
|
|
Vk.put_ptr(render3d_st.gsl_inputs, 0, render3d_st.gsl_vp)
|
|
let r = Vk.sl_evaluate_feature(GSL_DLSS, render3d_st.gsl_token, render3d_st.gsl_inputs, 1, cb)
|
|
for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] }
|
|
render3d_st.gsl_reset = false
|
|
# Streamline records its own pipeline and descriptors into the command buffer; nothing needs
|
|
# forgetting, because every gvk_draw binds its pipeline, view and set afresh
|
|
render3d_st.gsl_eval_ok = r == 0
|
|
if r != 0 {
|
|
if not render3d_st.gsl_said { gsl_say_eval_failed(r); render3d_st.gsl_said = true }
|
|
render3d_st.post_color_w = render3d_st.post_w; render3d_st.post_color_h = render3d_st.post_h
|
|
return render3d_st.post_hdr.color
|
|
}
|
|
render3d_st.post_color_w = gl_width(); render3d_st.post_color_h = gl_height()
|
|
return render3d_st.gsl_out.color
|
|
}
|
|
|
|
# messages, each built in a function of its own so the path that says it holds no allocation
|
|
@alloc_ok("a message, built only when it is said: a failure, a warning or a debug switch")
|
|
function gsl_say_eval_failed(r: int) -> void { print(`r3d: streamline: DLSS evaluate failed ({r}); drawing without it`) }
|