wip(0.S3): the packages migrated by ludic migrate state packages - every package test green
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
1e8b5b0523
commit
07505e7ef2
294 changed files with 14996 additions and 14675 deletions
|
|
@ -58,44 +58,35 @@ property AcProg {
|
|||
scene_frame: int = -1 # the frame its scene uniforms were bound
|
||||
}
|
||||
|
||||
var ac_lit: AcProg = null
|
||||
var ac_lit_cut: AcProg = null
|
||||
var ac_sh: AcProg = null
|
||||
var ac_sh_cut: AcProg = null
|
||||
var ac_out: AcProg = null
|
||||
var ac_out_cut: AcProg = null
|
||||
var ac_actors: []Actor = null
|
||||
var ac_next_id: int = 0
|
||||
var ac_frame: int = 0
|
||||
|
||||
function ac_prog_new(vs: string, fs: string, defs: string) -> AcProg {
|
||||
function ac_prog_new(render3d_st: mut Render3dState, vs: string, fs: string, defs: string) -> AcProg {
|
||||
let a = new AcProg
|
||||
a.prog = r3d_program(vs, fs, defs)
|
||||
a.prog = r3d_program(render3d_st, vs, fs, defs)
|
||||
let p = a.prog
|
||||
a.l_model = gpu_uniform(p, "u_model"); a.l_skin = gpu_uniform(p, "u_skinned"); a.l_lvp = gpu_uniform(p, "u_light_vp")
|
||||
a.l_bones = gpu_uniform(p, "u_bones[0]"); if a.l_bones < 0 { a.l_bones = gpu_uniform(p, "u_bones") }
|
||||
a.l_tint = gpu_uniform(p, "u_tint"); a.l_rough = gpu_uniform(p, "u_rough_scale"); a.l_emis = gpu_uniform(p, "u_emissive")
|
||||
a.l_view = gpu_uniform(p, "u_view"); a.l_proj = gpu_uniform(p, "u_proj"); a.l_mh = gpu_uniform(p, "u_model_h")
|
||||
a.l_out = gpu_uniform(p, "u_outline"); a.l_ocol = gpu_uniform(p, "u_outline_col")
|
||||
a.l_model = gpu_uniform(render3d_st, p, "u_model"); a.l_skin = gpu_uniform(render3d_st, p, "u_skinned"); a.l_lvp = gpu_uniform(render3d_st, p, "u_light_vp")
|
||||
a.l_bones = gpu_uniform(render3d_st, p, "u_bones[0]"); if a.l_bones < 0 { a.l_bones = gpu_uniform(render3d_st, p, "u_bones") }
|
||||
a.l_tint = gpu_uniform(render3d_st, p, "u_tint"); a.l_rough = gpu_uniform(render3d_st, p, "u_rough_scale"); a.l_emis = gpu_uniform(render3d_st, p, "u_emissive")
|
||||
a.l_view = gpu_uniform(render3d_st, p, "u_view"); a.l_proj = gpu_uniform(render3d_st, p, "u_proj"); a.l_mh = gpu_uniform(render3d_st, p, "u_model_h")
|
||||
a.l_out = gpu_uniform(render3d_st, p, "u_outline"); a.l_ocol = gpu_uniform(render3d_st, p, "u_outline_col")
|
||||
# the samplers never move: units 0, 1, 2
|
||||
gpu_use_program(p)
|
||||
u_i(gpu_uniform(p, "u_diff"), 0); u_i(gpu_uniform(p, "u_nrm"), 1); u_i(gpu_uniform(p, "u_arm"), 2)
|
||||
gpu_use_program(render3d_st, p)
|
||||
u_i(render3d_st, gpu_uniform(render3d_st, p, "u_diff"), 0); u_i(render3d_st, gpu_uniform(render3d_st, p, "u_nrm"), 1); u_i(render3d_st, gpu_uniform(render3d_st, p, "u_arm"), 2)
|
||||
return a
|
||||
}
|
||||
function actor_init() -> void {
|
||||
ac_lit = ac_prog_new("skin.vert", "model.frag", "")
|
||||
ac_lit_cut = ac_prog_new("skin.vert", "model.frag", "#define ALPHA_TEST\n")
|
||||
ac_sh = ac_prog_new("skin.vert", "shadow.frag", "#define SHADOW_PASS\n")
|
||||
ac_sh_cut = ac_prog_new("skin.vert", "shadow.frag", "#define SHADOW_PASS\n#define ALPHA_TEST\n")
|
||||
ac_out = ac_prog_new("skin.vert", "outline.frag", "#define OUTLINE\n")
|
||||
ac_out_cut = ac_prog_new("skin.vert", "outline.frag", "#define OUTLINE\n#define ALPHA_TEST\n")
|
||||
ac_actors = new []Actor
|
||||
function actor_init(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.ac_lit = ac_prog_new(render3d_st, "skin.vert", "model.frag", "")
|
||||
render3d_st.ac_lit_cut = ac_prog_new(render3d_st, "skin.vert", "model.frag", "#define ALPHA_TEST\n")
|
||||
render3d_st.ac_sh = ac_prog_new(render3d_st, "skin.vert", "shadow.frag", "#define SHADOW_PASS\n")
|
||||
render3d_st.ac_sh_cut = ac_prog_new(render3d_st, "skin.vert", "shadow.frag", "#define SHADOW_PASS\n#define ALPHA_TEST\n")
|
||||
render3d_st.ac_out = ac_prog_new(render3d_st, "skin.vert", "outline.frag", "#define OUTLINE\n")
|
||||
render3d_st.ac_out_cut = ac_prog_new(render3d_st, "skin.vert", "outline.frag", "#define OUTLINE\n#define ALPHA_TEST\n")
|
||||
render3d_st.ac_actors = new []Actor
|
||||
}
|
||||
function actor_remove(a: Actor) -> void {
|
||||
if ac_actors == null { return }
|
||||
function actor_remove(render3d_st: mut Render3dState, a: Actor) -> void {
|
||||
if render3d_st.ac_actors == null { return }
|
||||
let keep = new []Actor
|
||||
for i in 0 .. len(ac_actors) { if ac_actors[i].id != a.id { push(keep, ac_actors[i]) } }
|
||||
ac_actors = keep
|
||||
for i in 0 .. len(render3d_st.ac_actors) { if render3d_st.ac_actors[i].id != a.id { push(keep, render3d_st.ac_actors[i]) } }
|
||||
render3d_st.ac_actors = keep
|
||||
}
|
||||
# colour one named part of the model (a material name from the file)
|
||||
function actor_tint_part(a: Actor, name: string, r: float, g: float, b: float) -> void {
|
||||
|
|
@ -113,18 +104,18 @@ function actor_hide_part(a: Actor, name: string, hidden: bool) -> void {
|
|||
for i in 0 .. n { if a.model.prims[i].name == name { a.hide[i] = v } }
|
||||
}
|
||||
|
||||
function actor_new(model: Model) -> Actor {
|
||||
function actor_new(render3d_st: mut Render3dState, model: Model) -> Actor {
|
||||
let a = new Actor
|
||||
a.model = model
|
||||
a.scale = 1.0
|
||||
a.tint = v3_new(1.0, 1.0, 1.0)
|
||||
a.rough = 1.0
|
||||
a.mat = m4_new()
|
||||
ac_next_id += 1; a.id = ac_next_id
|
||||
render3d_st.ac_next_id += 1; a.id = render3d_st.ac_next_id
|
||||
if model != null { a.radius = Math.max(model.radius, model.height) + 1.0 }
|
||||
a.cull = 450.0
|
||||
if ac_actors == null { ac_actors = new []Actor }
|
||||
push(ac_actors, a)
|
||||
if render3d_st.ac_actors == null { render3d_st.ac_actors = new []Actor }
|
||||
push(render3d_st.ac_actors, a)
|
||||
return a
|
||||
}
|
||||
|
||||
|
|
@ -134,24 +125,24 @@ function actor_place(a: Actor, x: float, y: float, z: float, yaw: float) -> void
|
|||
}
|
||||
|
||||
# the scene's lighting for a lit program, once per frame
|
||||
function ac_bind_scene(ap: AcProg) -> void {
|
||||
if ap.scene_frame == ac_frame { return }
|
||||
ap.scene_frame = ac_frame
|
||||
function ac_bind_scene(render3d_st: mut Render3dState, ap: AcProg) -> void {
|
||||
if ap.scene_frame == render3d_st.ac_frame { return }
|
||||
ap.scene_frame = render3d_st.ac_frame
|
||||
let p = ap.prog
|
||||
gpu_use_program(p)
|
||||
u_mat4(ap.l_view, cam_view)
|
||||
u_mat4(ap.l_proj, cam_proj)
|
||||
u_f(ap.l_mh, 0.0)
|
||||
sky_bind_lighting(p)
|
||||
shadow_bind(p)
|
||||
fog_bind(p)
|
||||
gpu_use_program(render3d_st, p)
|
||||
u_mat4(render3d_st, ap.l_view, render3d_st.cam_view)
|
||||
u_mat4(render3d_st, ap.l_proj, render3d_st.cam_proj)
|
||||
u_f(render3d_st, ap.l_mh, 0.0)
|
||||
sky_bind_lighting(render3d_st, p)
|
||||
shadow_bind(render3d_st, p)
|
||||
fog_bind(render3d_st, p)
|
||||
}
|
||||
function ac_visible(a: Actor, shadow: bool) -> bool {
|
||||
function ac_visible(render3d_st: Render3dState, a: Actor, shadow: bool) -> bool {
|
||||
if a.model == null { return false }
|
||||
if not a.visible and not (shadow and a.cast_hidden) { return false }
|
||||
if shadow and not a.casts { return false }
|
||||
if a.cull != 0.0 {
|
||||
let dx = a.x - cam_pos[0]; let dz = a.z - cam_pos[2]
|
||||
let dx = a.x - render3d_st.cam_pos[0]; let dz = a.z - render3d_st.cam_pos[2]
|
||||
let d2 = dx * dx + dz * dz
|
||||
var c = a.cull
|
||||
if shadow { c = Math.min(c, 300.0) }
|
||||
|
|
@ -164,38 +155,38 @@ function ac_visible(a: Actor, shadow: bool) -> bool {
|
|||
# keeps the main camera's frustum planes, so the image is what has to be tested against them: the
|
||||
# actor itself was, and the town's people - far above the lake, their images far below the frame -
|
||||
# all drew again into the reflection.
|
||||
if ter_reflect { cy = 2.0 * water_level - cy }
|
||||
if not cam_sphere_visible(a.x, cy, a.z, r * 1.5) { return false }
|
||||
if render3d_st.ter_reflect { cy = 2.0 * render3d_st.water_level - cy }
|
||||
if not cam_sphere_visible(render3d_st, a.x, cy, a.z, r * 1.5) { return false }
|
||||
}
|
||||
return true
|
||||
}
|
||||
function actor_draw_one(a: Actor, ap: AcProg, shadow: bool) -> void {
|
||||
function actor_draw_one(render3d_st: mut Render3dState, a: Actor, ap: AcProg, shadow: bool) -> void {
|
||||
let p = ap.prog
|
||||
gpu_use_program(p)
|
||||
u_mat4(ap.l_model, a.mat)
|
||||
gpu_use_program(render3d_st, p)
|
||||
u_mat4(render3d_st, ap.l_model, a.mat)
|
||||
var skinned = 0.0
|
||||
if a.skin != null { skinned = 1.0; u_mat4n(ap.l_bones, a.skin.n_joints, a.skin.bones) }
|
||||
else if a.model.skin != null { skinned = 1.0; u_mat4n(ap.l_bones, a.model.skin.n_joints, a.model.skin.bones) }
|
||||
u_f(ap.l_skin, skinned)
|
||||
if a.skin != null { skinned = 1.0; u_mat4n(render3d_st, ap.l_bones, a.skin.n_joints, a.skin.bones) }
|
||||
else if a.model.skin != null { skinned = 1.0; u_mat4n(render3d_st, ap.l_bones, a.model.skin.n_joints, a.model.skin.bones) }
|
||||
u_f(render3d_st, ap.l_skin, skinned)
|
||||
if not shadow {
|
||||
u_f(ap.l_emis, a.emissive)
|
||||
u_f(ap.l_rough, a.rough)
|
||||
u_f(render3d_st, ap.l_emis, a.emissive)
|
||||
u_f(render3d_st, ap.l_rough, a.rough)
|
||||
}
|
||||
gpu_cull(false)
|
||||
gpu_cull(render3d_st, false)
|
||||
let model = a.model
|
||||
for i in 0 .. len(model.prims) {
|
||||
let pr = model.prims[i]
|
||||
if a.hide != null and a.hide[i] != 0 { continue }
|
||||
if not shadow {
|
||||
if a.ptint != null and a.ptint[i * 4] != 0 { u_f3(ap.l_tint, float_from_bits(a.ptint[i * 4 + 1]), float_from_bits(a.ptint[i * 4 + 2]), float_from_bits(a.ptint[i * 4 + 3])) }
|
||||
else { u_v3(ap.l_tint, a.tint) }
|
||||
if a.ptint != null and a.ptint[i * 4] != 0 { u_f3(render3d_st, ap.l_tint, float_from_bits(a.ptint[i * 4 + 1]), float_from_bits(a.ptint[i * 4 + 2]), float_from_bits(a.ptint[i * 4 + 3])) }
|
||||
else { u_v3(render3d_st, ap.l_tint, a.tint) }
|
||||
}
|
||||
gpu_tex_unit(0); gpu_tex_bind(GPU_TEX2D, pr.diff)
|
||||
gpu_tex_unit(render3d_st, 0); gpu_tex_bind(render3d_st, GPU_TEX2D, pr.diff)
|
||||
if not shadow {
|
||||
gpu_tex_unit(1); gpu_tex_bind(GPU_TEX2D, pr.nrm)
|
||||
gpu_tex_unit(2); gpu_tex_bind(GPU_TEX2D, pr.arm)
|
||||
gpu_tex_unit(render3d_st, 1); gpu_tex_bind(render3d_st, GPU_TEX2D, pr.nrm)
|
||||
gpu_tex_unit(render3d_st, 2); gpu_tex_bind(render3d_st, GPU_TEX2D, pr.arm)
|
||||
}
|
||||
mesh_draw(pr.mesh)
|
||||
mesh_draw(render3d_st, pr.mesh)
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -224,41 +215,38 @@ property OutlineReq {
|
|||
g: float = 0.0,
|
||||
b: float = 0.0
|
||||
}
|
||||
var ac_oq: []OutlineReq = null
|
||||
var ac_oq_n: int = 0 # live entries; the array is kept and reused
|
||||
# A frame is drawn by more than one pass - a shadow map, a water reflection, the scene -
|
||||
# and actor_draw runs in each of them. An Actor survives that because its rim is a field
|
||||
# on the actor; a queue does not, if the first pass empties it. So a flush CLOSES the
|
||||
# batch rather than clearing it, and the next add opens a new one: every pass in a frame
|
||||
# sees the same requests, and the caller needs no frame hook.
|
||||
var ac_oq_closed: bool = true
|
||||
function outline_model(m: Model, mat: floats, width: float, r: float, g: float, b: float) -> void {
|
||||
function outline_model(render3d_st: mut Render3dState, m: Model, mat: floats, width: float, r: float, g: float, b: float) -> void {
|
||||
if m == null or mat == null or width == 0.0 { return }
|
||||
if ac_oq_closed { ac_oq_n = 0; ac_oq_closed = false }
|
||||
if ac_oq == null { ac_oq = new []OutlineReq }
|
||||
if render3d_st.ac_oq_closed { render3d_st.ac_oq_n = 0; render3d_st.ac_oq_closed = false }
|
||||
if render3d_st.ac_oq == null { render3d_st.ac_oq = new []OutlineReq }
|
||||
var q: OutlineReq = null
|
||||
if ac_oq_n < len(ac_oq) { q = ac_oq[ac_oq_n] } else { q = new OutlineReq; q.mat = m4_new(); push(ac_oq, q) }
|
||||
if render3d_st.ac_oq_n < len(render3d_st.ac_oq) { q = render3d_st.ac_oq[render3d_st.ac_oq_n] } else { q = new OutlineReq; q.mat = m4_new(); push(render3d_st.ac_oq, q) }
|
||||
q.model = m; q.width = width; q.r = r; q.g = g; q.b = b
|
||||
m4_copy(q.mat, mat)
|
||||
ac_oq_n += 1
|
||||
render3d_st.ac_oq_n += 1
|
||||
}
|
||||
function outline_clear() -> void { ac_oq_n = 0; ac_oq_closed = true }
|
||||
function outline_clear(render3d_st: mut Render3dState) -> void { render3d_st.ac_oq_n = 0; render3d_st.ac_oq_closed = true }
|
||||
# Called at the top of every frame. A batch the last frame closed and nobody has reopened
|
||||
# since is stale: the caller stopped asking. Without this it was drawn for ever - flush
|
||||
# closes a batch but only the NEXT outline_model call empties it, so looking away from a
|
||||
# highlighted tree left its rim on until something else was highlighted.
|
||||
function outline_frame() -> void { if ac_oq_closed { ac_oq_n = 0 } }
|
||||
function outline_frame(render3d_st: mut Render3dState) -> void { if render3d_st.ac_oq_closed { render3d_st.ac_oq_n = 0 } }
|
||||
|
||||
# R3D_ACTOR_CENSUS=<frame>: at that frame, the lit pass's actors grouped by model - skinned or rigid,
|
||||
# how many share it, the primitives each draws - so instancing is aimed at what repeats
|
||||
function actor_census() -> void {
|
||||
function actor_census(render3d_st: Render3dState) -> void {
|
||||
let keys = new []Model
|
||||
let cnt = new []int
|
||||
let prims = new []int
|
||||
let rigid = new []int
|
||||
for i in 0 .. len(ac_actors) {
|
||||
let a = ac_actors[i]
|
||||
if not ac_visible(a, false) { continue }
|
||||
for i in 0 .. len(render3d_st.ac_actors) {
|
||||
let a = render3d_st.ac_actors[i]
|
||||
if not ac_visible(render3d_st, a, false) { continue }
|
||||
var k = -1
|
||||
for j in 0 .. len(keys) { if keys[j] == a.model { k = j } }
|
||||
if k < 0 { push(keys, a.model); push(cnt, 0); push(prims, 0); var r = 1; if a.model.skin != null { r = 0 }; push(rigid, r); k = len(keys) - 1 }
|
||||
|
|
@ -277,71 +265,70 @@ function actor_census() -> void {
|
|||
}
|
||||
print(`actor census: {len(keys)} models, {total} lit draws`)
|
||||
}
|
||||
var ac_census_at: int = -2
|
||||
function actor_draw() -> void {
|
||||
if ac_actors == null { return }
|
||||
ac_frame += 1
|
||||
if ac_census_at == -2 { ac_census_at = -1; if r3d_env_has("R3D_ACTOR_CENSUS") { ac_census_at = Text.to_int(r3d_env("R3D_ACTOR_CENSUS")) } }
|
||||
if ac_census_at > 0 and ac_frame == ac_census_at { actor_census() }
|
||||
for i in 0 .. len(ac_actors) {
|
||||
let a = ac_actors[i]
|
||||
if not ac_visible(a, false) { continue }
|
||||
var ap = ac_lit
|
||||
if a.cutout { ap = ac_lit_cut }
|
||||
ac_bind_scene(ap)
|
||||
actor_draw_one(a, ap, false)
|
||||
function actor_draw(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.ac_actors == null { return }
|
||||
render3d_st.ac_frame += 1
|
||||
if render3d_st.ac_census_at == -2 { render3d_st.ac_census_at = -1; if r3d_env_has(render3d_st, "R3D_ACTOR_CENSUS") { render3d_st.ac_census_at = Text.to_int(r3d_env(render3d_st, "R3D_ACTOR_CENSUS")) } }
|
||||
if render3d_st.ac_census_at > 0 and render3d_st.ac_frame == render3d_st.ac_census_at { actor_census(render3d_st) }
|
||||
for i in 0 .. len(render3d_st.ac_actors) {
|
||||
let a = render3d_st.ac_actors[i]
|
||||
if not ac_visible(render3d_st, a, false) { continue }
|
||||
var ap = render3d_st.ac_lit
|
||||
if a.cutout { ap = render3d_st.ac_lit_cut }
|
||||
ac_bind_scene(render3d_st, ap)
|
||||
actor_draw_one(render3d_st, a, ap, false)
|
||||
}
|
||||
actor_draw_outlines()
|
||||
gpu_cull(true)
|
||||
actor_draw_outlines(render3d_st)
|
||||
gpu_cull(render3d_st, true)
|
||||
}
|
||||
# The rim around a highlighted actor: the model again, its vertices pushed out along their
|
||||
# normals (skin.vert, OUTLINE) and its front faces culled, so only the far side of the
|
||||
# swollen shell survives — a silhouette exactly `outline` metres wide. Depth-tested
|
||||
# against the scene, so anything in front of the actor hides its rim too.
|
||||
function actor_draw_outlines() -> void {
|
||||
var any = ac_oq_n > 0
|
||||
for i in 0 .. len(ac_actors) { if ac_actors[i].outline != 0.0 and ac_visible(ac_actors[i], false) { any = true; break } }
|
||||
function actor_draw_outlines(render3d_st: mut Render3dState) -> void {
|
||||
var any = render3d_st.ac_oq_n > 0
|
||||
for i in 0 .. len(render3d_st.ac_actors) { if render3d_st.ac_actors[i].outline != 0.0 and ac_visible(render3d_st, render3d_st.ac_actors[i], false) { any = true; break } }
|
||||
if not any { return }
|
||||
gpu_cull(true)
|
||||
gpu_cull_face(GL_FRONT)
|
||||
for i in 0 .. len(ac_actors) {
|
||||
let a = ac_actors[i]
|
||||
if a.outline == 0.0 or not ac_visible(a, false) { continue }
|
||||
var ap = ac_out
|
||||
if a.cutout { ap = ac_out_cut }
|
||||
gpu_use_program(ap.prog)
|
||||
u_mat4(ap.l_view, cam_view); u_mat4(ap.l_proj, cam_proj)
|
||||
u_f(ap.l_out, a.outline * a.scale)
|
||||
if a.ocol != null { u_v3(ap.l_ocol, a.ocol) } else { u_f3(ap.l_ocol, 1.0, 1.0, 1.0) }
|
||||
actor_draw_outline_one(a, ap)
|
||||
gpu_cull(render3d_st, true)
|
||||
gpu_cull_face(render3d_st, GL_FRONT)
|
||||
for i in 0 .. len(render3d_st.ac_actors) {
|
||||
let a = render3d_st.ac_actors[i]
|
||||
if a.outline == 0.0 or not ac_visible(render3d_st, a, false) { continue }
|
||||
var ap = render3d_st.ac_out
|
||||
if a.cutout { ap = render3d_st.ac_out_cut }
|
||||
gpu_use_program(render3d_st, ap.prog)
|
||||
u_mat4(render3d_st, ap.l_view, render3d_st.cam_view); u_mat4(render3d_st, ap.l_proj, render3d_st.cam_proj)
|
||||
u_f(render3d_st, ap.l_out, a.outline * a.scale)
|
||||
if a.ocol != null { u_v3(render3d_st, ap.l_ocol, a.ocol) } else { u_f3(render3d_st, ap.l_ocol, 1.0, 1.0, 1.0) }
|
||||
actor_draw_outline_one(render3d_st, a, ap)
|
||||
}
|
||||
# and whatever asked for a rim without being an actor
|
||||
for i in 0 .. ac_oq_n {
|
||||
let q = ac_oq[i]
|
||||
let ap = ac_out
|
||||
gpu_use_program(ap.prog)
|
||||
u_mat4(ap.l_view, cam_view); u_mat4(ap.l_proj, cam_proj)
|
||||
u_f(ap.l_out, q.width)
|
||||
u_f3(ap.l_ocol, q.r, q.g, q.b)
|
||||
u_mat4(ap.l_model, q.mat)
|
||||
u_f(ap.l_skin, 0.0)
|
||||
for k in 0 .. len(q.model.prims) { mesh_draw(q.model.prims[k].mesh) }
|
||||
for i in 0 .. render3d_st.ac_oq_n {
|
||||
let q = render3d_st.ac_oq[i]
|
||||
let ap = render3d_st.ac_out
|
||||
gpu_use_program(render3d_st, ap.prog)
|
||||
u_mat4(render3d_st, ap.l_view, render3d_st.cam_view); u_mat4(render3d_st, ap.l_proj, render3d_st.cam_proj)
|
||||
u_f(render3d_st, ap.l_out, q.width)
|
||||
u_f3(render3d_st, ap.l_ocol, q.r, q.g, q.b)
|
||||
u_mat4(render3d_st, ap.l_model, q.mat)
|
||||
u_f(render3d_st, ap.l_skin, 0.0)
|
||||
for k in 0 .. len(q.model.prims) { mesh_draw(render3d_st, q.model.prims[k].mesh) }
|
||||
}
|
||||
ac_oq_closed = true
|
||||
gpu_cull_face(GL_BACK)
|
||||
render3d_st.ac_oq_closed = true
|
||||
gpu_cull_face(render3d_st, GL_BACK)
|
||||
}
|
||||
function actor_draw_outline_one(a: Actor, ap: AcProg) -> void {
|
||||
u_mat4(ap.l_model, a.mat)
|
||||
function actor_draw_outline_one(render3d_st: mut Render3dState, a: Actor, ap: AcProg) -> void {
|
||||
u_mat4(render3d_st, ap.l_model, a.mat)
|
||||
var skinned = 0.0
|
||||
if a.skin != null { skinned = 1.0; u_mat4n(ap.l_bones, a.skin.n_joints, a.skin.bones) }
|
||||
else if a.model.skin != null { skinned = 1.0; u_mat4n(ap.l_bones, a.model.skin.n_joints, a.model.skin.bones) }
|
||||
u_f(ap.l_skin, skinned)
|
||||
if a.skin != null { skinned = 1.0; u_mat4n(render3d_st, ap.l_bones, a.skin.n_joints, a.skin.bones) }
|
||||
else if a.model.skin != null { skinned = 1.0; u_mat4n(render3d_st, ap.l_bones, a.model.skin.n_joints, a.model.skin.bones) }
|
||||
u_f(render3d_st, ap.l_skin, skinned)
|
||||
let model = a.model
|
||||
for i in 0 .. len(model.prims) {
|
||||
let pr = model.prims[i]
|
||||
if a.hide != null and a.hide[i] != 0 { continue }
|
||||
if a.cutout { gpu_tex_unit(0); gpu_tex_bind(GPU_TEX2D, pr.diff) }
|
||||
mesh_draw(pr.mesh)
|
||||
if a.cutout { gpu_tex_unit(render3d_st, 0); gpu_tex_bind(render3d_st, GPU_TEX2D, pr.diff) }
|
||||
mesh_draw(render3d_st, pr.mesh)
|
||||
}
|
||||
}
|
||||
# Is a sphere inside this cascade's light box? The projection is orthographic, so a point's
|
||||
|
|
@ -349,74 +336,66 @@ function actor_draw_outline_one(a: Actor, ap: AcProg) -> void {
|
|||
# three axis offsets. An actor outside the box casts nothing into that layer: skipping it changes no
|
||||
# depth, only the draw. (Each actor used to draw into every cascade out to 300 m - in town that was
|
||||
# about 1500 skinned shadow draws a frame against 69 in the lit pass.)
|
||||
var ac_lp0: floats = null
|
||||
var ac_lpx: floats = null
|
||||
var ac_lpy: floats = null
|
||||
var ac_lpz: floats = null
|
||||
function ac_in_light(vp: floats, x: float, y: float, z: float, r: float) -> bool {
|
||||
if ac_lp0 == null { ac_lp0 = floats(3); ac_lpx = floats(3); ac_lpy = floats(3); ac_lpz = floats(3) }
|
||||
m4_xform_point(ac_lp0, vp, x, y, z)
|
||||
m4_xform_point(ac_lpx, vp, x + r, y, z)
|
||||
m4_xform_point(ac_lpy, vp, x, y + r, z)
|
||||
m4_xform_point(ac_lpz, vp, x, y, z + r)
|
||||
function ac_in_light(render3d_st: mut Render3dState, vp: floats, x: float, y: float, z: float, r: float) -> bool {
|
||||
if render3d_st.ac_lp0 == null { render3d_st.ac_lp0 = floats(3); render3d_st.ac_lpx = floats(3); render3d_st.ac_lpy = floats(3); render3d_st.ac_lpz = floats(3) }
|
||||
m4_xform_point(render3d_st.ac_lp0, vp, x, y, z)
|
||||
m4_xform_point(render3d_st.ac_lpx, vp, x + r, y, z)
|
||||
m4_xform_point(render3d_st.ac_lpy, vp, x, y + r, z)
|
||||
m4_xform_point(render3d_st.ac_lpz, vp, x, y, z + r)
|
||||
for k in 0 .. 2 {
|
||||
let c = ac_lp0[k]
|
||||
let e = ac_abs(ac_lpx[k] - c) + ac_abs(ac_lpy[k] - c) + ac_abs(ac_lpz[k] - c)
|
||||
let c = render3d_st.ac_lp0[k]
|
||||
let e = ac_abs(render3d_st.ac_lpx[k] - c) + ac_abs(render3d_st.ac_lpy[k] - c) + ac_abs(render3d_st.ac_lpz[k] - c)
|
||||
if ac_abs(c) - e > 1.0 { return false }
|
||||
}
|
||||
return true
|
||||
}
|
||||
function ac_abs(v: float) -> float { return Math.max(v, -v) }
|
||||
|
||||
function actor_draw_casters(light_vp: floats) -> void {
|
||||
if ac_actors == null { return }
|
||||
gpu_use_program(ac_sh.prog); u_mat4(ac_sh.l_lvp, light_vp)
|
||||
gpu_use_program(ac_sh_cut.prog); u_mat4(ac_sh_cut.l_lvp, light_vp)
|
||||
for i in 0 .. len(ac_actors) {
|
||||
let a = ac_actors[i]
|
||||
if not ac_visible(a, true) { continue }
|
||||
function actor_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void {
|
||||
if render3d_st.ac_actors == null { return }
|
||||
gpu_use_program(render3d_st, render3d_st.ac_sh.prog); u_mat4(render3d_st, render3d_st.ac_sh.l_lvp, light_vp)
|
||||
gpu_use_program(render3d_st, render3d_st.ac_sh_cut.prog); u_mat4(render3d_st, render3d_st.ac_sh_cut.l_lvp, light_vp)
|
||||
for i in 0 .. len(render3d_st.ac_actors) {
|
||||
let a = render3d_st.ac_actors[i]
|
||||
if not ac_visible(render3d_st, a, true) { continue }
|
||||
if a.radius != 0.0 {
|
||||
# the radius is the horizontal extent: a person is far taller than wide, so the sphere is
|
||||
# centred at half the model's height and reaches the larger of the two, with a margin for
|
||||
# the pose (a first version centred it at the radius and dropped a head at a cascade's edge)
|
||||
let hh = a.model.height * 0.5 * a.scale
|
||||
let r = Math.max(a.radius * a.scale, hh)
|
||||
if not ac_in_light(light_vp, a.x, a.y + hh, a.z, r * 1.5) { continue }
|
||||
if not ac_in_light(render3d_st, light_vp, a.x, a.y + hh, a.z, r * 1.5) { continue }
|
||||
}
|
||||
# Inside the light box is not the same as able to shade anything this cascade covers: every actor
|
||||
# within 300 m sits inside the far cascades' boxes, whose receivers start 250 and 1100 m out - at
|
||||
# the camp 160 actors cast 300 draws into each of those, against 41 into the nearest.
|
||||
let dxa = a.x - cam_pos[0]; let dza = a.z - cam_pos[2]
|
||||
let dxa = a.x - render3d_st.cam_pos[0]; let dza = a.z - render3d_st.cam_pos[2]
|
||||
let da = Math.sqrt(dxa * dxa + dza * dza)
|
||||
let ra = a.radius * a.scale
|
||||
if not cast_band_reaches(Math.max(da - ra, 0.0), da + ra, a.model.height * a.scale) { continue }
|
||||
var ap = ac_sh
|
||||
if a.cutout { ap = ac_sh_cut }
|
||||
if ac_census_at > 0 and ac_frame == ac_census_at { ac_caster_count(a) }
|
||||
actor_draw_one(a, ap, true)
|
||||
if not cast_band_reaches(render3d_st, Math.max(da - ra, 0.0), da + ra, a.model.height * a.scale) { continue }
|
||||
var ap = render3d_st.ac_sh
|
||||
if a.cutout { ap = render3d_st.ac_sh_cut }
|
||||
if render3d_st.ac_census_at > 0 and render3d_st.ac_frame == render3d_st.ac_census_at { ac_caster_count(render3d_st, a) }
|
||||
actor_draw_one(render3d_st, a, ap, true)
|
||||
}
|
||||
if ac_census_at > 0 and ac_frame == ac_census_at {
|
||||
print(`actor census, cascade {sh_cascade}: {ac_cc_actors} actors cast {ac_cc_draws} draws ({ac_cc_skinned} of them skinned, {ac_cc_cut} cutout)`)
|
||||
ac_cc_actors = 0; ac_cc_draws = 0; ac_cc_skinned = 0; ac_cc_cut = 0
|
||||
if render3d_st.ac_census_at > 0 and render3d_st.ac_frame == render3d_st.ac_census_at {
|
||||
print(`actor census, cascade {render3d_st.sh_cascade}: {render3d_st.ac_cc_actors} actors cast {render3d_st.ac_cc_draws} draws ({render3d_st.ac_cc_skinned} of them skinned, {render3d_st.ac_cc_cut} cutout)`)
|
||||
render3d_st.ac_cc_actors = 0; render3d_st.ac_cc_draws = 0; render3d_st.ac_cc_skinned = 0; render3d_st.ac_cc_cut = 0
|
||||
}
|
||||
prof_cpu_mark("shadow actors")
|
||||
prof_cpu_mark(render3d_st, "shadow actors")
|
||||
}
|
||||
var ac_cc_actors: int = 0
|
||||
var ac_cc_draws: int = 0
|
||||
var ac_cc_skinned: int = 0
|
||||
var ac_cc_cut: int = 0
|
||||
function ac_caster_count(a: Actor) -> void {
|
||||
function ac_caster_count(render3d_st: mut Render3dState, a: Actor) -> void {
|
||||
var shown = 0
|
||||
for p in 0 .. len(a.model.prims) { if a.hide == null or a.hide[p] == 0 { shown += 1 } }
|
||||
ac_cc_actors += 1
|
||||
ac_cc_draws += shown
|
||||
if a.skin != null or a.model.skin != null { ac_cc_skinned += shown }
|
||||
if a.cutout { ac_cc_cut += shown }
|
||||
render3d_st.ac_cc_actors += 1
|
||||
render3d_st.ac_cc_draws += shown
|
||||
if a.skin != null or a.model.skin != null { render3d_st.ac_cc_skinned += shown }
|
||||
if a.cutout { render3d_st.ac_cc_cut += shown }
|
||||
}
|
||||
|
||||
# every actor off the stage at once, for a world being replaced. Actors own no GL objects;
|
||||
# their models belong to whoever loaded them.
|
||||
function actor_clear_all() -> void {
|
||||
ac_actors = new []Actor
|
||||
outline_clear()
|
||||
function actor_clear_all(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.ac_actors = new []Actor
|
||||
outline_clear(render3d_st)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,75 +3,58 @@
|
|||
# projection matrices. Float bits throughout; see fmath.ludic.
|
||||
# ============================================================================
|
||||
|
||||
var cam_pos: floats = null # x, y, z
|
||||
var cam_yaw: float = 0.0 # radians, 0 = looking down -z
|
||||
var cam_pitch: float = 0.0
|
||||
var cam_fov: float = 0.0 # vertical, radians
|
||||
var cam_near: float = 0.0
|
||||
var cam_far: float = 0.0
|
||||
var cam_aspect: float = 0.0
|
||||
var cam_view: floats = null
|
||||
var cam_proj: floats = null
|
||||
var cam_vp: floats = null
|
||||
var cam_inv_vp: floats = null
|
||||
var cam_inv_proj: floats = null
|
||||
var cam_vp_clean: floats = null # view-projection, kept for depth reconstruction
|
||||
var cam_inv_vp_clean: floats = null
|
||||
var cam_fwd: floats = null
|
||||
var cam_right: floats = null
|
||||
# the view frustum's four side planes (a, b, c, d), float bits: left, right, bottom, top;
|
||||
# from the clean (unjittered) view-projection, column-major m[col * 4 + row]
|
||||
var cam_planes: floats = null
|
||||
|
||||
function cam_init(aspect: float) -> void {
|
||||
cam_pos = v3_new(0.0, 2.0, 0.0)
|
||||
cam_view = m4_new(); cam_proj = m4_new(); cam_vp = m4_new(); cam_inv_vp = m4_new(); cam_inv_proj = m4_new()
|
||||
cam_vp_clean = m4_new(); cam_inv_vp_clean = m4_new()
|
||||
cam_fwd = v3_new(0.0, 0.0, -1.0)
|
||||
cam_right = v3_new(1.0, 0.0, 0.0)
|
||||
cam_fov = Math.deg_to_rad(42.0)
|
||||
cam_near = 0.3
|
||||
cam_far = 14000.0
|
||||
cam_aspect = aspect
|
||||
cam_update()
|
||||
function cam_init(render3d_st: mut Render3dState, aspect: float) -> void {
|
||||
render3d_st.cam_pos = v3_new(0.0, 2.0, 0.0)
|
||||
render3d_st.cam_view = m4_new(); render3d_st.cam_proj = m4_new(); render3d_st.cam_vp = m4_new(); render3d_st.cam_inv_vp = m4_new(); render3d_st.cam_inv_proj = m4_new()
|
||||
render3d_st.cam_vp_clean = m4_new(); render3d_st.cam_inv_vp_clean = m4_new()
|
||||
render3d_st.cam_fwd = v3_new(0.0, 0.0, -1.0)
|
||||
render3d_st.cam_right = v3_new(1.0, 0.0, 0.0)
|
||||
render3d_st.cam_fov = Math.deg_to_rad(42.0)
|
||||
render3d_st.cam_near = 0.3
|
||||
render3d_st.cam_far = 14000.0
|
||||
render3d_st.cam_aspect = aspect
|
||||
cam_update(render3d_st)
|
||||
}
|
||||
function cam_begin_frame(n: int, w: int, h: int) -> void {
|
||||
gsl_jitter_frame()
|
||||
cam_update()
|
||||
function cam_begin_frame(render3d_st: mut Render3dState, n: int, w: int, h: int) -> void {
|
||||
gsl_jitter_frame(render3d_st)
|
||||
cam_update(render3d_st)
|
||||
}
|
||||
function cam_set(x: float, y: float, z: float, yaw_deg: float, pitch_deg: float) -> void {
|
||||
v3_set(cam_pos, x, y, z)
|
||||
cam_yaw = Math.deg_to_rad(yaw_deg)
|
||||
cam_pitch = Math.deg_to_rad(pitch_deg)
|
||||
cam_update()
|
||||
function cam_set(render3d_st: mut Render3dState, x: float, y: float, z: float, yaw_deg: float, pitch_deg: float) -> void {
|
||||
v3_set(render3d_st.cam_pos, x, y, z)
|
||||
render3d_st.cam_yaw = Math.deg_to_rad(yaw_deg)
|
||||
render3d_st.cam_pitch = Math.deg_to_rad(pitch_deg)
|
||||
cam_update(render3d_st)
|
||||
}
|
||||
function cam_update() -> void {
|
||||
let cy = Math.cos(cam_yaw); let sy = Math.sin(cam_yaw)
|
||||
let cp = Math.cos(cam_pitch); let sp = Math.sin(cam_pitch)
|
||||
v3_set(cam_fwd, -(sy * cp), sp, -(cy * cp))
|
||||
v3_set(cam_right, cy, 0.0, -sy)
|
||||
function cam_update(render3d_st: mut Render3dState) -> void {
|
||||
let cy = Math.cos(render3d_st.cam_yaw); let sy = Math.sin(render3d_st.cam_yaw)
|
||||
let cp = Math.cos(render3d_st.cam_pitch); let sp = Math.sin(render3d_st.cam_pitch)
|
||||
v3_set(render3d_st.cam_fwd, -(sy * cp), sp, -(cy * cp))
|
||||
v3_set(render3d_st.cam_right, cy, 0.0, -sy)
|
||||
let at = floats(3)
|
||||
v3_add(at, cam_pos, cam_fwd)
|
||||
v3_add(at, render3d_st.cam_pos, render3d_st.cam_fwd)
|
||||
let up = v3_new(0.0, 1.0, 0.0)
|
||||
m4_look_at(cam_view, cam_pos, at, up)
|
||||
m4_perspective(cam_proj, cam_fov, cam_aspect, cam_near, cam_far)
|
||||
m4_mul(cam_vp_clean, cam_proj, cam_view)
|
||||
m4_inverse(cam_inv_vp_clean, cam_vp_clean)
|
||||
m4_look_at(render3d_st.cam_view, render3d_st.cam_pos, at, up)
|
||||
m4_perspective(render3d_st.cam_proj, render3d_st.cam_fov, render3d_st.cam_aspect, render3d_st.cam_near, render3d_st.cam_far)
|
||||
m4_mul(render3d_st.cam_vp_clean, render3d_st.cam_proj, render3d_st.cam_view)
|
||||
m4_inverse(render3d_st.cam_inv_vp_clean, render3d_st.cam_vp_clean)
|
||||
# DLSS's sub-pixel jitter (streamline.ludic), in the projection the scene draws with only:
|
||||
# culling, depth reconstruction and last frame's matrix keep the clean one
|
||||
if gsl_jitter_x != 0.0 or gsl_jitter_y != 0.0 {
|
||||
cam_proj[8] = cam_proj[8] - gsl_jitter_x
|
||||
cam_proj[9] = cam_proj[9] - gsl_jitter_y
|
||||
if render3d_st.gsl_jitter_x != 0.0 or render3d_st.gsl_jitter_y != 0.0 {
|
||||
render3d_st.cam_proj[8] = render3d_st.cam_proj[8] - render3d_st.gsl_jitter_x
|
||||
render3d_st.cam_proj[9] = render3d_st.cam_proj[9] - render3d_st.gsl_jitter_y
|
||||
}
|
||||
m4_mul(cam_vp, cam_proj, cam_view)
|
||||
m4_inverse(cam_inv_vp, cam_vp)
|
||||
m4_inverse(cam_inv_proj, cam_proj)
|
||||
m4_mul(render3d_st.cam_vp, render3d_st.cam_proj, render3d_st.cam_view)
|
||||
m4_inverse(render3d_st.cam_inv_vp, render3d_st.cam_vp)
|
||||
m4_inverse(render3d_st.cam_inv_proj, render3d_st.cam_proj)
|
||||
free(at); free(up)
|
||||
cam_planes_update()
|
||||
cam_planes_update(render3d_st)
|
||||
}
|
||||
function cam_planes_update() -> void {
|
||||
if cam_planes == null { cam_planes = floats(16) }
|
||||
let m = cam_vp_clean
|
||||
function cam_planes_update(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.cam_planes == null { render3d_st.cam_planes = floats(16) }
|
||||
let m = render3d_st.cam_vp_clean
|
||||
for p in 0 .. 4 {
|
||||
var r = 0
|
||||
if p >= 2 { r = 1 }
|
||||
|
|
@ -82,27 +65,27 @@ function cam_planes_update() -> void {
|
|||
var c = m[11] + sg * m[8 + r]
|
||||
var d = m[15] + sg * m[12 + r]
|
||||
let inv = 1.0 / Math.sqrt(a * a + b * b + c * c)
|
||||
cam_planes[p * 4] = a * inv; cam_planes[p * 4 + 1] = b * inv
|
||||
cam_planes[p * 4 + 2] = c * inv; cam_planes[p * 4 + 3] = d * inv
|
||||
render3d_st.cam_planes[p * 4] = a * inv; render3d_st.cam_planes[p * 4 + 1] = b * inv
|
||||
render3d_st.cam_planes[p * 4 + 2] = c * inv; render3d_st.cam_planes[p * 4 + 3] = d * inv
|
||||
}
|
||||
}
|
||||
# is a sphere (float bits) at least partly inside the side planes of the view?
|
||||
function cam_sphere_visible(x: float, y: float, z: float, r: float) -> bool {
|
||||
if cam_planes == null { return true }
|
||||
function cam_sphere_visible(render3d_st: Render3dState, x: float, y: float, z: float, r: float) -> bool {
|
||||
if render3d_st.cam_planes == null { return true }
|
||||
let nr = -r
|
||||
for p in 0 .. 4 {
|
||||
let o = p * 4
|
||||
let dist = cam_planes[o] * x + cam_planes[o + 1] * y + cam_planes[o + 2] * z + cam_planes[o + 3]
|
||||
let dist = render3d_st.cam_planes[o] * x + render3d_st.cam_planes[o + 1] * y + render3d_st.cam_planes[o + 2] * z + render3d_st.cam_planes[o + 3]
|
||||
if dist < nr { return false }
|
||||
}
|
||||
return true
|
||||
}
|
||||
# fly: forward/strafe in metres, turn in radians
|
||||
function cam_move(fwd: float, strafe: float, up: float, dyaw: float, dpitch: float) -> void {
|
||||
cam_yaw = cam_yaw + dyaw
|
||||
cam_pitch = Math.clamp(cam_pitch + dpitch, -1.5, 1.5)
|
||||
v3_madd(cam_pos, cam_pos, cam_fwd, fwd)
|
||||
v3_madd(cam_pos, cam_pos, cam_right, strafe)
|
||||
cam_pos[1] = cam_pos[1] + up
|
||||
cam_update()
|
||||
function cam_move(render3d_st: mut Render3dState, fwd: float, strafe: float, up: float, dyaw: float, dpitch: float) -> void {
|
||||
render3d_st.cam_yaw = render3d_st.cam_yaw + dyaw
|
||||
render3d_st.cam_pitch = Math.clamp(render3d_st.cam_pitch + dpitch, -1.5, 1.5)
|
||||
v3_madd(render3d_st.cam_pos, render3d_st.cam_pos, render3d_st.cam_fwd, fwd)
|
||||
v3_madd(render3d_st.cam_pos, render3d_st.cam_pos, render3d_st.cam_right, strafe)
|
||||
render3d_st.cam_pos[1] = render3d_st.cam_pos[1] + up
|
||||
cam_update(render3d_st)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -9,89 +9,78 @@
|
|||
const COL_CELL: int = 16
|
||||
const COL_CAP: int = 120000
|
||||
|
||||
var col_x: floats = null
|
||||
var col_z: floats = null
|
||||
var col_r: floats = null
|
||||
# What a collider occupies VERTICALLY: y0 its base, y1 its top, metres, float bits. A circle used
|
||||
# to be an infinite pillar - you could not climb a boulder, and a knee-high rock stopped you dead,
|
||||
# because there was no height to compare against. col_add keeps that shape (a span from far below
|
||||
# to far above) so every existing caller behaves exactly as it did; col_add_h gives a real one.
|
||||
var col_y0: floats = null
|
||||
var col_y1: floats = null
|
||||
var col_n: int = 0
|
||||
var col_side: int = 0 # cells per side
|
||||
var col_start: words = null # per cell: first index into col_sorted (side*side + 1)
|
||||
var col_sorted: words = null
|
||||
var col_built: bool = false
|
||||
var col_out: floats = null # the resolved position (x, z)
|
||||
|
||||
const COL_LOW: float = -16777216.0 # -16777216.0: below any ground
|
||||
const COL_HIGH: float = 16777216.0 # 16777216.0: above any sky
|
||||
function col_add(x: float, z: float, r: float) -> void { col_add_h(x, z, r, COL_LOW, COL_HIGH) }
|
||||
function col_add(render3d_st: mut Render3dState, x: float, z: float, r: float) -> void { col_add_h(render3d_st, x, z, r, COL_LOW, COL_HIGH) }
|
||||
# a collider that occupies only y0 .. y1: a body above its top walks over it, a body below its base
|
||||
# passes under, and col_top_at reports it as something to stand on
|
||||
function col_add_h(x: float, z: float, r: float, y0: float, y1: float) -> void {
|
||||
if col_x == null {
|
||||
col_x = floats(COL_CAP); col_z = floats(COL_CAP); col_r = floats(COL_CAP)
|
||||
col_y0 = floats(COL_CAP); col_y1 = floats(COL_CAP); col_out = floats(2)
|
||||
function col_add_h(render3d_st: mut Render3dState, x: float, z: float, r: float, y0: float, y1: float) -> void {
|
||||
if render3d_st.col_x == null {
|
||||
render3d_st.col_x = floats(COL_CAP); render3d_st.col_z = floats(COL_CAP); render3d_st.col_r = floats(COL_CAP)
|
||||
render3d_st.col_y0 = floats(COL_CAP); render3d_st.col_y1 = floats(COL_CAP); render3d_st.col_out = floats(2)
|
||||
}
|
||||
if col_n >= COL_CAP { return }
|
||||
col_x[col_n] = x; col_z[col_n] = z; col_r[col_n] = r
|
||||
col_y0[col_n] = y0; col_y1[col_n] = y1
|
||||
col_n += 1
|
||||
col_built = false
|
||||
if render3d_st.col_n >= COL_CAP { return }
|
||||
render3d_st.col_x[render3d_st.col_n] = x; render3d_st.col_z[render3d_st.col_n] = z; render3d_st.col_r[render3d_st.col_n] = r
|
||||
render3d_st.col_y0[render3d_st.col_n] = y0; render3d_st.col_y1[render3d_st.col_n] = y1
|
||||
render3d_st.col_n += 1
|
||||
render3d_st.col_built = false
|
||||
}
|
||||
function col_cell_of(v: float, origin: float) -> int {
|
||||
var c = int(Math.floor((v - origin + float(TERRAIN_HALF)) / float(COL_CELL)))
|
||||
function col_cell_of(render3d_st: Render3dState, v: float, origin: float) -> int {
|
||||
var c = int(Math.floor((v - origin + float(render3d_st.TERRAIN_HALF)) / float(COL_CELL)))
|
||||
if c < 0 { c = 0 }
|
||||
if c > col_side - 1 { c = col_side - 1 }
|
||||
if c > render3d_st.col_side - 1 { c = render3d_st.col_side - 1 }
|
||||
return c
|
||||
}
|
||||
function col_build() -> void {
|
||||
col_side = (TERRAIN_HALF * 2) / COL_CELL
|
||||
let ncell = col_side * col_side
|
||||
if col_start == null { col_start = words(ncell + 1); col_sorted = words(COL_CAP) }
|
||||
for i in 0 .. ncell + 1 { col_start[i] = 0 }
|
||||
for i in 0 .. col_n { col_start[col_cell_of(col_z[i], ter_oz) * col_side + col_cell_of(col_x[i], ter_ox) + 1] += 1 }
|
||||
for c in 0 .. ncell { col_start[c + 1] += col_start[c] }
|
||||
function col_build(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.col_side = (render3d_st.TERRAIN_HALF * 2) / COL_CELL
|
||||
let ncell = render3d_st.col_side * render3d_st.col_side
|
||||
if render3d_st.col_start == null { render3d_st.col_start = words(ncell + 1); render3d_st.col_sorted = words(COL_CAP) }
|
||||
for i in 0 .. ncell + 1 { render3d_st.col_start[i] = 0 }
|
||||
for i in 0 .. render3d_st.col_n { render3d_st.col_start[col_cell_of(render3d_st, render3d_st.col_z[i], render3d_st.ter_oz) * render3d_st.col_side + col_cell_of(render3d_st, render3d_st.col_x[i], render3d_st.ter_ox) + 1] += 1 }
|
||||
for c in 0 .. ncell { render3d_st.col_start[c + 1] += render3d_st.col_start[c] }
|
||||
let fill = words(ncell)
|
||||
for c in 0 .. ncell { fill[c] = col_start[c] }
|
||||
for i in 0 .. col_n {
|
||||
let c = col_cell_of(col_z[i], ter_oz) * col_side + col_cell_of(col_x[i], ter_ox)
|
||||
col_sorted[fill[c]] = i
|
||||
for c in 0 .. ncell { fill[c] = render3d_st.col_start[c] }
|
||||
for i in 0 .. render3d_st.col_n {
|
||||
let c = col_cell_of(render3d_st, render3d_st.col_z[i], render3d_st.ter_oz) * render3d_st.col_side + col_cell_of(render3d_st, render3d_st.col_x[i], render3d_st.ter_ox)
|
||||
render3d_st.col_sorted[fill[c]] = i
|
||||
fill[c] += 1
|
||||
}
|
||||
free(fill)
|
||||
col_built = true
|
||||
print(`colliders: {col_n}`)
|
||||
render3d_st.col_built = true
|
||||
print(`colliders: {render3d_st.col_n}`)
|
||||
}
|
||||
|
||||
# push (px, pz) with radius pr out of every circle it overlaps; the result is in col_out
|
||||
function col_resolve(px: float, pz: float, pr: float) -> bool { return col_resolve_at(px, pz, pr, COL_LOW, COL_HIGH) }
|
||||
function col_resolve(render3d_st: mut Render3dState, px: float, pz: float, pr: float) -> bool { return col_resolve_at(render3d_st, px, pz, pr, COL_LOW, COL_HIGH) }
|
||||
# the same, for a body that occupies feet .. head: a collider whose span misses that is not in the
|
||||
# way at all. This is what lets a hiker stand on top of a boulder rather than inside it.
|
||||
function col_resolve_at(px: float, pz: float, pr: float, feet: float, head: float) -> bool {
|
||||
col_out[0] = px; col_out[1] = pz
|
||||
if not col_built or col_n == 0 { return false }
|
||||
function col_resolve_at(render3d_st: mut Render3dState, px: float, pz: float, pr: float, feet: float, head: float) -> bool {
|
||||
render3d_st.col_out[0] = px; render3d_st.col_out[1] = pz
|
||||
if not render3d_st.col_built or render3d_st.col_n == 0 { return false }
|
||||
var x = px; var z = pz
|
||||
var moved = false
|
||||
let cx = col_cell_of(px, ter_ox); let cz = col_cell_of(pz, ter_oz)
|
||||
let cx = col_cell_of(render3d_st, px, render3d_st.ter_ox); let cz = col_cell_of(render3d_st, pz, render3d_st.ter_oz)
|
||||
for pass in 0 .. 2 {
|
||||
for dz in 0 .. 3 {
|
||||
let zc = cz + dz - 1
|
||||
if zc < 0 or zc >= col_side { continue }
|
||||
if zc < 0 or zc >= render3d_st.col_side { continue }
|
||||
for dx in 0 .. 3 {
|
||||
let xc = cx + dx - 1
|
||||
if xc < 0 or xc >= col_side { continue }
|
||||
let c = zc * col_side + xc
|
||||
for k in col_start[c] .. col_start[c + 1] {
|
||||
let i = col_sorted[k]
|
||||
let ex = x - col_x[i]; let ez = z - col_z[i]
|
||||
if xc < 0 or xc >= render3d_st.col_side { continue }
|
||||
let c = zc * render3d_st.col_side + xc
|
||||
for k in render3d_st.col_start[c] .. render3d_st.col_start[c + 1] {
|
||||
let i = render3d_st.col_sorted[k]
|
||||
let ex = x - render3d_st.col_x[i]; let ez = z - render3d_st.col_z[i]
|
||||
let d2 = ex * ex + ez * ez
|
||||
let rr = col_r[i] + pr
|
||||
let rr = render3d_st.col_r[i] + pr
|
||||
# nothing to push out of if the body is wholly above its top or below its base
|
||||
if not (feet < col_y1[i]) { continue }
|
||||
if not (head > col_y0[i]) { continue }
|
||||
if not (feet < render3d_st.col_y1[i]) { continue }
|
||||
if not (head > render3d_st.col_y0[i]) { continue }
|
||||
if d2 < rr * rr {
|
||||
let d = Math.sqrt(d2)
|
||||
# The UNIT normal out of this circle. (ex, ez) / d is always unit for d > 0,
|
||||
|
|
@ -116,31 +105,31 @@ function col_resolve_at(px: float, pz: float, pr: float, feet: float, head: floa
|
|||
}
|
||||
}
|
||||
}
|
||||
col_out[0] = x; col_out[1] = z
|
||||
render3d_st.col_out[0] = x; render3d_st.col_out[1] = z
|
||||
return moved
|
||||
}
|
||||
# The highest collider top under (px, pz) that a body at `feet` could be standing on or step up to:
|
||||
# tops above `reach` are a wall, not a step. F_ZERO-safe: returns `floor` when there is nothing, so
|
||||
# a caller can pass the terrain height and use the answer directly as the ground.
|
||||
function col_top_at(px: float, pz: float, pr: float, feet: float, reach: float, floor: float) -> float {
|
||||
function col_top_at(render3d_st: Render3dState, px: float, pz: float, pr: float, feet: float, reach: float, floor: float) -> float {
|
||||
var top = floor
|
||||
if not col_built or col_n == 0 { return top }
|
||||
let cx = col_cell_of(px, ter_ox); let cz = col_cell_of(pz, ter_oz)
|
||||
if not render3d_st.col_built or render3d_st.col_n == 0 { return top }
|
||||
let cx = col_cell_of(render3d_st, px, render3d_st.ter_ox); let cz = col_cell_of(render3d_st, pz, render3d_st.ter_oz)
|
||||
let limit = feet + reach
|
||||
for dz in 0 .. 3 {
|
||||
let zc = cz + dz - 1
|
||||
if zc < 0 or zc >= col_side { continue }
|
||||
if zc < 0 or zc >= render3d_st.col_side { continue }
|
||||
for dx in 0 .. 3 {
|
||||
let xc = cx + dx - 1
|
||||
if xc < 0 or xc >= col_side { continue }
|
||||
let c = zc * col_side + xc
|
||||
for k in col_start[c] .. col_start[c + 1] {
|
||||
let i = col_sorted[k]
|
||||
let ex = px - col_x[i]; let ez = pz - col_z[i]
|
||||
if xc < 0 or xc >= render3d_st.col_side { continue }
|
||||
let c = zc * render3d_st.col_side + xc
|
||||
for k in render3d_st.col_start[c] .. render3d_st.col_start[c + 1] {
|
||||
let i = render3d_st.col_sorted[k]
|
||||
let ex = px - render3d_st.col_x[i]; let ez = pz - render3d_st.col_z[i]
|
||||
let d2 = ex * ex + ez * ez
|
||||
let rr = col_r[i] + pr
|
||||
let rr = render3d_st.col_r[i] + pr
|
||||
if not (d2 < rr * rr) { continue }
|
||||
let t = col_y1[i]
|
||||
let t = render3d_st.col_y1[i]
|
||||
if t > limit { continue } # too tall to step onto: it is a wall
|
||||
if t > top { top = t }
|
||||
}
|
||||
|
|
@ -149,21 +138,21 @@ function col_top_at(px: float, pz: float, pr: float, feet: float, reach: float,
|
|||
return top
|
||||
}
|
||||
# is the segment from (x0,z0) to (x1,z1) clear of every circle (a camera line of sight)?
|
||||
function col_clear(x0: float, z0: float, x1: float, z1: float, r: float) -> bool {
|
||||
function col_clear(render3d_st: mut Render3dState, x0: float, z0: float, x1: float, z1: float, r: float) -> bool {
|
||||
let steps = 6
|
||||
for s in 0 .. steps + 1 {
|
||||
let t = float(s) / float(steps)
|
||||
let x = Math.lerp(x0, x1, t); let z = Math.lerp(z0, z1, t)
|
||||
if col_resolve(x, z, r) { return false }
|
||||
if col_resolve(render3d_st, x, z, r) { return false }
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
# No colliders, ready for another map's. The cell index is sized from TERRAIN_HALF on its
|
||||
# first build and kept, so it is released too: a larger map would overrun the old one.
|
||||
function col_reset() -> void {
|
||||
col_n = 0
|
||||
col_built = false
|
||||
if col_start != null { free(col_start); col_start = null }
|
||||
if col_sorted != null { free(col_sorted); col_sorted = null }
|
||||
function col_reset(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.col_n = 0
|
||||
render3d_st.col_built = false
|
||||
if render3d_st.col_start != null { free(render3d_st.col_start); render3d_st.col_start = null }
|
||||
if render3d_st.col_sorted != null { free(render3d_st.col_sorted); render3d_st.col_sorted = null }
|
||||
}
|
||||
|
|
|
|||
|
|
@ -9,65 +9,35 @@
|
|||
# angle they were baked at; the height-field shadow rebakes as the sun moves.
|
||||
# ============================================================================
|
||||
|
||||
var day_on: bool = false
|
||||
var day_hours: float = 0.0 # float bits, 0 .. 24
|
||||
var day_light: float = 0.0 # 0 night .. 1 full day (float bits)
|
||||
var day_ibl: floats = null # rgb scale on the sky's light
|
||||
var day_sun_base: floats = null # the HDRI's sun radiance, kept
|
||||
var day_az0: float = 0.0 # the sun's azimuth at the reference hour (radians, yaw convention)
|
||||
var day_yaw0: float = 0.0 # the sky yaw the scene was tuned at
|
||||
var day_hour0: float = 0.0 # the hour the photograph was taken (10.5)
|
||||
var day_dir: float = 0.0 # +1 / -1: which way the sun travels in yaw
|
||||
var day_az: float = 0.0
|
||||
var day_el: float = 0.0
|
||||
var day_gen: int = 0 # bumps when the light moved enough to rebake the terrain shadow
|
||||
var day_baked_az: float = 0.0
|
||||
var day_baked_el: float = 0.0
|
||||
var day_sky_baked: float = 0.0 # the sky yaw the convolutions were baked at
|
||||
var fire_pos: floats = null
|
||||
var fire_color: floats = null
|
||||
var day_moon: bool = false
|
||||
# The moon's place in its month, 0 new .. 0.5 full .. 1 new again. It decides the disc's
|
||||
# terminator, how much light reaches the ground, and where in the sky it rides: a full
|
||||
# moon is opposite the sun and rises at sunset, a new one travels with it.
|
||||
var day_moon_phase: float = 0.5 # 0.5: full, which is where the game used to be
|
||||
var day_moon_illum: float = 1.0 # the lit fraction, derived from the phase
|
||||
var day_moon_dir: floats = null # toward the moon, whether or not it is up
|
||||
var hand_pos: floats = null
|
||||
var hand_color: floats = null
|
||||
var hand_dir: floats = null
|
||||
var hand_cone: float = 0.0
|
||||
var hand_reach: float = 0.0 # metres the hand light reaches (float bits)
|
||||
var day_overcast: float = 0.0 # 0 clear .. 1 a low grey sky (float bits): dims the sun and the sky's light
|
||||
var day_flash: float = 0.0 # a lightning flash this frame (0..1): the sun brightens for it
|
||||
var day_fog_mul: float = 1.0 # multiplies the base fog density (rain and snow thicken the air)
|
||||
var day_fog_base: float = 0.0
|
||||
|
||||
function daylight_init() -> void {
|
||||
day_ibl = v3_new(1.0, 1.0, 1.0)
|
||||
day_sun_base = v3_new(sun_color[0], sun_color[1], sun_color[2])
|
||||
fire_pos = v3_new(0.0, -1000.0, 0.0)
|
||||
fire_color = v3_new(0.0, 0.0, 0.0)
|
||||
day_moon_dir = v3_new(0.0, 1.0, 0.0)
|
||||
hand_pos = v3_new(0.0, -1000.0, 0.0); hand_color = v3_new(0.0, 0.0, 0.0); hand_dir = v3_new(0.0, 0.0, -1.0); hand_cone = -2.0; hand_reach = 12.0
|
||||
day_light = 1.0
|
||||
function daylight_init(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.day_ibl = v3_new(1.0, 1.0, 1.0)
|
||||
render3d_st.day_sun_base = v3_new(render3d_st.sun_color[0], render3d_st.sun_color[1], render3d_st.sun_color[2])
|
||||
render3d_st.fire_pos = v3_new(0.0, -1000.0, 0.0)
|
||||
render3d_st.fire_color = v3_new(0.0, 0.0, 0.0)
|
||||
render3d_st.day_moon_dir = v3_new(0.0, 1.0, 0.0)
|
||||
render3d_st.hand_pos = v3_new(0.0, -1000.0, 0.0); render3d_st.hand_color = v3_new(0.0, 0.0, 0.0); render3d_st.hand_dir = v3_new(0.0, 0.0, -1.0); render3d_st.hand_cone = -2.0; render3d_st.hand_reach = 12.0
|
||||
render3d_st.day_light = 1.0
|
||||
}
|
||||
|
||||
# Start the clock: the scene as tuned (sky yaw `yaw0`) is the photograph's hour `hour0`;
|
||||
# the sun rises and sets toward `set_yaw` (the direction it should be in at 19:00).
|
||||
function daylight_start(yaw0: float, hour0: float, set_yaw: float) -> void {
|
||||
day_yaw0 = yaw0; day_hour0 = hour0
|
||||
day_az0 = Math.atan2(-sun_dir[0], -sun_dir[2])
|
||||
day_sky_baked = yaw0
|
||||
function daylight_start(render3d_st: mut Render3dState, yaw0: float, hour0: float, set_yaw: float) -> void {
|
||||
render3d_st.day_yaw0 = yaw0; render3d_st.day_hour0 = hour0
|
||||
render3d_st.day_az0 = Math.atan2(-render3d_st.sun_dir[0], -render3d_st.sun_dir[2])
|
||||
render3d_st.day_sky_baked = yaw0
|
||||
# which way does the sun travel? the way that puts it nearest `set_yaw` at 19:00
|
||||
let step = (19.0 - hour0) * Math.deg_to_rad(15.0)
|
||||
let da = Math.abs(day_wrap(day_az0 + step - set_yaw))
|
||||
let db = Math.abs(day_wrap(day_az0 - step - set_yaw))
|
||||
day_dir = 1.0
|
||||
if db < da { day_dir = -1.0 }
|
||||
day_on = true
|
||||
day_baked_az = 1000.0
|
||||
daylight_set(hour0)
|
||||
let da = Math.abs(day_wrap(render3d_st.day_az0 + step - set_yaw))
|
||||
let db = Math.abs(day_wrap(render3d_st.day_az0 - step - set_yaw))
|
||||
render3d_st.day_dir = 1.0
|
||||
if db < da { render3d_st.day_dir = -1.0 }
|
||||
render3d_st.day_on = true
|
||||
render3d_st.day_baked_az = 1000.0
|
||||
daylight_set(render3d_st, hour0)
|
||||
}
|
||||
function day_wrap(a: float) -> float {
|
||||
let two_pi = 2.0 * PI
|
||||
|
|
@ -81,68 +51,68 @@ function smoothf(a: float, b: float, x: float) -> float {
|
|||
return t * t * (3.0 - 2.0 * t)
|
||||
}
|
||||
|
||||
function daylight_set(hours: float) -> void {
|
||||
function daylight_set(render3d_st: mut Render3dState, hours: float) -> void {
|
||||
var h = hours % 24.0
|
||||
if h < 0.0 { h = h + 24.0 }
|
||||
day_hours = h
|
||||
render3d_st.day_hours = h
|
||||
# the arc: up at 5:30, highest (about 57 degrees) at 12:45, down at 20:00
|
||||
let t = (h - 5.5) / 14.5 * PI
|
||||
let el = Math.deg_to_rad(-6.0 + 63.0 * Math.sin(t))
|
||||
let az = day_az0 + (h - day_hour0) * Math.deg_to_rad(15.0) * day_dir
|
||||
day_az = az; day_el = el
|
||||
let az = render3d_st.day_az0 + (h - render3d_st.day_hour0) * Math.deg_to_rad(15.0) * render3d_st.day_dir
|
||||
render3d_st.day_az = az; render3d_st.day_el = el
|
||||
let d = smoothf(Math.deg_to_rad(-8.0), Math.deg_to_rad(12.0), el)
|
||||
day_light = d
|
||||
render3d_st.day_light = d
|
||||
# the sun, warm and dim near the horizon; past dusk, the moon from across the sky
|
||||
var laz = az; var lel = Math.max(el, Math.deg_to_rad(3.0))
|
||||
var warm_r = 1.0; var warm_g = 1.0; var warm_b = 1.0
|
||||
let low = smoothf(0.0, Math.deg_to_rad(24.0), el)
|
||||
warm_g = Math.lerp(0.55, 1.0, low); warm_b = Math.lerp(0.28, 1.0, low)
|
||||
# an overcast sky: the sun goes diffuse and grey, a lightning flash brings it back white
|
||||
let oc = Math.clamp(day_overcast, 0.0, 1.0)
|
||||
let sunk = 1.0 - 0.92 * oc + 2.5 * day_flash
|
||||
var sr = day_sun_base[0] * (d * warm_r * sunk)
|
||||
var sg = day_sun_base[1] * (d * warm_g * sunk)
|
||||
var sb = day_sun_base[2] * (d * warm_b * sunk)
|
||||
let oc = Math.clamp(render3d_st.day_overcast, 0.0, 1.0)
|
||||
let sunk = 1.0 - 0.92 * oc + 2.5 * render3d_st.day_flash
|
||||
var sr = render3d_st.day_sun_base[0] * (d * warm_r * sunk)
|
||||
var sg = render3d_st.day_sun_base[1] * (d * warm_g * sunk)
|
||||
var sb = render3d_st.day_sun_base[2] * (d * warm_b * sunk)
|
||||
# The moon rides a lag behind the sun that is its phase: full is opposite (half a turn),
|
||||
# new is alongside. Its elevation follows the same arc, offset by the same amount, so a
|
||||
# full moon rises as the sun sets and a new moon is up all day and invisible.
|
||||
# The moon is where the sun was `lag` of a day ago: at full that is half a day, so it
|
||||
# rises as the sun sets; at new it is alongside the sun and up all day, invisible. The
|
||||
# sign matters — a waxing crescent has to set AFTER the sun, not before it.
|
||||
let moon_lag = 2.0 * PI * day_moon_phase
|
||||
let maz = az - moon_lag * day_dir
|
||||
let moon_lag = 2.0 * PI * render3d_st.day_moon_phase
|
||||
let maz = az - moon_lag * render3d_st.day_dir
|
||||
let mel = Math.deg_to_rad(-6.0 + 63.0 * Math.sin(t - moon_lag))
|
||||
let mce = Math.cos(mel)
|
||||
v3_set(day_moon_dir, -(Math.sin(maz) * mce), Math.sin(mel), -(Math.cos(maz) * mce))
|
||||
day_moon = false
|
||||
v3_set(render3d_st.day_moon_dir, -(Math.sin(maz) * mce), Math.sin(mel), -(Math.cos(maz) * mce))
|
||||
render3d_st.day_moon = false
|
||||
if el < Math.deg_to_rad(-7.0) {
|
||||
day_moon = true
|
||||
render3d_st.day_moon = true
|
||||
laz = maz
|
||||
lel = Math.clamp(mel, Math.deg_to_rad(6.0), Math.deg_to_rad(70.0))
|
||||
let m = smoothf(Math.deg_to_rad(-7.0), Math.deg_to_rad(-16.0), el)
|
||||
# what the moon is worth on the ground, by how much of it is lit. A new moon is a
|
||||
# properly dark night, which is what makes a torch and a lantern matter.
|
||||
let up = smoothf(Math.deg_to_rad(-4.0), Math.deg_to_rad(8.0), mel)
|
||||
let lit = m * up * (0.06 + 0.94 * day_moon_illum)
|
||||
sr = day_sun_base[0] * (0.0130 * lit)
|
||||
sg = day_sun_base[1] * (0.0165 * lit)
|
||||
sb = day_sun_base[2] * (0.0250 * lit)
|
||||
let lit = m * up * (0.06 + 0.94 * render3d_st.day_moon_illum)
|
||||
sr = render3d_st.day_sun_base[0] * (0.0130 * lit)
|
||||
sg = render3d_st.day_sun_base[1] * (0.0165 * lit)
|
||||
sb = render3d_st.day_sun_base[2] * (0.0250 * lit)
|
||||
}
|
||||
let ce = Math.cos(lel)
|
||||
v3_set(sun_dir, -(Math.sin(laz) * ce), Math.sin(lel), -(Math.cos(laz) * ce))
|
||||
v3_set(sun_color, sr, sg, sb)
|
||||
v3_set(render3d_st.sun_dir, -(Math.sin(laz) * ce), Math.sin(lel), -(Math.cos(laz) * ce))
|
||||
v3_set(render3d_st.sun_color, sr, sg, sb)
|
||||
# the sky's light: full by day, a deep blue by night, amber through the dusk
|
||||
let dusk = smoothf(Math.deg_to_rad(-10.0), Math.deg_to_rad(2.0), el) * (1.0 - smoothf(Math.deg_to_rad(2.0), Math.deg_to_rad(18.0), el))
|
||||
# a full moon lifts the night's own ambient nearly threefold; a new moon leaves it alone
|
||||
var moonlit = 1.0
|
||||
if day_moon { moonlit = 1.0 + 1.8 * day_moon_illum }
|
||||
v3_set(day_ibl, Math.lerp(0.020 * moonlit, 1.0, d), Math.lerp(0.026 * moonlit, 1.0, d), Math.lerp(0.045 * moonlit, 1.0, d))
|
||||
day_ibl[0] = day_ibl[0] * (1.0 + 0.35 * dusk)
|
||||
day_ibl[2] = day_ibl[2] * (1.0 - 0.25 * dusk)
|
||||
if render3d_st.day_moon { moonlit = 1.0 + 1.8 * render3d_st.day_moon_illum }
|
||||
v3_set(render3d_st.day_ibl, Math.lerp(0.020 * moonlit, 1.0, d), Math.lerp(0.026 * moonlit, 1.0, d), Math.lerp(0.045 * moonlit, 1.0, d))
|
||||
render3d_st.day_ibl[0] = render3d_st.day_ibl[0] * (1.0 + 0.35 * dusk)
|
||||
render3d_st.day_ibl[2] = render3d_st.day_ibl[2] * (1.0 - 0.25 * dusk)
|
||||
# clouds: less light, and greyer (the blue and the warmth both fade)
|
||||
let grey = (day_ibl[0] + (day_ibl[1] + day_ibl[2])) * 0.3333 + 0.6 * day_flash
|
||||
let grey = (render3d_st.day_ibl[0] + (render3d_st.day_ibl[1] + render3d_st.day_ibl[2])) * 0.3333 + 0.6 * render3d_st.day_flash
|
||||
let dim = 1.0 - 0.55 * oc
|
||||
for i in 0 .. 3 { day_ibl[i] = Math.lerp(day_ibl[i], grey, 0.7 * oc) * dim }
|
||||
for i in 0 .. 3 { render3d_st.day_ibl[i] = Math.lerp(render3d_st.day_ibl[i], grey, 0.7 * oc) * dim }
|
||||
# ---- the air ---------------------------------------------------------------------------
|
||||
# Aerial perspective is a CURVE, not the one constant it was. A low sun is shining through
|
||||
# far more air than a high one, and it is shining ALONG the ground rather than down onto it,
|
||||
|
|
@ -152,25 +122,25 @@ function daylight_set(hours: float) -> void {
|
|||
# gets a dawn's haze with no dawn to justify it. Overcast thickens the air and flattens the
|
||||
# scatter, because a grey sky has no disc to scatter from. The player's slider still
|
||||
# multiplies the result, so nobody loses the setting they chose.
|
||||
if day_fog_base == 0.0 { day_fog_base = r3d_fog_density }
|
||||
if render3d_st.day_fog_base == 0.0 { render3d_st.day_fog_base = render3d_st.r3d_fog_density }
|
||||
let fog_rise = smoothf(Math.deg_to_rad(-12.0), Math.deg_to_rad(1.0), el)
|
||||
let fog_high = smoothf(Math.deg_to_rad(2.0), Math.deg_to_rad(26.0), el)
|
||||
let lowsun = fog_rise * (1.0 - fog_high)
|
||||
var dens = day_fog_base * (1.0 + 1.5 * lowsun)
|
||||
var dens = render3d_st.day_fog_base * (1.0 + 1.5 * lowsun)
|
||||
dens = dens * (1.0 + 1.2 * oc)
|
||||
r3d_fog_density = dens * day_fog_mul * r3d_fog_scale
|
||||
render3d_st.r3d_fog_density = dens * render3d_st.day_fog_mul * render3d_st.r3d_fog_scale
|
||||
# how fast the haze thins with height: a settled morning's lies IN the valley, a noon sky is
|
||||
# thin all the way up, so the falloff rises as the sun drops
|
||||
r3d_fog_falloff = 0.002 * (1.0 + 1.6 * lowsun)
|
||||
render3d_st.r3d_fog_falloff = 0.002 * (1.0 + 1.6 * lowsun)
|
||||
# the glow a ridge is silhouetted against when you look into a low sun
|
||||
r3d_fog_inscatter = (0.012 + 0.16 * lowsun) * (1.0 - 0.7 * oc)
|
||||
render3d_st.r3d_fog_inscatter = (0.012 + 0.16 * lowsun) * (1.0 - 0.7 * oc)
|
||||
# and distance takes colour away at every hour, harder under cloud
|
||||
# 1.9 was tuned on a valley whose rock was grey anyway. Maroon's air is thin, dry and at
|
||||
# 2900 m, and the Bells are only a couple of kilometres from the lake - in the photograph
|
||||
# everybody knows, they are still plainly RED at that distance. Desaturating them to a pale
|
||||
# grey-pink is physically defensible and loses the one thing the range is named for, so the
|
||||
# basin's own air gets a gentler figure and overcast still takes colour away faster.
|
||||
r3d_fog_desat = 1.15 + 0.7 * oc
|
||||
render3d_st.r3d_fog_desat = 1.15 + 0.7 * oc
|
||||
# ---- the grade -------------------------------------------------------------------------
|
||||
# The look of an hour is not only how much air is in front of the mountain; it is what colour
|
||||
# the light is and what the shadows are filled with. Noon is the case worth naming: direct
|
||||
|
|
@ -183,29 +153,29 @@ function daylight_set(hours: float) -> void {
|
|||
# half-cooled. It only begins under the horizon.
|
||||
let night = 1.0 - smoothf(Math.deg_to_rad(-14.0), Math.deg_to_rad(-4.0), el)
|
||||
let hi = fog_high
|
||||
post_wb_r = 1.02 + (0.10 * lowsun - (0.03 * hi + 0.06 * night))
|
||||
post_wb_g = 1.0 + 0.012 * lowsun
|
||||
post_wb_b = 0.97 + (0.055 * hi + 0.13 * night - 0.09 * lowsun)
|
||||
render3d_st.post_wb_r = 1.02 + (0.10 * lowsun - (0.03 * hi + 0.06 * night))
|
||||
render3d_st.post_wb_g = 1.0 + 0.012 * lowsun
|
||||
render3d_st.post_wb_b = 0.97 + (0.055 * hi + 0.13 * night - 0.09 * lowsun)
|
||||
# the shadows' floor: warm and open into a low sun, blue at noon, cold and crushed at night
|
||||
post_lift_r = 0.004 + 0.013 * lowsun - 0.002 * (hi + night)
|
||||
post_lift_g = 0.004 + 0.008 * lowsun - 0.001 * night
|
||||
post_lift_b = 0.012 + (0.004 * lowsun + 0.013 * hi)
|
||||
render3d_st.post_lift_r = 0.004 + 0.013 * lowsun - 0.002 * (hi + night)
|
||||
render3d_st.post_lift_g = 0.004 + 0.008 * lowsun - 0.001 * night
|
||||
render3d_st.post_lift_b = 0.012 + (0.004 * lowsun + 0.013 * hi)
|
||||
# Gain is left alone on purpose. It looks like a highlight control and is not: the grade is
|
||||
# `c * gain + lift * (1 - c)`, so warming the gain warms the WHOLE frame, and warming it at
|
||||
# noon undid the cool white balance above and turned one o'clock yellower than seven in the
|
||||
# morning - the opposite of the thing this grade exists to do. Midday's punch comes from
|
||||
# contrast, and midday's colour from a cool balance over a blue shadow.
|
||||
post_gain_r = 0.99
|
||||
post_gain_g = 0.995
|
||||
post_gain_b = 1.0
|
||||
post_contrast = 1.12 + 0.11 * hi - (0.17 * oc + 0.06 * night)
|
||||
post_saturation = 1.04 + 0.12 * lowsun - (0.20 * oc + 0.12 * night)
|
||||
render3d_st.post_gain_r = 0.99
|
||||
render3d_st.post_gain_g = 0.995
|
||||
render3d_st.post_gain_b = 1.0
|
||||
render3d_st.post_contrast = 1.12 + 0.11 * hi - (0.17 * oc + 0.06 * night)
|
||||
render3d_st.post_saturation = 1.04 + 0.12 * lowsun - (0.20 * oc + 0.12 * night)
|
||||
|
||||
# The visible sky is relit by the hour (sky.frag). It fades out under the horizon, where the
|
||||
# night's own tint, stars and moon take over, and it eases off under heavy cloud - a relit
|
||||
# overcast is still overcast, and driving a clear-sky model hard through one paints a blue
|
||||
# zenith onto a grey day.
|
||||
r3d_sky_relight = smoothf(Math.deg_to_rad(-10.0), Math.deg_to_rad(1.0), el) * (1.0 - 0.75 * oc)
|
||||
render3d_st.r3d_sky_relight = smoothf(Math.deg_to_rad(-10.0), Math.deg_to_rad(1.0), el) * (1.0 - 0.75 * oc)
|
||||
|
||||
# ---- what is IN the air -------------------------------------------------------------
|
||||
# The volumetric march (post.ludic) needs the same story the analytic fog tells, or the
|
||||
|
|
@ -214,64 +184,64 @@ function daylight_set(hours: float) -> void {
|
|||
# the cold at either end of the day, lies ON the ground rather than filling the basin, and
|
||||
# burns off by mid-morning. Cloud thickens the air and kills the shafts, because a shaft
|
||||
# needs a disc to come from.
|
||||
post_vol_density = (0.00030 + 0.00070 * lowsun) * (1.0 - 0.55 * oc)
|
||||
post_vol_mist = 0.40 * lowsun * (1.0 - 0.4 * oc)
|
||||
post_vol_falloff = 0.006 + 0.004 * lowsun
|
||||
render3d_st.post_vol_density = (0.00030 + 0.00070 * lowsun) * (1.0 - 0.55 * oc)
|
||||
render3d_st.post_vol_mist = 0.40 * lowsun * (1.0 - 0.4 * oc)
|
||||
render3d_st.post_vol_falloff = 0.006 + 0.004 * lowsun
|
||||
|
||||
# R3D_NOAIR=1: the air as it was before any of the above - one density, one falloff, the
|
||||
# inscatter the shader used to hard-code, and no distance desaturation at all. It is here
|
||||
# so a before-and-after can be shot from ONE binary at one hour, which is the only kind of
|
||||
# comparison worth looking at, and it joins R3D_NOCLOUD / R3D_NOSHADOW / R3D_NOGI.
|
||||
if r3d_env_has("R3D_NOAIR") {
|
||||
r3d_fog_density = day_fog_base * day_fog_mul * r3d_fog_scale
|
||||
r3d_fog_falloff = 0.002
|
||||
r3d_fog_inscatter = 0.02
|
||||
r3d_fog_desat = 0.0
|
||||
post_wb_r = 1.02; post_wb_g = 1.0; post_wb_b = 0.97
|
||||
post_lift_r = 0.004; post_lift_g = 0.004; post_lift_b = 0.012
|
||||
post_gain_r = 0.99; post_gain_g = 0.995; post_gain_b = 1.0
|
||||
post_contrast = 1.12; post_saturation = 1.04
|
||||
r3d_sky_relight = 0.0
|
||||
post_vol_density = 0.0; post_vol_mist = 0.0
|
||||
if r3d_env_has(render3d_st, "R3D_NOAIR") {
|
||||
render3d_st.r3d_fog_density = render3d_st.day_fog_base * render3d_st.day_fog_mul * render3d_st.r3d_fog_scale
|
||||
render3d_st.r3d_fog_falloff = 0.002
|
||||
render3d_st.r3d_fog_inscatter = 0.02
|
||||
render3d_st.r3d_fog_desat = 0.0
|
||||
render3d_st.post_wb_r = 1.02; render3d_st.post_wb_g = 1.0; render3d_st.post_wb_b = 0.97
|
||||
render3d_st.post_lift_r = 0.004; render3d_st.post_lift_g = 0.004; render3d_st.post_lift_b = 0.012
|
||||
render3d_st.post_gain_r = 0.99; render3d_st.post_gain_g = 0.995; render3d_st.post_gain_b = 1.0
|
||||
render3d_st.post_contrast = 1.12; render3d_st.post_saturation = 1.04
|
||||
render3d_st.r3d_sky_relight = 0.0
|
||||
render3d_st.post_vol_density = 0.0; render3d_st.post_vol_mist = 0.0
|
||||
}
|
||||
# exposure: auto-exposure must not turn the night into day
|
||||
# the ceiling has to move with the moon or auto-exposure eats the difference between a
|
||||
# full-moon night and a new-moon one
|
||||
var night_max = 4.5
|
||||
if day_moon { night_max = 4.5 + 3.5 * day_moon_illum }
|
||||
post_exposure_max = Math.lerp(night_max, 20.0, d)
|
||||
if render3d_st.day_moon { night_max = 4.5 + 3.5 * render3d_st.day_moon_illum }
|
||||
render3d_st.post_exposure_max = Math.lerp(night_max, 20.0, d)
|
||||
# the visible sky turns with the sun (cheap); its convolutions rebake when far off
|
||||
let sky_yaw_now = day_yaw0 + (h - day_hour0) * Math.deg_to_rad(15.0) * day_dir
|
||||
sky_set_rot(sky_yaw_now)
|
||||
if Math.abs(day_wrap(sky_yaw_now - day_sky_baked)) > Math.deg_to_rad(35.0) and d > 0.05 {
|
||||
day_sky_baked = sky_yaw_now
|
||||
sky_precompute()
|
||||
let sky_yaw_now = render3d_st.day_yaw0 + (h - render3d_st.day_hour0) * Math.deg_to_rad(15.0) * render3d_st.day_dir
|
||||
sky_set_rot(render3d_st, sky_yaw_now)
|
||||
if Math.abs(day_wrap(sky_yaw_now - render3d_st.day_sky_baked)) > Math.deg_to_rad(35.0) and d > 0.05 {
|
||||
render3d_st.day_sky_baked = sky_yaw_now
|
||||
sky_precompute(render3d_st)
|
||||
}
|
||||
# the terrain's baked shadow follows the light in steps
|
||||
if Math.abs(day_wrap(laz - day_baked_az)) > Math.deg_to_rad(4.0) or Math.abs(lel - day_baked_el) > Math.deg_to_rad(3.0) {
|
||||
day_baked_az = laz; day_baked_el = lel
|
||||
day_gen += 1
|
||||
if Math.abs(day_wrap(laz - render3d_st.day_baked_az)) > Math.deg_to_rad(4.0) or Math.abs(lel - render3d_st.day_baked_el) > Math.deg_to_rad(3.0) {
|
||||
render3d_st.day_baked_az = laz; render3d_st.day_baked_el = lel
|
||||
render3d_st.day_gen += 1
|
||||
}
|
||||
}
|
||||
|
||||
# where the moon is in its month; the game advances this each morning
|
||||
function daylight_moon(phase: float) -> void {
|
||||
function daylight_moon(render3d_st: mut Render3dState, phase: float) -> void {
|
||||
var p = phase % 1.0
|
||||
if p < 0.0 { p = p + 1.0 }
|
||||
day_moon_phase = p
|
||||
render3d_st.day_moon_phase = p
|
||||
# illuminated fraction: (1 - cos(2 pi p)) / 2, which is 0 at new and 1 at full
|
||||
day_moon_illum = (1.0 - Math.cos(2.0 * PI * p)) * 0.5
|
||||
daylight_set(day_hours)
|
||||
render3d_st.day_moon_illum = (1.0 - Math.cos(2.0 * PI * p)) * 0.5
|
||||
daylight_set(render3d_st, render3d_st.day_hours)
|
||||
}
|
||||
|
||||
# the weather over the valley: overcast 0..1, a fog multiplier, a lightning flash 0..1
|
||||
function daylight_weather(overcast: float, fog_mul: float, flash: float) -> void {
|
||||
day_overcast = overcast; day_fog_mul = fog_mul; day_flash = flash
|
||||
function daylight_weather(render3d_st: mut Render3dState, overcast: float, fog_mul: float, flash: float) -> void {
|
||||
render3d_st.day_overcast = overcast; render3d_st.day_fog_mul = fog_mul; render3d_st.day_flash = flash
|
||||
}
|
||||
# the campfire: a point light at (x, y, z) of `strength` (0 = out)
|
||||
function daylight_fire(x: float, y: float, z: float, strength: float) -> void {
|
||||
v3_set(fire_pos, x, y, z)
|
||||
v3_set(fire_color, 9.0 * strength, 4.6 * strength, 1.4 * strength)
|
||||
function daylight_fire(render3d_st: Render3dState, x: float, y: float, z: float, strength: float) -> void {
|
||||
v3_set(render3d_st.fire_pos, x, y, z)
|
||||
v3_set(render3d_st.fire_color, 9.0 * strength, 4.6 * strength, 1.4 * strength)
|
||||
}
|
||||
|
||||
# the light in the hand: a point light (cone < -1) or a cone along dir (cone = cos half-angle)
|
||||
|
|
@ -280,23 +250,23 @@ function daylight_fire(x: float, y: float, z: float, strength: float) -> void {
|
|||
# is what a pool of firelight looks like. It used to be a windowed inverse square with the
|
||||
# window and the scale both hard-coded, so every hand light in every game had the reach of
|
||||
# a candle whatever it was meant to be.
|
||||
function daylight_hand(x: float, y: float, z: float, dx: float, dy: float, dz: float, cone: float, reach: float, r: float, g: float, b: float) -> void {
|
||||
v3_set(hand_pos, x, y, z); v3_set(hand_dir, dx, dy, dz); hand_cone = cone
|
||||
v3_set(hand_color, r, g, b)
|
||||
hand_reach = reach
|
||||
function daylight_hand(render3d_st: mut Render3dState, x: float, y: float, z: float, dx: float, dy: float, dz: float, cone: float, reach: float, r: float, g: float, b: float) -> void {
|
||||
v3_set(render3d_st.hand_pos, x, y, z); v3_set(render3d_st.hand_dir, dx, dy, dz); render3d_st.hand_cone = cone
|
||||
v3_set(render3d_st.hand_color, r, g, b)
|
||||
render3d_st.hand_reach = reach
|
||||
}
|
||||
function daylight_bind(prog: int) -> void {
|
||||
u_v3(gpu_uniform(prog, "u_hand_pos"), hand_pos)
|
||||
u_v3(gpu_uniform(prog, "u_hand_color"), hand_color)
|
||||
u_v3(gpu_uniform(prog, "u_hand_dir"), hand_dir)
|
||||
u_f(gpu_uniform(prog, "u_hand_cone"), hand_cone)
|
||||
u_f(gpu_uniform(prog, "u_hand_reach"), hand_reach)
|
||||
u_v3(gpu_uniform(prog, "u_ibl_scale"), day_ibl)
|
||||
u_v3(gpu_uniform(prog, "u_moon_dir"), day_moon_dir)
|
||||
u_f(gpu_uniform(prog, "u_moon_phase"), day_moon_phase)
|
||||
u_f(gpu_uniform(prog, "u_moon_illum"), day_moon_illum)
|
||||
u_f(gpu_uniform(prog, "u_moon_haze"), day_overcast)
|
||||
u_f(gpu_uniform(prog, "u_daylight"), day_light)
|
||||
u_v3(gpu_uniform(prog, "u_fire_pos"), fire_pos)
|
||||
u_v3(gpu_uniform(prog, "u_fire_color"), fire_color)
|
||||
function daylight_bind(render3d_st: mut Render3dState, prog: int) -> void {
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_hand_pos"), render3d_st.hand_pos)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_hand_color"), render3d_st.hand_color)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_hand_dir"), render3d_st.hand_dir)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_hand_cone"), render3d_st.hand_cone)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_hand_reach"), render3d_st.hand_reach)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_ibl_scale"), render3d_st.day_ibl)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_moon_dir"), render3d_st.day_moon_dir)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_moon_phase"), render3d_st.day_moon_phase)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_moon_illum"), render3d_st.day_moon_illum)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_moon_haze"), render3d_st.day_overcast)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_daylight"), render3d_st.day_light)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_fire_pos"), render3d_st.fire_pos)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_fire_color"), render3d_st.fire_color)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -26,103 +26,86 @@ const DS_RANGE: int = 3
|
|||
const DS_INDIRECT: int = 4
|
||||
const DS_KINDS: int = 5
|
||||
|
||||
var ds_on: int = -1 # -1: not read yet
|
||||
var ds_counting: bool = false
|
||||
var ds_from: int = 90
|
||||
var ds_frames: int = 60
|
||||
var ds_frame_no: int = 0
|
||||
var ds_outside: pointer = "(outside passes)"
|
||||
|
||||
# passes, in the order the frame first opens them
|
||||
var ds_pass_names: []pointer = null
|
||||
var ds_pass_prog: []long = null # program changes inside the pass
|
||||
var ds_pass_tex: []long = null # texture binds inside the pass
|
||||
var ds_pass_i: int = 0
|
||||
|
||||
# rows: (pass, program, kind)
|
||||
var ds_row_pass: []int = null
|
||||
var ds_row_prog: []int = null
|
||||
var ds_row_kind: []int = null
|
||||
var ds_row_draws: []long = null
|
||||
var ds_row_inst: []long = null
|
||||
var ds_row_idx: []long = null
|
||||
var ds_last: int = -1
|
||||
|
||||
function ds_enabled() -> bool {
|
||||
if ds_on < 0 {
|
||||
ds_on = 0
|
||||
if r3d_env_has("R3D_DRAWSTATS") {
|
||||
ds_on = 1
|
||||
ds_frames = Text.to_int(r3d_env("R3D_DRAWSTATS"))
|
||||
if ds_frames < 2 { ds_frames = 60 }
|
||||
if r3d_env_has("R3D_DRAWSTATS_FROM") { ds_from = Text.to_int(r3d_env("R3D_DRAWSTATS_FROM")) }
|
||||
ds_pass_names = new []pointer; ds_pass_prog = new []long; ds_pass_tex = new []long
|
||||
ds_row_pass = new []int; ds_row_prog = new []int; ds_row_kind = new []int
|
||||
ds_row_draws = new []long; ds_row_inst = new []long; ds_row_idx = new []long
|
||||
ds_set_pass(ds_outside)
|
||||
function ds_enabled(render3d_st: mut Render3dState) -> bool {
|
||||
if render3d_st.ds_on < 0 {
|
||||
render3d_st.ds_on = 0
|
||||
if r3d_env_has(render3d_st, "R3D_DRAWSTATS") {
|
||||
render3d_st.ds_on = 1
|
||||
render3d_st.ds_frames = Text.to_int(r3d_env(render3d_st, "R3D_DRAWSTATS"))
|
||||
if render3d_st.ds_frames < 2 { render3d_st.ds_frames = 60 }
|
||||
if r3d_env_has(render3d_st, "R3D_DRAWSTATS_FROM") { render3d_st.ds_from = Text.to_int(r3d_env(render3d_st, "R3D_DRAWSTATS_FROM")) }
|
||||
render3d_st.ds_pass_names = new []pointer; render3d_st.ds_pass_prog = new []long; render3d_st.ds_pass_tex = new []long
|
||||
render3d_st.ds_row_pass = new []int; render3d_st.ds_row_prog = new []int; render3d_st.ds_row_kind = new []int
|
||||
render3d_st.ds_row_draws = new []long; render3d_st.ds_row_inst = new []long; render3d_st.ds_row_idx = new []long
|
||||
ds_set_pass(render3d_st, render3d_st.ds_outside)
|
||||
}
|
||||
}
|
||||
return ds_on == 1
|
||||
return render3d_st.ds_on == 1
|
||||
}
|
||||
|
||||
# the profiler's pass boundaries (prof_begin / prof_end), whether or not R3D_PROF is on
|
||||
function ds_set_pass(name: pointer) -> void {
|
||||
if ds_on != 1 { return }
|
||||
ds_last = -1
|
||||
function ds_set_pass(render3d_st: mut Render3dState, name: pointer) -> void {
|
||||
if render3d_st.ds_on != 1 { return }
|
||||
render3d_st.ds_last = -1
|
||||
var i = 0
|
||||
while i < len(ds_pass_names) {
|
||||
if ds_pass_names[i] == name { ds_pass_i = i; return }
|
||||
while i < len(render3d_st.ds_pass_names) {
|
||||
if render3d_st.ds_pass_names[i] == name { render3d_st.ds_pass_i = i; return }
|
||||
i += 1
|
||||
}
|
||||
push(ds_pass_names, name); push(ds_pass_prog, 0); push(ds_pass_tex, 0)
|
||||
ds_pass_i = len(ds_pass_names) - 1
|
||||
push(render3d_st.ds_pass_names, name); push(render3d_st.ds_pass_prog, 0); push(render3d_st.ds_pass_tex, 0)
|
||||
render3d_st.ds_pass_i = len(render3d_st.ds_pass_names) - 1
|
||||
}
|
||||
|
||||
# once per render frame, before any of its work (prof_gen_frame)
|
||||
function ds_frame() -> void {
|
||||
if not ds_enabled() { return }
|
||||
ds_frame_no += 1
|
||||
ds_set_pass(ds_outside)
|
||||
if ds_frame_no == ds_from {
|
||||
ds_counting = true
|
||||
print(`r3d: drawstats counting frames {ds_from}..{ds_from + ds_frames - 1}`)
|
||||
function ds_frame(render3d_st: mut Render3dState) -> void {
|
||||
if not ds_enabled(render3d_st) { return }
|
||||
render3d_st.ds_frame_no += 1
|
||||
ds_set_pass(render3d_st, render3d_st.ds_outside)
|
||||
if render3d_st.ds_frame_no == render3d_st.ds_from {
|
||||
render3d_st.ds_counting = true
|
||||
print(`r3d: drawstats counting frames {render3d_st.ds_from}..{render3d_st.ds_from + render3d_st.ds_frames - 1}`)
|
||||
}
|
||||
if ds_frame_no == ds_from + ds_frames {
|
||||
ds_counting = false
|
||||
ds_report()
|
||||
if render3d_st.ds_frame_no == render3d_st.ds_from + render3d_st.ds_frames {
|
||||
render3d_st.ds_counting = false
|
||||
ds_report(render3d_st)
|
||||
}
|
||||
}
|
||||
|
||||
function ds_draw(kind: int, instances: int, indices: int) -> void {
|
||||
if ds_on != 1 or not ds_counting { return }
|
||||
let p = gpu_prog_cur
|
||||
var r = ds_last
|
||||
if r < 0 or ds_row_prog[r] != p or ds_row_kind[r] != kind or ds_row_pass[r] != ds_pass_i {
|
||||
function ds_draw(render3d_st: mut Render3dState, kind: int, instances: int, indices: int) -> void {
|
||||
if render3d_st.ds_on != 1 or not render3d_st.ds_counting { return }
|
||||
let p = render3d_st.gpu_prog_cur
|
||||
var r = render3d_st.ds_last
|
||||
if r < 0 or render3d_st.ds_row_prog[r] != p or render3d_st.ds_row_kind[r] != kind or render3d_st.ds_row_pass[r] != render3d_st.ds_pass_i {
|
||||
r = -1
|
||||
var i = 0
|
||||
while r < 0 and i < len(ds_row_pass) {
|
||||
if ds_row_pass[i] == ds_pass_i and ds_row_prog[i] == p and ds_row_kind[i] == kind { r = i }
|
||||
while r < 0 and i < len(render3d_st.ds_row_pass) {
|
||||
if render3d_st.ds_row_pass[i] == render3d_st.ds_pass_i and render3d_st.ds_row_prog[i] == p and render3d_st.ds_row_kind[i] == kind { r = i }
|
||||
i += 1
|
||||
}
|
||||
if r < 0 {
|
||||
push(ds_row_pass, ds_pass_i); push(ds_row_prog, p); push(ds_row_kind, kind)
|
||||
push(ds_row_draws, 0); push(ds_row_inst, 0); push(ds_row_idx, 0)
|
||||
r = len(ds_row_pass) - 1
|
||||
push(render3d_st.ds_row_pass, render3d_st.ds_pass_i); push(render3d_st.ds_row_prog, p); push(render3d_st.ds_row_kind, kind)
|
||||
push(render3d_st.ds_row_draws, 0); push(render3d_st.ds_row_inst, 0); push(render3d_st.ds_row_idx, 0)
|
||||
r = len(render3d_st.ds_row_pass) - 1
|
||||
}
|
||||
ds_last = r
|
||||
render3d_st.ds_last = r
|
||||
}
|
||||
ds_row_draws[r] = ds_row_draws[r] + 1
|
||||
ds_row_inst[r] = ds_row_inst[r] + instances
|
||||
ds_row_idx[r] = ds_row_idx[r] + indices
|
||||
render3d_st.ds_row_draws[r] = render3d_st.ds_row_draws[r] + 1
|
||||
render3d_st.ds_row_inst[r] = render3d_st.ds_row_inst[r] + instances
|
||||
render3d_st.ds_row_idx[r] = render3d_st.ds_row_idx[r] + indices
|
||||
}
|
||||
|
||||
function ds_program_change(from: int, to: int) -> void {
|
||||
if ds_on != 1 or not ds_counting or from == to { return }
|
||||
ds_pass_prog[ds_pass_i] = ds_pass_prog[ds_pass_i] + 1
|
||||
function ds_program_change(render3d_st: mut Render3dState, from: int, to: int) -> void {
|
||||
if render3d_st.ds_on != 1 or not render3d_st.ds_counting or from == to { return }
|
||||
render3d_st.ds_pass_prog[render3d_st.ds_pass_i] = render3d_st.ds_pass_prog[render3d_st.ds_pass_i] + 1
|
||||
}
|
||||
function ds_tex_bind() -> void {
|
||||
if ds_on != 1 or not ds_counting { return }
|
||||
ds_pass_tex[ds_pass_i] = ds_pass_tex[ds_pass_i] + 1
|
||||
function ds_tex_bind(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.ds_on != 1 or not render3d_st.ds_counting { return }
|
||||
render3d_st.ds_pass_tex[render3d_st.ds_pass_i] = render3d_st.ds_pass_tex[render3d_st.ds_pass_i] + 1
|
||||
}
|
||||
|
||||
function ds_kind_name(k: int) -> pointer {
|
||||
|
|
@ -133,18 +116,18 @@ function ds_kind_name(k: int) -> pointer {
|
|||
return "indirect"
|
||||
}
|
||||
# a per-frame mean with one decimal
|
||||
function ds_per(v: long) -> string {
|
||||
let x = (v * 10) / ds_frames
|
||||
function ds_per(render3d_st: Render3dState, v: long) -> string {
|
||||
let x = (v * 10) / render3d_st.ds_frames
|
||||
return `{string(x / 10)}.{string(x % 10)}`
|
||||
}
|
||||
function ds_backend() -> pointer {
|
||||
if gpu_kind == GPU_VK { return "vulkan" }
|
||||
function ds_backend(render3d_st: Render3dState) -> pointer {
|
||||
if render3d_st.gpu_kind == GPU_VK { return "vulkan" }
|
||||
return "opengl"
|
||||
}
|
||||
|
||||
function ds_report() -> void {
|
||||
let n_rows = len(ds_row_pass)
|
||||
let n_pass = len(ds_pass_names)
|
||||
function ds_report(render3d_st: mut Render3dState) -> void {
|
||||
let n_rows = len(render3d_st.ds_row_pass)
|
||||
let n_pass = len(render3d_st.ds_pass_names)
|
||||
var total: long = 0
|
||||
let by_kind = new []long
|
||||
for k in 0 .. DS_KINDS { push(by_kind, 0) }
|
||||
|
|
@ -152,20 +135,20 @@ function ds_report() -> void {
|
|||
let pass_inst = new []long
|
||||
for s in 0 .. n_pass { push(pass_draws, 0); push(pass_inst, 0) }
|
||||
for r in 0 .. n_rows {
|
||||
total = total + ds_row_draws[r]
|
||||
by_kind[ds_row_kind[r]] = by_kind[ds_row_kind[r]] + ds_row_draws[r]
|
||||
pass_draws[ds_row_pass[r]] = pass_draws[ds_row_pass[r]] + ds_row_draws[r]
|
||||
pass_inst[ds_row_pass[r]] = pass_inst[ds_row_pass[r]] + ds_row_inst[r]
|
||||
total = total + render3d_st.ds_row_draws[r]
|
||||
by_kind[render3d_st.ds_row_kind[r]] = by_kind[render3d_st.ds_row_kind[r]] + render3d_st.ds_row_draws[r]
|
||||
pass_draws[render3d_st.ds_row_pass[r]] = pass_draws[render3d_st.ds_row_pass[r]] + render3d_st.ds_row_draws[r]
|
||||
pass_inst[render3d_st.ds_row_pass[r]] = pass_inst[render3d_st.ds_row_pass[r]] + render3d_st.ds_row_inst[r]
|
||||
}
|
||||
print("")
|
||||
print(`draw calls per frame ({ds_backend()}, {gl_w}x{gl_h}, mean of {ds_frames} frames from {ds_from}): {ds_per(total)}`)
|
||||
print(`draw calls per frame ({ds_backend(render3d_st)}, {gl_w}x{gl_h}, mean of {render3d_st.ds_frames} frames from {render3d_st.ds_from}): {ds_per(render3d_st, total)}`)
|
||||
var kinds = ""
|
||||
for k in 0 .. DS_KINDS { kinds = kinds + ` {ds_kind_name(k)} {ds_per(by_kind[k])}` }
|
||||
for k in 0 .. DS_KINDS { kinds = kinds + ` {ds_kind_name(k)} {ds_per(render3d_st, by_kind[k])}` }
|
||||
print(` by kind:{kinds}`)
|
||||
print("")
|
||||
print(" pass draws instances program changes texture binds")
|
||||
for s in 0 .. n_pass {
|
||||
print(` {Text.pad_right(ds_pass_names[s], 24)} {Text.pad_left(ds_per(pass_draws[s]), 9)} {Text.pad_left(ds_per(pass_inst[s]), 11)} {Text.pad_left(ds_per(ds_pass_prog[s]), 17)} {Text.pad_left(ds_per(ds_pass_tex[s]), 15)}`)
|
||||
print(` {Text.pad_right(render3d_st.ds_pass_names[s], 24)} {Text.pad_left(ds_per(render3d_st, pass_draws[s]), 9)} {Text.pad_left(ds_per(render3d_st, pass_inst[s]), 11)} {Text.pad_left(ds_per(render3d_st, render3d_st.ds_pass_prog[s]), 17)} {Text.pad_left(ds_per(render3d_st, render3d_st.ds_pass_tex[s]), 15)}`)
|
||||
}
|
||||
# rows by draws, most first
|
||||
let order = new []int
|
||||
|
|
@ -174,7 +157,7 @@ function ds_report() -> void {
|
|||
while a < n_rows {
|
||||
let v = order[a]
|
||||
var b = a - 1
|
||||
while b >= 0 and ds_row_draws[order[b]] < ds_row_draws[v] { order[b + 1] = order[b]; b -= 1 }
|
||||
while b >= 0 and render3d_st.ds_row_draws[order[b]] < render3d_st.ds_row_draws[v] { order[b + 1] = order[b]; b -= 1 }
|
||||
order[b + 1] = v
|
||||
a += 1
|
||||
}
|
||||
|
|
@ -184,19 +167,19 @@ function ds_report() -> void {
|
|||
var shown = 0
|
||||
while shown < 40 and shown < n_rows {
|
||||
let r = order[shown]
|
||||
print(` {Text.pad_left(ds_per(ds_row_draws[r]), 9)} {Text.pad_left(ds_per(ds_row_inst[r]), 11)} {Text.pad_left(ds_per(ds_row_idx[r]), 12)} {Text.pad_right(ds_pass_names[ds_row_pass[r]], 24)} {Text.pad_right(ds_kind_name(ds_row_kind[r]), 14)} {gpu_program_key(ds_row_prog[r])}`)
|
||||
print(` {Text.pad_left(ds_per(render3d_st, render3d_st.ds_row_draws[r]), 9)} {Text.pad_left(ds_per(render3d_st, render3d_st.ds_row_inst[r]), 11)} {Text.pad_left(ds_per(render3d_st, render3d_st.ds_row_idx[r]), 12)} {Text.pad_right(render3d_st.ds_pass_names[render3d_st.ds_row_pass[r]], 24)} {Text.pad_right(ds_kind_name(render3d_st.ds_row_kind[r]), 14)} {gpu_program_key(render3d_st, render3d_st.ds_row_prog[r])}`)
|
||||
shown += 1
|
||||
}
|
||||
var path = "build/drawstats.csv"
|
||||
if r3d_env_has("R3D_DRAWSTATS_CSV") { path = r3d_env("R3D_DRAWSTATS_CSV") }
|
||||
if r3d_env_has(render3d_st, "R3D_DRAWSTATS_CSV") { path = r3d_env(render3d_st, "R3D_DRAWSTATS_CSV") }
|
||||
let f = file_open(path, "wb")
|
||||
if f == null { print(`r3d: drawstats could not write {path}`); return }
|
||||
let head = "backend,width,height,frames,pass,kind,program,draws_per_frame,instances_per_frame,indices_per_frame,pass_program_changes_per_frame,pass_texture_binds_per_frame\n"
|
||||
file_write(f, head, len(head))
|
||||
for i in 0 .. n_rows {
|
||||
let r = order[i]
|
||||
let s = ds_row_pass[r]
|
||||
let line = `{ds_backend()},{gl_w},{gl_h},{ds_frames},"{ds_pass_names[s]}",{ds_kind_name(ds_row_kind[r])},"{gpu_program_key(ds_row_prog[r])}",{ds_per(ds_row_draws[r])},{ds_per(ds_row_inst[r])},{ds_per(ds_row_idx[r])},{ds_per(ds_pass_prog[s])},{ds_per(ds_pass_tex[s])}\n`
|
||||
let s = render3d_st.ds_row_pass[r]
|
||||
let line = `{ds_backend(render3d_st)},{gl_w},{gl_h},{render3d_st.ds_frames},"{render3d_st.ds_pass_names[s]}",{ds_kind_name(render3d_st.ds_row_kind[r])},"{gpu_program_key(render3d_st, render3d_st.ds_row_prog[r])}",{ds_per(render3d_st, render3d_st.ds_row_draws[r])},{ds_per(render3d_st, render3d_st.ds_row_inst[r])},{ds_per(render3d_st, render3d_st.ds_row_idx[r])},{ds_per(render3d_st, render3d_st.ds_pass_prog[s])},{ds_per(render3d_st, render3d_st.ds_pass_tex[s])}\n`
|
||||
file_write(f, line, len(line))
|
||||
}
|
||||
file_close(f)
|
||||
|
|
|
|||
|
|
@ -4,16 +4,862 @@
|
|||
# windowed build - which is what a player runs - only when R3D_DEV is set to
|
||||
# anything but "0". Read every R3D_* through r3d_env_has / r3d_env, never
|
||||
# Os.has_env / Os.env.
|
||||
var r3d_dev_v: int = -1
|
||||
function r3d_dev() -> bool {
|
||||
if r3d_dev_v < 0 { if Os.has_env("R3D_DEV") and Os.env("R3D_DEV") != "0" { r3d_dev_v = 1 } else { r3d_dev_v = 0 } }
|
||||
return r3d_dev_v == 1
|
||||
export state Render3dState {
|
||||
ac_lit: AcProg = null
|
||||
ac_lit_cut: AcProg = null
|
||||
ac_sh: AcProg = null
|
||||
ac_sh_cut: AcProg = null
|
||||
ac_out: AcProg = null
|
||||
ac_out_cut: AcProg = null
|
||||
ac_actors: []Actor = null
|
||||
ac_next_id: int = 0
|
||||
ac_frame: int = 0
|
||||
ac_oq: []OutlineReq = null
|
||||
ac_oq_n: int = 0 # live entries; the array is kept and reused
|
||||
ac_oq_closed: bool = true
|
||||
ac_census_at: int = -2
|
||||
ac_lp0: floats = null
|
||||
ac_lpx: floats = null
|
||||
ac_lpy: floats = null
|
||||
ac_lpz: floats = null
|
||||
ac_cc_actors: int = 0
|
||||
ac_cc_draws: int = 0
|
||||
ac_cc_skinned: int = 0
|
||||
ac_cc_cut: int = 0
|
||||
cam_pos: floats = null # x, y, z
|
||||
cam_yaw: float = 0.0 # radians, 0 = looking down -z
|
||||
cam_pitch: float = 0.0
|
||||
cam_fov: float = 0.0 # vertical, radians
|
||||
cam_near: float = 0.0
|
||||
cam_far: float = 0.0
|
||||
cam_aspect: float = 0.0
|
||||
cam_view: floats = null
|
||||
cam_proj: floats = null
|
||||
cam_vp: floats = null
|
||||
cam_inv_vp: floats = null
|
||||
cam_inv_proj: floats = null
|
||||
cam_vp_clean: floats = null # view-projection, kept for depth reconstruction
|
||||
cam_inv_vp_clean: floats = null
|
||||
cam_fwd: floats = null
|
||||
cam_right: floats = null
|
||||
cam_planes: floats = null
|
||||
col_x: floats = null
|
||||
col_z: floats = null
|
||||
col_r: floats = null
|
||||
col_y0: floats = null
|
||||
col_y1: floats = null
|
||||
col_n: int = 0
|
||||
col_side: int = 0 # cells per side
|
||||
col_start: words = null # per cell: first index into col_sorted (side*side + 1)
|
||||
col_sorted: words = null
|
||||
col_built: bool = false
|
||||
col_out: floats = null # the resolved position (x, z)
|
||||
day_on: bool = false
|
||||
day_hours: float = 0.0 # float bits, 0 .. 24
|
||||
day_light: float = 0.0 # 0 night .. 1 full day (float bits)
|
||||
day_ibl: floats = null # rgb scale on the sky's light
|
||||
day_sun_base: floats = null # the HDRI's sun radiance, kept
|
||||
day_az0: float = 0.0 # the sun's azimuth at the reference hour (radians, yaw convention)
|
||||
day_yaw0: float = 0.0 # the sky yaw the scene was tuned at
|
||||
day_hour0: float = 0.0 # the hour the photograph was taken (10.5)
|
||||
day_dir: float = 0.0 # +1 / -1: which way the sun travels in yaw
|
||||
day_az: float = 0.0
|
||||
day_el: float = 0.0
|
||||
day_gen: int = 0 # bumps when the light moved enough to rebake the terrain shadow
|
||||
day_baked_az: float = 0.0
|
||||
day_baked_el: float = 0.0
|
||||
day_sky_baked: float = 0.0 # the sky yaw the convolutions were baked at
|
||||
fire_pos: floats = null
|
||||
fire_color: floats = null
|
||||
day_moon: bool = false
|
||||
day_moon_phase: float = 0.5 # 0.5: full, which is where the game used to be
|
||||
day_moon_illum: float = 1.0 # the lit fraction, derived from the phase
|
||||
day_moon_dir: floats = null # toward the moon, whether or not it is up
|
||||
hand_pos: floats = null
|
||||
hand_color: floats = null
|
||||
hand_dir: floats = null
|
||||
hand_cone: float = 0.0
|
||||
hand_reach: float = 0.0 # metres the hand light reaches (float bits)
|
||||
day_overcast: float = 0.0 # 0 clear .. 1 a low grey sky (float bits): dims the sun and the sky's light
|
||||
day_flash: float = 0.0 # a lightning flash this frame (0..1): the sun brightens for it
|
||||
day_fog_mul: float = 1.0 # multiplies the base fog density (rain and snow thicken the air)
|
||||
day_fog_base: float = 0.0
|
||||
ds_on: int = -1 # -1: not read yet
|
||||
ds_counting: bool = false
|
||||
ds_from: int = 90
|
||||
ds_frames: int = 60
|
||||
ds_frame_no: int = 0
|
||||
ds_outside: pointer = "(outside passes)"
|
||||
ds_pass_names: []pointer = null
|
||||
ds_pass_prog: []long = null # program changes inside the pass
|
||||
ds_pass_tex: []long = null # texture binds inside the pass
|
||||
ds_pass_i: int = 0
|
||||
ds_row_pass: []int = null
|
||||
ds_row_prog: []int = null
|
||||
ds_row_kind: []int = null
|
||||
ds_row_draws: []long = null
|
||||
ds_row_inst: []long = null
|
||||
ds_row_idx: []long = null
|
||||
ds_last: int = -1
|
||||
r3d_dev_v: int = -1
|
||||
gltf_dir: string = null
|
||||
gltf_bin: pointer = null
|
||||
gltf_doc: Val = null
|
||||
gltf_count: int = 0
|
||||
gltf_ctype: int = 0
|
||||
gltf_comps: int = 0
|
||||
gltf_tex_paths: []pointer = null
|
||||
gltf_tex_ids: words = null
|
||||
gltf_tex_n: int = 0
|
||||
gltf_white: int = 0
|
||||
gltf_flat: int = 0
|
||||
gltf_mat_names: []string = null
|
||||
gltf_mat_diff: words = null
|
||||
gltf_mat_nrm: words = null
|
||||
gltf_mat_arm: words = null
|
||||
gltf_mat_n: int = 0
|
||||
gltf_cutout: bool = false # the material being loaded is alpha-blended (a cut-out atlas)
|
||||
gpu_kind: int = 0 # the backend running; 0 until gpu_select
|
||||
gpu_wanted: int = 0 # what was asked for (setting or R3D_GFX)
|
||||
gpu_fallback_reason: string = null # why the wanted backend is not the one running
|
||||
gpu_s_depth_test: int = -1
|
||||
gpu_s_depth_func: int = -1
|
||||
gpu_s_depth_write: int = -1
|
||||
gpu_s_blend: int = -1
|
||||
gpu_s_blend_src: int = -1
|
||||
gpu_s_blend_dst: int = -1
|
||||
gpu_s_cull: int = -1
|
||||
gpu_s_cull_face: int = -1
|
||||
gpu_s_color_write: int = -1
|
||||
gpu_s_a2c: int = -1
|
||||
gpu_s_bias: int = -1
|
||||
gpu_s_bias_f: float = 0.0 # float bits, for a backend that bakes the bias into a pipeline
|
||||
gpu_s_bias_u: float = 0.0
|
||||
gpu_s_scissor: int = -1
|
||||
gpu_prog_ids: []int = null
|
||||
gpu_prog_keys: []string = null
|
||||
gpu_prog_cur: int = 0
|
||||
gpu_u_tmp: words = null
|
||||
gpu_cap_probed: bool = false
|
||||
gpu_cap_windows: bool = false # the advanced features exist on this platform at all
|
||||
gpu_cap_vulkan: bool = false # a Vulkan loader, an instance and a device
|
||||
gpu_cap_floor: bool = false # Vulkan 1.3 with everything the renderer's floor needs
|
||||
gpu_cap_rt: bool = false # ray query + acceleration structures
|
||||
gpu_cap_mesh: bool = false # VK_EXT_mesh_shader
|
||||
gpu_cap_nvidia: bool = false
|
||||
gpu_cap_rtx: int = 0 # the RTX generation (20, 30, 40, 50); 0 = not RTX
|
||||
gpu_cap_reflex: bool = false # VK_NV_low_latency2
|
||||
gpu_cap_hdr: bool = false # the instance offers HDR colour spaces
|
||||
gpu_cap_device: string = ""
|
||||
gpu_tx: words = null
|
||||
gpu_tx_cap: int = 0
|
||||
gpu_bound_2d: int = 0
|
||||
gpu_bound_array: int = 0
|
||||
gpu_unit_2d: words = null
|
||||
gpu_unit_cur: int = 0
|
||||
gpu_fb: words = null
|
||||
gpu_fb_cap: int = 0
|
||||
gpu_fb_cur: int = 0
|
||||
gpu_glcheck: int = -1
|
||||
gpu_rb_samples: int = 0
|
||||
gpu_drawbufs: words = null
|
||||
gpu_glcheck_seen: []string = null
|
||||
gpu_variants: []GpuVariant = null
|
||||
gvk_ready: bool = false
|
||||
gvk_inst: pointer = null
|
||||
gvk_pd: pointer = null
|
||||
gvk_dev: pointer = null
|
||||
gvk_queue: pointer = null
|
||||
gvk_family: int = -1
|
||||
gvk_mp: bytes = null # VkPhysicalDeviceMemoryProperties
|
||||
gvk_device_name: string = ""
|
||||
gvk_why: string = "" # why the device did not come up, for the fallback notice
|
||||
gvk_max_aniso: int = 0x3F800000 # the device's anisotropy limit as float bits; 1.0 when it has none
|
||||
gvk_want_surface: bool = false # set before gvk_init when the renderer will present to a window
|
||||
gvk_has_surface: bool = false # the instance and device can make a surface and a swapchain
|
||||
gvk_errlog: string = null
|
||||
gvk_errlog_read: bool = false
|
||||
gvk_has_mdi: bool = false # multiDrawIndirect + drawIndirectFirstInstance
|
||||
gvk_has_dic: bool = false # drawIndirectCount
|
||||
gvk_has_mesh: bool = false # VK_EXT_mesh_shader with its meshShader feature on
|
||||
gvk_mesh_fn: pointer = null # vkCmdDrawMeshTasksEXT: the device's own, the interposer exports none
|
||||
gvk_n_allocs: int = 0
|
||||
gvk_blk_mem: []long = null
|
||||
gvk_blk_kind: []int = null # memory type * 2, + 1 for images
|
||||
gvk_blk_size: []int = null
|
||||
gvk_blk_map: []pointer = null
|
||||
gvk_fr_blk: []int = null # free ranges: block, offset, length (0 = an unused slot)
|
||||
gvk_fr_off: []int = null
|
||||
gvk_fr_len: []int = null
|
||||
gvk_al_blk: []int = null # per allocation id: its block, or -1 for its own memory
|
||||
gvk_al_mem: []long = null # its own memory when it has one
|
||||
gvk_al_off: []int = null
|
||||
gvk_al_len: []int = null
|
||||
gvk_al_map: []pointer = null
|
||||
gvk_al_spare: []int = null # ids to reuse
|
||||
gvk_pool: long = 0
|
||||
gvk_fence: bytes = null
|
||||
gvk_has_hqr: bool = false
|
||||
gvk_ts_period: float = 0.0 # float bits: nanoseconds per timestamp tick
|
||||
gvk_qpool: long = 0
|
||||
gvk_q_active: int = -1
|
||||
gvk_prog_var: []GpuVariant = null
|
||||
gvk_prog_vs: []long = null
|
||||
gvk_prog_fs: []long = null
|
||||
gvk_prog_dsl: []long = null
|
||||
gvk_prog_layout: []long = null
|
||||
gvk_spv_dir: string = ""
|
||||
gvk_spv_len: int = 0
|
||||
gvk_prog_mesh: []int = null
|
||||
gvk_cp_pipe: []long = null
|
||||
gvk_cp_layout: []long = null
|
||||
gvk_cp_dsl: []long = null
|
||||
gvk_cp_nbuf: []int = null
|
||||
gvk_pipe_keys: []string = null
|
||||
gvk_pipe: []long = null
|
||||
gvk_zero_vbuf: int = 0
|
||||
gvk_skip_said: words = null # per program: its skipped draws have been reported
|
||||
gvk_ublk_v: []bytes = null # per program: the vertex stage's block as last set
|
||||
gvk_ublk_f: []bytes = null
|
||||
gvk_prog_tex: []words = null # per program: the texture bound to each manifest sampler
|
||||
gvk_prog_unit: []words = null # per program: the unit + 1 each sampler was pointed at with u_i
|
||||
gvk_prog_dirty: []int = null # per program: its blocks or textures changed since its last set
|
||||
gvk_set_last: []long = null # per program: the descriptor set its last draw used
|
||||
gvk_set_frame: []int = null # per program: the frame that set belongs to
|
||||
gvk_set_tex: []words = null # per program: the textures that set was filled with
|
||||
gvk_ring_buf: int = 0 # a gvk_buf handle
|
||||
gvk_ring_off: int = 0
|
||||
gvk_ring_align: int = 256
|
||||
gvk_dpool: long = 0
|
||||
gvk_white: int = 0 # a 1x1 white texture for a sampler nothing was bound to
|
||||
gvk_msaa_max: int = 0 # samples the scene may use (gvk_frame_init reads the device's limits)
|
||||
gvk_kpool: long = 0
|
||||
gvk_sc_prog: []int = null # per cached set: its program
|
||||
gvk_sc_koff: []int = null # per cached set: where its key starts in gvk_sc_keys
|
||||
gvk_sc_set: []long = null
|
||||
gvk_sc_next: []int = null # the program's next cached set, -1 at the end
|
||||
gvk_sc_keys: []long = null # per texture of a key: handle * 65536 + generation, sampler
|
||||
gvk_sc_head: words = null # per program (< 4096): its most recently made set, -1 if none
|
||||
gvk_sc_tmp: []long = null # this draw's key while it is looked up
|
||||
gvk_set_offs: bytes = null # this draw's dynamic offsets, in binding order
|
||||
gvk_set_ndyn: int = 0
|
||||
gvk_ub_frame: []int = null # per program: the frame its blocks were last copied into the ring
|
||||
gvk_ub_offv: []int = null
|
||||
gvk_ub_offf: []int = null
|
||||
gvk_sc_made: int = 0 # sets made since start (R3D_VK_PROF)
|
||||
gvk_default_smp: long = 0 # linear, clamped: for a texture slot nothing was bound to
|
||||
gvk_cb: pointer = null
|
||||
gvk_in_pass: bool = false
|
||||
gvk_fb_cur: int = 0
|
||||
gvk_fb_read: int = 0
|
||||
gvk_fb_draw: int = 0
|
||||
gvk_pass_w: int = 0
|
||||
gvk_pass_h: int = 0
|
||||
gvk_pass_ncolor: int = 0
|
||||
gvk_pass_cfmt: int = 0
|
||||
gvk_pass_dfmt: int = 0
|
||||
gvk_pass_samples: int = 1 # the attachments' sample count, which every pipeline in the pass must match
|
||||
gvk_pass_col: words = null # the pass's colour attachments: texture, layer + 1
|
||||
gvk_pass_dep: words = null # its depth attachment: texture, layer + 1
|
||||
gvk_clear_bits: int = 0 # GL_COLOR_BUFFER_BIT / GL_DEPTH_BUFFER_BIT waiting for the pass
|
||||
gvk_clear_rgba: floats = null # float bits
|
||||
gvk_vp: words = null # x, y, w, h
|
||||
gvk_sc: words = null # scissor on, x, y (OpenGL rows), w, h
|
||||
gvk_screen_color: int = 0
|
||||
gvk_screen_depth: int = 0
|
||||
gvk_screen_w: int = 0
|
||||
gvk_screen_h: int = 0
|
||||
gvk_layer_views: []string = null # "tex:layer" -> index into gvk_layer_view
|
||||
gvk_layer_view: []long = null
|
||||
gvk_fb_ncolor: words = null # per framebuffer: colour slots drawn (draw buffers; 0 = none)
|
||||
gvk_hdr_want: bool = false # r3d_hdr: the setting
|
||||
gvk_hdr_on: bool = false # the swapchain is HDR10 now
|
||||
gvk_has_colorspace: bool = false # the instance took VK_EXT_swapchain_colorspace
|
||||
gvk_has_hdr_meta: bool = false # the device took VK_EXT_hdr_metadata
|
||||
r3d_hdr_peak: float = 0.0 # 0 until r3d_hdr_calibrate: 1000
|
||||
r3d_hdr_paper: float = 0.0 # 200
|
||||
r3d_hdr_black: float = 0.0 # 0
|
||||
gvk_fb_counter: int = 0
|
||||
gvk_prog_counter: int = 0
|
||||
gvk_wireframe: int = 0
|
||||
gvk_state: GvkState = null
|
||||
gvk_ind_buf: int = 0
|
||||
gvk_ind_off: int = 0
|
||||
gvk_ind_n: int = 0
|
||||
gvk_ind_cbuf: int = 0
|
||||
gvk_ind_coff: int = 0
|
||||
gvk_mesh_x: int = 0
|
||||
gvk_mesh_y: int = 0
|
||||
gvk_mesh_z: int = 0
|
||||
gvk_surface: long = 0
|
||||
gvk_swap: long = 0
|
||||
gvk_swap_n: int = 0
|
||||
gvk_swap_images: bytes = null
|
||||
gvk_swap_fmt: int = 0
|
||||
gvk_swap_w: int = 0
|
||||
gvk_swap_h: int = 0
|
||||
gvk_vsync: bool = true
|
||||
gvk_acq_fence: bytes = null
|
||||
gvk_swap_stale: bool = false
|
||||
gvk_size_buf: words = null
|
||||
gvk_prof_state: int = -1
|
||||
gvk_us_pipe: long = 0
|
||||
gvk_us_set: long = 0
|
||||
gvk_us_draw: long = 0
|
||||
gvk_n_draws: int = 0
|
||||
gvk_n_asked: int = 0 # draws that reached gvk_draw_now: every one the renderer asked for
|
||||
gvk_n_flush: int = 0
|
||||
gvk_prof_frames: int = 0
|
||||
gvk_n_pipe_new: int = 0 # pipelines made this profile window
|
||||
gvk_layout_keys: []string = null
|
||||
gvk_pc_prog: []int = null
|
||||
gvk_pc_layout: []int = null
|
||||
gvk_pc_state: []int = null
|
||||
gvk_pc_bias: []int = null
|
||||
gvk_pc_pass: []int = null
|
||||
gvk_pc_pipe: []long = null
|
||||
gvk_pc_last: words = null
|
||||
gvk_tex_image: []long = null
|
||||
gvk_tex_view: []long = null
|
||||
gvk_tex_mem: []long = null
|
||||
gvk_tex_levels: []int = null
|
||||
gvk_tex_layers: []int = null
|
||||
gvk_tex_vkfmt: []int = null
|
||||
gvk_tex_dims_w: []int = null
|
||||
gvk_tex_dims_h: []int = null
|
||||
gvk_tex_glfmt: []int = null # the format the renderer asked for (gpu.ludic's vocabulary)
|
||||
gvk_tex_array: []int = null # 1 for an array image
|
||||
gvk_tex_gen: []int = null # bumped whenever the image behind a handle is replaced
|
||||
gvk_tex_smp_sig: []int = null # the sampling parameters the cached sampler was made for
|
||||
gvk_tex_smp: []long = null
|
||||
gvk_tex_samples: []int = null # 1, or the sample count of a multisampled renderbuffer
|
||||
gvk_storage_samples: int = 1 # what the next gvk_tex_storage makes: gpu_rb_storage sets it around its call
|
||||
gvk_unpack_swap: bool = false # GL_UNPACK_SWAP_BYTES: 16-bit PNG samples arrive big-endian
|
||||
gvk_st_buf: long = 0
|
||||
gvk_st_mem: long = 0
|
||||
gvk_smp_keys: []string = null
|
||||
gvk_smp: []long = null
|
||||
gvk_buf: []long = null
|
||||
gvk_buf_mem: []long = null
|
||||
gvk_buf_size: []int = null
|
||||
gvk_buf_map: []pointer = null # host-visible buffers stay mapped for their whole life
|
||||
gvk_buf_used: []int = null # the frame (gvk_frame_no) a draw last bound the buffer in
|
||||
gvk_frame_no: int = 1
|
||||
gvk_retired_buf: []long = null
|
||||
gvk_retired_mem: []long = null
|
||||
gvk_buf_gpu: []int = null
|
||||
grass_prog: int = 0
|
||||
grass_mesh: Mesh = null
|
||||
grass_on: bool = true
|
||||
grass_wind: float = 0.0
|
||||
grass_blade_tex: int = 0
|
||||
grass_blade_cols: int = 8
|
||||
grass_push_x: float = 0.0
|
||||
grass_push_z: float = 0.0
|
||||
grass_push_r: float = 0.0
|
||||
grass_s0: float = 0.0 # float bits: blade spacing at the camera (m)
|
||||
grass_d0: float = 0.0 # the distance at which the spacing has doubled (m)
|
||||
grass_radius: float = 0.0 # no blades past this (m)
|
||||
grass_draws: int = 0
|
||||
grass_dbg: int = 0
|
||||
grass_merge: bool = false
|
||||
grass_rec: words = null # VkDrawIndexedIndirectCommand records, 5 words each
|
||||
grass_tv: floats = null # per record: corner x, corner z, indices per cell, 0 (float bits)
|
||||
grass_chunk_tv: floats = null
|
||||
grass_n: int = 0
|
||||
grass_cmds: int = 0
|
||||
grass_band_start: words = null # per band: its first record
|
||||
grass_band_cells: words = null # ... and its 16 m cells per tile side
|
||||
grass_band_n: int = 0
|
||||
grass_mesh_on: bool = false
|
||||
grass_mesh_prog: int = 0
|
||||
r3d_draw_cb: fn() = null
|
||||
r3d_casters_cb: fn(floats) = null
|
||||
r3d_fill_cb: fn(Stream, int, int, int) = null
|
||||
ov_prog: int = 0
|
||||
ov_mesh: Mesh = null
|
||||
ov_vbo: int = 0
|
||||
ov_buf: pointer = null
|
||||
ov_n: int = 0
|
||||
ov_tex: int = 0
|
||||
ov_mode: int = 2 # what the next quads read: 0 the image texture, 1 the font, 2 a flat colour
|
||||
ov_voff: float = 0.0 # float bits: 4 x ov_mode, added to v (overlay.frag)
|
||||
ov_white: int = 0
|
||||
ov_font: int = 0
|
||||
ov_font_adv: floats = null # float bits, em units, one per glyph in the atlas
|
||||
ov_font_n: int = 0 # glyphs in the atlas
|
||||
ov_font_map: words = null
|
||||
ov_font_hi_cp: words = null # code points >= OV_MAP_N, ascending
|
||||
ov_font_hi_g: words = null
|
||||
ov_font_hi_n: int = 0
|
||||
ov_font_space: int = 0 # the glyph for ' ', and for what has none: '?'
|
||||
ov_font_qmark: int = 0
|
||||
ov_font_cols: int = 16
|
||||
ov_font_rows: int = 6
|
||||
ov_font_cell: int = 128 # px per cell in the atlas
|
||||
ov_font_em: int = 100 # px per em in the atlas
|
||||
ov_pad_x: float = 0.0 # float bits, em
|
||||
ov_base_y: float = 0.0
|
||||
ov_ready: bool = false
|
||||
ov_open: bool = false
|
||||
ov_dbg: bool = false
|
||||
ov_ranges: words = null
|
||||
ov_nr: int = 0
|
||||
ov_range_start: int = 0
|
||||
ov_clip_x: int = 0 # the current clip rectangle in screen pixels (top-left origin)
|
||||
ov_clip_y: int = 0
|
||||
ov_clip_w: int = 0
|
||||
ov_clip_h: int = 0
|
||||
ov_prog_sdr: int = 0
|
||||
ov_prog_hdr: int = 0
|
||||
ov_hdr_scale: float = 0.0 # float bits; 0 until ov_begin sets 1
|
||||
ov_u8_len: int = 1
|
||||
post_hdr: Target = null
|
||||
post_ms_fbo: int = 0 # 4x multisampled scene target, resolved into post_hdr
|
||||
post_ms_samples: int = 1 # temporal AA carries the edges; R3D_MSAA=n to compare
|
||||
post_bloom: []Target = null
|
||||
post_p_down: int = 0
|
||||
post_p_up: int = 0
|
||||
post_p_tone: int = 0
|
||||
post_fs: Mesh = null
|
||||
post_exposure: float = 0.0
|
||||
post_bloom_strength: int = 0
|
||||
post_vignette: float = 0.0
|
||||
post_saturation: float = 0.0
|
||||
post_contrast: float = 0.0
|
||||
post_w: int = 0
|
||||
post_h: int = 0
|
||||
post_auto: bool = true
|
||||
post_key: float = 0.0 # target mean luminance after exposure (float bits)
|
||||
post_lum: words = null
|
||||
post_mips: int = 0
|
||||
post_adapt: float = 0.0 # smoothed exposure (float bits)
|
||||
post_exposure_max: float = 20.0 # 20: the ceiling auto-exposure may reach (night lowers it)
|
||||
post_wb_r: float = 1.02 # 1.02
|
||||
post_wb_g: float = 1.0 # 1.0
|
||||
post_wb_b: float = 0.97 # 0.97
|
||||
post_lift_r: float = 0.004 # 0.004
|
||||
post_lift_g: float = 0.004 # 0.004
|
||||
post_lift_b: float = 0.012 # 0.012
|
||||
post_gain_r: float = 0.99 # 0.99
|
||||
post_gain_g: float = 0.995 # 0.995
|
||||
post_gain_b: float = 1.0 # 1.0
|
||||
post_ao: Target = null
|
||||
post_ao_blur: Target = null
|
||||
post_p_ao: int = 0
|
||||
post_p_ao_blur: int = 0
|
||||
post_ao_radius: float = 0.0
|
||||
post_contact: float = 1.25 # 1.25: how hard a thing is darkened where it meets the ground
|
||||
post_ao_intensity: float = 0.0
|
||||
post_ao_strength: float = 0.0
|
||||
post_gi_strength: float = 0.4 # 0.4
|
||||
post_no_gi: bool = false
|
||||
post_ldr: Target = null
|
||||
post_depth_copy: Target = null
|
||||
post_vol: Target = null
|
||||
post_p_vol: int = 0
|
||||
post_vol_steps: float = 24.0 # 24
|
||||
post_vol_density: float = 0.002 # 0.002
|
||||
post_vol_falloff: float = 0.005 # 0.005
|
||||
post_vol_far: float = 800.0 # 800 m
|
||||
post_vol_g: float = 0.6 # 0.6: air throws light forward
|
||||
post_vol_mist: float = 0.0 # the day sets these two
|
||||
post_vol_mist_h: float = 40.0 # 40 m
|
||||
post_prev: Target = null # last frame's scene colour, for the SSGI bounce only
|
||||
post_scene: Target = null # this frame's scene colour before the water, for refraction
|
||||
post_frame: int = 0
|
||||
post_color_w: int = 0 # its size: the display's when DLSS upscaled it
|
||||
post_color_h: int = 0
|
||||
post_color: int = 0 # the HDR colour the rest of post reads # the resolved depth, copied so passes can read it while drawing into the frame
|
||||
post_p_sharp: int = 0
|
||||
post_fxaa: float = 1.0
|
||||
post_p_dof: int = 0
|
||||
post_dof: Target = null
|
||||
post_dof_focus: float = 10.0 # 10 m
|
||||
post_dof_aperture: float = 0.0 # 0 = no lens, and no pass
|
||||
post_dof_max: float = 12.0 # 12 px
|
||||
post_p_tone_hdr: int = 0 # the tonemap's HDR10 variant, made the first time HDR is on
|
||||
post_ldr_hdr: bool = false # post_ldr was made for HDR10 output (10-bit)
|
||||
post_sharpen: float = 0.0
|
||||
post_grain: int = 0
|
||||
post_adapt_t: []Target = null
|
||||
post_adapt_i: int = 0
|
||||
post_p_adapt: int = 0
|
||||
post_adapt_reset: bool = true
|
||||
prof_on: bool = false
|
||||
prof_names: []pointer = null
|
||||
prof_ids: words = null # PROF_SLOTS * PROF_RING query objects
|
||||
prof_ns: []long = null # accumulated nanoseconds per slot
|
||||
prof_hits: []long = null # samples accumulated per slot
|
||||
prof_n: int = 0 # slots in use
|
||||
prof_frame: int = 0
|
||||
prof_active: int = -1 # slot whose query is currently open
|
||||
prof_scratch: words = null
|
||||
prof_gen: []int = null
|
||||
prof_gen_cur: int = 0
|
||||
prof_ft: []long = null # real frame times, microseconds
|
||||
prof_last_us: long = 0
|
||||
prof_chunk_us: []long = null
|
||||
prof_chunk_kind: []int = null
|
||||
prof_chunk_band: []int = null
|
||||
prof_chunk_n: []int = null
|
||||
prof_gen_us_cur: long = 0
|
||||
prof_gen_us: []long = null # generation microseconds per frame
|
||||
prof_layer_us_cur: long = 0 # rebuilding and uploading instance buffers
|
||||
prof_bake_us_cur: long = 0 # the height-field shadow rebake
|
||||
prof_layer_us: []long = null
|
||||
prof_bake_us: []long = null
|
||||
prof_dt: []long = null # frame time, aligned with the arrays above
|
||||
prof_up_cur: long = 0 # instance bytes uploaded this frame
|
||||
prof_up: []long = null
|
||||
prof_pre_swap: long = 0
|
||||
prof_cpu_us: []long = null
|
||||
prof_sum_dt: long = 0 # sums over the frames past the first eight, for the mean split
|
||||
prof_sum_cpu: long = 0
|
||||
prof_sum_game: long = 0
|
||||
prof_sum_n: int = 0
|
||||
prof_mark_last: long = 0
|
||||
prof_mark_best: long = 0
|
||||
prof_mark_name: pointer = null
|
||||
prof_mark_us: []long = null
|
||||
prof_mark_who: []pointer = null
|
||||
prof_mk_names: []pointer = null
|
||||
prof_mk_us: []long = null
|
||||
prof_mk_n: int = 0
|
||||
prof_game_us_cur: long = 0
|
||||
prof_game_us: []long = null
|
||||
r3d_root: string = "packages/ludic.render3d" # where shaders/ lives
|
||||
r3d_assets: string = "assets/polyhaven" # where the CC0 assets live
|
||||
r3d_root_found: bool = false
|
||||
r3d_global_defs: string = ""
|
||||
r3d_noise_src: string = null
|
||||
r3d_lighting_src: string = null
|
||||
r3d_wind_src: string = null
|
||||
r3d_prog_log: int = -1
|
||||
q_scratch: floats = null
|
||||
q_sy: floats = null # views of q_scratch's second, third and fourth
|
||||
q_sz: floats = null
|
||||
q_st: floats = null
|
||||
r3d_sky_prog: int = 0
|
||||
r3d_fog_density: float = 0.0
|
||||
r3d_fog_scale: float = 1.0 # float bits: a setting's multiplier over the density the day sets
|
||||
r3d_fog_falloff: float = 0.0
|
||||
r3d_fog_base: float = 0.0 # the height fog is measured from (float bits); 0 = y = 0
|
||||
r3d_fog_inscatter: float = 0.02 # 0.02, what the shader used to hard-code
|
||||
r3d_fog_desat: float = 1.35 # 1.35
|
||||
r3d_sky_relight: float = 0.0
|
||||
r3d_ground_alb_r: float = 0.3
|
||||
r3d_ground_alb_g: float = 0.34
|
||||
r3d_ground_alb_b: float = 0.14
|
||||
r3d_ground_hi_r: float = 0.289
|
||||
r3d_ground_hi_g: float = 0.28
|
||||
r3d_ground_hi_b: float = 0.26
|
||||
r3d_ground_hi_y: float = 480.0 # 480
|
||||
r3d_ground_hi_w: float = 150.0 # 150
|
||||
r3d_time: float = 0.0
|
||||
r3d_ready: bool = false
|
||||
r3d_dem_path: string = null
|
||||
r3d_dem_min: float = 0.0
|
||||
r3d_dem_max: float = 0.0
|
||||
r3d_dem_base: float = 0.0
|
||||
r3d_dem_ox: float = 0.0
|
||||
r3d_dem_oz: float = 0.0
|
||||
r3d_ortho_path: string = null
|
||||
r3d_plate: bool = false
|
||||
r3d_sky_path: string = null
|
||||
r3d_debug: bool = false
|
||||
r3d_debug_shadow: bool = false
|
||||
r3d_debug_max: bool = false
|
||||
r3d_cloud_shadow: float = 0.5 # 0.5
|
||||
r3d_clip_y: float = -2147483600.0 # -2^31: no clipping
|
||||
r3d_prepass_env: int = -1
|
||||
r3d_test_frame: int = 0
|
||||
r3d_test_resize: int = 0
|
||||
r3d_no_shadow: bool = false
|
||||
r3d_no_trees: bool = false
|
||||
r3d_no_refl: bool = false
|
||||
r3d_cam_log: int = -1
|
||||
sc_prog: int = 0
|
||||
sc_prog_fol: int = 0
|
||||
sc_prog_wind: int = 0
|
||||
sc_prog_blade: int = 0
|
||||
sc_prog_flower: int = 0
|
||||
sc_prog_card: int = 0
|
||||
sc_prog_card_shadow: int = 0
|
||||
sc_prog_card_cheap: int = 0
|
||||
sc_blade_base: floats = null
|
||||
sc_blade_tip: floats = null
|
||||
sc_blade_tint: floats = null
|
||||
sc_prog_shadow: int = 0
|
||||
sc_prog_shadow_wind: int = 0
|
||||
sc_prog_shadow_fol: int = 0 # foliage meshes: alpha-tested casters
|
||||
sc_prog_fol_depth: int = 0
|
||||
sc_prog_fol_eq: int = 0
|
||||
sc_prepass: bool = false # the main pass is drawing over what the prepass laid down
|
||||
sc_imp_prog: int = 0
|
||||
sc_imp_prog_shadow: int = 0
|
||||
sc_bake_prog: int = 0
|
||||
sc_bake_card_prog: int = 0
|
||||
sc_bake_flower_prog: int = 0
|
||||
sc_card: Mesh = null
|
||||
sc_ident_buf: int = 0
|
||||
sc_layers: []Layer = null
|
||||
sc_debug_dump: bool = false
|
||||
sc_printed: bool = false
|
||||
sc_a2c: bool = true
|
||||
sc_bake_flower: bool = false
|
||||
sc_dump_n: int = 0
|
||||
sc_view_gen: int = 1
|
||||
sc_view_pos: floats = null
|
||||
sc_view_fwd: floats = null
|
||||
sc_cull_prog: int = 0
|
||||
sc_cull_tried: bool = false
|
||||
sc_ind_base: int = -1 # >= 0: layer_draw_model / layer_draw_depth draw records from here
|
||||
sc_ind_n: int = 1 # records each of those draws covers: one per level of a material
|
||||
sc_ind_stride: int = 4 # records between one material's first and the next's (3 for the shadow LOD)
|
||||
sc_freeze: bool = false # a secondary pass (reflection) reuses the partition
|
||||
sc_dbg_blade: int = 0
|
||||
sc_dbg_level: int = -1
|
||||
sc_dbg_lod: bool = false
|
||||
sc_dbg_tint: floats = null
|
||||
sc_cast_all: int = -1
|
||||
sc_cast_gen: int = -1
|
||||
sc_cast_dy: float = 0.0
|
||||
sc_cast_k: float = 0.0
|
||||
sc_skip_blade: bool = false
|
||||
sc_skip_flower: bool = false
|
||||
sc_skip_card: bool = false
|
||||
cb_state: int = 12345
|
||||
shadow_res: int = 2048
|
||||
shadow_refused: int = 0 # the last size the graphics card had no memory for (0: none)
|
||||
sh_tex: int = 0
|
||||
sh_fbo: int = 0
|
||||
sh_vp: floats = null # the cascades' view-projections, 16 floats each
|
||||
sh_vp_v: [][]float = null # a view of each
|
||||
sh_split: floats = null # view-space far distance of each cascade
|
||||
sh_range: floats = null # 4 light-frustum depth extents (metres)
|
||||
sh_texel: floats = null # 4 shadow texel sizes (metres)
|
||||
sh_tmp_proj: floats = null
|
||||
sh_tmp_vp: floats = null
|
||||
sh_tmp_inv: floats = null
|
||||
sh_tmp_view: floats = null
|
||||
sh_corner: floats = null
|
||||
sh_cascade: int = 0 # the cascade being rendered (for casters that skip far ones)
|
||||
sh_printed2: bool = false
|
||||
sh_printed3: bool = false
|
||||
sh_enabled: bool = true
|
||||
sh_force: int = -1 # R3D_FORCE=<c> pins every pixel to cascade c (debug)
|
||||
sh_skip_terrain: bool = false
|
||||
sh_probe_x: float = 0.0
|
||||
sh_probe_y: float = 0.0
|
||||
sh_probe_z: float = 0.0
|
||||
sh_printed: bool = false
|
||||
sky_tex: int = 0
|
||||
sky_w: int = 0
|
||||
sky_h: int = 0
|
||||
sky_irradiance: int = 0
|
||||
sky_prefilter: int = 0 # GL_TEXTURE_2D_ARRAY
|
||||
sky_brdf: int = 0
|
||||
sun_dir: floats = null # toward the sun (float bits)
|
||||
sun_color: floats = null # radiance (float bits)
|
||||
sky_fullscreen: Mesh = null
|
||||
sky_yaw: float = 0.0 # radians: the HDRI is turned by this about y
|
||||
sky_sun_boost: float = 2.3 # 2.3: the photograph's thin cloud dims its sun; a crisper day wants more
|
||||
sky_rot_s: float = 0.0
|
||||
sky_rot_c: float = 0.0
|
||||
sun_hdri: floats = null # the sun direction as found in the file
|
||||
sky_start_yaw: float = 0.0
|
||||
sky_baked: bool = false # the IBL set exists, baked at sky_baked_yaw
|
||||
sky_baked_yaw: float = 0.0
|
||||
sky_p_irr: int = 0
|
||||
sky_p_pre: int = 0
|
||||
sky_p_brdf: int = 0
|
||||
sky_prefilter_w: int = 512
|
||||
STREAM_MAX_CHUNKS: int = 4096
|
||||
stream_all: []Stream = null
|
||||
stream_cap_read: bool = false
|
||||
stream_no_evict: bool = false # R3D_NOEVICT: the old behaviour, for comparison
|
||||
stream_walk_no: int = 0 # counts ring walks; a chunk's age is measured in these
|
||||
stream_evictions: int = 0
|
||||
stream_us_gen: long = 0 # generating new chunks (stream_fill)
|
||||
stream_us_gather: long = 0 # copying cached chunks into the layer buffer
|
||||
stream_us_walk: long = 0 # the ring walk itself
|
||||
stream_walks: int = 0 # streams that walked their whole ring this frame
|
||||
stream_debug_n: int = 0
|
||||
stream_scratch: floats = null
|
||||
STREAM_BUDGET_US: int = 2500 # microseconds of generation per frame; a setting may move it
|
||||
stream_deadline: long = 0
|
||||
stream_budget_left: int = 0
|
||||
gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in
|
||||
gsl_dlss_ok: bool = false # what this adapter can run, from slIsFeatureSupported
|
||||
gsl_rr_ok: bool = false
|
||||
gsl_fg_ok: bool = false
|
||||
gsl_reflex_ok: bool = false
|
||||
gsl_pcl_ok: bool = false
|
||||
gsl_feats: bytes = null # kept alive: Preferences points at it
|
||||
gsl_logdir: bytes = null # and at this
|
||||
gsl_vp: bytes = null # sl::ViewportHandle 0, the one view
|
||||
gsl_tok_buf: bytes = null
|
||||
gsl_idx_buf: bytes = null
|
||||
gsl_token: pointer = null # this frame's sl::FrameToken, owned by Streamline
|
||||
gsl_frame_n: int = 0
|
||||
gsl_said: bool = false # one failure message, not one a frame
|
||||
gsl_fresh: bool = false # a token was taken since the last frame started (a present took it)
|
||||
gsl_reflex_mode: int = 0 # 0 off, 1 low latency, 2 low latency + boost
|
||||
gsl_reflex_applied: int = -1
|
||||
gsl_f_sleep: pointer = null
|
||||
gsl_f_marker: pointer = null
|
||||
gsl_dlss_mode: int = 0 # 0 off, 1 DLAA, 2 quality, 3 balanced, 4 performance
|
||||
gsl_opts: bytes = null
|
||||
gsl_opt_mode: int = -1
|
||||
gsl_opt_w: int = 0
|
||||
gsl_opt_h: int = 0
|
||||
gsl_rw: int = 0 # the render size DLSS asked for
|
||||
gsl_rh: int = 0
|
||||
gsl_set_mode: int = -1
|
||||
gsl_set_w: int = 0
|
||||
gsl_set_h: int = 0
|
||||
gsl_mv: Target = null
|
||||
gsl_out: Target = null
|
||||
gsl_consts: bytes = null
|
||||
gsl_tags: bytes = null
|
||||
gsl_res: bytes = null
|
||||
gsl_inputs: bytes = null
|
||||
gsl_prev_vp: floats = null
|
||||
gsl_reset: bool = true
|
||||
gsl_jitter_x: float = 0.0 # float bits, NDC offsets the projection carries this frame
|
||||
gsl_jitter_y: float = 0.0
|
||||
gsl_jpx: float = 0.0 # float bits, the same in pixels
|
||||
gsl_jpy: float = 0.0
|
||||
gsl_eval_ok: bool = true # the last evaluate worked: only then is the next frame jittered
|
||||
TERRAIN_HALF: int = 4096 # world half-size in metres (8 km square)
|
||||
cd_mesh: Mesh = null
|
||||
cd_range: floats = null # float bits: how far each level is drawn
|
||||
cd_min: []floats = null # per level: min height of each patch (float bits)
|
||||
cd_max: []floats = null
|
||||
cd_draws: int = 0
|
||||
cd_far_draws: int = 0
|
||||
cd_near_draws: int = 0
|
||||
ter_force_far: bool = false
|
||||
ter_no_split: bool = false
|
||||
ter_force_near: bool = false
|
||||
ter_skip: bool = false
|
||||
ter_height_tex: int = 0
|
||||
ter_heights: floats = null # CPU copy, float bits, TERRAIN_RES^2
|
||||
ter_reflect: bool = false # drawing the reflection: the mid mesh is plenty
|
||||
ter_prog: int = 0
|
||||
ter_prog_far: int = 0
|
||||
ter_prog_near: int = 0 # the detailed tier alone (NEAR_ONLY)
|
||||
ter_sun_prog: int = 0 # tersun.frag: sun visibility into a screen buffer
|
||||
ter_sun_inline: bool = false # the ground's shader reads the cascades itself (Vulkan); no sun pass
|
||||
ter_sun_tgt: Target = null # that buffer, at the frame's size
|
||||
ter_sun_refl: Target = null # and at the reflection's, which is smaller
|
||||
ter_sun_dumped: bool = false
|
||||
ter_sun_checked: bool = false
|
||||
ter_sun_done: bool = false # the caller already ran the pass (so it can time it)
|
||||
ter_sun_pass: bool = false # selection is drawing the visibility pass
|
||||
ter_sun_tex: int = 0
|
||||
ter_prog_cur: int = 0 # the program currently bound during selection
|
||||
ter_smooth: bool = false # generate the analytic test ground instead of a survey
|
||||
ter_far_split: float = 0.0 # metres: beyond this the terrain takes its cheap far path (R3D_TFAR)
|
||||
ter_far_band: float = 0.0 # half-width of the near/far blend (R3D_TBAND)
|
||||
ter_snow_line: float = 0.0
|
||||
ter_wet: float = 0.0
|
||||
ter_tex: words = null # 11 material textures (see terrain_bind)
|
||||
ter_ox: float = 0.0 # world x/z of the terrain centre (float bits)
|
||||
ter_oz: float = 0.0
|
||||
ter_dem_tex: int = 0 # a real height map (16-bit), or 0 for the procedural valley
|
||||
ter_dem_blur: float = 0.0 # gaussian texels applied to the survey (0 for lidar; ~3 for 30 m data)
|
||||
ter_dem_min: float = 0.0
|
||||
ter_dem_max: float = 0.0
|
||||
ter_dem_base: float = 0.0
|
||||
ter_ortho_tex: int = 0 # a photograph of the same window, draped with distance
|
||||
ter_carpet: int = 0 # the distant-grass carpet (carpet_bake), 0 = none
|
||||
ter_shadow_tex: int = 0 # height-field sun shadow: RG32F (lowest lit height, occluder distance)
|
||||
ter_shadow_yaw: float = 1000000000.0 # the sky yaw it was baked for
|
||||
ter_shadow_gen: int = -1 # the daylight generation it was baked for
|
||||
ter_shadow_prog: int = 0
|
||||
ter_lake_level: float = 0.0 # a lake carved into the height map (float bits; ex = 0 → none)
|
||||
ter_lake_cx: float = 0.0
|
||||
ter_lake_cz: float = 0.0
|
||||
ter_lake_ex: float = 0.0
|
||||
ter_lake_ez: float = 0.0
|
||||
ter_lake_carve: bool = true
|
||||
ter_sea_level: float = 0.0
|
||||
ter_sea_set: bool = false
|
||||
ter_isle_cx: float = 0.0
|
||||
ter_isle_cz: float = 0.0
|
||||
ter_isle_r: float = 0.0
|
||||
ter_isle_fall: float = 0.0
|
||||
ter_isle_mode: int = 0
|
||||
ter_ortho_px: pointer = null # the photograph on the CPU (RGB8, TERRAIN_RES^2) for placement rules
|
||||
ter_ortho_w: int = 0
|
||||
ter_ortho_c: int = 3
|
||||
ter_h_scale: float = 0.0
|
||||
ter_o_scale: float = 0.0
|
||||
ter_bw: floats = null
|
||||
ter_wire: bool = false
|
||||
ter_printed: bool = false
|
||||
ter_scr: floats = null
|
||||
tex_w: int = 0 # the last decoded image
|
||||
tex_h: int = 0
|
||||
tex_channels: int = 0
|
||||
tex_depth: int = 0 # bits per sample (8 or 16)
|
||||
tex_file_len: int = 0
|
||||
tex_anisotropy: fixed = 16.0
|
||||
tex_size_ids: []int = new []int
|
||||
tex_size_ws: []int = new []int
|
||||
tex_size_hs: []int = new []int
|
||||
hdr_max_lum: float = 0.0 # float bits of the brightest texel (sun finding)
|
||||
hdr_max_x: int = 0
|
||||
hdr_max_y: int = 0
|
||||
hdr_sun_r: float = 0.0 # irradiance (float bits) of everything above the IBL clip: the sun
|
||||
hdr_sun_g: float = 0.0
|
||||
hdr_sun_b: float = 0.0
|
||||
hdr_clip: float = 0.0 # float bits; texels above this (per channel) feed the sun, not the IBL
|
||||
tex_dump_alpha: bool = false
|
||||
water_mesh: Mesh = null
|
||||
water_prog: int = 0
|
||||
water_level: float = 0.0
|
||||
water_cx: float = 0.0
|
||||
water_cz: float = 0.0
|
||||
water_ex: float = 0.0
|
||||
water_ez: float = 0.0
|
||||
water_on: bool = false
|
||||
wt_wade_x: float = 0.0
|
||||
wt_wade_z: float = 0.0
|
||||
wt_wade_s: float = 0.0
|
||||
water_refl: Target = null # the world mirrored in the surface, half resolution
|
||||
water_refl_div: int = 2 # R3D_REFLDIV overrides: 2 = half res, 4 = quarter
|
||||
water_saved: floats = null # the real camera's matrices, restored after the pass
|
||||
water_saved_vp: floats = null # views of its second and third matrices
|
||||
water_saved_ivp: floats = null
|
||||
water_dumped: bool = false
|
||||
wb_n: int = 0
|
||||
wb_level: floats = null
|
||||
wb_cx: floats = null
|
||||
wb_cz: floats = null
|
||||
wb_ex: floats = null
|
||||
wb_ez: floats = null
|
||||
wb_reflect: words = null
|
||||
wb_primary: int = -1 # the body the reflection pass mirrors, or -1
|
||||
wb_cells: floats = null # x0, z0, x1, z1 of each rectangle with water showing
|
||||
wb_ncells: int = -1 # -1: not built for the current mirrored body
|
||||
wb_blocks: words = null # x0, z0, x1, z1, first cell, cells past the last - of each block
|
||||
wb_nblocks: int = 0
|
||||
wb_refl_always: int = -1
|
||||
wb_dbg: int = -1
|
||||
wb_build_us: long = 0 # how long the last water_cells_build took (R3D_REFL_DBG prints it)
|
||||
}
|
||||
function r3d_env_has(name: string) -> bool {
|
||||
if is_windowed() and not r3d_dev() { return false }
|
||||
function r3d_dev(render3d_st: mut Render3dState) -> bool {
|
||||
if render3d_st.r3d_dev_v < 0 { if Os.has_env("R3D_DEV") and Os.env("R3D_DEV") != "0" { render3d_st.r3d_dev_v = 1 } else { render3d_st.r3d_dev_v = 0 } }
|
||||
return render3d_st.r3d_dev_v == 1
|
||||
}
|
||||
function r3d_env_has(render3d_st: mut Render3dState, name: string) -> bool {
|
||||
if is_windowed() and not r3d_dev(render3d_st) { return false }
|
||||
return Os.has_env(name)
|
||||
}
|
||||
function r3d_env(name: string) -> string {
|
||||
if is_windowed() and not r3d_dev() { return "" }
|
||||
function r3d_env(render3d_st: mut Render3dState, name: string) -> string {
|
||||
if is_windowed() and not r3d_dev(render3d_st) { return "" }
|
||||
return Os.env(name)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -22,17 +22,6 @@ property Model {
|
|||
skin: Skin # the skeleton, for a skinned node (skin.ludic); null for a rigid model
|
||||
}
|
||||
|
||||
var gltf_dir: string = null
|
||||
var gltf_bin: pointer = null
|
||||
var gltf_doc: Val = null
|
||||
var gltf_count: int = 0
|
||||
var gltf_ctype: int = 0
|
||||
var gltf_comps: int = 0
|
||||
var gltf_tex_paths: []pointer = null
|
||||
var gltf_tex_ids: words = null
|
||||
var gltf_tex_n: int = 0
|
||||
var gltf_white: int = 0
|
||||
var gltf_flat: int = 0
|
||||
|
||||
# a JSON number as float bits (ints and fixed-point decimals both)
|
||||
function jnum(v: Val) -> float {
|
||||
|
|
@ -48,41 +37,35 @@ function jint(v: Val, key: pointer, fallback: int) -> int {
|
|||
# lod_export.py) carries the material names but no images — the blend links textures
|
||||
# that are not downloaded with it — so its primitives take the textures the LOD0
|
||||
# download's material of the same name loaded.
|
||||
var gltf_mat_names: []string = null
|
||||
var gltf_mat_diff: words = null
|
||||
var gltf_mat_nrm: words = null
|
||||
var gltf_mat_arm: words = null
|
||||
var gltf_mat_n: int = 0
|
||||
function gltf_mat_find(name: string) -> int {
|
||||
if gltf_mat_names == null { return -1 }
|
||||
for i in 0 .. gltf_mat_n { if gltf_mat_names[i] == name { return i } }
|
||||
function gltf_mat_find(render3d_st: Render3dState, name: string) -> int {
|
||||
if render3d_st.gltf_mat_names == null { return -1 }
|
||||
for i in 0 .. render3d_st.gltf_mat_n { if render3d_st.gltf_mat_names[i] == name { return i } }
|
||||
return -1
|
||||
}
|
||||
function gltf_mat_remember(name: string, diff: int, nrm: int, arm: int) -> void {
|
||||
if gltf_mat_names == null { gltf_mat_names = new []string; gltf_mat_diff = words(256); gltf_mat_nrm = words(256); gltf_mat_arm = words(256) }
|
||||
if gltf_mat_find(name) >= 0 or gltf_mat_n >= 256 { return }
|
||||
push(gltf_mat_names, name); gltf_mat_diff[gltf_mat_n] = diff; gltf_mat_nrm[gltf_mat_n] = nrm; gltf_mat_arm[gltf_mat_n] = arm
|
||||
gltf_mat_n += 1
|
||||
function gltf_mat_remember(render3d_st: mut Render3dState, name: string, diff: int, nrm: int, arm: int) -> void {
|
||||
if render3d_st.gltf_mat_names == null { render3d_st.gltf_mat_names = new []string; render3d_st.gltf_mat_diff = words(256); render3d_st.gltf_mat_nrm = words(256); render3d_st.gltf_mat_arm = words(256) }
|
||||
if gltf_mat_find(render3d_st, name) >= 0 or render3d_st.gltf_mat_n >= 256 { return }
|
||||
push(render3d_st.gltf_mat_names, name); render3d_st.gltf_mat_diff[render3d_st.gltf_mat_n] = diff; render3d_st.gltf_mat_nrm[render3d_st.gltf_mat_n] = nrm; render3d_st.gltf_mat_arm[render3d_st.gltf_mat_n] = arm
|
||||
render3d_st.gltf_mat_n += 1
|
||||
}
|
||||
|
||||
var gltf_cutout: bool = false # the material being loaded is alpha-blended (a cut-out atlas)
|
||||
function gltf_texture(uri: string, srgb: bool) -> int {
|
||||
if gltf_tex_paths == null { gltf_tex_paths = new []pointer; gltf_tex_ids = words(256) }
|
||||
function gltf_texture(render3d_st: mut Render3dState, uri: string, srgb: bool) -> int {
|
||||
if render3d_st.gltf_tex_paths == null { render3d_st.gltf_tex_paths = new []pointer; render3d_st.gltf_tex_ids = words(256) }
|
||||
var png: string = uri
|
||||
let n = len(uri)
|
||||
if n > 4 and uri[n - 4] == '.' and uri[n - 3] == 'j' { png = uri[0 .. n - 4] + ".png" }
|
||||
let path = gltf_dir + "/" + png
|
||||
let path = render3d_st.gltf_dir + "/" + png
|
||||
var i = 0
|
||||
while i < gltf_tex_n { if gltf_tex_paths[i] == path { return gltf_tex_ids[i] }; i += 1 }
|
||||
while i < render3d_st.gltf_tex_n { if render3d_st.gltf_tex_paths[i] == path { return render3d_st.gltf_tex_ids[i] }; i += 1 }
|
||||
var dil = 0
|
||||
if gltf_cutout { dil = 24 }
|
||||
let id = tex_load_ex(path, srgb, dil)
|
||||
if gltf_tex_n < 256 { push(gltf_tex_paths, path); gltf_tex_ids[gltf_tex_n] = id; gltf_tex_n += 1 }
|
||||
if render3d_st.gltf_cutout { dil = 24 }
|
||||
let id = tex_load_ex(render3d_st, path, srgb, dil)
|
||||
if render3d_st.gltf_tex_n < 256 { push(render3d_st.gltf_tex_paths, path); render3d_st.gltf_tex_ids[render3d_st.gltf_tex_n] = id; render3d_st.gltf_tex_n += 1 }
|
||||
return id
|
||||
}
|
||||
|
||||
# the image uri behind materials[m].<slot>.index, or null
|
||||
function gltf_mat_uri(mat: Val, slot: pointer) -> string {
|
||||
function gltf_mat_uri(render3d_st: Render3dState, mat: Val, slot: pointer) -> string {
|
||||
var holder = mat
|
||||
if slot == "baseColorTexture" or slot == "metallicRoughnessTexture" {
|
||||
if value_has(mat, "pbrMetallicRoughness") == 0 { return null }
|
||||
|
|
@ -90,84 +73,84 @@ function gltf_mat_uri(mat: Val, slot: pointer) -> string {
|
|||
}
|
||||
if value_has(holder, slot) == 0 { return null }
|
||||
let ti = value_as_int(value_get(value_get(holder, slot), "index"))
|
||||
let tex = value_at(value_get(gltf_doc, "textures"), ti)
|
||||
let tex = value_at(value_get(render3d_st.gltf_doc, "textures"), ti)
|
||||
let src = value_as_int(value_get(tex, "source"))
|
||||
let img = value_at(value_get(gltf_doc, "images"), src)
|
||||
let img = value_at(value_get(render3d_st.gltf_doc, "images"), src)
|
||||
return value_as_str(value_get(img, "uri"))
|
||||
}
|
||||
|
||||
# read one accessor's bytes from the .bin; sets gltf_count / gltf_ctype / gltf_comps
|
||||
function gltf_accessor(idx: int) -> pointer {
|
||||
let acc = value_at(value_get(gltf_doc, "accessors"), idx)
|
||||
let bv = value_at(value_get(gltf_doc, "bufferViews"), jint(acc, "bufferView", 0))
|
||||
function gltf_accessor(render3d_st: mut Render3dState, idx: int) -> pointer {
|
||||
let acc = value_at(value_get(render3d_st.gltf_doc, "accessors"), idx)
|
||||
let bv = value_at(value_get(render3d_st.gltf_doc, "bufferViews"), jint(acc, "bufferView", 0))
|
||||
let off = jint(bv, "byteOffset", 0) + jint(acc, "byteOffset", 0)
|
||||
gltf_count = jint(acc, "count", 0)
|
||||
gltf_ctype = jint(acc, "componentType", 5126)
|
||||
render3d_st.gltf_count = jint(acc, "count", 0)
|
||||
render3d_st.gltf_ctype = jint(acc, "componentType", 5126)
|
||||
let ty = value_as_str(value_get(acc, "type"))
|
||||
gltf_comps = 1
|
||||
if ty == "VEC2" { gltf_comps = 2 }
|
||||
if ty == "VEC3" { gltf_comps = 3 }
|
||||
if ty == "VEC4" { gltf_comps = 4 }
|
||||
if ty == "MAT2" { gltf_comps = 4 }
|
||||
if ty == "MAT3" { gltf_comps = 9 }
|
||||
if ty == "MAT4" { gltf_comps = 16 } # a skin's inverse bind matrices
|
||||
render3d_st.gltf_comps = 1
|
||||
if ty == "VEC2" { render3d_st.gltf_comps = 2 }
|
||||
if ty == "VEC3" { render3d_st.gltf_comps = 3 }
|
||||
if ty == "VEC4" { render3d_st.gltf_comps = 4 }
|
||||
if ty == "MAT2" { render3d_st.gltf_comps = 4 }
|
||||
if ty == "MAT3" { render3d_st.gltf_comps = 9 }
|
||||
if ty == "MAT4" { render3d_st.gltf_comps = 16 } # a skin's inverse bind matrices
|
||||
var csz = 4
|
||||
if gltf_ctype == 5123 or gltf_ctype == 5122 { csz = 2 }
|
||||
if gltf_ctype == 5121 or gltf_ctype == 5120 { csz = 1 }
|
||||
let n = gltf_count * gltf_comps * csz
|
||||
if render3d_st.gltf_ctype == 5123 or render3d_st.gltf_ctype == 5122 { csz = 2 }
|
||||
if render3d_st.gltf_ctype == 5121 or render3d_st.gltf_ctype == 5120 { csz = 1 }
|
||||
let n = render3d_st.gltf_count * render3d_st.gltf_comps * csz
|
||||
let buf = bytes(n + 8)
|
||||
file_seek(gltf_bin, off, 0)
|
||||
file_read(gltf_bin, buf, n)
|
||||
file_seek(render3d_st.gltf_bin, off, 0)
|
||||
file_read(render3d_st.gltf_bin, buf, n)
|
||||
return buf
|
||||
}
|
||||
|
||||
function gltf_attrib(m: Mesh, attrs: Val, name: pointer, loc: int) -> bool {
|
||||
function gltf_attrib(render3d_st: mut Render3dState, m: Mesh, attrs: Val, name: pointer, loc: int) -> bool {
|
||||
if value_has(attrs, name) == 0 { return false }
|
||||
let data = gltf_accessor(value_as_int(value_get(attrs, name)))
|
||||
gpu_mesh_vertices(m, data, gltf_count * gltf_comps * 4, GPU_STATIC)
|
||||
gpu_mesh_attr(m, loc, gltf_comps, GPU_F32, 0, 0, false)
|
||||
let data = gltf_accessor(render3d_st, value_as_int(value_get(attrs, name)))
|
||||
gpu_mesh_vertices(render3d_st, m, data, render3d_st.gltf_count * render3d_st.gltf_comps * 4, GPU_STATIC)
|
||||
gpu_mesh_attr(render3d_st, m, loc, render3d_st.gltf_comps, GPU_F32, 0, 0, false)
|
||||
free(data)
|
||||
return true
|
||||
}
|
||||
|
||||
function gltf_prim(p: Val) -> Prim {
|
||||
function gltf_prim(render3d_st: mut Render3dState, p: Val) -> Prim {
|
||||
let pr = new Prim
|
||||
let m = gpu_mesh_new()
|
||||
let m = gpu_mesh_new(render3d_st)
|
||||
let attrs = value_get(p, "attributes")
|
||||
gltf_attrib(m, attrs, "POSITION", 0)
|
||||
let nverts = gltf_count
|
||||
gltf_attrib(m, attrs, "NORMAL", 1)
|
||||
gltf_attrib(m, attrs, "TEXCOORD_0", 2)
|
||||
skin_attribs(m, attrs) # JOINTS_0 / WEIGHTS_0 onto 5 / 6, when the mesh has them
|
||||
let idx = gltf_accessor(value_as_int(value_get(p, "indices")))
|
||||
gltf_attrib(render3d_st, m, attrs, "POSITION", 0)
|
||||
let nverts = render3d_st.gltf_count
|
||||
gltf_attrib(render3d_st, m, attrs, "NORMAL", 1)
|
||||
gltf_attrib(render3d_st, m, attrs, "TEXCOORD_0", 2)
|
||||
skin_attribs(render3d_st, m, attrs) # JOINTS_0 / WEIGHTS_0 onto 5 / 6, when the mesh has them
|
||||
let idx = gltf_accessor(render3d_st, value_as_int(value_get(p, "indices")))
|
||||
var isz = 4
|
||||
if gltf_ctype == 5123 { isz = 2 }
|
||||
gpu_mesh_indices(m, idx, gltf_count * isz, isz)
|
||||
if render3d_st.gltf_ctype == 5123 { isz = 2 }
|
||||
gpu_mesh_indices(render3d_st, m, idx, render3d_st.gltf_count * isz, isz)
|
||||
free(idx)
|
||||
m.count = gltf_count
|
||||
gpu_mesh_done(m)
|
||||
m.count = render3d_st.gltf_count
|
||||
gpu_mesh_done(render3d_st, m)
|
||||
pr.mesh = m
|
||||
pr.verts = nverts
|
||||
# material textures
|
||||
if gltf_white == 0 { gltf_white = tex_solid(200, 200, 200, 255); gltf_flat = tex_solid(128, 128, 255, 255) }
|
||||
pr.diff = gltf_white; pr.nrm = gltf_flat; pr.arm = gltf_white
|
||||
if render3d_st.gltf_white == 0 { render3d_st.gltf_white = tex_solid(render3d_st, 200, 200, 200, 255); render3d_st.gltf_flat = tex_solid(render3d_st, 128, 128, 255, 255) }
|
||||
pr.diff = render3d_st.gltf_white; pr.nrm = render3d_st.gltf_flat; pr.arm = render3d_st.gltf_white
|
||||
if value_has(p, "material") != 0 {
|
||||
let mat = value_at(value_get(gltf_doc, "materials"), value_as_int(value_get(p, "material")))
|
||||
gltf_cutout = value_has(mat, "alphaMode") != 0
|
||||
let ud = gltf_mat_uri(mat, "baseColorTexture")
|
||||
let un = gltf_mat_uri(mat, "normalTexture")
|
||||
let ua = gltf_mat_uri(mat, "metallicRoughnessTexture")
|
||||
if ud != null { pr.diff = gltf_texture(ud, true) }
|
||||
if un != null { pr.nrm = gltf_texture(un, false) }
|
||||
if ua != null { pr.arm = gltf_texture(ua, false) }
|
||||
let mat = value_at(value_get(render3d_st.gltf_doc, "materials"), value_as_int(value_get(p, "material")))
|
||||
render3d_st.gltf_cutout = value_has(mat, "alphaMode") != 0
|
||||
let ud = gltf_mat_uri(render3d_st, mat, "baseColorTexture")
|
||||
let un = gltf_mat_uri(render3d_st, mat, "normalTexture")
|
||||
let ua = gltf_mat_uri(render3d_st, mat, "metallicRoughnessTexture")
|
||||
if ud != null { pr.diff = gltf_texture(render3d_st, ud, true) }
|
||||
if un != null { pr.nrm = gltf_texture(render3d_st, un, false) }
|
||||
if ua != null { pr.arm = gltf_texture(render3d_st, ua, false) }
|
||||
var mname: string = null
|
||||
if value_has(mat, "name") != 0 { mname = value_as_str(value_get(mat, "name")) }
|
||||
pr.name = mname
|
||||
if mname != null {
|
||||
if ud != null { gltf_mat_remember(mname, pr.diff, pr.nrm, pr.arm) }
|
||||
if ud != null { gltf_mat_remember(render3d_st, mname, pr.diff, pr.nrm, pr.arm) }
|
||||
else {
|
||||
let k = gltf_mat_find(mname)
|
||||
if k >= 0 { pr.diff = gltf_mat_diff[k]; pr.nrm = gltf_mat_nrm[k]; pr.arm = gltf_mat_arm[k] }
|
||||
let k = gltf_mat_find(render3d_st, mname)
|
||||
if k >= 0 { pr.diff = render3d_st.gltf_mat_diff[k]; pr.nrm = render3d_st.gltf_mat_nrm[k]; pr.arm = render3d_st.gltf_mat_arm[k] }
|
||||
else { print(`gltf: material {mname} has no textures and none were loaded before it`) }
|
||||
}
|
||||
}
|
||||
|
|
@ -176,34 +159,34 @@ function gltf_prim(p: Val) -> Prim {
|
|||
}
|
||||
|
||||
# Load the mesh of the node called `node_name` from dir/file.
|
||||
function gltf_load(dir: string, file: string, node_name: string) -> Model {
|
||||
gltf_dir = dir
|
||||
function gltf_load(render3d_st: mut Render3dState, dir: string, file: string, node_name: string) -> Model {
|
||||
render3d_st.gltf_dir = dir
|
||||
let text = Fs.read_text(dir + "/" + file)
|
||||
if text == null { print(`gltf: cannot read {dir}/{file}`); return null }
|
||||
gltf_doc = Json.parse(text)
|
||||
let buffers = value_get(gltf_doc, "buffers")
|
||||
render3d_st.gltf_doc = Json.parse(text)
|
||||
let buffers = value_get(render3d_st.gltf_doc, "buffers")
|
||||
let bin_uri = value_as_str(value_get(value_at(buffers, 0), "uri"))
|
||||
gltf_bin = file_open(dir + "/" + bin_uri, "rb")
|
||||
if gltf_bin == null { print(`gltf: cannot open {bin_uri}`); return null }
|
||||
let nodes = value_get(gltf_doc, "nodes")
|
||||
render3d_st.gltf_bin = file_open(dir + "/" + bin_uri, "rb")
|
||||
if render3d_st.gltf_bin == null { print(`gltf: cannot open {bin_uri}`); return null }
|
||||
let nodes = value_get(render3d_st.gltf_doc, "nodes")
|
||||
var mesh_idx = -1
|
||||
var skin_idx = -1
|
||||
for i in 0 .. value_count(nodes) {
|
||||
let nd = value_at(nodes, i)
|
||||
if value_as_str(value_get(nd, "name")) == node_name { mesh_idx = jint(nd, "mesh", -1); skin_idx = jint(nd, "skin", -1) }
|
||||
}
|
||||
if mesh_idx < 0 { print(`gltf: no node {node_name} in {file}`); file_close(gltf_bin); return null }
|
||||
if mesh_idx < 0 { print(`gltf: no node {node_name} in {file}`); file_close(render3d_st.gltf_bin); return null }
|
||||
let model = new Model
|
||||
model.prims = new []Prim
|
||||
let mesh = value_at(value_get(gltf_doc, "meshes"), mesh_idx)
|
||||
let mesh = value_at(value_get(render3d_st.gltf_doc, "meshes"), mesh_idx)
|
||||
let prims = value_get(mesh, "primitives")
|
||||
var r2 = 0.0; var ymin = 1000.0; var ymax = -1000.0
|
||||
for i in 0 .. value_count(prims) {
|
||||
let p = value_at(prims, i)
|
||||
push(model.prims, gltf_prim(p))
|
||||
model.tris += gltf_count / 3
|
||||
push(model.prims, gltf_prim(render3d_st, p))
|
||||
model.tris += render3d_st.gltf_count / 3
|
||||
# bounds from the accessor min/max
|
||||
let acc = value_at(value_get(gltf_doc, "accessors"), value_as_int(value_get(value_get(p, "attributes"), "POSITION")))
|
||||
let acc = value_at(value_get(render3d_st.gltf_doc, "accessors"), value_as_int(value_get(value_get(p, "attributes"), "POSITION")))
|
||||
let mn = value_get(acc, "min"); let mx = value_get(acc, "max")
|
||||
let x0 = Math.abs(jnum(value_at(mn, 0))); let x1 = Math.abs(jnum(value_at(mx, 0)))
|
||||
let z0 = Math.abs(jnum(value_at(mn, 2))); let z1 = Math.abs(jnum(value_at(mx, 2)))
|
||||
|
|
@ -214,8 +197,8 @@ function gltf_load(dir: string, file: string, node_name: string) -> Model {
|
|||
if y0 < ymin { ymin = y0 }
|
||||
if y1 > ymax { ymax = y1 }
|
||||
}
|
||||
if skin_idx >= 0 { model.skin = skin_load(skin_idx) }
|
||||
file_close(gltf_bin)
|
||||
if skin_idx >= 0 { model.skin = skin_load(render3d_st, skin_idx) }
|
||||
file_close(render3d_st.gltf_bin)
|
||||
model.radius = Math.sqrt(r2)
|
||||
model.ymin = ymin
|
||||
model.height = ymax - ymin
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -34,7 +34,6 @@ property GpuVariant {
|
|||
i_type: []string
|
||||
}
|
||||
|
||||
var gpu_variants: []GpuVariant = null
|
||||
|
||||
function gpu_variant_new(id: string, vs: string, fs: string, defs: string) -> GpuVariant {
|
||||
let v = new GpuVariant
|
||||
|
|
@ -47,8 +46,8 @@ function gpu_variant_new(id: string, vs: string, fs: string, defs: string) -> Gp
|
|||
}
|
||||
|
||||
# Read a manifest; returns how many programs it holds (0: none, or no file).
|
||||
function gpu_manifest_load(path: string) -> int {
|
||||
gpu_variants = new []GpuVariant
|
||||
function gpu_manifest_load(render3d_st: mut Render3dState, path: string) -> int {
|
||||
render3d_st.gpu_variants = new []GpuVariant
|
||||
let text = Fs.read_text(path)
|
||||
if text == null { return 0 }
|
||||
let lines = Text.split(text, "\n")
|
||||
|
|
@ -68,7 +67,7 @@ function gpu_manifest_load(path: string) -> int {
|
|||
k += 1
|
||||
}
|
||||
cur = gpu_variant_new(f[1], f[2], f[3], defs)
|
||||
push(gpu_variants, cur)
|
||||
push(render3d_st.gpu_variants, cur)
|
||||
continue
|
||||
}
|
||||
if cur == null { continue }
|
||||
|
|
@ -85,18 +84,18 @@ function gpu_manifest_load(path: string) -> int {
|
|||
push(cur.i_name, f[2]); push(cur.i_loc, Text.to_int(f[3])); push(cur.i_type, f[4])
|
||||
}
|
||||
}
|
||||
return len(gpu_variants)
|
||||
return len(render3d_st.gpu_variants)
|
||||
}
|
||||
|
||||
# the variant for a program as r3d_program names it (defines with real newlines)
|
||||
function gpu_variant_find(vs: string, fs: string, defines: string) -> GpuVariant {
|
||||
return gpu_variant_find_key(vs, fs, Text.replace(defines, "\n", ";"))
|
||||
function gpu_variant_find(render3d_st: Render3dState, vs: string, fs: string, defines: string) -> GpuVariant {
|
||||
return gpu_variant_find_key(render3d_st, vs, fs, Text.replace(defines, "\n", ";"))
|
||||
}
|
||||
# ... or as variants.list writes it (';' for newlines)
|
||||
function gpu_variant_find_key(vs: string, fs: string, defs: string) -> GpuVariant {
|
||||
if gpu_variants == null { return null }
|
||||
for i in 0 .. len(gpu_variants) {
|
||||
let v = gpu_variants[i]
|
||||
function gpu_variant_find_key(render3d_st: Render3dState, vs: string, fs: string, defs: string) -> GpuVariant {
|
||||
if render3d_st.gpu_variants == null { return null }
|
||||
for i in 0 .. len(render3d_st.gpu_variants) {
|
||||
let v = render3d_st.gpu_variants[i]
|
||||
if v.vs == vs and v.fs == fs and v.defs == defs { return v }
|
||||
}
|
||||
return null
|
||||
|
|
|
|||
|
|
@ -7,38 +7,24 @@
|
|||
# here. Nothing in this file runs unless the Vulkan backend was chosen and came up.
|
||||
|
||||
# ---- the device ---------------------------------------------------------------------------
|
||||
var gvk_ready: bool = false
|
||||
var gvk_inst: pointer = null
|
||||
var gvk_pd: pointer = null
|
||||
var gvk_dev: pointer = null
|
||||
var gvk_queue: pointer = null
|
||||
var gvk_family: int = -1
|
||||
var gvk_mp: bytes = null # VkPhysicalDeviceMemoryProperties
|
||||
var gvk_device_name: string = ""
|
||||
var gvk_why: string = "" # why the device did not come up, for the fallback notice
|
||||
var gvk_max_aniso: int = 0x3F800000 # the device's anisotropy limit as float bits; 1.0 when it has none
|
||||
var gvk_want_surface: bool = false # set before gvk_init when the renderer will present to a window
|
||||
var gvk_has_surface: bool = false # the instance and device can make a surface and a swapchain
|
||||
|
||||
function gvk_fail(what: string, r: int) -> bool {
|
||||
gvk_why = `{what} failed (VkResult {r})`
|
||||
gvk_note(`r3d: vulkan: {gvk_why}`)
|
||||
function gvk_fail(render3d_st: mut Render3dState, what: string, r: int) -> bool {
|
||||
render3d_st.gvk_why = `{what} failed (VkResult {r})`
|
||||
gvk_note(render3d_st, `r3d: vulkan: {render3d_st.gvk_why}`)
|
||||
return false
|
||||
}
|
||||
|
||||
# A Vulkan failure is printed, and with R3D_VK_ERRLOG=<file> also appended to that file, opened and
|
||||
# closed for each line: stdout is buffered, and a crash straight after a failure (an image that was
|
||||
# never made, then used) used to take the one line that explained it with it.
|
||||
var gvk_errlog: string = null
|
||||
var gvk_errlog_read: bool = false
|
||||
function gvk_note(msg: string) -> void {
|
||||
function gvk_note(render3d_st: mut Render3dState, msg: string) -> void {
|
||||
print(msg)
|
||||
if not gvk_errlog_read {
|
||||
gvk_errlog_read = true
|
||||
if r3d_env_has("R3D_VK_ERRLOG") { gvk_errlog = r3d_env("R3D_VK_ERRLOG") }
|
||||
if not render3d_st.gvk_errlog_read {
|
||||
render3d_st.gvk_errlog_read = true
|
||||
if r3d_env_has(render3d_st, "R3D_VK_ERRLOG") { render3d_st.gvk_errlog = r3d_env(render3d_st, "R3D_VK_ERRLOG") }
|
||||
}
|
||||
if gvk_errlog == null { return }
|
||||
let f = file_open(gvk_errlog, "ab")
|
||||
if render3d_st.gvk_errlog == null { return }
|
||||
let f = file_open(render3d_st.gvk_errlog, "ab")
|
||||
if f == null { return }
|
||||
let line = msg + "\n"
|
||||
file_write(f, line, len(line))
|
||||
|
|
@ -56,16 +42,12 @@ function gvk_ext_in(props: bytes, n: int, want: string) -> bool {
|
|||
# Instance, the first discrete GPU (else the first listed), its first graphics queue, and a
|
||||
# device with the Tier 1 floor switched on. False, with gvk_why set, if any of it is missing:
|
||||
# the caller stays on OpenGL.
|
||||
var gvk_has_mdi: bool = false # multiDrawIndirect + drawIndirectFirstInstance
|
||||
var gvk_has_dic: bool = false # drawIndirectCount
|
||||
var gvk_has_mesh: bool = false # VK_EXT_mesh_shader with its meshShader feature on
|
||||
var gvk_mesh_fn: pointer = null # vkCmdDrawMeshTasksEXT: the device's own, the interposer exports none
|
||||
|
||||
function gvk_init() -> bool {
|
||||
if gvk_ready { return true }
|
||||
gsl_boot()
|
||||
if Vk.open() == 0 { gvk_why = "no Vulkan loader"; return false }
|
||||
gsl_init()
|
||||
function gvk_init(render3d_st: mut Render3dState) -> bool {
|
||||
if render3d_st.gvk_ready { return true }
|
||||
gsl_boot(render3d_st)
|
||||
if Vk.open() == 0 { render3d_st.gvk_why = "no Vulkan loader"; return false }
|
||||
gsl_init(render3d_st)
|
||||
|
||||
let cnt = bytes(4)
|
||||
Vk.put_i32(cnt, 0, 0)
|
||||
|
|
@ -91,50 +73,50 @@ function gvk_init() -> bool {
|
|||
Vk.put_i32(ici, VkInstanceCreateInfo_flags, VK_INSTANCE_CREATE_ENUMERATE_PORTABILITY_BIT_KHR)
|
||||
}
|
||||
# a window's surface: only the Win32 one is built, so a window elsewhere stays on OpenGL
|
||||
gvk_has_surface = gvk_want_surface and gvk_ext_in(iexts, nie, VK_KHR_SURFACE_EXTENSION_NAME) and gvk_ext_in(iexts, nie, VK_KHR_WIN32_SURFACE_EXTENSION_NAME)
|
||||
if gvk_want_surface and not gvk_has_surface { gvk_why = "no Win32 surface in this Vulkan instance"; return false }
|
||||
if gvk_has_surface {
|
||||
render3d_st.gvk_has_surface = render3d_st.gvk_want_surface and gvk_ext_in(iexts, nie, VK_KHR_SURFACE_EXTENSION_NAME) and gvk_ext_in(iexts, nie, VK_KHR_WIN32_SURFACE_EXTENSION_NAME)
|
||||
if render3d_st.gvk_want_surface and not render3d_st.gvk_has_surface { render3d_st.gvk_why = "no Win32 surface in this Vulkan instance"; return false }
|
||||
if render3d_st.gvk_has_surface {
|
||||
Vk.put_ptr(iext_names, n_iext * 8, VK_KHR_SURFACE_EXTENSION_NAME); n_iext += 1
|
||||
Vk.put_ptr(iext_names, n_iext * 8, VK_KHR_WIN32_SURFACE_EXTENSION_NAME); n_iext += 1
|
||||
}
|
||||
# HDR output: an instance only lists the HDR colour spaces when it asks for them
|
||||
gvk_has_colorspace = gvk_has_surface and gvk_ext_in(iexts, nie, VK_EXT_SWAPCHAIN_COLOR_SPACE_EXTENSION_NAME)
|
||||
if gvk_has_colorspace { Vk.put_ptr(iext_names, n_iext * 8, VK_EXT_SWAPCHAIN_COLOR_SPACE_EXTENSION_NAME); n_iext += 1 }
|
||||
render3d_st.gvk_has_colorspace = render3d_st.gvk_has_surface and gvk_ext_in(iexts, nie, VK_EXT_SWAPCHAIN_COLOR_SPACE_EXTENSION_NAME)
|
||||
if render3d_st.gvk_has_colorspace { Vk.put_ptr(iext_names, n_iext * 8, VK_EXT_SWAPCHAIN_COLOR_SPACE_EXTENSION_NAME); n_iext += 1 }
|
||||
if n_iext > 0 {
|
||||
Vk.put_i32(ici, VkInstanceCreateInfo_enabledExtensionCount, n_iext)
|
||||
Vk.put_ptr(ici, VkInstanceCreateInfo_ppEnabledExtensionNames, iext_names)
|
||||
}
|
||||
let out = bytes(8)
|
||||
var r = Vk.create_instance(ici, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkCreateInstance", r) }
|
||||
gvk_inst = Vk.get_ptr(out, 0)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateInstance", r) }
|
||||
render3d_st.gvk_inst = Vk.get_ptr(out, 0)
|
||||
|
||||
Vk.put_i32(cnt, 0, 0)
|
||||
Vk.enumerate_physical_devices(gvk_inst, cnt, null)
|
||||
Vk.enumerate_physical_devices(render3d_st.gvk_inst, cnt, null)
|
||||
let nd = Vk.get_i32(cnt, 0)
|
||||
if nd == 0 { gvk_why = "no Vulkan device"; return false }
|
||||
if nd == 0 { render3d_st.gvk_why = "no Vulkan device"; return false }
|
||||
let devs = bytes(nd * 8 + 8)
|
||||
Vk.enumerate_physical_devices(gvk_inst, cnt, devs)
|
||||
Vk.enumerate_physical_devices(render3d_st.gvk_inst, cnt, devs)
|
||||
let props = bytes(VkPhysicalDeviceProperties_sizeof)
|
||||
var pick = 0
|
||||
for d in 0 .. nd {
|
||||
Vk.get_physical_device_properties(Vk.get_ptr(devs, d * 8), props)
|
||||
if Vk.get_i32(props, VkPhysicalDeviceProperties_deviceType) == VK_PHYSICAL_DEVICE_TYPE_DISCRETE_GPU { pick = d; break }
|
||||
}
|
||||
gvk_pd = Vk.get_ptr(devs, pick * 8)
|
||||
Vk.get_physical_device_properties(gvk_pd, props)
|
||||
gvk_device_name = string(Vk.at(props, VkPhysicalDeviceProperties_deviceName))
|
||||
render3d_st.gvk_pd = Vk.get_ptr(devs, pick * 8)
|
||||
Vk.get_physical_device_properties(render3d_st.gvk_pd, props)
|
||||
render3d_st.gvk_device_name = string(Vk.at(props, VkPhysicalDeviceProperties_deviceName))
|
||||
|
||||
Vk.put_i32(cnt, 0, 0)
|
||||
Vk.get_physical_device_queue_family_properties(gvk_pd, cnt, null)
|
||||
Vk.get_physical_device_queue_family_properties(render3d_st.gvk_pd, cnt, null)
|
||||
let nq = Vk.get_i32(cnt, 0)
|
||||
let qprops = bytes(nq * VkQueueFamilyProperties_sizeof + 8)
|
||||
Vk.get_physical_device_queue_family_properties(gvk_pd, cnt, qprops)
|
||||
Vk.get_physical_device_queue_family_properties(render3d_st.gvk_pd, cnt, qprops)
|
||||
for q in 0 .. nq {
|
||||
let flags = Vk.get_i32(qprops, q * VkQueueFamilyProperties_sizeof + VkQueueFamilyProperties_queueFlags)
|
||||
if gvk_family < 0 and (flags & VK_QUEUE_GRAPHICS_BIT) != 0 { gvk_family = q }
|
||||
if render3d_st.gvk_family < 0 and (flags & VK_QUEUE_GRAPHICS_BIT) != 0 { render3d_st.gvk_family = q }
|
||||
}
|
||||
if gvk_family < 0 { gvk_why = `{gvk_device_name} has no graphics queue`; return false }
|
||||
if render3d_st.gvk_family < 0 { render3d_st.gvk_why = `{render3d_st.gvk_device_name} has no graphics queue`; return false }
|
||||
|
||||
# what the device offers, then the same structs handed back asking for the floor
|
||||
let f13 = bytes(VkPhysicalDeviceVulkan13Features_sizeof)
|
||||
|
|
@ -148,16 +130,16 @@ function gvk_init() -> bool {
|
|||
Vk.zero(f2, VkPhysicalDeviceFeatures2_sizeof)
|
||||
Vk.put_i32(f2, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
|
||||
Vk.put_ptr(f2, VkPhysicalDeviceFeatures2_pNext, f12)
|
||||
Vk.get_physical_device_features2(gvk_pd, f2)
|
||||
Vk.get_physical_device_features2(render3d_st.gvk_pd, f2)
|
||||
var missing = ""
|
||||
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_dynamicRendering) != 1 { missing = missing + " dynamicRendering" }
|
||||
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_synchronization2) != 1 { missing = missing + " synchronization2" }
|
||||
if Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_descriptorIndexing) != 1 { missing = missing + " descriptorIndexing" }
|
||||
if Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_timelineSemaphore) != 1 { missing = missing + " timelineSemaphore" }
|
||||
if len(missing) > 0 { gvk_why = `{gvk_device_name} lacks{missing}`; return false }
|
||||
if len(missing) > 0 { render3d_st.gvk_why = `{render3d_st.gvk_device_name} lacks{missing}`; return false }
|
||||
# anisotropic filtering is a 1.0 feature every desktop GPU has; ask for it when it is there
|
||||
let aniso = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_samplerAnisotropy) == 1
|
||||
if aniso { gvk_max_aniso = Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_maxSamplerAnisotropy) }
|
||||
if aniso { render3d_st.gvk_max_aniso = Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_maxSamplerAnisotropy) }
|
||||
let want13 = bytes(VkPhysicalDeviceVulkan13Features_sizeof)
|
||||
Vk.zero(want13, VkPhysicalDeviceVulkan13Features_sizeof)
|
||||
Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES)
|
||||
|
|
@ -167,7 +149,7 @@ function gvk_init() -> bool {
|
|||
if Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_maintenance4) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_maintenance4, 1) }
|
||||
# Streamline's hooks keep their own data on our objects (private data slots), and ask the device
|
||||
# to have the feature rather than turning it on themselves
|
||||
if gsl_on and Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_privateData) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_privateData, 1) }
|
||||
if render3d_st.gsl_on and Vk.get_i32(f13, VkPhysicalDeviceVulkan13Features_privateData) == 1 { Vk.put_i32(want13, VkPhysicalDeviceVulkan13Features_privateData, 1) }
|
||||
let want12 = bytes(VkPhysicalDeviceVulkan12Features_sizeof)
|
||||
Vk.zero(want12, VkPhysicalDeviceVulkan12Features_sizeof)
|
||||
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES)
|
||||
|
|
@ -182,30 +164,30 @@ function gvk_init() -> bool {
|
|||
# GPU-driven drawing: one indirect buffer holds a layer's draws, and a compute pass may write
|
||||
# how many of them there are. Asked for where the device has them; the renderer checks
|
||||
# gvk_has_mdi / gvk_has_dic before it takes that path.
|
||||
gvk_has_mdi = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect) == 1 and Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance) == 1
|
||||
if gvk_has_mdi {
|
||||
render3d_st.gvk_has_mdi = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect) == 1 and Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance) == 1
|
||||
if render3d_st.gvk_has_mdi {
|
||||
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect, 1)
|
||||
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance, 1)
|
||||
}
|
||||
gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1
|
||||
if gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) }
|
||||
render3d_st.gvk_has_dic = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_drawIndirectCount) == 1
|
||||
if render3d_st.gvk_has_dic { Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_drawIndirectCount, 1) }
|
||||
# R3D_PROF: per-pass GPU time from timestamp queries (gvk_query_*). The slots are reused every few
|
||||
# frames, so they are reset from the host - which is a feature to ask for.
|
||||
gvk_has_hqr = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_hostQueryReset) == 1 and Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampComputeAndGraphics) == 1
|
||||
if gvk_has_hqr {
|
||||
render3d_st.gvk_has_hqr = Vk.get_i32(f12, VkPhysicalDeviceVulkan12Features_hostQueryReset) == 1 and Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampComputeAndGraphics) == 1
|
||||
if render3d_st.gvk_has_hqr {
|
||||
Vk.put_i32(want12, VkPhysicalDeviceVulkan12Features_hostQueryReset, 1)
|
||||
gvk_ts_period = float_from_bits(Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampPeriod))
|
||||
render3d_st.gvk_ts_period = float_from_bits(Vk.get_i32(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_timestampPeriod))
|
||||
}
|
||||
|
||||
Vk.put_i32(cnt, 0, 0)
|
||||
Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, null)
|
||||
Vk.enumerate_device_extension_properties(render3d_st.gvk_pd, null, cnt, null)
|
||||
let nde = Vk.get_i32(cnt, 0)
|
||||
let dexts = bytes(nde * VkExtensionProperties_sizeof + 8)
|
||||
Vk.enumerate_device_extension_properties(gvk_pd, null, cnt, dexts)
|
||||
Vk.enumerate_device_extension_properties(render3d_st.gvk_pd, null, cnt, dexts)
|
||||
# mesh-shader grass: the extension, and its meshShader feature asked for where the device has it.
|
||||
# R3D_NO_MESH=1 leaves it off.
|
||||
gvk_has_mesh = false
|
||||
if gvk_ext_in(dexts, nde, VK_EXT_MESH_SHADER_EXTENSION_NAME) and not r3d_env_has("R3D_NO_MESH") {
|
||||
render3d_st.gvk_has_mesh = false
|
||||
if gvk_ext_in(dexts, nde, VK_EXT_MESH_SHADER_EXTENSION_NAME) and not r3d_env_has(render3d_st, "R3D_NO_MESH") {
|
||||
let fm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
||||
Vk.zero(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
||||
Vk.put_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT)
|
||||
|
|
@ -213,14 +195,14 @@ function gvk_init() -> bool {
|
|||
Vk.zero(qm, VkPhysicalDeviceFeatures2_sizeof)
|
||||
Vk.put_i32(qm, VkPhysicalDeviceFeatures2_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2)
|
||||
Vk.put_ptr(qm, VkPhysicalDeviceFeatures2_pNext, fm)
|
||||
Vk.get_physical_device_features2(gvk_pd, qm)
|
||||
Vk.get_physical_device_features2(render3d_st.gvk_pd, qm)
|
||||
if Vk.get_i32(fm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader) == 1 {
|
||||
let wm = bytes(VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
||||
Vk.zero(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sizeof)
|
||||
Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_sType, VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MESH_SHADER_FEATURES_EXT)
|
||||
Vk.put_i32(wm, VkPhysicalDeviceMeshShaderFeaturesEXT_meshShader, 1)
|
||||
Vk.put_ptr(want13, VkPhysicalDeviceVulkan13Features_pNext, wm)
|
||||
gvk_has_mesh = true
|
||||
render3d_st.gvk_has_mesh = true
|
||||
}
|
||||
}
|
||||
let prio = bytes(4)
|
||||
|
|
@ -228,7 +210,7 @@ function gvk_init() -> bool {
|
|||
let qci = bytes(VkDeviceQueueCreateInfo_sizeof)
|
||||
Vk.zero(qci, VkDeviceQueueCreateInfo_sizeof)
|
||||
Vk.put_i32(qci, VkDeviceQueueCreateInfo_sType, VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO)
|
||||
Vk.put_i32(qci, VkDeviceQueueCreateInfo_queueFamilyIndex, gvk_family)
|
||||
Vk.put_i32(qci, VkDeviceQueueCreateInfo_queueFamilyIndex, render3d_st.gvk_family)
|
||||
Vk.put_i32(qci, VkDeviceQueueCreateInfo_queueCount, 1)
|
||||
Vk.put_ptr(qci, VkDeviceQueueCreateInfo_pQueuePriorities, prio)
|
||||
let dext_names = bytes(48)
|
||||
|
|
@ -240,38 +222,38 @@ function gvk_init() -> bool {
|
|||
Vk.put_i32(dci, VkDeviceCreateInfo_queueCreateInfoCount, 1)
|
||||
Vk.put_ptr(dci, VkDeviceCreateInfo_pQueueCreateInfos, qci)
|
||||
if gvk_ext_in(dexts, nde, "VK_KHR_portability_subset") { Vk.put_ptr(dext_names, n_dext * 8, "VK_KHR_portability_subset"); n_dext += 1 }
|
||||
if gvk_has_surface {
|
||||
if not gvk_ext_in(dexts, nde, VK_KHR_SWAPCHAIN_EXTENSION_NAME) { gvk_why = `{gvk_device_name} has no swapchain`; return false }
|
||||
if render3d_st.gvk_has_surface {
|
||||
if not gvk_ext_in(dexts, nde, VK_KHR_SWAPCHAIN_EXTENSION_NAME) { render3d_st.gvk_why = `{render3d_st.gvk_device_name} has no swapchain`; return false }
|
||||
Vk.put_ptr(dext_names, n_dext * 8, VK_KHR_SWAPCHAIN_EXTENSION_NAME); n_dext += 1
|
||||
# Reflex: Streamline adds VK_NV_low_latency2 to this device, which needs present ids it does not add
|
||||
if gsl_on and gvk_ext_in(dexts, nde, "VK_KHR_present_id") { Vk.put_ptr(dext_names, n_dext * 8, "VK_KHR_present_id"); n_dext += 1 }
|
||||
if render3d_st.gsl_on and gvk_ext_in(dexts, nde, "VK_KHR_present_id") { Vk.put_ptr(dext_names, n_dext * 8, "VK_KHR_present_id"); n_dext += 1 }
|
||||
# HDR output tells the display what the picture holds
|
||||
# and only where the loader has the command: Streamline's interposer exports no vkSetHdrMetadataEXT,
|
||||
# and calling its thunk there crashed the game the moment the swapchain came up HDR10
|
||||
gvk_has_hdr_meta = gvk_ext_in(dexts, nde, VK_EXT_HDR_METADATA_EXTENSION_NAME) and Vk.has("vkSetHdrMetadataEXT") == 1
|
||||
if gvk_has_hdr_meta { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_HDR_METADATA_EXTENSION_NAME); n_dext += 1 }
|
||||
render3d_st.gvk_has_hdr_meta = gvk_ext_in(dexts, nde, VK_EXT_HDR_METADATA_EXTENSION_NAME) and Vk.has("vkSetHdrMetadataEXT") == 1
|
||||
if render3d_st.gvk_has_hdr_meta { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_HDR_METADATA_EXTENSION_NAME); n_dext += 1 }
|
||||
}
|
||||
if gvk_has_mesh { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_MESH_SHADER_EXTENSION_NAME); n_dext += 1 }
|
||||
if render3d_st.gvk_has_mesh { Vk.put_ptr(dext_names, n_dext * 8, VK_EXT_MESH_SHADER_EXTENSION_NAME); n_dext += 1 }
|
||||
if n_dext > 0 {
|
||||
Vk.put_i32(dci, VkDeviceCreateInfo_enabledExtensionCount, n_dext)
|
||||
Vk.put_ptr(dci, VkDeviceCreateInfo_ppEnabledExtensionNames, dext_names)
|
||||
}
|
||||
r = Vk.create_device(gvk_pd, dci, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkCreateDevice", r) }
|
||||
gvk_dev = Vk.get_ptr(out, 0)
|
||||
Vk.get_device_queue(gvk_dev, gvk_family, 0, out)
|
||||
gvk_queue = Vk.get_ptr(out, 0)
|
||||
gsl_probe_device(gvk_pd)
|
||||
if gvk_has_mesh {
|
||||
gvk_mesh_fn = Vk.get_device_proc_addr(gvk_dev, "vkCmdDrawMeshTasksEXT")
|
||||
if gvk_mesh_fn == null { gvk_has_mesh = false }
|
||||
r = Vk.create_device(render3d_st.gvk_pd, dci, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateDevice", r) }
|
||||
render3d_st.gvk_dev = Vk.get_ptr(out, 0)
|
||||
Vk.get_device_queue(render3d_st.gvk_dev, render3d_st.gvk_family, 0, out)
|
||||
render3d_st.gvk_queue = Vk.get_ptr(out, 0)
|
||||
gsl_probe_device(render3d_st, render3d_st.gvk_pd)
|
||||
if render3d_st.gvk_has_mesh {
|
||||
render3d_st.gvk_mesh_fn = Vk.get_device_proc_addr(render3d_st.gvk_dev, "vkCmdDrawMeshTasksEXT")
|
||||
if render3d_st.gvk_mesh_fn == null { render3d_st.gvk_has_mesh = false }
|
||||
}
|
||||
gvk_mp = bytes(VkPhysicalDeviceMemoryProperties_sizeof)
|
||||
Vk.get_physical_device_memory_properties(gvk_pd, gvk_mp)
|
||||
render3d_st.gvk_mp = bytes(VkPhysicalDeviceMemoryProperties_sizeof)
|
||||
Vk.get_physical_device_memory_properties(render3d_st.gvk_pd, render3d_st.gvk_mp)
|
||||
|
||||
if not gvk_cmd_init() { return false }
|
||||
gvk_ready = true
|
||||
print(`r3d: vulkan on {gvk_device_name}, anisotropy up to {gvk_aniso_x(gvk_max_aniso)}x, multi-draw indirect {gvk_has_mdi}, indirect count {gvk_has_dic}, mesh shaders {gvk_has_mesh}`)
|
||||
if not gvk_cmd_init(render3d_st) { return false }
|
||||
render3d_st.gvk_ready = true
|
||||
print(`r3d: vulkan on {render3d_st.gvk_device_name}, anisotropy up to {gvk_aniso_x(render3d_st.gvk_max_aniso)}x, multi-draw indirect {render3d_st.gvk_has_mdi}, indirect count {render3d_st.gvk_has_dic}, mesh shaders {render3d_st.gvk_has_mesh}`)
|
||||
return true
|
||||
}
|
||||
|
||||
|
|
@ -284,9 +266,9 @@ function gvk_aniso_x(bits: int) -> int {
|
|||
|
||||
# ---- memory -------------------------------------------------------------------------------
|
||||
# The first memory type a resource allows with every property wanted; -1 if there is none.
|
||||
function gvk_mem_type(allowed: int, want: int) -> int {
|
||||
for t in 0 .. Vk.get_i32(gvk_mp, VkPhysicalDeviceMemoryProperties_memoryTypeCount) {
|
||||
let pf = Vk.get_i32(gvk_mp, VkPhysicalDeviceMemoryProperties_memoryTypes + t * VkMemoryType_sizeof + VkMemoryType_propertyFlags)
|
||||
function gvk_mem_type(render3d_st: Render3dState, allowed: int, want: int) -> int {
|
||||
for t in 0 .. Vk.get_i32(render3d_st.gvk_mp, VkPhysicalDeviceMemoryProperties_memoryTypeCount) {
|
||||
let pf = Vk.get_i32(render3d_st.gvk_mp, VkPhysicalDeviceMemoryProperties_memoryTypes + t * VkMemoryType_sizeof + VkMemoryType_propertyFlags)
|
||||
if ((allowed >> t) & 1) == 1 and (pf & want) == want { return t }
|
||||
}
|
||||
return -1
|
||||
|
|
@ -294,12 +276,11 @@ function gvk_mem_type(allowed: int, want: int) -> int {
|
|||
# Memory for one resource, from its requirements (a VkMemoryRequirements). One allocation per
|
||||
# resource while the backend comes up; the block allocator with sub-allocation replaces it
|
||||
# before the forest and the streams are on this backend, which allocate thousands.
|
||||
var gvk_n_allocs: int = 0
|
||||
function gvk_alloc(req: bytes, want: int) -> long {
|
||||
function gvk_alloc(render3d_st: mut Render3dState, req: bytes, want: int) -> long {
|
||||
let allowed = Vk.get_i32(req, VkMemoryRequirements_memoryTypeBits)
|
||||
var t = gvk_mem_type(allowed, want)
|
||||
var t = gvk_mem_type(render3d_st, allowed, want)
|
||||
# device-local is a preference; host-visible is a need
|
||||
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(allowed, 0) }
|
||||
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(render3d_st, allowed, 0) }
|
||||
let zero: long = 0
|
||||
if t < 0 { print(`r3d: vulkan: no memory type for properties {want}`); return zero }
|
||||
let mai = bytes(VkMemoryAllocateInfo_sizeof)
|
||||
|
|
@ -308,9 +289,9 @@ function gvk_alloc(req: bytes, want: int) -> long {
|
|||
Vk.put_i64(mai, VkMemoryAllocateInfo_allocationSize, Vk.get_i64(req, VkMemoryRequirements_size))
|
||||
Vk.put_i32(mai, VkMemoryAllocateInfo_memoryTypeIndex, t)
|
||||
let out = bytes(8)
|
||||
let r = Vk.allocate_memory(gvk_dev, mai, null, out)
|
||||
let r = Vk.allocate_memory(render3d_st.gvk_dev, mai, null, out)
|
||||
if r != VK_SUCCESS { print(`r3d: vulkan: vkAllocateMemory failed (VkResult {r})`); return zero }
|
||||
gvk_n_allocs += 1
|
||||
render3d_st.gvk_n_allocs += 1
|
||||
return gvk_handle(out)
|
||||
}
|
||||
|
||||
|
|
@ -327,32 +308,19 @@ function gvk_alloc(req: bytes, want: int) -> long {
|
|||
const GVK_BLOCK_DEVICE: int = 67108864
|
||||
const GVK_BLOCK_HOST: int = 67108864
|
||||
const GVK_OWN_OVER: int = 16777216
|
||||
var gvk_blk_mem: []long = null
|
||||
var gvk_blk_kind: []int = null # memory type * 2, + 1 for images
|
||||
var gvk_blk_size: []int = null
|
||||
var gvk_blk_map: []pointer = null
|
||||
var gvk_fr_blk: []int = null # free ranges: block, offset, length (0 = an unused slot)
|
||||
var gvk_fr_off: []int = null
|
||||
var gvk_fr_len: []int = null
|
||||
var gvk_al_blk: []int = null # per allocation id: its block, or -1 for its own memory
|
||||
var gvk_al_mem: []long = null # its own memory when it has one
|
||||
var gvk_al_off: []int = null
|
||||
var gvk_al_len: []int = null
|
||||
var gvk_al_map: []pointer = null
|
||||
var gvk_al_spare: []int = null # ids to reuse
|
||||
|
||||
function gvk_mem_init() -> void {
|
||||
if gvk_al_blk != null { return }
|
||||
function gvk_mem_init(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gvk_al_blk != null { return }
|
||||
let zero: long = 0
|
||||
gvk_blk_mem = new []long; gvk_blk_kind = new []int; gvk_blk_size = new []int; gvk_blk_map = new []pointer
|
||||
gvk_fr_blk = new []int; gvk_fr_off = new []int; gvk_fr_len = new []int
|
||||
gvk_al_blk = new []int; gvk_al_mem = new []long; gvk_al_off = new []int; gvk_al_len = new []int
|
||||
gvk_al_map = new []pointer; gvk_al_spare = new []int
|
||||
push(gvk_al_blk, -1); push(gvk_al_mem, zero); push(gvk_al_off, 0); push(gvk_al_len, 0); push(gvk_al_map, null)
|
||||
render3d_st.gvk_blk_mem = new []long; render3d_st.gvk_blk_kind = new []int; render3d_st.gvk_blk_size = new []int; render3d_st.gvk_blk_map = new []pointer
|
||||
render3d_st.gvk_fr_blk = new []int; render3d_st.gvk_fr_off = new []int; render3d_st.gvk_fr_len = new []int
|
||||
render3d_st.gvk_al_blk = new []int; render3d_st.gvk_al_mem = new []long; render3d_st.gvk_al_off = new []int; render3d_st.gvk_al_len = new []int
|
||||
render3d_st.gvk_al_map = new []pointer; render3d_st.gvk_al_spare = new []int
|
||||
push(render3d_st.gvk_al_blk, -1); push(render3d_st.gvk_al_mem, zero); push(render3d_st.gvk_al_off, 0); push(render3d_st.gvk_al_len, 0); push(render3d_st.gvk_al_map, null)
|
||||
}
|
||||
|
||||
# one vkAllocateMemory of `size` bytes of memory type t, mapped when host-visible; 0 on failure
|
||||
function gvk_mem_raw(t: int, size: int, host: bool, out_map: []pointer) -> long {
|
||||
function gvk_mem_raw(render3d_st: mut Render3dState, t: int, size: int, host: bool, out_map: []pointer) -> long {
|
||||
let zero: long = 0
|
||||
let mai = bytes(VkMemoryAllocateInfo_sizeof)
|
||||
Vk.zero(mai, VkMemoryAllocateInfo_sizeof)
|
||||
|
|
@ -361,32 +329,32 @@ function gvk_mem_raw(t: int, size: int, host: bool, out_map: []pointer) -> long
|
|||
Vk.put_i64(mai, VkMemoryAllocateInfo_allocationSize, size_l)
|
||||
Vk.put_i32(mai, VkMemoryAllocateInfo_memoryTypeIndex, t)
|
||||
let out = bytes(8)
|
||||
let r = Vk.allocate_memory(gvk_dev, mai, null, out)
|
||||
if r != VK_SUCCESS { gvk_note(`r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {gvk_n_allocs} allocations live)`); return zero }
|
||||
gvk_n_allocs += 1
|
||||
let r = Vk.allocate_memory(render3d_st.gvk_dev, mai, null, out)
|
||||
if r != VK_SUCCESS { gvk_note(render3d_st, `r3d: vulkan: vkAllocateMemory of {size} bytes failed (VkResult {r}, {render3d_st.gvk_n_allocs} allocations live)`); return zero }
|
||||
render3d_st.gvk_n_allocs += 1
|
||||
let mem = gvk_handle(out)
|
||||
out_map[0] = null
|
||||
if host {
|
||||
let pp = bytes(8)
|
||||
if Vk.map_memory(gvk_dev, mem, zero, size_l, 0, pp) != VK_SUCCESS { gvk_note("r3d: vulkan: vkMapMemory failed"); return zero }
|
||||
if Vk.map_memory(render3d_st.gvk_dev, mem, zero, size_l, 0, pp) != VK_SUCCESS { gvk_note(render3d_st, "r3d: vulkan: vkMapMemory failed"); return zero }
|
||||
out_map[0] = Vk.get_ptr(pp, 0)
|
||||
}
|
||||
return mem
|
||||
}
|
||||
|
||||
function gvk_fr_put(b: int, off: int, len_: int) -> void {
|
||||
function gvk_fr_put(render3d_st: mut Render3dState, b: int, off: int, len_: int) -> void {
|
||||
if len_ <= 0 { return }
|
||||
for i in 0 .. len(gvk_fr_blk) { if gvk_fr_len[i] == 0 { gvk_fr_blk[i] = b; gvk_fr_off[i] = off; gvk_fr_len[i] = len_; return } }
|
||||
push(gvk_fr_blk, b); push(gvk_fr_off, off); push(gvk_fr_len, len_)
|
||||
for i in 0 .. len(render3d_st.gvk_fr_blk) { if render3d_st.gvk_fr_len[i] == 0 { render3d_st.gvk_fr_blk[i] = b; render3d_st.gvk_fr_off[i] = off; render3d_st.gvk_fr_len[i] = len_; return } }
|
||||
push(render3d_st.gvk_fr_blk, b); push(render3d_st.gvk_fr_off, off); push(render3d_st.gvk_fr_len, len_)
|
||||
}
|
||||
|
||||
# Memory for a resource from its requirements (a VkMemoryRequirements): an allocation id, 0 if none.
|
||||
function gvk_mem_new(req: bytes, want: int, image: bool) -> int {
|
||||
gvk_mem_init()
|
||||
function gvk_mem_new(render3d_st: mut Render3dState, req: bytes, want: int, image: bool) -> int {
|
||||
gvk_mem_init(render3d_st)
|
||||
let allowed = Vk.get_i32(req, VkMemoryRequirements_memoryTypeBits)
|
||||
var t = gvk_mem_type(allowed, want)
|
||||
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(allowed, 0) }
|
||||
if t < 0 { gvk_note(`r3d: vulkan: no memory type for properties {want}`); return 0 }
|
||||
var t = gvk_mem_type(render3d_st, allowed, want)
|
||||
if t < 0 and want == VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT { t = gvk_mem_type(render3d_st, allowed, 0) }
|
||||
if t < 0 { gvk_note(render3d_st, `r3d: vulkan: no memory type for properties {want}`); return 0 }
|
||||
let host = (want & VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT) != 0
|
||||
let size = Text.to_int(string(Vk.get_i64(req, VkMemoryRequirements_size)))
|
||||
var align = Text.to_int(string(Vk.get_i64(req, VkMemoryRequirements_alignment)))
|
||||
|
|
@ -400,17 +368,17 @@ function gvk_mem_new(req: bytes, want: int, image: bool) -> int {
|
|||
var kind = t * 2
|
||||
if image { kind += 1 }
|
||||
var i = 0
|
||||
while i < len(gvk_fr_blk) and blk < 0 {
|
||||
let n = gvk_fr_len[i]
|
||||
if n > 0 and gvk_blk_kind[gvk_fr_blk[i]] == kind {
|
||||
let start = (gvk_fr_off[i] + align - 1) / align * align
|
||||
let end = gvk_fr_off[i] + n
|
||||
while i < len(render3d_st.gvk_fr_blk) and blk < 0 {
|
||||
let n = render3d_st.gvk_fr_len[i]
|
||||
if n > 0 and render3d_st.gvk_blk_kind[render3d_st.gvk_fr_blk[i]] == kind {
|
||||
let start = (render3d_st.gvk_fr_off[i] + align - 1) / align * align
|
||||
let end = render3d_st.gvk_fr_off[i] + n
|
||||
if start + size <= end {
|
||||
blk = gvk_fr_blk[i]; off = start
|
||||
let front = start - gvk_fr_off[i]
|
||||
blk = render3d_st.gvk_fr_blk[i]; off = start
|
||||
let front = start - render3d_st.gvk_fr_off[i]
|
||||
let back = end - (start + size)
|
||||
if front > 0 { gvk_fr_len[i] = front } else { gvk_fr_len[i] = 0 }
|
||||
gvk_fr_put(blk, start + size, back)
|
||||
if front > 0 { render3d_st.gvk_fr_len[i] = front } else { render3d_st.gvk_fr_len[i] = 0 }
|
||||
gvk_fr_put(render3d_st, blk, start + size, back)
|
||||
}
|
||||
}
|
||||
i += 1
|
||||
|
|
@ -418,32 +386,32 @@ function gvk_mem_new(req: bytes, want: int, image: bool) -> int {
|
|||
if blk < 0 {
|
||||
var bsize = GVK_BLOCK_DEVICE
|
||||
if host { bsize = GVK_BLOCK_HOST }
|
||||
let mem = gvk_mem_raw(t, bsize, host, mp)
|
||||
let mem = gvk_mem_raw(render3d_st, t, bsize, host, mp)
|
||||
if mem != 0 {
|
||||
push(gvk_blk_mem, mem); push(gvk_blk_kind, kind); push(gvk_blk_size, bsize); push(gvk_blk_map, mp[0])
|
||||
blk = len(gvk_blk_mem) - 1; off = 0
|
||||
gvk_fr_put(blk, size, bsize - size)
|
||||
push(render3d_st.gvk_blk_mem, mem); push(render3d_st.gvk_blk_kind, kind); push(render3d_st.gvk_blk_size, bsize); push(render3d_st.gvk_blk_map, mp[0])
|
||||
blk = len(render3d_st.gvk_blk_mem) - 1; off = 0
|
||||
gvk_fr_put(render3d_st, blk, size, bsize - size)
|
||||
}
|
||||
}
|
||||
}
|
||||
var own: long = 0
|
||||
var map: pointer = null
|
||||
if blk < 0 {
|
||||
own = gvk_mem_raw(t, size, host, mp)
|
||||
own = gvk_mem_raw(render3d_st, t, size, host, mp)
|
||||
if own == 0 { return 0 }
|
||||
map = mp[0]
|
||||
} else if gvk_blk_map[blk] != null {
|
||||
map = mem_off(gvk_blk_map[blk], off)
|
||||
} else if render3d_st.gvk_blk_map[blk] != null {
|
||||
map = mem_off(render3d_st.gvk_blk_map[blk], off)
|
||||
}
|
||||
var a = 0
|
||||
if len(gvk_al_spare) > 0 {
|
||||
a = gvk_al_spare[len(gvk_al_spare) - 1]
|
||||
gvk_al_spare = gvk_list_drop_last(gvk_al_spare)
|
||||
if len(render3d_st.gvk_al_spare) > 0 {
|
||||
a = render3d_st.gvk_al_spare[len(render3d_st.gvk_al_spare) - 1]
|
||||
render3d_st.gvk_al_spare = gvk_list_drop_last(render3d_st.gvk_al_spare)
|
||||
} else {
|
||||
push(gvk_al_blk, -1); push(gvk_al_mem, zero); push(gvk_al_off, 0); push(gvk_al_len, 0); push(gvk_al_map, null)
|
||||
a = len(gvk_al_blk) - 1
|
||||
push(render3d_st.gvk_al_blk, -1); push(render3d_st.gvk_al_mem, zero); push(render3d_st.gvk_al_off, 0); push(render3d_st.gvk_al_len, 0); push(render3d_st.gvk_al_map, null)
|
||||
a = len(render3d_st.gvk_al_blk) - 1
|
||||
}
|
||||
gvk_al_blk[a] = blk; gvk_al_mem[a] = own; gvk_al_off[a] = off; gvk_al_len[a] = size; gvk_al_map[a] = map
|
||||
render3d_st.gvk_al_blk[a] = blk; render3d_st.gvk_al_mem[a] = own; render3d_st.gvk_al_off[a] = off; render3d_st.gvk_al_len[a] = size; render3d_st.gvk_al_map[a] = map
|
||||
return a
|
||||
}
|
||||
|
||||
|
|
@ -453,87 +421,85 @@ function gvk_list_drop_last(l: []int) -> []int {
|
|||
return out
|
||||
}
|
||||
|
||||
function gvk_mem_handle(a: int) -> long {
|
||||
if gvk_al_blk[a] < 0 { return gvk_al_mem[a] }
|
||||
return gvk_blk_mem[gvk_al_blk[a]]
|
||||
function gvk_mem_handle(render3d_st: Render3dState, a: int) -> long {
|
||||
if render3d_st.gvk_al_blk[a] < 0 { return render3d_st.gvk_al_mem[a] }
|
||||
return render3d_st.gvk_blk_mem[render3d_st.gvk_al_blk[a]]
|
||||
}
|
||||
function gvk_mem_offset(a: int) -> long {
|
||||
let o: long = gvk_al_off[a]
|
||||
function gvk_mem_offset(render3d_st: Render3dState, a: int) -> long {
|
||||
let o: long = render3d_st.gvk_al_off[a]
|
||||
return o
|
||||
}
|
||||
function gvk_mem_ptr(a: int) -> pointer { return gvk_al_map[a] }
|
||||
function gvk_mem_ptr(render3d_st: Render3dState, a: int) -> pointer { return render3d_st.gvk_al_map[a] }
|
||||
|
||||
# give an allocation back: its own memory is freed, a range returns to its block and merges
|
||||
function gvk_mem_free(a: int) -> void {
|
||||
if gvk_al_blk == null or a <= 0 or a >= len(gvk_al_blk) or gvk_al_len[a] == 0 { return }
|
||||
let b = gvk_al_blk[a]
|
||||
function gvk_mem_free(render3d_st: mut Render3dState, a: int) -> void {
|
||||
if render3d_st.gvk_al_blk == null or a <= 0 or a >= len(render3d_st.gvk_al_blk) or render3d_st.gvk_al_len[a] == 0 { return }
|
||||
let b = render3d_st.gvk_al_blk[a]
|
||||
let zero: long = 0
|
||||
if b < 0 {
|
||||
if gvk_al_map[a] != null { Vk.unmap_memory(gvk_dev, gvk_al_mem[a]) }
|
||||
Vk.free_memory(gvk_dev, gvk_al_mem[a], null)
|
||||
gvk_n_allocs -= 1
|
||||
if render3d_st.gvk_al_map[a] != null { Vk.unmap_memory(render3d_st.gvk_dev, render3d_st.gvk_al_mem[a]) }
|
||||
Vk.free_memory(render3d_st.gvk_dev, render3d_st.gvk_al_mem[a], null)
|
||||
render3d_st.gvk_n_allocs -= 1
|
||||
} else {
|
||||
var off = gvk_al_off[a]
|
||||
var n = gvk_al_len[a]
|
||||
var off = render3d_st.gvk_al_off[a]
|
||||
var n = render3d_st.gvk_al_len[a]
|
||||
# merge with a free range that ends where this starts, and one that starts where this ends
|
||||
for i in 0 .. len(gvk_fr_blk) {
|
||||
if gvk_fr_len[i] > 0 and gvk_fr_blk[i] == b and gvk_fr_off[i] + gvk_fr_len[i] == off {
|
||||
off = gvk_fr_off[i]; n += gvk_fr_len[i]; gvk_fr_len[i] = 0
|
||||
for i in 0 .. len(render3d_st.gvk_fr_blk) {
|
||||
if render3d_st.gvk_fr_len[i] > 0 and render3d_st.gvk_fr_blk[i] == b and render3d_st.gvk_fr_off[i] + render3d_st.gvk_fr_len[i] == off {
|
||||
off = render3d_st.gvk_fr_off[i]; n += render3d_st.gvk_fr_len[i]; render3d_st.gvk_fr_len[i] = 0
|
||||
}
|
||||
}
|
||||
for i in 0 .. len(gvk_fr_blk) {
|
||||
if gvk_fr_len[i] > 0 and gvk_fr_blk[i] == b and gvk_fr_off[i] == off + n {
|
||||
n += gvk_fr_len[i]; gvk_fr_len[i] = 0
|
||||
for i in 0 .. len(render3d_st.gvk_fr_blk) {
|
||||
if render3d_st.gvk_fr_len[i] > 0 and render3d_st.gvk_fr_blk[i] == b and render3d_st.gvk_fr_off[i] == off + n {
|
||||
n += render3d_st.gvk_fr_len[i]; render3d_st.gvk_fr_len[i] = 0
|
||||
}
|
||||
}
|
||||
if off == 0 and n == gvk_blk_size[b] {
|
||||
if off == 0 and n == render3d_st.gvk_blk_size[b] {
|
||||
# the block is empty again: give it back, or a world swapped out keeps its memory for good
|
||||
if gvk_blk_map[b] != null { Vk.unmap_memory(gvk_dev, gvk_blk_mem[b]) }
|
||||
Vk.free_memory(gvk_dev, gvk_blk_mem[b], null)
|
||||
gvk_n_allocs -= 1
|
||||
gvk_blk_mem[b] = zero; gvk_blk_map[b] = null; gvk_blk_kind[b] = -1; gvk_blk_size[b] = 0
|
||||
if render3d_st.gvk_blk_map[b] != null { Vk.unmap_memory(render3d_st.gvk_dev, render3d_st.gvk_blk_mem[b]) }
|
||||
Vk.free_memory(render3d_st.gvk_dev, render3d_st.gvk_blk_mem[b], null)
|
||||
render3d_st.gvk_n_allocs -= 1
|
||||
render3d_st.gvk_blk_mem[b] = zero; render3d_st.gvk_blk_map[b] = null; render3d_st.gvk_blk_kind[b] = -1; render3d_st.gvk_blk_size[b] = 0
|
||||
} else {
|
||||
gvk_fr_put(b, off, n)
|
||||
gvk_fr_put(render3d_st, b, off, n)
|
||||
}
|
||||
}
|
||||
gvk_al_len[a] = 0; gvk_al_mem[a] = zero; gvk_al_map[a] = null; gvk_al_blk[a] = -1
|
||||
push(gvk_al_spare, a)
|
||||
render3d_st.gvk_al_len[a] = 0; render3d_st.gvk_al_mem[a] = zero; render3d_st.gvk_al_map[a] = null; render3d_st.gvk_al_blk[a] = -1
|
||||
push(render3d_st.gvk_al_spare, a)
|
||||
}
|
||||
function gvk_mem_id(x: long) -> int { return Text.to_int(string(x)) }
|
||||
|
||||
# ---- one-shot commands --------------------------------------------------------------------
|
||||
# Uploads, bakes and read-backs record into a command buffer, submit it and wait. The frame
|
||||
# itself does not go through here.
|
||||
var gvk_pool: long = 0
|
||||
var gvk_fence: bytes = null
|
||||
function gvk_cmd_init() -> bool {
|
||||
function gvk_cmd_init(render3d_st: mut Render3dState) -> bool {
|
||||
let cpi = bytes(VkCommandPoolCreateInfo_sizeof)
|
||||
Vk.zero(cpi, VkCommandPoolCreateInfo_sizeof)
|
||||
Vk.put_i32(cpi, VkCommandPoolCreateInfo_sType, VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO)
|
||||
Vk.put_i32(cpi, VkCommandPoolCreateInfo_flags, VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT)
|
||||
Vk.put_i32(cpi, VkCommandPoolCreateInfo_queueFamilyIndex, gvk_family)
|
||||
Vk.put_i32(cpi, VkCommandPoolCreateInfo_queueFamilyIndex, render3d_st.gvk_family)
|
||||
let out = bytes(8)
|
||||
var r = Vk.create_command_pool(gvk_dev, cpi, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkCreateCommandPool", r) }
|
||||
gvk_pool = gvk_handle(out)
|
||||
var r = Vk.create_command_pool(render3d_st.gvk_dev, cpi, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateCommandPool", r) }
|
||||
render3d_st.gvk_pool = gvk_handle(out)
|
||||
let fci = bytes(VkFenceCreateInfo_sizeof)
|
||||
Vk.zero(fci, VkFenceCreateInfo_sizeof)
|
||||
Vk.put_i32(fci, VkFenceCreateInfo_sType, VK_STRUCTURE_TYPE_FENCE_CREATE_INFO)
|
||||
gvk_fence = bytes(8)
|
||||
r = Vk.create_fence(gvk_dev, fci, null, gvk_fence)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkCreateFence", r) }
|
||||
render3d_st.gvk_fence = bytes(8)
|
||||
r = Vk.create_fence(render3d_st.gvk_dev, fci, null, render3d_st.gvk_fence)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateFence", r) }
|
||||
return true
|
||||
}
|
||||
# a command buffer, begun
|
||||
function gvk_once_begin() -> pointer {
|
||||
function gvk_once_begin(render3d_st: Render3dState) -> pointer {
|
||||
let cbai = bytes(VkCommandBufferAllocateInfo_sizeof)
|
||||
Vk.zero(cbai, VkCommandBufferAllocateInfo_sizeof)
|
||||
Vk.put_i32(cbai, VkCommandBufferAllocateInfo_sType, VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO)
|
||||
Vk.put_i64(cbai, VkCommandBufferAllocateInfo_commandPool, gvk_pool)
|
||||
Vk.put_i64(cbai, VkCommandBufferAllocateInfo_commandPool, render3d_st.gvk_pool)
|
||||
Vk.put_i32(cbai, VkCommandBufferAllocateInfo_level, VK_COMMAND_BUFFER_LEVEL_PRIMARY)
|
||||
Vk.put_i32(cbai, VkCommandBufferAllocateInfo_commandBufferCount, 1)
|
||||
let cbs = bytes(8)
|
||||
if Vk.allocate_command_buffers(gvk_dev, cbai, cbs) != VK_SUCCESS { return null }
|
||||
if Vk.allocate_command_buffers(render3d_st.gvk_dev, cbai, cbs) != VK_SUCCESS { return null }
|
||||
let cb = Vk.get_ptr(cbs, 0)
|
||||
let cbbi = bytes(VkCommandBufferBeginInfo_sizeof)
|
||||
Vk.zero(cbbi, VkCommandBufferBeginInfo_sizeof)
|
||||
|
|
@ -543,36 +509,36 @@ function gvk_once_begin() -> pointer {
|
|||
return cb
|
||||
}
|
||||
# end it, submit it, wait for it, free it
|
||||
function gvk_once_end(cb: pointer) -> bool {
|
||||
function gvk_once_end(render3d_st: mut Render3dState, cb: pointer) -> bool {
|
||||
var r = Vk.end_command_buffer(cb)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkEndCommandBuffer", r) }
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkEndCommandBuffer", r) }
|
||||
let cbs = bytes(8)
|
||||
Vk.put_ptr(cbs, 0, cb)
|
||||
Vk.reset_fences(gvk_dev, 1, gvk_fence)
|
||||
Vk.reset_fences(render3d_st.gvk_dev, 1, render3d_st.gvk_fence)
|
||||
let si = bytes(VkSubmitInfo_sizeof)
|
||||
Vk.zero(si, VkSubmitInfo_sizeof)
|
||||
Vk.put_i32(si, VkSubmitInfo_sType, VK_STRUCTURE_TYPE_SUBMIT_INFO)
|
||||
Vk.put_i32(si, VkSubmitInfo_commandBufferCount, 1)
|
||||
Vk.put_ptr(si, VkSubmitInfo_pCommandBuffers, cbs)
|
||||
r = Vk.queue_submit(gvk_queue, 1, si, Vk.get_i64(gvk_fence, 0))
|
||||
if r != VK_SUCCESS { return gvk_fail("vkQueueSubmit", r) }
|
||||
r = Vk.queue_submit(render3d_st.gvk_queue, 1, si, Vk.get_i64(render3d_st.gvk_fence, 0))
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkQueueSubmit", r) }
|
||||
let forever: long = -1
|
||||
r = Vk.wait_for_fences(gvk_dev, 1, gvk_fence, 1, forever)
|
||||
Vk.free_command_buffers(gvk_dev, gvk_pool, 1, cbs)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkWaitForFences", r) }
|
||||
r = Vk.wait_for_fences(render3d_st.gvk_dev, 1, render3d_st.gvk_fence, 1, forever)
|
||||
Vk.free_command_buffers(render3d_st.gvk_dev, render3d_st.gvk_pool, 1, cbs)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkWaitForFences", r) }
|
||||
return true
|
||||
}
|
||||
|
||||
# ---- teardown -----------------------------------------------------------------------------
|
||||
function gvk_shutdown() -> void {
|
||||
if not gvk_ready { return }
|
||||
Vk.device_wait_idle(gvk_dev)
|
||||
gsl_shutdown()
|
||||
Vk.destroy_fence(gvk_dev, Vk.get_i64(gvk_fence, 0), null)
|
||||
Vk.destroy_command_pool(gvk_dev, gvk_pool, null)
|
||||
Vk.destroy_device(gvk_dev, null)
|
||||
Vk.destroy_instance(gvk_inst, null)
|
||||
gvk_ready = false
|
||||
function gvk_shutdown(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.gvk_ready { return }
|
||||
Vk.device_wait_idle(render3d_st.gvk_dev)
|
||||
gsl_shutdown(render3d_st)
|
||||
Vk.destroy_fence(render3d_st.gvk_dev, Vk.get_i64(render3d_st.gvk_fence, 0), null)
|
||||
Vk.destroy_command_pool(render3d_st.gvk_dev, render3d_st.gvk_pool, null)
|
||||
Vk.destroy_device(render3d_st.gvk_dev, null)
|
||||
Vk.destroy_instance(render3d_st.gvk_inst, null)
|
||||
render3d_st.gvk_ready = false
|
||||
}
|
||||
|
||||
# ---- GPU timestamps (R3D_PROF) ------------------------------------------------------------
|
||||
|
|
@ -580,42 +546,38 @@ function gvk_shutdown() -> void {
|
|||
# ends. A slot is read back PROF_RING frames after it was written and reset just before it is written
|
||||
# again. On a device without host query reset or timestamps, every read says "not yet" and the report
|
||||
# stays empty.
|
||||
var gvk_has_hqr: bool = false
|
||||
var gvk_ts_period: float = 0.0 # float bits: nanoseconds per timestamp tick
|
||||
var gvk_qpool: long = 0
|
||||
var gvk_q_active: int = -1
|
||||
function gvk_query_new(n: int, ids: words) -> void {
|
||||
function gvk_query_new(render3d_st: mut Render3dState, n: int, ids: words) -> void {
|
||||
for i in 0 .. n { ids[i] = i }
|
||||
if not gvk_has_hqr or gvk_dev == null { return }
|
||||
if not render3d_st.gvk_has_hqr or render3d_st.gvk_dev == null { return }
|
||||
let qci = bytes(VkQueryPoolCreateInfo_sizeof)
|
||||
Vk.zero(qci, VkQueryPoolCreateInfo_sizeof)
|
||||
Vk.put_i32(qci, VkQueryPoolCreateInfo_sType, VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO)
|
||||
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryType, VK_QUERY_TYPE_TIMESTAMP)
|
||||
Vk.put_i32(qci, VkQueryPoolCreateInfo_queryCount, n * 2)
|
||||
let out = bytes(8)
|
||||
if Vk.create_query_pool(gvk_dev, qci, null, out) != VK_SUCCESS { return }
|
||||
gvk_qpool = gvk_handle(out)
|
||||
Vk.reset_query_pool(gvk_dev, gvk_qpool, 0, n * 2)
|
||||
if Vk.create_query_pool(render3d_st.gvk_dev, qci, null, out) != VK_SUCCESS { return }
|
||||
render3d_st.gvk_qpool = gvk_handle(out)
|
||||
Vk.reset_query_pool(render3d_st.gvk_dev, render3d_st.gvk_qpool, 0, n * 2)
|
||||
}
|
||||
function gvk_query_begin(id: int) -> void {
|
||||
if gvk_qpool == 0 { return }
|
||||
Vk.reset_query_pool(gvk_dev, gvk_qpool, id * 2, 2)
|
||||
Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, id * 2)
|
||||
gvk_q_active = id
|
||||
function gvk_query_begin(render3d_st: mut Render3dState, id: int) -> void {
|
||||
if render3d_st.gvk_qpool == 0 { return }
|
||||
Vk.reset_query_pool(render3d_st.gvk_dev, render3d_st.gvk_qpool, id * 2, 2)
|
||||
Vk.cmd_write_timestamp(gvk_frame_cb(render3d_st), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, render3d_st.gvk_qpool, id * 2)
|
||||
render3d_st.gvk_q_active = id
|
||||
}
|
||||
function gvk_query_end() -> void {
|
||||
if gvk_qpool == 0 or gvk_q_active < 0 { return }
|
||||
Vk.cmd_write_timestamp(gvk_frame_cb(), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, gvk_qpool, gvk_q_active * 2 + 1)
|
||||
gvk_q_active = -1
|
||||
function gvk_query_end(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gvk_qpool == 0 or render3d_st.gvk_q_active < 0 { return }
|
||||
Vk.cmd_write_timestamp(gvk_frame_cb(render3d_st), VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, render3d_st.gvk_qpool, render3d_st.gvk_q_active * 2 + 1)
|
||||
render3d_st.gvk_q_active = -1
|
||||
}
|
||||
function gvk_query_result(id: int, out: words) -> bool {
|
||||
if gvk_qpool == 0 { return false }
|
||||
function gvk_query_result(render3d_st: Render3dState, id: int, out: words) -> bool {
|
||||
if render3d_st.gvk_qpool == 0 { return false }
|
||||
let data = bytes(16)
|
||||
let size: long = 16
|
||||
let stride: long = 8
|
||||
if Vk.get_query_pool_results(gvk_dev, gvk_qpool, id * 2, 2, size, data, stride, VK_QUERY_RESULT_64_BIT) != VK_SUCCESS { return false }
|
||||
if Vk.get_query_pool_results(render3d_st.gvk_dev, render3d_st.gvk_qpool, id * 2, 2, size, data, stride, VK_QUERY_RESULT_64_BIT) != VK_SUCCESS { return false }
|
||||
let ticks = Text.to_int(string(Vk.get_i64(data, 8) - Vk.get_i64(data, 0)))
|
||||
if ticks < 0 { return false }
|
||||
out[0] = int(float(ticks) * gvk_ts_period)
|
||||
out[0] = int(float(ticks) * render3d_st.gvk_ts_period)
|
||||
return true
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -45,46 +45,30 @@ function gvk_channel_bytes(ifmt: int) -> int {
|
|||
# objects. Pixels uploaded with the image get a full mip chain, allocated up front, because the
|
||||
# renderer asks for mipmaps after the upload; an image made without pixels is a target and gets
|
||||
# one level. Every level sits in SHADER_READ_ONLY_OPTIMAL between uses.
|
||||
var gvk_tex_image: []long = null
|
||||
var gvk_tex_view: []long = null
|
||||
var gvk_tex_mem: []long = null
|
||||
var gvk_tex_levels: []int = null
|
||||
var gvk_tex_layers: []int = null
|
||||
var gvk_tex_vkfmt: []int = null
|
||||
var gvk_tex_dims_w: []int = null
|
||||
var gvk_tex_dims_h: []int = null
|
||||
var gvk_tex_glfmt: []int = null # the format the renderer asked for (gpu.ludic's vocabulary)
|
||||
var gvk_tex_array: []int = null # 1 for an array image
|
||||
var gvk_tex_gen: []int = null # bumped whenever the image behind a handle is replaced
|
||||
var gvk_tex_smp_sig: []int = null # the sampling parameters the cached sampler was made for
|
||||
var gvk_tex_smp: []long = null
|
||||
var gvk_tex_samples: []int = null # 1, or the sample count of a multisampled renderbuffer
|
||||
var gvk_storage_samples: int = 1 # what the next gvk_tex_storage makes: gpu_rb_storage sets it around its call
|
||||
var gvk_unpack_swap: bool = false # GL_UNPACK_SWAP_BYTES: 16-bit PNG samples arrive big-endian
|
||||
|
||||
function gvk_tex_new() -> int {
|
||||
function gvk_tex_new(render3d_st: mut Render3dState) -> int {
|
||||
let zero: long = 0
|
||||
if gvk_tex_image == null {
|
||||
gvk_tex_image = new []long; gvk_tex_view = new []long; gvk_tex_mem = new []long
|
||||
gvk_tex_levels = new []int; gvk_tex_layers = new []int; gvk_tex_vkfmt = new []int
|
||||
gvk_tex_dims_w = new []int; gvk_tex_dims_h = new []int
|
||||
gvk_tex_glfmt = new []int; gvk_tex_array = new []int; gvk_tex_gen = new []int
|
||||
gvk_tex_smp_sig = new []int; gvk_tex_smp = new []long
|
||||
gvk_tex_samples = new []int; push(gvk_tex_samples, 0)
|
||||
push(gvk_tex_dims_w, 0); push(gvk_tex_dims_h, 0)
|
||||
push(gvk_tex_glfmt, 0); push(gvk_tex_array, 0); push(gvk_tex_gen, 0)
|
||||
push(gvk_tex_smp_sig, 0); push(gvk_tex_smp, zero)
|
||||
if render3d_st.gvk_tex_image == null {
|
||||
render3d_st.gvk_tex_image = new []long; render3d_st.gvk_tex_view = new []long; render3d_st.gvk_tex_mem = new []long
|
||||
render3d_st.gvk_tex_levels = new []int; render3d_st.gvk_tex_layers = new []int; render3d_st.gvk_tex_vkfmt = new []int
|
||||
render3d_st.gvk_tex_dims_w = new []int; render3d_st.gvk_tex_dims_h = new []int
|
||||
render3d_st.gvk_tex_glfmt = new []int; render3d_st.gvk_tex_array = new []int; render3d_st.gvk_tex_gen = new []int
|
||||
render3d_st.gvk_tex_smp_sig = new []int; render3d_st.gvk_tex_smp = new []long
|
||||
render3d_st.gvk_tex_samples = new []int; push(render3d_st.gvk_tex_samples, 0)
|
||||
push(render3d_st.gvk_tex_dims_w, 0); push(render3d_st.gvk_tex_dims_h, 0)
|
||||
push(render3d_st.gvk_tex_glfmt, 0); push(render3d_st.gvk_tex_array, 0); push(render3d_st.gvk_tex_gen, 0)
|
||||
push(render3d_st.gvk_tex_smp_sig, 0); push(render3d_st.gvk_tex_smp, zero)
|
||||
# handle 0 is "no texture", as it is on OpenGL
|
||||
push(gvk_tex_image, zero); push(gvk_tex_view, zero); push(gvk_tex_mem, zero)
|
||||
push(gvk_tex_levels, 0); push(gvk_tex_layers, 0); push(gvk_tex_vkfmt, 0)
|
||||
push(render3d_st.gvk_tex_image, zero); push(render3d_st.gvk_tex_view, zero); push(render3d_st.gvk_tex_mem, zero)
|
||||
push(render3d_st.gvk_tex_levels, 0); push(render3d_st.gvk_tex_layers, 0); push(render3d_st.gvk_tex_vkfmt, 0)
|
||||
}
|
||||
push(gvk_tex_image, zero); push(gvk_tex_view, zero); push(gvk_tex_mem, zero)
|
||||
push(gvk_tex_levels, 0); push(gvk_tex_layers, 0); push(gvk_tex_vkfmt, 0)
|
||||
push(gvk_tex_dims_w, 0); push(gvk_tex_dims_h, 0)
|
||||
push(gvk_tex_glfmt, 0); push(gvk_tex_array, 0); push(gvk_tex_gen, 0)
|
||||
push(gvk_tex_smp_sig, 0); push(gvk_tex_smp, zero)
|
||||
push(gvk_tex_samples, 0)
|
||||
return len(gvk_tex_image) - 1
|
||||
push(render3d_st.gvk_tex_image, zero); push(render3d_st.gvk_tex_view, zero); push(render3d_st.gvk_tex_mem, zero)
|
||||
push(render3d_st.gvk_tex_levels, 0); push(render3d_st.gvk_tex_layers, 0); push(render3d_st.gvk_tex_vkfmt, 0)
|
||||
push(render3d_st.gvk_tex_dims_w, 0); push(render3d_st.gvk_tex_dims_h, 0)
|
||||
push(render3d_st.gvk_tex_glfmt, 0); push(render3d_st.gvk_tex_array, 0); push(render3d_st.gvk_tex_gen, 0)
|
||||
push(render3d_st.gvk_tex_smp_sig, 0); push(render3d_st.gvk_tex_smp, zero)
|
||||
push(render3d_st.gvk_tex_samples, 0)
|
||||
return len(render3d_st.gvk_tex_image) - 1
|
||||
}
|
||||
|
||||
function gvk_mip_levels(w: int, h: int) -> int {
|
||||
|
|
@ -129,9 +113,7 @@ function gvk_barrier(cb: pointer, image: long, depth: bool, base: int, n: int, l
|
|||
}
|
||||
|
||||
# a host-visible buffer of n bytes, mapped; gvk_staging_free unmaps and frees it
|
||||
var gvk_st_buf: long = 0
|
||||
var gvk_st_mem: long = 0
|
||||
function gvk_staging(n: int, usage: int) -> pointer {
|
||||
function gvk_staging(render3d_st: mut Render3dState, n: int, usage: int) -> pointer {
|
||||
let size: long = n
|
||||
let bci = bytes(VkBufferCreateInfo_sizeof)
|
||||
Vk.zero(bci, VkBufferCreateInfo_sizeof)
|
||||
|
|
@ -140,41 +122,41 @@ function gvk_staging(n: int, usage: int) -> pointer {
|
|||
Vk.put_i32(bci, VkBufferCreateInfo_usage, usage)
|
||||
Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
|
||||
let out = bytes(8)
|
||||
if Vk.create_buffer(gvk_dev, bci, null, out) != VK_SUCCESS { return null }
|
||||
gvk_st_buf = gvk_handle(out)
|
||||
if Vk.create_buffer(render3d_st.gvk_dev, bci, null, out) != VK_SUCCESS { return null }
|
||||
render3d_st.gvk_st_buf = gvk_handle(out)
|
||||
let req = bytes(VkMemoryRequirements_sizeof)
|
||||
Vk.get_buffer_memory_requirements(gvk_dev, gvk_st_buf, req)
|
||||
gvk_st_mem = gvk_alloc(req, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)
|
||||
Vk.get_buffer_memory_requirements(render3d_st.gvk_dev, render3d_st.gvk_st_buf, req)
|
||||
render3d_st.gvk_st_mem = gvk_alloc(render3d_st, req, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)
|
||||
let zero: long = 0
|
||||
Vk.bind_buffer_memory(gvk_dev, gvk_st_buf, gvk_st_mem, zero)
|
||||
Vk.bind_buffer_memory(render3d_st.gvk_dev, render3d_st.gvk_st_buf, render3d_st.gvk_st_mem, zero)
|
||||
let pp = bytes(8)
|
||||
if Vk.map_memory(gvk_dev, gvk_st_mem, zero, size, 0, pp) != VK_SUCCESS { return null }
|
||||
if Vk.map_memory(render3d_st.gvk_dev, render3d_st.gvk_st_mem, zero, size, 0, pp) != VK_SUCCESS { return null }
|
||||
return Vk.get_ptr(pp, 0)
|
||||
}
|
||||
function gvk_staging_free() -> void {
|
||||
Vk.unmap_memory(gvk_dev, gvk_st_mem)
|
||||
Vk.destroy_buffer(gvk_dev, gvk_st_buf, null)
|
||||
Vk.free_memory(gvk_dev, gvk_st_mem, null)
|
||||
gvk_n_allocs -= 1
|
||||
function gvk_staging_free(render3d_st: mut Render3dState) -> void {
|
||||
Vk.unmap_memory(render3d_st.gvk_dev, render3d_st.gvk_st_mem)
|
||||
Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_st_buf, null)
|
||||
Vk.free_memory(render3d_st.gvk_dev, render3d_st.gvk_st_mem, null)
|
||||
render3d_st.gvk_n_allocs -= 1
|
||||
}
|
||||
|
||||
# (Re)make the image behind a handle: glTexImage2D on a texture that already has one replaces it.
|
||||
function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layers: int, with_mips: bool) -> bool {
|
||||
if tex <= 0 or tex >= len(gvk_tex_image) { return false }
|
||||
function gvk_tex_storage(render3d_st: mut Render3dState, tex: int, array: bool, ifmt: int, w: int, h: int, layers: int, with_mips: bool) -> bool {
|
||||
if tex <= 0 or tex >= len(render3d_st.gvk_tex_image) { return false }
|
||||
let vkfmt = gvk_format(ifmt)
|
||||
if vkfmt == VK_FORMAT_UNDEFINED { print(`r3d: vulkan: no image format for GL format {ifmt}`); return false }
|
||||
gvk_tex_release(tex)
|
||||
gvk_tex_release(render3d_st, tex)
|
||||
let depth = gvk_is_depth(ifmt)
|
||||
var levels = 1
|
||||
if with_mips and not depth { levels = gvk_mip_levels(w, h) }
|
||||
var samples = gvk_storage_samples
|
||||
var samples = render3d_st.gvk_storage_samples
|
||||
if samples < 1 { samples = 1 }
|
||||
if samples > 1 { levels = 1 }
|
||||
var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT
|
||||
if depth { usage = usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } else { usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT }
|
||||
# DLSS reads and writes the frame from compute: a float colour target is storage too while
|
||||
# Streamline is running (8-bit sRGB formats cannot be, so only the float ones)
|
||||
if gsl_on and samples == 1 and (vkfmt == VK_FORMAT_R16G16B16A16_SFLOAT or vkfmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }
|
||||
if render3d_st.gsl_on and samples == 1 and (vkfmt == VK_FORMAT_R16G16B16A16_SFLOAT or vkfmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }
|
||||
let ici = bytes(VkImageCreateInfo_sizeof)
|
||||
Vk.zero(ici, VkImageCreateInfo_sizeof)
|
||||
Vk.put_i32(ici, VkImageCreateInfo_sType, VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO)
|
||||
|
|
@ -191,22 +173,22 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer
|
|||
Vk.put_i32(ici, VkImageCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
|
||||
Vk.put_i32(ici, VkImageCreateInfo_initialLayout, VK_IMAGE_LAYOUT_UNDEFINED)
|
||||
let out = bytes(8)
|
||||
var r = Vk.create_image(gvk_dev, ici, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(`vkCreateImage {w}x{h}x{layers} format {vkfmt}`, r) }
|
||||
var r = Vk.create_image(render3d_st.gvk_dev, ici, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, `vkCreateImage {w}x{h}x{layers} format {vkfmt}`, r) }
|
||||
let image = gvk_handle(out)
|
||||
let req = bytes(VkMemoryRequirements_sizeof)
|
||||
Vk.get_image_memory_requirements(gvk_dev, image, req)
|
||||
let ma = gvk_mem_new(req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, true)
|
||||
Vk.get_image_memory_requirements(render3d_st.gvk_dev, image, req)
|
||||
let ma = gvk_mem_new(render3d_st, req, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, true)
|
||||
let mem: long = ma # the allocation id (gvk_mem_new), kept where the memory was
|
||||
let zero: long = 0
|
||||
# no memory for it (one allocation per resource meets the driver's allocation limit long before
|
||||
# the card is full): say so instead of binding a null allocation, which the driver may accept
|
||||
if mem == 0 {
|
||||
Vk.destroy_image(gvk_dev, image, null)
|
||||
return gvk_fail(`no device memory for a {w}x{h}x{layers} image ({gvk_n_allocs} allocations live)`, VK_ERROR_OUT_OF_DEVICE_MEMORY)
|
||||
Vk.destroy_image(render3d_st.gvk_dev, image, null)
|
||||
return gvk_fail(render3d_st, `no device memory for a {w}x{h}x{layers} image ({render3d_st.gvk_n_allocs} allocations live)`, VK_ERROR_OUT_OF_DEVICE_MEMORY)
|
||||
}
|
||||
r = Vk.bind_image_memory(gvk_dev, image, gvk_mem_handle(ma), gvk_mem_offset(ma))
|
||||
if r != VK_SUCCESS { return gvk_fail("vkBindImageMemory", r) }
|
||||
r = Vk.bind_image_memory(render3d_st.gvk_dev, image, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkBindImageMemory", r) }
|
||||
let vci = bytes(VkImageViewCreateInfo_sizeof)
|
||||
Vk.zero(vci, VkImageViewCreateInfo_sizeof)
|
||||
Vk.put_i32(vci, VkImageViewCreateInfo_sType, VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO)
|
||||
|
|
@ -217,18 +199,18 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer
|
|||
if depth { Vk.put_i32(vci, sr + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_DEPTH_BIT) } else { Vk.put_i32(vci, sr + VkImageSubresourceRange_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT) }
|
||||
Vk.put_i32(vci, sr + VkImageSubresourceRange_levelCount, levels)
|
||||
Vk.put_i32(vci, sr + VkImageSubresourceRange_layerCount, layers)
|
||||
r = Vk.create_image_view(gvk_dev, vci, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail("vkCreateImageView", r) }
|
||||
gvk_tex_image[tex] = image; gvk_tex_view[tex] = gvk_handle(out); gvk_tex_mem[tex] = mem
|
||||
gvk_tex_levels[tex] = levels; gvk_tex_layers[tex] = layers; gvk_tex_vkfmt[tex] = vkfmt
|
||||
gvk_tex_dims_w[tex] = w; gvk_tex_dims_h[tex] = h
|
||||
gvk_tex_glfmt[tex] = ifmt; gvk_tex_gen[tex] = gvk_tex_gen[tex] + 1
|
||||
gvk_tex_samples[tex] = samples
|
||||
if array { gvk_tex_array[tex] = 1 } else { gvk_tex_array[tex] = 0 }
|
||||
r = Vk.create_image_view(render3d_st.gvk_dev, vci, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkCreateImageView", r) }
|
||||
render3d_st.gvk_tex_image[tex] = image; render3d_st.gvk_tex_view[tex] = gvk_handle(out); render3d_st.gvk_tex_mem[tex] = mem
|
||||
render3d_st.gvk_tex_levels[tex] = levels; render3d_st.gvk_tex_layers[tex] = layers; render3d_st.gvk_tex_vkfmt[tex] = vkfmt
|
||||
render3d_st.gvk_tex_dims_w[tex] = w; render3d_st.gvk_tex_dims_h[tex] = h
|
||||
render3d_st.gvk_tex_glfmt[tex] = ifmt; render3d_st.gvk_tex_gen[tex] = render3d_st.gvk_tex_gen[tex] + 1
|
||||
render3d_st.gvk_tex_samples[tex] = samples
|
||||
if array { render3d_st.gvk_tex_array[tex] = 1 } else { render3d_st.gvk_tex_array[tex] = 0 }
|
||||
# a target starts cleared-to-nothing but readable: every level in the layout samplers expect
|
||||
let cb = gvk_once_begin()
|
||||
let cb = gvk_once_begin(render3d_st)
|
||||
gvk_barrier(cb, image, depth, 0, levels, layers, VK_IMAGE_LAYOUT_UNDEFINED, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
||||
return gvk_once_end(cb)
|
||||
return gvk_once_end(render3d_st, cb)
|
||||
}
|
||||
|
||||
# IEEE single (as its bits) to IEEE half: out-of-range values saturate to infinity, tiny ones to zero
|
||||
|
|
@ -255,17 +237,17 @@ function gvk_gl_type_bytes(ty: int) -> int {
|
|||
# Level 0 of every layer from pixels laid out the OpenGL way (fmt channels of type ty), converted
|
||||
# to what the image stores: a missing alpha filled opaque, 16-bit samples byte-swapped when the
|
||||
# unpack state says so, 32-bit floats into a half-float image halved.
|
||||
function gvk_tex_upload(tex: int, ifmt: int, w: int, h: int, layers: int, fmt: int, ty: int, data: pointer) -> bool {
|
||||
function gvk_tex_upload(render3d_st: mut Render3dState, tex: int, ifmt: int, w: int, h: int, layers: int, fmt: int, ty: int, data: pointer) -> bool {
|
||||
let cin = gvk_gl_channels(fmt)
|
||||
let bin = gvk_gl_type_bytes(ty)
|
||||
let cout = gvk_channels(ifmt)
|
||||
let bout = gvk_channel_bytes(ifmt)
|
||||
let texels = w * h * layers
|
||||
let n = texels * cout * bout
|
||||
let dst = gvk_staging(n, VK_BUFFER_USAGE_TRANSFER_SRC_BIT)
|
||||
let dst = gvk_staging(render3d_st, n, VK_BUFFER_USAGE_TRANSFER_SRC_BIT)
|
||||
if dst == null { print("r3d: vulkan: no staging buffer for an upload"); return false }
|
||||
let buf = bytes(n)
|
||||
if cin == cout and bin == bout and not (bin == 2 and gvk_unpack_swap) {
|
||||
if cin == cout and bin == bout and not (bin == 2 and render3d_st.gvk_unpack_swap) {
|
||||
mem_copy(buf, data, n)
|
||||
} else {
|
||||
let src: pointer = data
|
||||
|
|
@ -282,7 +264,7 @@ function gvk_tex_upload(tex: int, ifmt: int, w: int, h: int, layers: int, fmt: i
|
|||
if bin == bout {
|
||||
if bin == 1 { buf[o] = src[i] }
|
||||
if bin == 2 {
|
||||
if gvk_unpack_swap { buf[o] = src[i + 1]; buf[o + 1] = src[i] } else { buf[o] = src[i]; buf[o + 1] = src[i + 1] }
|
||||
if render3d_st.gvk_unpack_swap { buf[o] = src[i + 1]; buf[o + 1] = src[i] } else { buf[o] = src[i]; buf[o + 1] = src[i + 1] }
|
||||
}
|
||||
if bin == 4 { Vk.put_i32(buf, o, Vk.get_i32(src, i)) }
|
||||
} else if bin == 4 and bout == 2 {
|
||||
|
|
@ -290,7 +272,7 @@ function gvk_tex_upload(tex: int, ifmt: int, w: int, h: int, layers: int, fmt: i
|
|||
buf[o] = hv & 255; buf[o + 1] = (hv >> 8) & 255
|
||||
} else {
|
||||
print(`r3d: vulkan: no conversion from {bin}-byte to {bout}-byte samples`)
|
||||
gvk_staging_free()
|
||||
gvk_staging_free(render3d_st)
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
|
@ -298,10 +280,10 @@ function gvk_tex_upload(tex: int, ifmt: int, w: int, h: int, layers: int, fmt: i
|
|||
}
|
||||
}
|
||||
mem_copy(dst, buf, n)
|
||||
let image = gvk_tex_image[tex]
|
||||
let image = render3d_st.gvk_tex_image[tex]
|
||||
let depth = gvk_is_depth(ifmt)
|
||||
let levels = gvk_tex_levels[tex]
|
||||
let cb = gvk_once_begin()
|
||||
let levels = render3d_st.gvk_tex_levels[tex]
|
||||
let cb = gvk_once_begin(render3d_st)
|
||||
gvk_barrier(cb, image, depth, 0, levels, layers, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
||||
let bic = bytes(VkBufferImageCopy_sizeof)
|
||||
Vk.zero(bic, VkBufferImageCopy_sizeof)
|
||||
|
|
@ -311,27 +293,27 @@ function gvk_tex_upload(tex: int, ifmt: int, w: int, h: int, layers: int, fmt: i
|
|||
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_width, w)
|
||||
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_height, h)
|
||||
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_depth, 1)
|
||||
Vk.cmd_copy_buffer_to_image(cb, gvk_st_buf, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, bic)
|
||||
Vk.cmd_copy_buffer_to_image(cb, render3d_st.gvk_st_buf, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, bic)
|
||||
gvk_barrier(cb, image, depth, 0, levels, layers, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
||||
let ok = gvk_once_end(cb)
|
||||
gvk_staging_free()
|
||||
let ok = gvk_once_end(render3d_st, cb)
|
||||
gvk_staging_free(render3d_st)
|
||||
return ok
|
||||
}
|
||||
|
||||
# The mip chain from level 0, each level blitted down from the one above it
|
||||
# A target made without pixels has one level; the renderer asking it for mipmaps (the exposure
|
||||
# measure reads the HDR scene's smallest level every frame) grows it a full chain, level 0 kept.
|
||||
function gvk_tex_grow_mips(tex: int, w: int, h: int) -> bool {
|
||||
let old_image = gvk_tex_image[tex]
|
||||
let old_view = gvk_tex_view[tex]
|
||||
let old_mem = gvk_tex_mem[tex]
|
||||
let layers = gvk_tex_layers[tex]
|
||||
function gvk_tex_grow_mips(render3d_st: mut Render3dState, tex: int, w: int, h: int) -> bool {
|
||||
let old_image = render3d_st.gvk_tex_image[tex]
|
||||
let old_view = render3d_st.gvk_tex_view[tex]
|
||||
let old_mem = render3d_st.gvk_tex_mem[tex]
|
||||
let layers = render3d_st.gvk_tex_layers[tex]
|
||||
let zero: long = 0
|
||||
gvk_tex_image[tex] = zero
|
||||
if not gvk_tex_storage(tex, gvk_tex_array[tex] == 1, gvk_tex_glfmt[tex], w, h, layers, true) { return false }
|
||||
let cb = gvk_once_begin()
|
||||
render3d_st.gvk_tex_image[tex] = zero
|
||||
if not gvk_tex_storage(render3d_st, tex, render3d_st.gvk_tex_array[tex] == 1, render3d_st.gvk_tex_glfmt[tex], w, h, layers, true) { return false }
|
||||
let cb = gvk_once_begin(render3d_st)
|
||||
gvk_barrier(cb, old_image, false, 0, 1, layers, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
|
||||
gvk_barrier(cb, gvk_tex_image[tex], false, 0, 1, layers, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
||||
gvk_barrier(cb, render3d_st.gvk_tex_image[tex], false, 0, 1, layers, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
|
||||
let ic = bytes(VkImageCopy_sizeof)
|
||||
Vk.zero(ic, VkImageCopy_sizeof)
|
||||
Vk.put_i32(ic, VkImageCopy_srcSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
|
||||
|
|
@ -341,31 +323,31 @@ function gvk_tex_grow_mips(tex: int, w: int, h: int) -> bool {
|
|||
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_width, w)
|
||||
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_height, h)
|
||||
Vk.put_i32(ic, VkImageCopy_extent + VkExtent3D_depth, 1)
|
||||
Vk.cmd_copy_image(cb, old_image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, gvk_tex_image[tex], VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ic)
|
||||
gvk_barrier(cb, gvk_tex_image[tex], false, 0, 1, layers, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
||||
let ok = gvk_once_end(cb)
|
||||
Vk.destroy_image_view(gvk_dev, old_view, null)
|
||||
Vk.destroy_image(gvk_dev, old_image, null)
|
||||
gvk_mem_free(gvk_mem_id(old_mem))
|
||||
Vk.cmd_copy_image(cb, old_image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, render3d_st.gvk_tex_image[tex], VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ic)
|
||||
gvk_barrier(cb, render3d_st.gvk_tex_image[tex], false, 0, 1, layers, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
||||
let ok = gvk_once_end(render3d_st, cb)
|
||||
Vk.destroy_image_view(render3d_st.gvk_dev, old_view, null)
|
||||
Vk.destroy_image(render3d_st.gvk_dev, old_image, null)
|
||||
gvk_mem_free(render3d_st, gvk_mem_id(old_mem))
|
||||
return ok
|
||||
}
|
||||
|
||||
function gvk_tex_mips(tex: int, w: int, h: int) -> bool {
|
||||
if gvk_tex_levels[tex] <= 1 and (w > 1 or h > 1) and not gvk_is_depth(gvk_tex_glfmt[tex]) {
|
||||
if not gvk_tex_grow_mips(tex, w, h) { return false }
|
||||
function gvk_tex_mips(render3d_st: mut Render3dState, tex: int, w: int, h: int) -> bool {
|
||||
if render3d_st.gvk_tex_levels[tex] <= 1 and (w > 1 or h > 1) and not gvk_is_depth(render3d_st.gvk_tex_glfmt[tex]) {
|
||||
if not gvk_tex_grow_mips(render3d_st, tex, w, h) { return false }
|
||||
}
|
||||
let levels = gvk_tex_levels[tex]
|
||||
let levels = render3d_st.gvk_tex_levels[tex]
|
||||
if levels <= 1 { return true }
|
||||
let cb = gvk_once_begin()
|
||||
gvk_tex_mips_into(cb, tex, w, h)
|
||||
return gvk_once_end(cb)
|
||||
let cb = gvk_once_begin(render3d_st)
|
||||
gvk_tex_mips_into(render3d_st, cb, tex, w, h)
|
||||
return gvk_once_end(render3d_st, cb)
|
||||
}
|
||||
# the chain's blits and barriers recorded into cb (the frame's own, when one is open)
|
||||
function gvk_tex_mips_into(cb: pointer, tex: int, w: int, h: int) -> void {
|
||||
let levels = gvk_tex_levels[tex]
|
||||
function gvk_tex_mips_into(render3d_st: Render3dState, cb: pointer, tex: int, w: int, h: int) -> void {
|
||||
let levels = render3d_st.gvk_tex_levels[tex]
|
||||
if levels <= 1 { return }
|
||||
let image = gvk_tex_image[tex]
|
||||
let layers = gvk_tex_layers[tex]
|
||||
let image = render3d_st.gvk_tex_image[tex]
|
||||
let layers = render3d_st.gvk_tex_layers[tex]
|
||||
var sw = w
|
||||
var sh = h
|
||||
for lv in 1 .. levels {
|
||||
|
|
@ -399,7 +381,7 @@ function gvk_tex_mips_into(cb: pointer, tex: int, w: int, h: int) -> void {
|
|||
|
||||
# Level 0 of layer 0 back to the CPU, laid out the OpenGL way when the layouts agree (the terrain
|
||||
# reads its height field back as R32F into GL_RED / GL_FLOAT)
|
||||
function gvk_tex_read(tex: int, ifmt: int, w: int, h: int, fmt: int, ty: int, out: pointer) -> bool {
|
||||
function gvk_tex_read(render3d_st: mut Render3dState, tex: int, ifmt: int, w: int, h: int, fmt: int, ty: int, out: pointer) -> bool {
|
||||
let cout = gvk_channels(ifmt)
|
||||
let bout = gvk_channel_bytes(ifmt)
|
||||
let cwant = gvk_gl_channels(fmt)
|
||||
|
|
@ -410,11 +392,11 @@ function gvk_tex_read(tex: int, ifmt: int, w: int, h: int, fmt: int, ty: int, ou
|
|||
return false
|
||||
}
|
||||
let n = w * h * cout * bout
|
||||
let src = gvk_staging(n, VK_BUFFER_USAGE_TRANSFER_DST_BIT)
|
||||
let src = gvk_staging(render3d_st, n, VK_BUFFER_USAGE_TRANSFER_DST_BIT)
|
||||
if src == null { return false }
|
||||
let image = gvk_tex_image[tex]
|
||||
let image = render3d_st.gvk_tex_image[tex]
|
||||
let depth = gvk_is_depth(ifmt)
|
||||
let cb = gvk_once_begin()
|
||||
let cb = gvk_once_begin(render3d_st)
|
||||
gvk_barrier(cb, image, depth, 0, 1, 1, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
|
||||
let bic = bytes(VkBufferImageCopy_sizeof)
|
||||
Vk.zero(bic, VkBufferImageCopy_sizeof)
|
||||
|
|
@ -424,36 +406,34 @@ function gvk_tex_read(tex: int, ifmt: int, w: int, h: int, fmt: int, ty: int, ou
|
|||
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_width, w)
|
||||
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_height, h)
|
||||
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_depth, 1)
|
||||
Vk.cmd_copy_image_to_buffer(cb, image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, gvk_st_buf, 1, bic)
|
||||
Vk.cmd_copy_image_to_buffer(cb, image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, render3d_st.gvk_st_buf, 1, bic)
|
||||
gvk_barrier(cb, image, depth, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
|
||||
let ok = gvk_once_end(cb)
|
||||
let ok = gvk_once_end(render3d_st, cb)
|
||||
if ok and cwant == cout { mem_copy(out, src, n) }
|
||||
if ok and cwant < cout {
|
||||
let texel = cout * bout
|
||||
let keep = cwant * bout
|
||||
for t in 0 .. w * h { mem_copy(mem_off(out, t * keep), mem_off(src, t * texel), keep) }
|
||||
}
|
||||
gvk_staging_free()
|
||||
gvk_staging_free(render3d_st)
|
||||
return ok
|
||||
}
|
||||
|
||||
function gvk_tex_release(tex: int) -> void {
|
||||
if gvk_tex_image[tex] == 0 { return }
|
||||
function gvk_tex_release(render3d_st: mut Render3dState, tex: int) -> void {
|
||||
if render3d_st.gvk_tex_image[tex] == 0 { return }
|
||||
let zero: long = 0
|
||||
Vk.destroy_image_view(gvk_dev, gvk_tex_view[tex], null)
|
||||
Vk.destroy_image(gvk_dev, gvk_tex_image[tex], null)
|
||||
gvk_mem_free(gvk_mem_id(gvk_tex_mem[tex]))
|
||||
gvk_tex_image[tex] = zero; gvk_tex_view[tex] = zero; gvk_tex_mem[tex] = zero
|
||||
gvk_tex_levels[tex] = 0; gvk_tex_layers[tex] = 0; gvk_tex_vkfmt[tex] = 0
|
||||
gvk_tex_gen[tex] = gvk_tex_gen[tex] + 1
|
||||
Vk.destroy_image_view(render3d_st.gvk_dev, render3d_st.gvk_tex_view[tex], null)
|
||||
Vk.destroy_image(render3d_st.gvk_dev, render3d_st.gvk_tex_image[tex], null)
|
||||
gvk_mem_free(render3d_st, gvk_mem_id(render3d_st.gvk_tex_mem[tex]))
|
||||
render3d_st.gvk_tex_image[tex] = zero; render3d_st.gvk_tex_view[tex] = zero; render3d_st.gvk_tex_mem[tex] = zero
|
||||
render3d_st.gvk_tex_levels[tex] = 0; render3d_st.gvk_tex_layers[tex] = 0; render3d_st.gvk_tex_vkfmt[tex] = 0
|
||||
render3d_st.gvk_tex_gen[tex] = render3d_st.gvk_tex_gen[tex] + 1
|
||||
}
|
||||
|
||||
# ---- samplers -----------------------------------------------------------------------------
|
||||
# One VkSampler per distinct way of reading a texture, made the first time it is asked for.
|
||||
# The arguments are what the renderer set through gpu_tex_param (0 where it set nothing, which
|
||||
# means GL's own default).
|
||||
var gvk_smp_keys: []string = null
|
||||
var gvk_smp: []long = null
|
||||
|
||||
function gvk_filter(f: int) -> int {
|
||||
if f == GL_NEAREST or f == GL_NEAREST_MIPMAP_NEAREST or f == GL_NEAREST_MIPMAP_LINEAR { return VK_FILTER_NEAREST }
|
||||
|
|
@ -480,19 +460,19 @@ function gvk_compare_op(f: int) -> int {
|
|||
# which is -2.0 at Performance and about -1.6 at Quality. It is applied ONLY while DLSS is live: a
|
||||
# plain spatial upscale has no temporal accumulation to hide the aliasing a negative bias brings,
|
||||
# so biasing there would trade blur for shimmer.
|
||||
function gvk_mip_bias() -> float {
|
||||
function gvk_mip_bias(render3d_st: mut Render3dState) -> float {
|
||||
# R3D_NO_MIPBIAS=1 puts it back the way it was, so one build can be compared against itself
|
||||
if r3d_env_has("R3D_NO_MIPBIAS") { return 0.0 }
|
||||
if not r3d_dlss_live() { return 0.0 }
|
||||
let rw = r3d_dlss_render_w()
|
||||
if r3d_env_has(render3d_st, "R3D_NO_MIPBIAS") { return 0.0 }
|
||||
if not r3d_dlss_live(render3d_st) { return 0.0 }
|
||||
let rw = r3d_dlss_render_w(render3d_st)
|
||||
if rw <= 0 or gl_w <= 0 or rw >= gl_w { return 0.0 }
|
||||
# log2 from the natural log the runtime has: log2(x) = ln(x) * 1/ln(2)
|
||||
return Math.log(float(rw) / float(gl_w)) * 1.4426950408889634 - 1.0
|
||||
}
|
||||
# the sampler for texture tex's parameters, from the texture's own one-entry cache when they have
|
||||
# not changed since it last asked - a draw asks for every texture it binds
|
||||
function gvk_tex_sampler(tex: int, min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare: int, aniso: int) -> long {
|
||||
let bias = float_bits(gvk_mip_bias())
|
||||
function gvk_tex_sampler(render3d_st: mut Render3dState, tex: int, min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare: int, aniso: int) -> long {
|
||||
let bias = float_bits(gvk_mip_bias(render3d_st))
|
||||
var sig = min_f * 31 + mag_f
|
||||
sig = sig * 31 + wrap_s
|
||||
sig = sig * 31 + wrap_t
|
||||
|
|
@ -501,16 +481,16 @@ function gvk_tex_sampler(tex: int, min_f: int, mag_f: int, wrap_s: int, wrap_t:
|
|||
# the bias is part of what the sampler IS, so it has to invalidate this cache too - otherwise
|
||||
# turning DLSS on mid-session keeps every sampler already made at the old bias
|
||||
sig = (sig * 31 + bias) | 1
|
||||
if tex > 0 and tex < len(gvk_tex_smp_sig) and gvk_tex_smp_sig[tex] == sig { return gvk_tex_smp[tex] }
|
||||
let s = gvk_sampler(min_f, mag_f, wrap_s, wrap_t, compare, aniso)
|
||||
if tex > 0 and tex < len(gvk_tex_smp_sig) { gvk_tex_smp_sig[tex] = sig; gvk_tex_smp[tex] = s }
|
||||
if tex > 0 and tex < len(render3d_st.gvk_tex_smp_sig) and render3d_st.gvk_tex_smp_sig[tex] == sig { return render3d_st.gvk_tex_smp[tex] }
|
||||
let s = gvk_sampler(render3d_st, min_f, mag_f, wrap_s, wrap_t, compare, aniso)
|
||||
if tex > 0 and tex < len(render3d_st.gvk_tex_smp_sig) { render3d_st.gvk_tex_smp_sig[tex] = sig; render3d_st.gvk_tex_smp[tex] = s }
|
||||
return s
|
||||
}
|
||||
function gvk_sampler(min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare: int, aniso: int) -> long {
|
||||
let bias = float_bits(gvk_mip_bias())
|
||||
function gvk_sampler(render3d_st: mut Render3dState, min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare: int, aniso: int) -> long {
|
||||
let bias = float_bits(gvk_mip_bias(render3d_st))
|
||||
let key = `{min_f}/{mag_f}/{wrap_s}/{wrap_t}/{compare}/{aniso}/{bias}`
|
||||
if gvk_smp_keys == null { gvk_smp_keys = new []string; gvk_smp = new []long }
|
||||
for i in 0 .. len(gvk_smp_keys) { if gvk_smp_keys[i] == key { return gvk_smp[i] } }
|
||||
if render3d_st.gvk_smp_keys == null { render3d_st.gvk_smp_keys = new []string; render3d_st.gvk_smp = new []long }
|
||||
for i in 0 .. len(render3d_st.gvk_smp_keys) { if render3d_st.gvk_smp_keys[i] == key { return render3d_st.gvk_smp[i] } }
|
||||
var mn = min_f
|
||||
if mn == 0 { mn = GL_NEAREST_MIPMAP_LINEAR }
|
||||
var mg = mag_f
|
||||
|
|
@ -531,7 +511,7 @@ function gvk_sampler(min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare:
|
|||
if mipmapped { Vk.put_i32(sci, VkSamplerCreateInfo_mipLodBias, bias) }
|
||||
if aniso > 0x3F800000 and mipmapped {
|
||||
var a = aniso
|
||||
if a > gvk_max_aniso { a = gvk_max_aniso }
|
||||
if a > render3d_st.gvk_max_aniso { a = render3d_st.gvk_max_aniso }
|
||||
Vk.put_i32(sci, VkSamplerCreateInfo_anisotropyEnable, 1)
|
||||
Vk.put_i32(sci, VkSamplerCreateInfo_maxAnisotropy, a)
|
||||
}
|
||||
|
|
@ -542,11 +522,11 @@ function gvk_sampler(min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare:
|
|||
Vk.put_i32(sci, VkSamplerCreateInfo_borderColor, VK_BORDER_COLOR_FLOAT_OPAQUE_WHITE)
|
||||
let out = bytes(8)
|
||||
let zero: long = 0
|
||||
let r = Vk.create_sampler(gvk_dev, sci, null, out)
|
||||
if r != VK_SUCCESS { gvk_fail("vkCreateSampler", r); return zero }
|
||||
let r = Vk.create_sampler(render3d_st.gvk_dev, sci, null, out)
|
||||
if r != VK_SUCCESS { gvk_fail(render3d_st, "vkCreateSampler", r); return zero }
|
||||
let s = gvk_handle(out)
|
||||
push(gvk_smp_keys, key)
|
||||
push(gvk_smp, s)
|
||||
push(render3d_st.gvk_smp_keys, key)
|
||||
push(render3d_st.gvk_smp, s)
|
||||
return s
|
||||
}
|
||||
|
||||
|
|
@ -555,69 +535,60 @@ function gvk_sampler(min_f: int, mag_f: int, wrap_s: int, wrap_t: int, compare:
|
|||
# indexes these lists. Every buffer can be read as vertices or as indices, so one handle serves
|
||||
# whichever the renderer binds it as. While the backend comes up every buffer is host-visible and
|
||||
# an upload maps and copies; staged device-local buffers arrive with the block allocator.
|
||||
var gvk_buf: []long = null
|
||||
var gvk_buf_mem: []long = null
|
||||
var gvk_buf_size: []int = null
|
||||
var gvk_buf_map: []pointer = null # host-visible buffers stay mapped for their whole life
|
||||
var gvk_buf_used: []int = null # the frame (gvk_frame_no) a draw last bound the buffer in
|
||||
var gvk_frame_no: int = 1
|
||||
# Storage a buffer was moved off, or freed, while the frame that used it has not been submitted:
|
||||
# destroyed at gvk_retire_flush, after the frame's work is done.
|
||||
var gvk_retired_buf: []long = null
|
||||
var gvk_retired_mem: []long = null
|
||||
|
||||
function gvk_buf_new() -> int {
|
||||
function gvk_buf_new(render3d_st: mut Render3dState) -> int {
|
||||
let zero: long = 0
|
||||
if gvk_buf == null {
|
||||
gvk_buf = new []long; gvk_buf_mem = new []long; gvk_buf_size = new []int; gvk_buf_map = new []pointer
|
||||
gvk_buf_used = new []int; gvk_retired_buf = new []long; gvk_retired_mem = new []long
|
||||
if render3d_st.gvk_buf == null {
|
||||
render3d_st.gvk_buf = new []long; render3d_st.gvk_buf_mem = new []long; render3d_st.gvk_buf_size = new []int; render3d_st.gvk_buf_map = new []pointer
|
||||
render3d_st.gvk_buf_used = new []int; render3d_st.gvk_retired_buf = new []long; render3d_st.gvk_retired_mem = new []long
|
||||
# handle 0 is "no buffer", as it is on OpenGL
|
||||
push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null); push(gvk_buf_used, 0)
|
||||
push(render3d_st.gvk_buf, zero); push(render3d_st.gvk_buf_mem, zero); push(render3d_st.gvk_buf_size, 0); push(render3d_st.gvk_buf_map, null); push(render3d_st.gvk_buf_used, 0)
|
||||
}
|
||||
push(gvk_buf, zero); push(gvk_buf_mem, zero); push(gvk_buf_size, 0); push(gvk_buf_map, null); push(gvk_buf_used, 0)
|
||||
return len(gvk_buf) - 1
|
||||
push(render3d_st.gvk_buf, zero); push(render3d_st.gvk_buf_mem, zero); push(render3d_st.gvk_buf_size, 0); push(render3d_st.gvk_buf_map, null); push(render3d_st.gvk_buf_used, 0)
|
||||
return len(render3d_st.gvk_buf) - 1
|
||||
}
|
||||
|
||||
function gvk_buf_release(b: int) -> void {
|
||||
if gvk_buf[b] == 0 { return }
|
||||
function gvk_buf_release(render3d_st: mut Render3dState, b: int) -> void {
|
||||
if render3d_st.gvk_buf[b] == 0 { return }
|
||||
let zero: long = 0
|
||||
if gvk_buf_used[b] == gvk_frame_no {
|
||||
if render3d_st.gvk_buf_used[b] == render3d_st.gvk_frame_no {
|
||||
# a draw recorded this frame still reads it: destroy it once the frame has been submitted
|
||||
push(gvk_retired_buf, gvk_buf[b]); push(gvk_retired_mem, gvk_buf_mem[b])
|
||||
push(render3d_st.gvk_retired_buf, render3d_st.gvk_buf[b]); push(render3d_st.gvk_retired_mem, render3d_st.gvk_buf_mem[b])
|
||||
} else {
|
||||
Vk.destroy_buffer(gvk_dev, gvk_buf[b], null)
|
||||
gvk_mem_free(gvk_mem_id(gvk_buf_mem[b]))
|
||||
Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_buf[b], null)
|
||||
gvk_mem_free(render3d_st, gvk_mem_id(render3d_st.gvk_buf_mem[b]))
|
||||
}
|
||||
gvk_buf[b] = zero; gvk_buf_mem[b] = zero; gvk_buf_size[b] = 0; gvk_buf_map[b] = null; gvk_buf_used[b] = 0
|
||||
render3d_st.gvk_buf[b] = zero; render3d_st.gvk_buf_mem[b] = zero; render3d_st.gvk_buf_size[b] = 0; render3d_st.gvk_buf_map[b] = null; render3d_st.gvk_buf_used[b] = 0
|
||||
}
|
||||
# the frame's submitted work is done: storage retired during it can go
|
||||
function gvk_retire_flush() -> void {
|
||||
if gvk_retired_buf == null { return }
|
||||
for i in 0 .. len(gvk_retired_buf) {
|
||||
Vk.destroy_buffer(gvk_dev, gvk_retired_buf[i], null)
|
||||
gvk_mem_free(gvk_mem_id(gvk_retired_mem[i]))
|
||||
function gvk_retire_flush(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gvk_retired_buf == null { return }
|
||||
for i in 0 .. len(render3d_st.gvk_retired_buf) {
|
||||
Vk.destroy_buffer(render3d_st.gvk_dev, render3d_st.gvk_retired_buf[i], null)
|
||||
gvk_mem_free(render3d_st, gvk_mem_id(render3d_st.gvk_retired_mem[i]))
|
||||
}
|
||||
gvk_retired_buf = new []long
|
||||
gvk_retired_mem = new []long
|
||||
render3d_st.gvk_retired_buf = new []long
|
||||
render3d_st.gvk_retired_mem = new []long
|
||||
}
|
||||
|
||||
# A buffer a compute pass writes: a draw later in the same frame reads what the GPU put there,
|
||||
# so it is never swapped for fresh storage because it was used this frame (gvk_buf_reserve).
|
||||
var gvk_buf_gpu: []int = null
|
||||
function gvk_buf_gpu_owned(b: int) -> void {
|
||||
if gvk_buf_gpu == null { gvk_buf_gpu = new []int }
|
||||
while len(gvk_buf_gpu) <= b { push(gvk_buf_gpu, 0) }
|
||||
gvk_buf_gpu[b] = 1
|
||||
function gvk_buf_gpu_owned(render3d_st: mut Render3dState, b: int) -> void {
|
||||
if render3d_st.gvk_buf_gpu == null { render3d_st.gvk_buf_gpu = new []int }
|
||||
while len(render3d_st.gvk_buf_gpu) <= b { push(render3d_st.gvk_buf_gpu, 0) }
|
||||
render3d_st.gvk_buf_gpu[b] = 1
|
||||
}
|
||||
function gvk_buf_is_gpu(b: int) -> bool { return gvk_buf_gpu != null and b < len(gvk_buf_gpu) and gvk_buf_gpu[b] == 1 }
|
||||
function gvk_buf_is_gpu(render3d_st: Render3dState, b: int) -> bool { return render3d_st.gvk_buf_gpu != null and b < len(render3d_st.gvk_buf_gpu) and render3d_st.gvk_buf_gpu[b] == 1 }
|
||||
|
||||
# Room for at least n bytes behind handle b; a buffer that is already big enough is kept, so a
|
||||
# stream re-filled every frame allocates once.
|
||||
function gvk_buf_reserve(b: int, n: int) -> bool {
|
||||
if b <= 0 or b >= len(gvk_buf) { return false }
|
||||
function gvk_buf_reserve(render3d_st: mut Render3dState, b: int, n: int) -> bool {
|
||||
if b <= 0 or b >= len(render3d_st.gvk_buf) { return false }
|
||||
# already big enough, and no draw this frame reads what is there: fill it in place
|
||||
if gvk_buf[b] != 0 and gvk_buf_size[b] >= n and (gvk_buf_used[b] != gvk_frame_no or gvk_buf_is_gpu(b)) { return true }
|
||||
gvk_buf_release(b)
|
||||
if render3d_st.gvk_buf[b] != 0 and render3d_st.gvk_buf_size[b] >= n and (render3d_st.gvk_buf_used[b] != render3d_st.gvk_frame_no or gvk_buf_is_gpu(render3d_st, b)) { return true }
|
||||
gvk_buf_release(render3d_st, b)
|
||||
var size = n
|
||||
if size < 64 { size = 64 }
|
||||
let size_l: long = size
|
||||
|
|
@ -628,24 +599,24 @@ function gvk_buf_reserve(b: int, n: int) -> bool {
|
|||
Vk.put_i32(bci, VkBufferCreateInfo_usage, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT | VK_BUFFER_USAGE_INDEX_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT | VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT)
|
||||
Vk.put_i32(bci, VkBufferCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE)
|
||||
let out = bytes(8)
|
||||
var r = Vk.create_buffer(gvk_dev, bci, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(`vkCreateBuffer ({size} bytes)`, r) }
|
||||
var r = Vk.create_buffer(render3d_st.gvk_dev, bci, null, out)
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, `vkCreateBuffer ({size} bytes)`, r) }
|
||||
let buf = gvk_handle(out)
|
||||
let req = bytes(VkMemoryRequirements_sizeof)
|
||||
Vk.get_buffer_memory_requirements(gvk_dev, buf, req)
|
||||
let ma = gvk_mem_new(req, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, false)
|
||||
Vk.get_buffer_memory_requirements(render3d_st.gvk_dev, buf, req)
|
||||
let ma = gvk_mem_new(render3d_st, req, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, false)
|
||||
let zero: long = 0
|
||||
if ma == 0 { Vk.destroy_buffer(gvk_dev, buf, null); return false }
|
||||
if ma == 0 { Vk.destroy_buffer(render3d_st.gvk_dev, buf, null); return false }
|
||||
let mem: long = ma
|
||||
r = Vk.bind_buffer_memory(gvk_dev, buf, gvk_mem_handle(ma), gvk_mem_offset(ma))
|
||||
if r != VK_SUCCESS { return gvk_fail("vkBindBufferMemory", r) }
|
||||
gvk_buf[b] = buf; gvk_buf_mem[b] = mem; gvk_buf_size[b] = size; gvk_buf_map[b] = gvk_mem_ptr(ma)
|
||||
r = Vk.bind_buffer_memory(render3d_st.gvk_dev, buf, gvk_mem_handle(render3d_st, ma), gvk_mem_offset(render3d_st, ma))
|
||||
if r != VK_SUCCESS { return gvk_fail(render3d_st, "vkBindBufferMemory", r) }
|
||||
render3d_st.gvk_buf[b] = buf; render3d_st.gvk_buf_mem[b] = mem; render3d_st.gvk_buf_size[b] = size; render3d_st.gvk_buf_map[b] = gvk_mem_ptr(render3d_st, ma)
|
||||
return true
|
||||
}
|
||||
|
||||
# glBufferData: the whole buffer, from data (or storage only when data is null)
|
||||
function gvk_buf_upload(b: int, n: int, data: pointer) -> bool {
|
||||
if not gvk_buf_reserve(b, n) { return false }
|
||||
if data != null and n > 0 { mem_copy(gvk_buf_map[b], data, n) }
|
||||
function gvk_buf_upload(render3d_st: mut Render3dState, b: int, n: int, data: pointer) -> bool {
|
||||
if not gvk_buf_reserve(render3d_st, b, n) { return false }
|
||||
if data != null and n > 0 { mem_copy(render3d_st.gvk_buf_map[b], data, n) }
|
||||
return true
|
||||
}
|
||||
|
|
|
|||
|
|
@ -11,60 +11,35 @@
|
|||
# ============================================================================
|
||||
|
||||
const GRASS_CELL: int = 16
|
||||
var grass_prog: int = 0
|
||||
var grass_mesh: Mesh = null
|
||||
var grass_on: bool = true
|
||||
var grass_wind: float = 0.0
|
||||
# A photographed blade, as an atlas of straightened blades side by side (the game sets
|
||||
# this; the renderer does not name a game asset). 0 = the procedural gradient, which is
|
||||
# what this was for a year: a two-tone ramp with a hard edge, and every blade in the
|
||||
# valley the same blade. A real blade has a midrib, a colour that runs olive to straw,
|
||||
# browning where it has dried and a tip that is its own shape - none of which can be
|
||||
# written down, only photographed.
|
||||
var grass_blade_tex: int = 0
|
||||
var grass_blade_cols: int = 8
|
||||
# Where a body is standing, and how wide it pushes. The grass has never known the player was
|
||||
# in it: you walked through a meadow and every blade ignored you, which is the single most
|
||||
# noticeable thing missing from every step the game asks you to take. The game sets this each
|
||||
# frame; radius 0 means nobody is there.
|
||||
var grass_push_x: float = 0.0
|
||||
var grass_push_z: float = 0.0
|
||||
var grass_push_r: float = 0.0
|
||||
var grass_s0: float = 0.0 # float bits: blade spacing at the camera (m)
|
||||
var grass_d0: float = 0.0 # the distance at which the spacing has doubled (m)
|
||||
var grass_radius: float = 0.0 # no blades past this (m)
|
||||
var grass_draws: int = 0
|
||||
var grass_dbg: int = 0
|
||||
# Vulkan with multi-draw indirect: every visible tile is a record in one buffer, uploaded once a
|
||||
# frame, and each band draws its records GRASS_CHUNK at a time - one draw for up to 256 tiles. A
|
||||
# record's firstInstance is its place in its chunk times 65536; grass.vert's TILES variant reads that
|
||||
# place's corner and indices per cell from u_tiles. R3D_GRASS_TILES=1 keeps a draw per tile.
|
||||
const GRASS_CHUNK: int = 256
|
||||
const GRASS_RECS: int = 8192
|
||||
var grass_merge: bool = false
|
||||
var grass_rec: words = null # VkDrawIndexedIndirectCommand records, 5 words each
|
||||
var grass_tv: floats = null # per record: corner x, corner z, indices per cell, 0 (float bits)
|
||||
var grass_chunk_tv: floats = null
|
||||
var grass_n: int = 0
|
||||
var grass_cmds: int = 0
|
||||
var grass_band_start: words = null # per band: its first record
|
||||
var grass_band_cells: words = null # ... and its 16 m cells per tile side
|
||||
var grass_band_n: int = 0
|
||||
# Mesh-shader grass (Settings, Video, Advanced): the chunked path's records, but each chunk is one
|
||||
# mesh dispatch - work group y a tile, x a batch of GRASS_MESH_BLADES of its blades - so a blade the
|
||||
# tests reject emits nothing instead of eight degenerate vertices. R3D_MESH_GRASS=1 / 0 overrides.
|
||||
const GRASS_MESH_BLADES: int = 16
|
||||
var grass_mesh_on: bool = false
|
||||
var grass_mesh_prog: int = 0
|
||||
function r3d_mesh_grass(on: bool) -> void {
|
||||
grass_mesh_on = on
|
||||
if r3d_env_has("R3D_MESH_GRASS") { grass_mesh_on = Text.to_int(r3d_env("R3D_MESH_GRASS")) != 0 }
|
||||
function r3d_mesh_grass(render3d_st: mut Render3dState, on: bool) -> void {
|
||||
render3d_st.grass_mesh_on = on
|
||||
if r3d_env_has(render3d_st, "R3D_MESH_GRASS") { render3d_st.grass_mesh_on = Text.to_int(r3d_env(render3d_st, "R3D_MESH_GRASS")) != 0 }
|
||||
}
|
||||
function grass_mesh_live() -> bool { return grass_mesh_on and grass_merge and grass_mesh_prog != 0 and gpu_has_mesh() }
|
||||
function grass_mesh_live(render3d_st: Render3dState) -> bool { return render3d_st.grass_mesh_on and render3d_st.grass_merge and render3d_st.grass_mesh_prog != 0 and gpu_has_mesh(render3d_st) }
|
||||
|
||||
# a blade: `rows` rows of 2 vertices (x across, y along, z bend), attribute 2 = uv
|
||||
function grass_blade_mesh(rows: int) -> Mesh {
|
||||
let m = gpu_mesh_new()
|
||||
function grass_blade_mesh(render3d_st: mut Render3dState, rows: int) -> Mesh {
|
||||
let m = gpu_mesh_new(render3d_st)
|
||||
let v = gl_floats(rows * 2 * 5)
|
||||
var k = 0
|
||||
# The blade's PROFILE, and it is the whole difference between grass and a green spike.
|
||||
|
|
@ -94,9 +69,9 @@ function grass_blade_mesh(rows: int) -> Mesh {
|
|||
k += 5
|
||||
}
|
||||
}
|
||||
gpu_mesh_vertices(m, v, gl_bytes_of(rows * 2 * 5), GPU_STATIC)
|
||||
gpu_mesh_attr(m, 0, 3, GPU_F32, 20, 0, false)
|
||||
gpu_mesh_attr(m, 2, 2, GPU_F32, 20, 12, false)
|
||||
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(rows * 2 * 5), GPU_STATIC)
|
||||
gpu_mesh_attr(render3d_st, m, 0, 3, GPU_F32, 20, 0, false)
|
||||
gpu_mesh_attr(render3d_st, m, 2, 2, GPU_F32, 20, 12, false)
|
||||
free(v)
|
||||
let nq = rows - 1
|
||||
let idx = words(nq * 6)
|
||||
|
|
@ -105,65 +80,65 @@ function grass_blade_mesh(rows: int) -> Mesh {
|
|||
idx[q * 6] = b; idx[q * 6 + 1] = b + 1; idx[q * 6 + 2] = b + 2
|
||||
idx[q * 6 + 3] = b + 1; idx[q * 6 + 4] = b + 3; idx[q * 6 + 5] = b + 2
|
||||
}
|
||||
gpu_mesh_indices(m, data_of(idx), nq * 6 * 4, 4)
|
||||
gpu_mesh_indices(render3d_st, m, data_of(idx), nq * 6 * 4, 4)
|
||||
free(idx)
|
||||
m.count = nq * 6
|
||||
gpu_mesh_done(m)
|
||||
gpu_mesh_done(render3d_st, m)
|
||||
return m
|
||||
}
|
||||
|
||||
function grass_init() -> void {
|
||||
grass_merge = gpu_has_mdi() and not r3d_env_has("R3D_GRASS_TILES")
|
||||
function grass_init(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.grass_merge = gpu_has_mdi(render3d_st) and not r3d_env_has(render3d_st, "R3D_GRASS_TILES")
|
||||
var defs = "#define FOLIAGE\n#define BLADE\n"
|
||||
if grass_merge {
|
||||
if render3d_st.grass_merge {
|
||||
defs = defs + "#define TILES\n"
|
||||
grass_rec = words(GRASS_RECS * 5); grass_tv = floats(GRASS_RECS * 4); grass_chunk_tv = floats(GRASS_CHUNK * 4)
|
||||
grass_band_start = words(4); grass_band_cells = words(4)
|
||||
grass_cmds = gpu_buffer_new()
|
||||
render3d_st.grass_rec = words(GRASS_RECS * 5); render3d_st.grass_tv = floats(GRASS_RECS * 4); render3d_st.grass_chunk_tv = floats(GRASS_CHUNK * 4)
|
||||
render3d_st.grass_band_start = words(4); render3d_st.grass_band_cells = words(4)
|
||||
render3d_st.grass_cmds = gpu_buffer_new(render3d_st)
|
||||
}
|
||||
grass_prog = r3d_program("grass.vert", "model.frag", defs)
|
||||
if grass_merge and gpu_has_mesh() { grass_mesh_prog = r3d_program("grass.mesh", "model.frag", "#define FOLIAGE\n#define BLADE\n#define MESH\n") }
|
||||
render3d_st.grass_prog = r3d_program(render3d_st, "grass.vert", "model.frag", defs)
|
||||
if render3d_st.grass_merge and gpu_has_mesh(render3d_st) { render3d_st.grass_mesh_prog = r3d_program(render3d_st, "grass.mesh", "model.frag", "#define FOLIAGE\n#define BLADE\n#define MESH\n") }
|
||||
# five rows, four quads: the arch above needs somewhere to bend, and at four rows a
|
||||
# blade that leans over is three straight segments and shows every join
|
||||
grass_mesh = grass_blade_mesh(5)
|
||||
grass_wind = 2.4
|
||||
render3d_st.grass_mesh = grass_blade_mesh(render3d_st, 5)
|
||||
render3d_st.grass_wind = 2.4
|
||||
# Matched to the blade's width: a 1 cm blade at 0.11 m spacing covers a third of what a
|
||||
# 2.8 cm blade did, and the meadow goes bare. The game's graphics settings override this
|
||||
# (gfx_grass_spacing), but only once game_init has run - a plain headless render never
|
||||
# gets there, so the two have to agree or a shot shows something no player will see.
|
||||
# That is exactly how the last change measured as "no effect": the render was identical
|
||||
# because this line, not the settings, was deciding.
|
||||
grass_s0 = 0.066
|
||||
grass_d0 = 45.0
|
||||
grass_radius = 1600.0
|
||||
if r3d_env_has("R3D_NOBLADES") { grass_on = false }
|
||||
if r3d_env_has("R3D_GRASS_R") { grass_radius = float(Text.to_int(r3d_env("R3D_GRASS_R"))) }
|
||||
if r3d_env_has("R3D_GRASS_DBG") { grass_dbg = Text.to_int(r3d_env("R3D_GRASS_DBG")) }
|
||||
render3d_st.grass_s0 = 0.066
|
||||
render3d_st.grass_d0 = 45.0
|
||||
render3d_st.grass_radius = 1600.0
|
||||
if r3d_env_has(render3d_st, "R3D_NOBLADES") { render3d_st.grass_on = false }
|
||||
if r3d_env_has(render3d_st, "R3D_GRASS_R") { render3d_st.grass_radius = float(Text.to_int(r3d_env(render3d_st, "R3D_GRASS_R"))) }
|
||||
if r3d_env_has(render3d_st, "R3D_GRASS_DBG") { render3d_st.grass_dbg = Text.to_int(r3d_env(render3d_st, "R3D_GRASS_DBG")) }
|
||||
}
|
||||
|
||||
# indices per 16 m cell that could exist at distance d (the count the shader computes)
|
||||
function grass_count_at(d: float) -> int {
|
||||
let spacing = grass_s0 * (1.0 + d / grass_d0)
|
||||
function grass_count_at(render3d_st: Render3dState, d: float) -> int {
|
||||
let spacing = render3d_st.grass_s0 * (1.0 + d / render3d_st.grass_d0)
|
||||
let n = float(GRASS_CELL * GRASS_CELL) / (spacing * spacing)
|
||||
return int(n) + 1
|
||||
}
|
||||
|
||||
# one tile size over one distance band
|
||||
function grass_tiles(size: int, d_min: float, d_max: float) -> void {
|
||||
let p = grass_prog
|
||||
function grass_tiles(render3d_st: mut Render3dState, size: int, d_min: float, d_max: float) -> void {
|
||||
let p = render3d_st.grass_prog
|
||||
let cells = size / GRASS_CELL
|
||||
if grass_merge {
|
||||
grass_band_start[grass_band_n] = grass_n; grass_band_cells[grass_band_n] = cells; grass_band_n += 1
|
||||
if render3d_st.grass_merge {
|
||||
render3d_st.grass_band_start[render3d_st.grass_band_n] = render3d_st.grass_n; render3d_st.grass_band_cells[render3d_st.grass_band_n] = cells; render3d_st.grass_band_n += 1
|
||||
} else {
|
||||
u_i(gpu_uniform(p, "u_tile_cells"), cells)
|
||||
u_i(render3d_st, gpu_uniform(render3d_st, p, "u_tile_cells"), cells)
|
||||
}
|
||||
let sz = float(size)
|
||||
let half = sz * 0.5
|
||||
let reach = d_max + half * 1.5
|
||||
let tx0 = int(Math.floor((cam_pos[0] - reach) / sz))
|
||||
let tx1 = int(Math.floor((cam_pos[0] + reach) / sz))
|
||||
let tz0 = int(Math.floor((cam_pos[2] - reach) / sz))
|
||||
let tz1 = int(Math.floor((cam_pos[2] + reach) / sz))
|
||||
let tx0 = int(Math.floor((render3d_st.cam_pos[0] - reach) / sz))
|
||||
let tx1 = int(Math.floor((render3d_st.cam_pos[0] + reach) / sz))
|
||||
let tz0 = int(Math.floor((render3d_st.cam_pos[2] - reach) / sz))
|
||||
let tz1 = int(Math.floor((render3d_st.cam_pos[2] + reach) / sz))
|
||||
let corner_r = half * 1.42
|
||||
var tz = tz0
|
||||
while tz <= tz1 {
|
||||
|
|
@ -171,30 +146,30 @@ function grass_tiles(size: int, d_min: float, d_max: float) -> void {
|
|||
while tx <= tx1 {
|
||||
let ox = float(tx) * sz; let oz = float(tz) * sz
|
||||
let cx = ox + half; let cz = oz + half
|
||||
let dx = cx - cam_pos[0]; let dz = cz - cam_pos[2]
|
||||
let dx = cx - render3d_st.cam_pos[0]; let dz = cz - render3d_st.cam_pos[2]
|
||||
let dc = Math.sqrt(dx * dx + dz * dz)
|
||||
# the tile's nearest and farthest points decide which band it belongs to
|
||||
let dnear = Math.max(dc - corner_r, 0.0)
|
||||
if dc < d_min or not (dnear < d_max) { tx += 1; continue }
|
||||
let cy = terrain_height(cx, cz)
|
||||
if cam_sphere_visible(cx, cy, cz, corner_r + 6.0) {
|
||||
let per = grass_count_at(dnear)
|
||||
if per > 0 and grass_merge {
|
||||
if grass_n < GRASS_RECS {
|
||||
let cy = terrain_height(render3d_st, cx, cz)
|
||||
if cam_sphere_visible(render3d_st, cx, cy, cz, corner_r + 6.0) {
|
||||
let per = grass_count_at(render3d_st, dnear)
|
||||
if per > 0 and render3d_st.grass_merge {
|
||||
if render3d_st.grass_n < GRASS_RECS {
|
||||
var inst = per * cells * cells
|
||||
if inst > 65535 { inst = 65535 }
|
||||
let r = grass_n * 5
|
||||
grass_rec[r] = grass_mesh.count; grass_rec[r + 1] = inst; grass_rec[r + 2] = 0; grass_rec[r + 3] = 0
|
||||
grass_rec[r + 4] = ((grass_n - grass_band_start[grass_band_n - 1]) % GRASS_CHUNK) * 65536
|
||||
let t = grass_n * 4
|
||||
grass_tv[t] = ox; grass_tv[t + 1] = oz; grass_tv[t + 2] = float(per); grass_tv[t + 3] = 0.0
|
||||
grass_n += 1
|
||||
let r = render3d_st.grass_n * 5
|
||||
render3d_st.grass_rec[r] = render3d_st.grass_mesh.count; render3d_st.grass_rec[r + 1] = inst; render3d_st.grass_rec[r + 2] = 0; render3d_st.grass_rec[r + 3] = 0
|
||||
render3d_st.grass_rec[r + 4] = ((render3d_st.grass_n - render3d_st.grass_band_start[render3d_st.grass_band_n - 1]) % GRASS_CHUNK) * 65536
|
||||
let t = render3d_st.grass_n * 4
|
||||
render3d_st.grass_tv[t] = ox; render3d_st.grass_tv[t + 1] = oz; render3d_st.grass_tv[t + 2] = float(per); render3d_st.grass_tv[t + 3] = 0.0
|
||||
render3d_st.grass_n += 1
|
||||
}
|
||||
} else if per > 0 {
|
||||
u_f2(gpu_uniform(p, "u_tile"), ox, oz)
|
||||
u_i(gpu_uniform(p, "u_per_cell"), per)
|
||||
mesh_draw_instanced(grass_mesh, per * cells * cells)
|
||||
grass_draws += 1
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, p, "u_tile"), ox, oz)
|
||||
u_i(render3d_st, gpu_uniform(render3d_st, p, "u_per_cell"), per)
|
||||
mesh_draw_instanced(render3d_st, render3d_st.grass_mesh, per * cells * cells)
|
||||
render3d_st.grass_draws += 1
|
||||
}
|
||||
}
|
||||
tx += 1
|
||||
|
|
@ -203,104 +178,104 @@ function grass_tiles(size: int, d_min: float, d_max: float) -> void {
|
|||
}
|
||||
}
|
||||
|
||||
function grass_draw() -> void {
|
||||
if not grass_on or ter_reflect or grass_prog == 0 { return }
|
||||
var p = grass_prog
|
||||
if grass_mesh_live() { p = grass_mesh_prog }
|
||||
gpu_use_program(p)
|
||||
u_mat4(gpu_uniform(p, "u_view"), cam_view)
|
||||
u_mat4(gpu_uniform(p, "u_proj"), cam_proj)
|
||||
u_mat4(gpu_uniform(p, "u_vp"), cam_vp_clean)
|
||||
u_f(gpu_uniform(p, "u_wind"), grass_wind)
|
||||
function grass_draw(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.grass_on or render3d_st.ter_reflect or render3d_st.grass_prog == 0 { return }
|
||||
var p = render3d_st.grass_prog
|
||||
if grass_mesh_live(render3d_st) { p = render3d_st.grass_mesh_prog }
|
||||
gpu_use_program(render3d_st, p)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_vp"), render3d_st.cam_vp_clean)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_wind"), render3d_st.grass_wind)
|
||||
# copied into a local first: a global reaching a uniform call is the codegen fault
|
||||
# CLAUDE.md records against u_wade and u_flutter, and it costs a day every time
|
||||
let btex = grass_blade_tex
|
||||
let bcols = grass_blade_cols
|
||||
u_f(gpu_uniform(p, "u_blade_cols"), float(bcols))
|
||||
let btex = render3d_st.grass_blade_tex
|
||||
let bcols = render3d_st.grass_blade_cols
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_blade_cols"), float(bcols))
|
||||
var bon = 0.0
|
||||
if btex != 0 { bon = 1.0; r3d_bind_2d(p, "u_blade_tex", 12, btex) }
|
||||
u_f(gpu_uniform(p, "u_blade_tex_on"), bon)
|
||||
u_f3(gpu_uniform(p, "u_push"), grass_push_x, grass_push_z, grass_push_r)
|
||||
u_f(gpu_uniform(p, "u_rough_scale"), 1.0)
|
||||
u_v3(gpu_uniform(p, "u_tint"), sc_blade_tint)
|
||||
u_v3(gpu_uniform(p, "u_blade_base"), sc_blade_base)
|
||||
u_v3(gpu_uniform(p, "u_blade_tip"), sc_blade_tip)
|
||||
u_f(gpu_uniform(p, "u_cull"), grass_radius)
|
||||
u_f(gpu_uniform(p, "u_model_h"), 0.0)
|
||||
u_f(gpu_uniform(p, "u_s0"), grass_s0)
|
||||
u_f(gpu_uniform(p, "u_d0"), grass_d0)
|
||||
u_f(gpu_uniform(p, "u_radius"), grass_radius)
|
||||
u_i(gpu_uniform(p, "u_dbg"), grass_dbg)
|
||||
var orthotex = ter_ortho_tex
|
||||
if btex != 0 { bon = 1.0; r3d_bind_2d(render3d_st, p, "u_blade_tex", 12, btex) }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_blade_tex_on"), bon)
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, p, "u_push"), render3d_st.grass_push_x, render3d_st.grass_push_z, render3d_st.grass_push_r)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_rough_scale"), 1.0)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_tint"), render3d_st.sc_blade_tint)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_blade_base"), render3d_st.sc_blade_base)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, p, "u_blade_tip"), render3d_st.sc_blade_tip)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_cull"), render3d_st.grass_radius)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_model_h"), 0.0)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_s0"), render3d_st.grass_s0)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_d0"), render3d_st.grass_d0)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_radius"), render3d_st.grass_radius)
|
||||
u_i(render3d_st, gpu_uniform(render3d_st, p, "u_dbg"), render3d_st.grass_dbg)
|
||||
var orthotex = render3d_st.ter_ortho_tex
|
||||
var oon = 1.0
|
||||
if orthotex == 0 { orthotex = ter_height_tex; oon = 0.0 }
|
||||
r3d_bind_2d(p, "u_ortho", 4, orthotex)
|
||||
u_f(gpu_uniform(p, "u_ortho_on"), oon)
|
||||
if orthotex == 0 { orthotex = render3d_st.ter_height_tex; oon = 0.0 }
|
||||
r3d_bind_2d(render3d_st, p, "u_ortho", 4, orthotex)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_ortho_on"), oon)
|
||||
var lake = -100000.0
|
||||
if ter_lake_ex != 0.0 { lake = ter_lake_level }
|
||||
u_f(gpu_uniform(p, "u_lake_level"), lake)
|
||||
if render3d_st.ter_lake_ex != 0.0 { lake = render3d_st.ter_lake_level }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_lake_level"), lake)
|
||||
var sea = lake
|
||||
if ter_sea_set { sea = ter_sea_level }
|
||||
u_f(gpu_uniform(p, "u_sea_level"), sea)
|
||||
u_f4(gpu_uniform(p, "u_lake"), ter_lake_cx, ter_lake_cz, ter_lake_ex, ter_lake_ez)
|
||||
u_f(gpu_uniform(p, "u_snow_line"), ter_snow_line)
|
||||
sky_bind_lighting(p)
|
||||
shadow_bind(p)
|
||||
fog_bind(p)
|
||||
if render3d_st.ter_sea_set { sea = render3d_st.ter_sea_level }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_sea_level"), sea)
|
||||
u_f4(render3d_st, gpu_uniform(render3d_st, p, "u_lake"), render3d_st.ter_lake_cx, render3d_st.ter_lake_cz, render3d_st.ter_lake_ex, render3d_st.ter_lake_ez)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_snow_line"), render3d_st.ter_snow_line)
|
||||
sky_bind_lighting(render3d_st, p)
|
||||
shadow_bind(render3d_st, p)
|
||||
fog_bind(render3d_st, p)
|
||||
# 0.15 was enough to put a hard white highlight down the length of a blade whenever it
|
||||
# caught the sun, and a white blade of grass is the one thing grass is never. Measured:
|
||||
# at 0.15, 0.70% of a near-ground frame was over 210 of 255; at 0.0 it is 0.05%. A blade
|
||||
# does have a faint sheen, so this is small rather than nothing - the foliage layers have
|
||||
# used 0.05 all along and never showed the fault.
|
||||
u_f(gpu_uniform(p, "u_spec_scale"), 0.008)
|
||||
gpu_cull(false)
|
||||
grass_draws = 0
|
||||
grass_n = 0
|
||||
grass_band_n = 0
|
||||
gpu_mesh_bind(grass_mesh)
|
||||
grass_tiles(16, 0.0, 300.0)
|
||||
grass_tiles(64, 300.0, 1200.0)
|
||||
grass_tiles(256, 1200.0, grass_radius)
|
||||
if grass_merge { grass_flush() }
|
||||
gpu_cull(true)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_spec_scale"), 0.008)
|
||||
gpu_cull(render3d_st, false)
|
||||
render3d_st.grass_draws = 0
|
||||
render3d_st.grass_n = 0
|
||||
render3d_st.grass_band_n = 0
|
||||
gpu_mesh_bind(render3d_st, render3d_st.grass_mesh)
|
||||
grass_tiles(render3d_st, 16, 0.0, 300.0)
|
||||
grass_tiles(render3d_st, 64, 300.0, 1200.0)
|
||||
grass_tiles(render3d_st, 256, 1200.0, render3d_st.grass_radius)
|
||||
if render3d_st.grass_merge { grass_flush(render3d_st) }
|
||||
gpu_cull(render3d_st, true)
|
||||
}
|
||||
|
||||
# the records gathered this frame: uploaded once, then each band GRASS_CHUNK records a draw
|
||||
function grass_flush() -> void {
|
||||
if grass_n == 0 { return }
|
||||
var p = grass_prog
|
||||
let mesh = grass_mesh_live()
|
||||
if mesh { p = grass_mesh_prog } else { gpu_buffer_upload(grass_cmds, grass_n * 20, data_of(grass_rec), GPU_DYNAMIC) }
|
||||
for b in 0 .. grass_band_n {
|
||||
let s = grass_band_start[b]
|
||||
var e = grass_n
|
||||
if b + 1 < grass_band_n { e = grass_band_start[b + 1] }
|
||||
if e > s { u_i(gpu_uniform(p, "u_tile_cells"), grass_band_cells[b]) }
|
||||
function grass_flush(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.grass_n == 0 { return }
|
||||
var p = render3d_st.grass_prog
|
||||
let mesh = grass_mesh_live(render3d_st)
|
||||
if mesh { p = render3d_st.grass_mesh_prog } else { gpu_buffer_upload(render3d_st, render3d_st.grass_cmds, render3d_st.grass_n * 20, data_of(render3d_st.grass_rec), GPU_DYNAMIC) }
|
||||
for b in 0 .. render3d_st.grass_band_n {
|
||||
let s = render3d_st.grass_band_start[b]
|
||||
var e = render3d_st.grass_n
|
||||
if b + 1 < render3d_st.grass_band_n { e = render3d_st.grass_band_start[b + 1] }
|
||||
if e > s { u_i(render3d_st, gpu_uniform(render3d_st, p, "u_tile_cells"), render3d_st.grass_band_cells[b]) }
|
||||
var k = s
|
||||
while k < e {
|
||||
var m = e - k
|
||||
if m > GRASS_CHUNK { m = GRASS_CHUNK }
|
||||
for q in 0 .. m * 4 { grass_chunk_tv[q] = grass_tv[k * 4 + q] }
|
||||
for q in 0 .. m * 4 { render3d_st.grass_chunk_tv[q] = render3d_st.grass_tv[k * 4 + q] }
|
||||
if mesh {
|
||||
# one dispatch per tile, as many blade batches as that tile has: sized for a chunk's largest
|
||||
# tile, the far tiles beside a near one ran thousands of empty invocations (9.8 ms against the
|
||||
# chunked path's 2 at 4K on an RTX 3070 Ti)
|
||||
let cells = grass_band_cells[b]
|
||||
let cells = render3d_st.grass_band_cells[b]
|
||||
for q in 0 .. m {
|
||||
var total = int(grass_tv[(k + q) * 4 + 2]) * cells * cells
|
||||
var total = int(render3d_st.grass_tv[(k + q) * 4 + 2]) * cells * cells
|
||||
if total > 65535 { total = 65535 }
|
||||
if total > 0 {
|
||||
for c in 0 .. 4 { grass_chunk_tv[c] = grass_tv[(k + q) * 4 + c] }
|
||||
u_f4v(gpu_uniform(p, "u_tiles"), 1, grass_chunk_tv)
|
||||
gpu_draw_mesh_tasks((total + GRASS_MESH_BLADES - 1) / GRASS_MESH_BLADES, 1, 1)
|
||||
grass_draws += 1
|
||||
for c in 0 .. 4 { render3d_st.grass_chunk_tv[c] = render3d_st.grass_tv[(k + q) * 4 + c] }
|
||||
u_f4v(render3d_st, gpu_uniform(render3d_st, p, "u_tiles"), 1, render3d_st.grass_chunk_tv)
|
||||
gpu_draw_mesh_tasks(render3d_st, (total + GRASS_MESH_BLADES - 1) / GRASS_MESH_BLADES, 1, 1)
|
||||
render3d_st.grass_draws += 1
|
||||
}
|
||||
}
|
||||
} else {
|
||||
u_f4v(gpu_uniform(p, "u_tiles"), m, grass_chunk_tv)
|
||||
gpu_draw_mesh_indirect(grass_mesh, grass_cmds, k * 20, m, 0, 0)
|
||||
u_f4v(render3d_st, gpu_uniform(render3d_st, p, "u_tiles"), m, render3d_st.grass_chunk_tv)
|
||||
gpu_draw_mesh_indirect(render3d_st, render3d_st.grass_mesh, render3d_st.grass_cmds, k * 20, m, 0, 0)
|
||||
}
|
||||
if not mesh { grass_draws += 1 }
|
||||
if not mesh { render3d_st.grass_draws += 1 }
|
||||
k += m
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,17 +1,14 @@
|
|||
# hooks.ludic - the scene a frame draws, as callbacks the program registers. They used to be
|
||||
# functions every program had to define by name (scene_draw, scene_draw_casters, stream_fill),
|
||||
# a contract only the linker enforced and that allowed one scene per program.
|
||||
var r3d_draw_cb: fn() = null
|
||||
var r3d_casters_cb: fn(floats) = null
|
||||
var r3d_fill_cb: fn(Stream, int, int, int) = null
|
||||
|
||||
# what a frame draws, for every pass that draws the scene (the view, the reflection)
|
||||
function r3d_on_draw(f: fn()) -> void { r3d_draw_cb = f }
|
||||
function r3d_on_draw(render3d_st: mut Render3dState, f: fn()) -> void { render3d_st.r3d_draw_cb = f }
|
||||
# what casts a shadow, drawn into the light's view `light_vp`
|
||||
function r3d_on_casters(f: fn(floats)) -> void { r3d_casters_cb = f }
|
||||
function r3d_on_casters(render3d_st: mut Render3dState, f: fn(floats)) -> void { render3d_st.r3d_casters_cb = f }
|
||||
# a streamed layer's chunk (cx, cz) at detail `band`, filled with its instances
|
||||
function r3d_on_stream_fill(f: fn(Stream, int, int, int)) -> void { r3d_fill_cb = f }
|
||||
function r3d_on_stream_fill(render3d_st: mut Render3dState, f: fn(Stream, int, int, int)) -> void { render3d_st.r3d_fill_cb = f }
|
||||
|
||||
function r3d_scene_draw() -> void { if r3d_draw_cb != null { r3d_draw_cb() } }
|
||||
function r3d_scene_casters(light_vp: floats) -> void { if r3d_casters_cb != null { r3d_casters_cb(light_vp) } }
|
||||
function r3d_stream_fill(s: Stream, cx: int, cz: int, band: int) -> void { if r3d_fill_cb != null { r3d_fill_cb(s, cx, cz, band) } }
|
||||
function r3d_scene_draw(render3d_st: Render3dState) -> void { if render3d_st.r3d_draw_cb != null { render3d_st.r3d_draw_cb() } }
|
||||
function r3d_scene_casters(render3d_st: Render3dState, light_vp: floats) -> void { if render3d_st.r3d_casters_cb != null { render3d_st.r3d_casters_cb(light_vp) } }
|
||||
function r3d_stream_fill(render3d_st: Render3dState, s: Stream, cx: int, cz: int, band: int) -> void { if render3d_st.r3d_fill_cb != null { render3d_st.r3d_fill_cb(s, cx, cz, band) } }
|
||||
|
|
|
|||
|
|
@ -20,13 +20,13 @@ property Mesh {
|
|||
vk_layout: int = 0 # the Vulkan backend's interned id for the recorded layout (0: not yet)
|
||||
}
|
||||
|
||||
function mesh_draw(m: Mesh) -> void { gpu_draw_mesh(m) }
|
||||
function mesh_draw_instanced(m: Mesh, n: int) -> void { gpu_draw_mesh_instanced(m, n) }
|
||||
function mesh_draw(render3d_st: mut Render3dState, m: Mesh) -> void { gpu_draw_mesh(render3d_st, m) }
|
||||
function mesh_draw_instanced(render3d_st: mut Render3dState, m: Mesh, n: int) -> void { gpu_draw_mesh_instanced(render3d_st, m, n) }
|
||||
|
||||
# A flat n x n vertex grid over [-half, half]^2 in x/z, y = 0. Attribute 0 = (x, z).
|
||||
# The terrain vertex shader lifts it with the height map.
|
||||
function mesh_grid(n: int, half: float) -> Mesh {
|
||||
let m = gpu_mesh_new()
|
||||
function mesh_grid(render3d_st: mut Render3dState, n: int, half: float) -> Mesh {
|
||||
let m = gpu_mesh_new(render3d_st)
|
||||
let nv = n * n
|
||||
let v = gl_floats(nv * 2)
|
||||
var k = 0
|
||||
|
|
@ -38,8 +38,8 @@ function mesh_grid(n: int, half: float) -> Mesh {
|
|||
k += 2
|
||||
}
|
||||
}
|
||||
gpu_mesh_vertices(m, v, gl_bytes_of(nv * 2), GPU_STATIC)
|
||||
gpu_mesh_attr(m, 0, 2, GPU_F32, 8, 0, false)
|
||||
gpu_mesh_vertices(render3d_st, m, v, gl_bytes_of(nv * 2), GPU_STATIC)
|
||||
gpu_mesh_attr(render3d_st, m, 0, 2, GPU_F32, 8, 0, false)
|
||||
free(v)
|
||||
let ni = (n - 1) * (n - 1) * 6
|
||||
let idx = words(ni)
|
||||
|
|
@ -52,41 +52,41 @@ function mesh_grid(n: int, half: float) -> Mesh {
|
|||
k += 6
|
||||
}
|
||||
}
|
||||
gpu_mesh_indices(m, data_of(idx), ni * 4, 4)
|
||||
gpu_mesh_indices(render3d_st, m, data_of(idx), ni * 4, 4)
|
||||
free(idx)
|
||||
m.count = ni
|
||||
gpu_mesh_done(m)
|
||||
gpu_mesh_done(render3d_st, m)
|
||||
return m
|
||||
}
|
||||
|
||||
# A full-screen triangle with no attributes (the vertex shader uses gl_VertexID).
|
||||
function mesh_fullscreen() -> Mesh {
|
||||
let m = gpu_mesh_new()
|
||||
gpu_mesh_done(m)
|
||||
function mesh_fullscreen(render3d_st: Render3dState) -> Mesh {
|
||||
let m = gpu_mesh_new(render3d_st)
|
||||
gpu_mesh_done(render3d_st, m)
|
||||
m.count = 3
|
||||
return m
|
||||
}
|
||||
|
||||
# A unit quad in x/y ([-0.5, 0.5] x [0, 1]) with uv, attribute 0 = xy, 1 = uv.
|
||||
function mesh_card() -> Mesh {
|
||||
let m = gpu_mesh_new()
|
||||
function mesh_card(render3d_st: mut Render3dState) -> Mesh {
|
||||
let m = gpu_mesh_new(render3d_st)
|
||||
let v = gl_floats(16)
|
||||
gl_put(v, 0, -0.5); gl_put(v, 1, 0.0); gl_put(v, 2, 0.0); gl_put(v, 3, 0.0)
|
||||
gl_put(v, 4, 0.5); gl_put(v, 5, 0.0); gl_put(v, 6, 1.0); gl_put(v, 7, 0.0)
|
||||
gl_put(v, 8, 0.5); gl_put(v, 9, 1.0); gl_put(v, 10, 1.0); gl_put(v, 11, 1.0)
|
||||
gl_put(v, 12, -0.5); gl_put(v, 13, 1.0); gl_put(v, 14, 0.0); gl_put(v, 15, 1.0)
|
||||
gpu_mesh_vertices(m, v, 64, GPU_STATIC)
|
||||
gpu_mesh_attr(m, 0, 2, GPU_F32, 16, 0, false)
|
||||
gpu_mesh_attr(m, 1, 2, GPU_F32, 16, 8, false)
|
||||
gpu_mesh_vertices(render3d_st, m, v, 64, GPU_STATIC)
|
||||
gpu_mesh_attr(render3d_st, m, 0, 2, GPU_F32, 16, 0, false)
|
||||
gpu_mesh_attr(render3d_st, m, 1, 2, GPU_F32, 16, 8, false)
|
||||
free(v)
|
||||
let idx = words(6)
|
||||
idx[0] = 0; idx[1] = 1; idx[2] = 2; idx[3] = 0; idx[4] = 2; idx[5] = 3
|
||||
gpu_mesh_indices(m, data_of(idx), 24, 4)
|
||||
gpu_mesh_indices(render3d_st, m, data_of(idx), 24, 4)
|
||||
free(idx)
|
||||
m.count = 6
|
||||
gpu_mesh_done(m)
|
||||
gpu_mesh_done(render3d_st, m)
|
||||
return m
|
||||
}
|
||||
|
||||
# release a mesh's GPU objects (a mesh this package built for something being thrown away)
|
||||
function mesh_free(m: Mesh) -> void { gpu_mesh_free(m) }
|
||||
function mesh_free(render3d_st: mut Render3dState, m: Mesh) -> void { gpu_mesh_free(render3d_st, m) }
|
||||
|
|
|
|||
|
|
@ -11,360 +11,322 @@
|
|||
const OV_MAX_QUADS: int = 6000
|
||||
const OV_FLOATS: int = 8 # x, y, u, v, r, g, b, a
|
||||
|
||||
var ov_prog: int = 0
|
||||
var ov_mesh: Mesh = null
|
||||
var ov_vbo: int = 0
|
||||
var ov_buf: pointer = null
|
||||
var ov_n: int = 0
|
||||
var ov_tex: int = 0
|
||||
var ov_mode: int = 2 # what the next quads read: 0 the image texture, 1 the font, 2 a flat colour
|
||||
var ov_voff: float = 0.0 # float bits: 4 x ov_mode, added to v (overlay.frag)
|
||||
var ov_white: int = 0
|
||||
var ov_font: int = 0
|
||||
var ov_font_adv: floats = null # float bits, em units, one per glyph in the atlas
|
||||
var ov_font_n: int = 0 # glyphs in the atlas
|
||||
# Text is UTF-8, and the atlas says which code points it holds. font.json's "codes" lists
|
||||
# them in atlas order; an older atlas without it is ASCII from "first" (32) on, which is
|
||||
# what every font built before this was. ov_font_map[cp] is glyph + 1 for a code point
|
||||
# under OV_MAP_N (0: not in the atlas); anything above is looked up in the sorted tail.
|
||||
const OV_MAP_N: int = 8192
|
||||
var ov_font_map: words = null
|
||||
var ov_font_hi_cp: words = null # code points >= OV_MAP_N, ascending
|
||||
var ov_font_hi_g: words = null
|
||||
var ov_font_hi_n: int = 0
|
||||
var ov_font_space: int = 0 # the glyph for ' ', and for what has none: '?'
|
||||
var ov_font_qmark: int = 0
|
||||
var ov_font_cols: int = 16
|
||||
var ov_font_rows: int = 6
|
||||
var ov_font_cell: int = 128 # px per cell in the atlas
|
||||
var ov_font_em: int = 100 # px per em in the atlas
|
||||
var ov_pad_x: float = 0.0 # float bits, em
|
||||
var ov_base_y: float = 0.0
|
||||
var ov_ready: bool = false
|
||||
var ov_open: bool = false
|
||||
var ov_dbg: bool = false
|
||||
const OV_MAX_RANGES: int = 512
|
||||
const OV_RANGE_W: int = 7 # texture, first quad, quad count, clip x, y, w, h (w = 0: none)
|
||||
var ov_ranges: words = null
|
||||
var ov_nr: int = 0
|
||||
var ov_range_start: int = 0
|
||||
var ov_clip_x: int = 0 # the current clip rectangle in screen pixels (top-left origin)
|
||||
var ov_clip_y: int = 0
|
||||
var ov_clip_w: int = 0
|
||||
var ov_clip_h: int = 0
|
||||
|
||||
# The interface is sRGB. While the output is HDR10 the overlay draws with its HDR10 variant, which puts
|
||||
# the interface at the picture's paper white; the SDR program there sent sRGB values as PQ, and the
|
||||
# menu's yellow read as red on an HDR display.
|
||||
var ov_prog_sdr: int = 0
|
||||
var ov_prog_hdr: int = 0
|
||||
# How bright the interface's white is while HDR10 is on, as a multiple of paper white. It is 1 for the
|
||||
# interface; a calibration screen draws its test patches at a number of nits with ov_hdr_nits, which
|
||||
# closes the batch so far - the multiple is one uniform per flush. On an SDR frame it does nothing.
|
||||
var ov_hdr_scale: float = 0.0 # float bits; 0 until ov_begin sets 1
|
||||
function ov_hdr_nits(nits: float) -> void {
|
||||
function ov_hdr_nits(render3d_st: mut Render3dState, nits: float) -> void {
|
||||
var s = 1.0
|
||||
if nits != 0.0 { s = nits / r3d_hdr_paper_nits() }
|
||||
if s == ov_hdr_scale { return }
|
||||
if ov_open { ov_flush() }
|
||||
ov_hdr_scale = s
|
||||
if nits != 0.0 { s = nits / r3d_hdr_paper_nits(render3d_st) }
|
||||
if s == render3d_st.ov_hdr_scale { return }
|
||||
if render3d_st.ov_open { ov_flush(render3d_st) }
|
||||
render3d_st.ov_hdr_scale = s
|
||||
}
|
||||
# back to the interface's own white
|
||||
function ov_hdr_paper() -> void { ov_hdr_nits(0.0) }
|
||||
function ov_pick_prog() -> void {
|
||||
if gpu_hdr_active() {
|
||||
if ov_prog_hdr == 0 {
|
||||
r3d_program_log("overlay.vert", "overlay.frag", "#define HDR10\n")
|
||||
ov_prog_hdr = gpu_program("#version 410 core\n#define HDR10\n" + r3d_shader_file("overlay.vert"), "#version 410 core\n#define HDR10\n" + r3d_shader_file("overlay.frag"), "overlay.vert", "overlay.frag", "#define HDR10\n")
|
||||
function ov_hdr_paper(render3d_st: mut Render3dState) -> void { ov_hdr_nits(render3d_st, 0.0) }
|
||||
function ov_pick_prog(render3d_st: mut Render3dState) -> void {
|
||||
if gpu_hdr_active(render3d_st) {
|
||||
if render3d_st.ov_prog_hdr == 0 {
|
||||
r3d_program_log(render3d_st, "overlay.vert", "overlay.frag", "#define HDR10\n")
|
||||
render3d_st.ov_prog_hdr = gpu_program(render3d_st, "#version 410 core\n#define HDR10\n" + r3d_shader_file(render3d_st, "overlay.vert"), "#version 410 core\n#define HDR10\n" + r3d_shader_file(render3d_st, "overlay.frag"), "overlay.vert", "overlay.frag", "#define HDR10\n")
|
||||
}
|
||||
if ov_prog_hdr != 0 { ov_prog = ov_prog_hdr; return }
|
||||
if render3d_st.ov_prog_hdr != 0 { render3d_st.ov_prog = render3d_st.ov_prog_hdr; return }
|
||||
}
|
||||
ov_prog = ov_prog_sdr
|
||||
render3d_st.ov_prog = render3d_st.ov_prog_sdr
|
||||
}
|
||||
|
||||
function overlay_init(font_dir: string) -> bool {
|
||||
r3d_program_log("overlay.vert", "overlay.frag", "")
|
||||
ov_prog = gpu_program("#version 410 core\n" + r3d_shader_file("overlay.vert"), "#version 410 core\n" + r3d_shader_file("overlay.frag"), "overlay.vert", "overlay.frag", "")
|
||||
if ov_prog == 0 { print("overlay: program failed"); return false }
|
||||
ov_prog_sdr = ov_prog
|
||||
ov_mesh = gpu_mesh_new()
|
||||
ov_vbo = gpu_mesh_vertices(ov_mesh, null, gl_bytes_of(OV_MAX_QUADS * 6 * OV_FLOATS), GPU_DYNAMIC)
|
||||
gpu_mesh_attr(ov_mesh, 0, 2, GPU_F32, OV_FLOATS * 4, 0, false)
|
||||
gpu_mesh_attr(ov_mesh, 1, 2, GPU_F32, OV_FLOATS * 4, 8, false)
|
||||
gpu_mesh_attr(ov_mesh, 2, 4, GPU_F32, OV_FLOATS * 4, 16, false)
|
||||
gpu_mesh_done(ov_mesh)
|
||||
ov_buf = gl_floats(OV_MAX_QUADS * 6 * OV_FLOATS)
|
||||
ov_ranges = words(OV_MAX_RANGES * OV_RANGE_W)
|
||||
ov_white = tex_solid(255, 255, 255, 255)
|
||||
overlay_font(font_dir)
|
||||
if ov_font == 0 { print("overlay: no font atlas, text disabled") }
|
||||
ov_dbg = r3d_env_has("R3D_FONTDBG")
|
||||
ov_ready = true
|
||||
function overlay_init(render3d_st: mut Render3dState, font_dir: string) -> bool {
|
||||
r3d_program_log(render3d_st, "overlay.vert", "overlay.frag", "")
|
||||
render3d_st.ov_prog = gpu_program(render3d_st, "#version 410 core\n" + r3d_shader_file(render3d_st, "overlay.vert"), "#version 410 core\n" + r3d_shader_file(render3d_st, "overlay.frag"), "overlay.vert", "overlay.frag", "")
|
||||
if render3d_st.ov_prog == 0 { print("overlay: program failed"); return false }
|
||||
render3d_st.ov_prog_sdr = render3d_st.ov_prog
|
||||
render3d_st.ov_mesh = gpu_mesh_new(render3d_st)
|
||||
render3d_st.ov_vbo = gpu_mesh_vertices(render3d_st, render3d_st.ov_mesh, null, gl_bytes_of(OV_MAX_QUADS * 6 * OV_FLOATS), GPU_DYNAMIC)
|
||||
gpu_mesh_attr(render3d_st, render3d_st.ov_mesh, 0, 2, GPU_F32, OV_FLOATS * 4, 0, false)
|
||||
gpu_mesh_attr(render3d_st, render3d_st.ov_mesh, 1, 2, GPU_F32, OV_FLOATS * 4, 8, false)
|
||||
gpu_mesh_attr(render3d_st, render3d_st.ov_mesh, 2, 4, GPU_F32, OV_FLOATS * 4, 16, false)
|
||||
gpu_mesh_done(render3d_st, render3d_st.ov_mesh)
|
||||
render3d_st.ov_buf = gl_floats(OV_MAX_QUADS * 6 * OV_FLOATS)
|
||||
render3d_st.ov_ranges = words(OV_MAX_RANGES * OV_RANGE_W)
|
||||
render3d_st.ov_white = tex_solid(render3d_st, 255, 255, 255, 255)
|
||||
overlay_font(render3d_st, font_dir)
|
||||
if render3d_st.ov_font == 0 { print("overlay: no font atlas, text disabled") }
|
||||
render3d_st.ov_dbg = r3d_env_has(render3d_st, "R3D_FONTDBG")
|
||||
render3d_st.ov_ready = true
|
||||
return true
|
||||
}
|
||||
|
||||
# Load (or swap to) the font atlas in `font_dir`: font.png and font.json. A game calls it
|
||||
# again to change fonts at run time - a language whose script the default atlas does not
|
||||
# carry brings its own. Returns false, and keeps the font it had, if there is none there.
|
||||
function overlay_font(font_dir: string) -> bool {
|
||||
function overlay_font(render3d_st: mut Render3dState, font_dir: string) -> bool {
|
||||
let meta = Fs.read_text(font_dir + "/font.json")
|
||||
if meta == null { return false }
|
||||
let tex = tex_load_ex(font_dir + "/font.png", false, 0)
|
||||
let tex = tex_load_ex(render3d_st, font_dir + "/font.png", false, 0)
|
||||
if tex == 0 { return false }
|
||||
let j = Json.parse(meta)
|
||||
ov_font_cols = jint(j, "cols", 16); ov_font_rows = jint(j, "rows", 6)
|
||||
ov_font_cell = int(jnum(value_get(j, "cell"))); ov_font_em = int(jnum(value_get(j, "em")))
|
||||
ov_pad_x = 0.14; ov_base_y = 0.30
|
||||
if value_has(j, "pad_x") != 0 { ov_pad_x = jnum(value_get(j, "pad_x")) }
|
||||
if value_has(j, "base_y") != 0 { ov_base_y = jnum(value_get(j, "base_y")) }
|
||||
render3d_st.ov_font_cols = jint(j, "cols", 16); render3d_st.ov_font_rows = jint(j, "rows", 6)
|
||||
render3d_st.ov_font_cell = int(jnum(value_get(j, "cell"))); render3d_st.ov_font_em = int(jnum(value_get(j, "em")))
|
||||
render3d_st.ov_pad_x = 0.14; render3d_st.ov_base_y = 0.30
|
||||
if value_has(j, "pad_x") != 0 { render3d_st.ov_pad_x = jnum(value_get(j, "pad_x")) }
|
||||
if value_has(j, "base_y") != 0 { render3d_st.ov_base_y = jnum(value_get(j, "base_y")) }
|
||||
let adv = value_get(j, "adv")
|
||||
let n = value_count(adv)
|
||||
let first = jint(j, "first", 32)
|
||||
var codes: Val = null
|
||||
if value_has(j, "codes") != 0 { codes = value_get(j, "codes") }
|
||||
ov_font_n = n
|
||||
ov_font_adv = floats(n + 1)
|
||||
for i in 0 .. n { ov_font_adv[i] = jnum(value_at(adv, i)) }
|
||||
ov_font_map = words(OV_MAP_N)
|
||||
for c in 0 .. OV_MAP_N { ov_font_map[c] = 0 }
|
||||
ov_font_hi_cp = words(n + 1); ov_font_hi_g = words(n + 1); ov_font_hi_n = 0
|
||||
render3d_st.ov_font_n = n
|
||||
render3d_st.ov_font_adv = floats(n + 1)
|
||||
for i in 0 .. n { render3d_st.ov_font_adv[i] = jnum(value_at(adv, i)) }
|
||||
render3d_st.ov_font_map = words(OV_MAP_N)
|
||||
for c in 0 .. OV_MAP_N { render3d_st.ov_font_map[c] = 0 }
|
||||
render3d_st.ov_font_hi_cp = words(n + 1); render3d_st.ov_font_hi_g = words(n + 1); render3d_st.ov_font_hi_n = 0
|
||||
for i in 0 .. n {
|
||||
var cp = first + i
|
||||
if codes != null and i < value_count(codes) { cp = int(jnum(value_at(codes, i))) }
|
||||
if cp >= 0 and cp < OV_MAP_N { ov_font_map[cp] = i + 1 }
|
||||
if cp >= 0 and cp < OV_MAP_N { render3d_st.ov_font_map[cp] = i + 1 }
|
||||
else if cp >= OV_MAP_N {
|
||||
# kept ascending: the builder writes code points in order, so this is an append
|
||||
ov_font_hi_cp[ov_font_hi_n] = cp; ov_font_hi_g[ov_font_hi_n] = i; ov_font_hi_n += 1
|
||||
render3d_st.ov_font_hi_cp[render3d_st.ov_font_hi_n] = cp; render3d_st.ov_font_hi_g[render3d_st.ov_font_hi_n] = i; render3d_st.ov_font_hi_n += 1
|
||||
}
|
||||
}
|
||||
ov_font_space = 0
|
||||
if ov_font_map[32] > 0 { ov_font_space = ov_font_map[32] - 1 }
|
||||
ov_font_qmark = ov_font_space
|
||||
if ov_font_map[63] > 0 { ov_font_qmark = ov_font_map[63] - 1 }
|
||||
ov_font = tex
|
||||
render3d_st.ov_font_space = 0
|
||||
if render3d_st.ov_font_map[32] > 0 { render3d_st.ov_font_space = render3d_st.ov_font_map[32] - 1 }
|
||||
render3d_st.ov_font_qmark = render3d_st.ov_font_space
|
||||
if render3d_st.ov_font_map[63] > 0 { render3d_st.ov_font_qmark = render3d_st.ov_font_map[63] - 1 }
|
||||
render3d_st.ov_font = tex
|
||||
return true
|
||||
}
|
||||
|
||||
# The code point starting at sp[i], with the bytes it took in ov_u8_len. A malformed or
|
||||
# cut-off sequence is one byte of '?', so a string sliced mid-character still draws.
|
||||
var ov_u8_len: int = 1
|
||||
function ov_u8(sp: pointer, i: int, n: int) -> int {
|
||||
function ov_u8(render3d_st: mut Render3dState, sp: pointer, i: int, n: int) -> int {
|
||||
let b0 = sp[i] & 255
|
||||
ov_u8_len = 1
|
||||
render3d_st.ov_u8_len = 1
|
||||
if b0 < 128 { return b0 }
|
||||
if b0 >= 240 and b0 < 248 and i + 3 < n {
|
||||
ov_u8_len = 4
|
||||
render3d_st.ov_u8_len = 4
|
||||
return ((b0 & 7) << 18) | ((sp[i + 1] & 63) << 12) | ((sp[i + 2] & 63) << 6) | (sp[i + 3] & 63)
|
||||
}
|
||||
if b0 >= 224 and b0 < 240 and i + 2 < n {
|
||||
ov_u8_len = 3
|
||||
render3d_st.ov_u8_len = 3
|
||||
return ((b0 & 15) << 12) | ((sp[i + 1] & 63) << 6) | (sp[i + 2] & 63)
|
||||
}
|
||||
if b0 >= 192 and b0 < 224 and i + 1 < n {
|
||||
ov_u8_len = 2
|
||||
render3d_st.ov_u8_len = 2
|
||||
return ((b0 & 31) << 6) | (sp[i + 1] & 63)
|
||||
}
|
||||
return 63
|
||||
}
|
||||
# the atlas glyph for a code point: a control character is a space, a missing one is '?'
|
||||
function ov_glyph(cp: int) -> int {
|
||||
if cp < 32 { return ov_font_space }
|
||||
function ov_glyph(render3d_st: Render3dState, cp: int) -> int {
|
||||
if cp < 32 { return render3d_st.ov_font_space }
|
||||
if cp < OV_MAP_N {
|
||||
let g = ov_font_map[cp]
|
||||
let g = render3d_st.ov_font_map[cp]
|
||||
if g > 0 { return g - 1 }
|
||||
return ov_font_qmark
|
||||
return render3d_st.ov_font_qmark
|
||||
}
|
||||
var lo = 0
|
||||
var hi = ov_font_hi_n - 1
|
||||
var hi = render3d_st.ov_font_hi_n - 1
|
||||
while lo <= hi {
|
||||
let mid = (lo + hi) / 2
|
||||
if ov_font_hi_cp[mid] == cp { return ov_font_hi_g[mid] }
|
||||
if ov_font_hi_cp[mid] < cp { lo = mid + 1 } else { hi = mid - 1 }
|
||||
if render3d_st.ov_font_hi_cp[mid] == cp { return render3d_st.ov_font_hi_g[mid] }
|
||||
if render3d_st.ov_font_hi_cp[mid] < cp { lo = mid + 1 } else { hi = mid - 1 }
|
||||
}
|
||||
return ov_font_qmark
|
||||
return render3d_st.ov_font_qmark
|
||||
}
|
||||
|
||||
# start drawing onto the screen: blending on, depth off
|
||||
function ov_begin() -> void {
|
||||
if not ov_ready { return }
|
||||
gpu_fb_bind(gpu_screen_fb())
|
||||
gpu_viewport(0, 0, gl_w, gl_h)
|
||||
gpu_depth_test(false)
|
||||
gpu_cull(false)
|
||||
gpu_blend(true)
|
||||
gpu_blend_func(GL_SRC_ALPHA, GL_ONE_MINUS_SRC_ALPHA)
|
||||
ov_pick_prog()
|
||||
gpu_use_program(ov_prog)
|
||||
u_f2(gpu_uniform(ov_prog, "u_screen"), float(gl_w), float(gl_h))
|
||||
ov_n = 0; ov_nr = 0; ov_range_start = 0
|
||||
ov_tex = ov_white
|
||||
ov_mode = 2; ov_voff = 8.0
|
||||
ov_clip_w = 0
|
||||
ov_hdr_scale = 1.0
|
||||
ov_open = true
|
||||
function ov_begin(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.ov_ready { return }
|
||||
gpu_fb_bind(render3d_st, gpu_screen_fb(render3d_st))
|
||||
gpu_viewport(render3d_st, 0, 0, gl_w, gl_h)
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_cull(render3d_st, false)
|
||||
gpu_blend(render3d_st, true)
|
||||
gpu_blend_func(render3d_st, GL_SRC_ALPHA, GL_ONE_MINUS_SRC_ALPHA)
|
||||
ov_pick_prog(render3d_st)
|
||||
gpu_use_program(render3d_st, render3d_st.ov_prog)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.ov_prog, "u_screen"), float(gl_w), float(gl_h))
|
||||
render3d_st.ov_n = 0; render3d_st.ov_nr = 0; render3d_st.ov_range_start = 0
|
||||
render3d_st.ov_tex = render3d_st.ov_white
|
||||
render3d_st.ov_mode = 2; render3d_st.ov_voff = 8.0
|
||||
render3d_st.ov_clip_w = 0
|
||||
render3d_st.ov_hdr_scale = 1.0
|
||||
render3d_st.ov_open = true
|
||||
}
|
||||
# everything drawn until ov_unclip stays inside this rectangle (a scrolling list)
|
||||
function ov_clip(x: int, y: int, w: int, h: int) -> void {
|
||||
ov_close_range()
|
||||
ov_clip_x = x; ov_clip_y = y; ov_clip_w = w; ov_clip_h = h
|
||||
function ov_clip(render3d_st: mut Render3dState, x: int, y: int, w: int, h: int) -> void {
|
||||
ov_close_range(render3d_st)
|
||||
render3d_st.ov_clip_x = x; render3d_st.ov_clip_y = y; render3d_st.ov_clip_w = w; render3d_st.ov_clip_h = h
|
||||
}
|
||||
function ov_unclip() -> void { ov_close_range(); ov_clip_w = 0 }
|
||||
function ov_unclip(render3d_st: mut Render3dState) -> void { ov_close_range(render3d_st); render3d_st.ov_clip_w = 0 }
|
||||
# close the current draw range (quads since its start, with the current texture)
|
||||
function ov_close_range() -> void {
|
||||
let n = ov_n - ov_range_start
|
||||
function ov_close_range(render3d_st: mut Render3dState) -> void {
|
||||
let n = render3d_st.ov_n - render3d_st.ov_range_start
|
||||
if n <= 0 { return }
|
||||
if ov_nr >= OV_MAX_RANGES { ov_flush(); return }
|
||||
let o = ov_nr * OV_RANGE_W
|
||||
ov_ranges[o] = ov_tex; ov_ranges[o + 1] = ov_range_start; ov_ranges[o + 2] = n
|
||||
ov_ranges[o + 3] = ov_clip_x; ov_ranges[o + 4] = ov_clip_y; ov_ranges[o + 5] = ov_clip_w; ov_ranges[o + 6] = ov_clip_h
|
||||
ov_nr += 1
|
||||
ov_range_start = ov_n
|
||||
if render3d_st.ov_nr >= OV_MAX_RANGES { ov_flush(render3d_st); return }
|
||||
let o = render3d_st.ov_nr * OV_RANGE_W
|
||||
render3d_st.ov_ranges[o] = render3d_st.ov_tex; render3d_st.ov_ranges[o + 1] = render3d_st.ov_range_start; render3d_st.ov_ranges[o + 2] = n
|
||||
render3d_st.ov_ranges[o + 3] = render3d_st.ov_clip_x; render3d_st.ov_ranges[o + 4] = render3d_st.ov_clip_y; render3d_st.ov_ranges[o + 5] = render3d_st.ov_clip_w; render3d_st.ov_ranges[o + 6] = render3d_st.ov_clip_h
|
||||
render3d_st.ov_nr += 1
|
||||
render3d_st.ov_range_start = render3d_st.ov_n
|
||||
}
|
||||
# one upload of everything batched so far, then a draw per range
|
||||
function ov_flush() -> void {
|
||||
ov_close_range()
|
||||
if ov_n == 0 { ov_nr = 0; ov_range_start = 0; return }
|
||||
function ov_flush(render3d_st: mut Render3dState) -> void {
|
||||
ov_close_range(render3d_st)
|
||||
if render3d_st.ov_n == 0 { render3d_st.ov_nr = 0; render3d_st.ov_range_start = 0; return }
|
||||
# The overlay draws onto the screen, whatever was bound since ov_begin: a render-scale change
|
||||
# rebuilds the scene targets mid-frame and leaves framebuffer 0 bound, and on a headless run
|
||||
# (where the screen is an offscreen framebuffer) every overlay draw after it was an invalid
|
||||
# framebuffer operation. Saying the target at each flush is what a render pass says anyway.
|
||||
gpu_fb_bind(gpu_screen_fb())
|
||||
gpu_viewport(0, 0, gl_w, gl_h)
|
||||
ov_pick_prog()
|
||||
gpu_use_program(ov_prog)
|
||||
gpu_mesh_bind(ov_mesh)
|
||||
gpu_buffer_upload(ov_vbo, gl_bytes_of(ov_n * 6 * OV_FLOATS), ov_buf, GPU_STREAM)
|
||||
var font = ov_font
|
||||
if font == 0 { font = ov_white }
|
||||
r3d_bind_2d(ov_prog, "u_font", 1, font)
|
||||
var hs = ov_hdr_scale
|
||||
gpu_fb_bind(render3d_st, gpu_screen_fb(render3d_st))
|
||||
gpu_viewport(render3d_st, 0, 0, gl_w, gl_h)
|
||||
ov_pick_prog(render3d_st)
|
||||
gpu_use_program(render3d_st, render3d_st.ov_prog)
|
||||
gpu_mesh_bind(render3d_st, render3d_st.ov_mesh)
|
||||
gpu_buffer_upload(render3d_st, render3d_st.ov_vbo, gl_bytes_of(render3d_st.ov_n * 6 * OV_FLOATS), render3d_st.ov_buf, GPU_STREAM)
|
||||
var font = render3d_st.ov_font
|
||||
if font == 0 { font = render3d_st.ov_white }
|
||||
r3d_bind_2d(render3d_st, render3d_st.ov_prog, "u_font", 1, font)
|
||||
var hs = render3d_st.ov_hdr_scale
|
||||
if hs == 0.0 { hs = 1.0 }
|
||||
u_f(gpu_uniform(ov_prog, "u_hdr_paper"), r3d_hdr_paper_nits())
|
||||
u_f(gpu_uniform(ov_prog, "u_hdr_scale"), hs)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.ov_prog, "u_hdr_paper"), r3d_hdr_paper_nits(render3d_st))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.ov_prog, "u_hdr_scale"), hs)
|
||||
var last = -1
|
||||
var clipped = false
|
||||
for i in 0 .. ov_nr {
|
||||
for i in 0 .. render3d_st.ov_nr {
|
||||
let o = i * OV_RANGE_W
|
||||
let t = ov_ranges[o]
|
||||
if t != last { r3d_bind_2d(ov_prog, "u_tex", 0, t); last = t }
|
||||
if ov_ranges[o + 5] > 0 {
|
||||
let t = render3d_st.ov_ranges[o]
|
||||
if t != last { r3d_bind_2d(render3d_st, render3d_st.ov_prog, "u_tex", 0, t); last = t }
|
||||
if render3d_st.ov_ranges[o + 5] > 0 {
|
||||
clipped = true
|
||||
gpu_scissor(ov_ranges[o + 3], ov_ranges[o + 4], ov_ranges[o + 5], ov_ranges[o + 6])
|
||||
} else if clipped { gpu_scissor_off(); clipped = false }
|
||||
gpu_draw_range(ov_mesh, ov_ranges[o + 1] * 6, ov_ranges[o + 2] * 6)
|
||||
gpu_scissor(render3d_st, render3d_st.ov_ranges[o + 3], render3d_st.ov_ranges[o + 4], render3d_st.ov_ranges[o + 5], render3d_st.ov_ranges[o + 6])
|
||||
} else if clipped { gpu_scissor_off(render3d_st); clipped = false }
|
||||
gpu_draw_range(render3d_st, render3d_st.ov_mesh, render3d_st.ov_ranges[o + 1] * 6, render3d_st.ov_ranges[o + 2] * 6)
|
||||
}
|
||||
if clipped { gpu_scissor_off() }
|
||||
gpu_mesh_unbind()
|
||||
ov_n = 0; ov_nr = 0; ov_range_start = 0
|
||||
if clipped { gpu_scissor_off(render3d_st) }
|
||||
gpu_mesh_unbind(render3d_st)
|
||||
render3d_st.ov_n = 0; render3d_st.ov_nr = 0; render3d_st.ov_range_start = 0
|
||||
}
|
||||
function ov_end() -> void {
|
||||
if not ov_open { return }
|
||||
ov_flush()
|
||||
gpu_blend(false)
|
||||
gpu_depth_test(true)
|
||||
ov_open = false
|
||||
function ov_end(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.ov_open { return }
|
||||
ov_flush(render3d_st)
|
||||
gpu_blend(render3d_st, false)
|
||||
gpu_depth_test(render3d_st, true)
|
||||
render3d_st.ov_open = false
|
||||
}
|
||||
# Text and flat panels never end a draw: the font atlas stays bound beside the image texture and each
|
||||
# vertex says which it reads (ov_voff). Only a different image texture closes the range. A panel and
|
||||
# its label used to alternate the white and font textures, and the HUD was 77-91 draws a frame.
|
||||
function ov_use_tex(t: int) -> void {
|
||||
if t != 0 and t == ov_font { ov_mode = 1; ov_voff = 4.0; return }
|
||||
if t == ov_white { ov_mode = 2; ov_voff = 8.0; return }
|
||||
ov_mode = 0; ov_voff = 0.0
|
||||
if t != ov_tex { ov_close_range(); ov_tex = t }
|
||||
function ov_use_tex(render3d_st: mut Render3dState, t: int) -> void {
|
||||
if t != 0 and t == render3d_st.ov_font { render3d_st.ov_mode = 1; render3d_st.ov_voff = 4.0; return }
|
||||
if t == render3d_st.ov_white { render3d_st.ov_mode = 2; render3d_st.ov_voff = 8.0; return }
|
||||
render3d_st.ov_mode = 0; render3d_st.ov_voff = 0.0
|
||||
if t != render3d_st.ov_tex { ov_close_range(render3d_st); render3d_st.ov_tex = t }
|
||||
}
|
||||
|
||||
# one vertex into the batch
|
||||
function ov_vert(k: int, x: float, y: float, u: float, v: float, r: float, g: float, b: float, a: float) -> void {
|
||||
function ov_vert(render3d_st: Render3dState, k: int, x: float, y: float, u: float, v: float, r: float, g: float, b: float, a: float) -> void {
|
||||
let o = k * OV_FLOATS
|
||||
gl_put_bits(ov_buf, o, float_bits(x)); gl_put_bits(ov_buf, o + 1, float_bits(y))
|
||||
gl_put_bits(ov_buf, o + 2, float_bits(u)); gl_put_bits(ov_buf, o + 3, float_bits(v + ov_voff))
|
||||
gl_put_bits(ov_buf, o + 4, float_bits(r)); gl_put_bits(ov_buf, o + 5, float_bits(g)); gl_put_bits(ov_buf, o + 6, float_bits(b)); gl_put_bits(ov_buf, o + 7, float_bits(a))
|
||||
gl_put_bits(render3d_st.ov_buf, o, float_bits(x)); gl_put_bits(render3d_st.ov_buf, o + 1, float_bits(y))
|
||||
gl_put_bits(render3d_st.ov_buf, o + 2, float_bits(u)); gl_put_bits(render3d_st.ov_buf, o + 3, float_bits(v + render3d_st.ov_voff))
|
||||
gl_put_bits(render3d_st.ov_buf, o + 4, float_bits(r)); gl_put_bits(render3d_st.ov_buf, o + 5, float_bits(g)); gl_put_bits(render3d_st.ov_buf, o + 6, float_bits(b)); gl_put_bits(render3d_st.ov_buf, o + 7, float_bits(a))
|
||||
}
|
||||
# a textured quad, float-bit pixel corners and uvs
|
||||
function ov_quad(x0: float, y0: float, x1: float, y1: float, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
if ov_n >= OV_MAX_QUADS { ov_flush() }
|
||||
let k = ov_n * 6
|
||||
ov_vert(k, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(k + 1, x1, y0, u1, v0, r, g, b, a)
|
||||
ov_vert(k + 2, x1, y1, u1, v1, r, g, b, a)
|
||||
ov_vert(k + 3, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(k + 4, x1, y1, u1, v1, r, g, b, a)
|
||||
ov_vert(k + 5, x0, y1, u0, v1, r, g, b, a)
|
||||
ov_n += 1
|
||||
function ov_quad(render3d_st: mut Render3dState, x0: float, y0: float, x1: float, y1: float, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
if render3d_st.ov_n >= OV_MAX_QUADS { ov_flush(render3d_st) }
|
||||
let k = render3d_st.ov_n * 6
|
||||
ov_vert(render3d_st, k, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 1, x1, y0, u1, v0, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 2, x1, y1, u1, v1, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 3, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 4, x1, y1, u1, v1, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 5, x0, y1, u0, v1, r, g, b, a)
|
||||
render3d_st.ov_n += 1
|
||||
}
|
||||
|
||||
# a filled rectangle at integer pixels; colour as float bits 0..1
|
||||
function ov_rect(x: int, y: int, w: int, h: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(ov_white)
|
||||
ov_quad(float(x), float(y), float(x + w), float(y + h), 0.0, 0.0, 1.0, 1.0, r, g, b, a)
|
||||
function ov_rect(render3d_st: mut Render3dState, x: int, y: int, w: int, h: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(render3d_st, render3d_st.ov_white)
|
||||
ov_quad(render3d_st, float(x), float(y), float(x + w), float(y + h), 0.0, 0.0, 1.0, 1.0, r, g, b, a)
|
||||
}
|
||||
function ov_frame(x: int, y: int, w: int, h: int, t: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_rect(x, y, w, t, r, g, b, a)
|
||||
ov_rect(x, y + h - t, w, t, r, g, b, a)
|
||||
ov_rect(x, y, t, h, r, g, b, a)
|
||||
ov_rect(x + w - t, y, t, h, r, g, b, a)
|
||||
function ov_frame(render3d_st: mut Render3dState, x: int, y: int, w: int, h: int, t: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_rect(render3d_st, x, y, w, t, r, g, b, a)
|
||||
ov_rect(render3d_st, x, y + h - t, w, t, r, g, b, a)
|
||||
ov_rect(render3d_st, x, y, t, h, r, g, b, a)
|
||||
ov_rect(render3d_st, x + w - t, y, t, h, r, g, b, a)
|
||||
}
|
||||
# a whole texture at integer pixels
|
||||
function ov_image(tex: int, x: int, y: int, w: int, h: int, a: float) -> void {
|
||||
ov_use_tex(tex)
|
||||
ov_quad(float(x), float(y), float(x + w), float(y + h), 0.0, 0.0, 1.0, 1.0, 1.0, 1.0, 1.0, a)
|
||||
function ov_image(render3d_st: mut Render3dState, tex: int, x: int, y: int, w: int, h: int, a: float) -> void {
|
||||
ov_use_tex(render3d_st, tex)
|
||||
ov_quad(render3d_st, float(x), float(y), float(x + w), float(y + h), 0.0, 0.0, 1.0, 1.0, 1.0, 1.0, 1.0, a)
|
||||
}
|
||||
|
||||
# the width in pixels of `s` at `size` pixels per em
|
||||
function ov_text_w(size: int, s: string) -> int {
|
||||
if ov_font_adv == null { return 0 }
|
||||
function ov_text_w(render3d_st: mut Render3dState, size: int, s: string) -> int {
|
||||
if render3d_st.ov_font_adv == null { return 0 }
|
||||
var w = 0.0
|
||||
let sp: pointer = s # UTF-8 bytes, not one-character strings
|
||||
let n = len(sp)
|
||||
var i = 0
|
||||
while i < n {
|
||||
let g = ov_glyph(ov_u8(sp, i, n))
|
||||
i += ov_u8_len
|
||||
w = w + ov_font_adv[g] * float(size)
|
||||
let g = ov_glyph(render3d_st, ov_u8(render3d_st, sp, i, n))
|
||||
i += render3d_st.ov_u8_len
|
||||
w = w + render3d_st.ov_font_adv[g] * float(size)
|
||||
}
|
||||
return int(w)
|
||||
}
|
||||
# text with its top-left at (x, y); returns the pen x after it
|
||||
function ov_text(x: int, y: int, size: int, s: string, r: float, g: float, b: float, a: float) -> int {
|
||||
if ov_font == 0 { return x }
|
||||
ov_use_tex(ov_font)
|
||||
let k = float(size) / float(ov_font_em) # atlas px -> screen px
|
||||
let cell = float(ov_font_cell) * k
|
||||
function ov_text(render3d_st: mut Render3dState, x: int, y: int, size: int, s: string, r: float, g: float, b: float, a: float) -> int {
|
||||
if render3d_st.ov_font == 0 { return x }
|
||||
ov_use_tex(render3d_st, render3d_st.ov_font)
|
||||
let k = float(size) / float(render3d_st.ov_font_em) # atlas px -> screen px
|
||||
let cell = float(render3d_st.ov_font_cell) * k
|
||||
var pen = float(x)
|
||||
let base = float(y) + float(size) * 0.80
|
||||
let px = ov_pad_x * float(ov_font_em) * k
|
||||
let py = ov_base_y * float(ov_font_em) * k
|
||||
let px = render3d_st.ov_pad_x * float(render3d_st.ov_font_em) * k
|
||||
let py = render3d_st.ov_base_y * float(render3d_st.ov_font_em) * k
|
||||
let sp: pointer = s
|
||||
let n = len(sp)
|
||||
var i = 0
|
||||
while i < n {
|
||||
let c = ov_glyph(ov_u8(sp, i, n))
|
||||
i += ov_u8_len
|
||||
if c != ov_font_space {
|
||||
let cx = c - (c / ov_font_cols) * ov_font_cols
|
||||
let cy = c / ov_font_cols
|
||||
let u0 = float(cx) / float(ov_font_cols); let u1 = float(cx + 1) / float(ov_font_cols)
|
||||
let v0 = float(cy) / float(ov_font_rows); let v1 = float(cy + 1) / float(ov_font_rows)
|
||||
let c = ov_glyph(render3d_st, ov_u8(render3d_st, sp, i, n))
|
||||
i += render3d_st.ov_u8_len
|
||||
if c != render3d_st.ov_font_space {
|
||||
let cx = c - (c / render3d_st.ov_font_cols) * render3d_st.ov_font_cols
|
||||
let cy = c / render3d_st.ov_font_cols
|
||||
let u0 = float(cx) / float(render3d_st.ov_font_cols); let u1 = float(cx + 1) / float(render3d_st.ov_font_cols)
|
||||
let v0 = float(cy) / float(render3d_st.ov_font_rows); let v1 = float(cy + 1) / float(render3d_st.ov_font_rows)
|
||||
let x0 = pen - px; let y1 = base + py
|
||||
ov_quad(x0, y1 - cell, x0 + cell, y1, u0, v0, u1, v1, r, g, b, a)
|
||||
ov_quad(render3d_st, x0, y1 - cell, x0 + cell, y1, u0, v0, u1, v1, r, g, b, a)
|
||||
}
|
||||
pen = pen + ov_font_adv[c] * float(size)
|
||||
pen = pen + render3d_st.ov_font_adv[c] * float(size)
|
||||
}
|
||||
return int(pen)
|
||||
}
|
||||
# text with a soft dark shadow under it (HUD over a bright meadow)
|
||||
function ov_text_sh(x: int, y: int, size: int, s: string, r: float, g: float, b: float, a: float) -> int {
|
||||
function ov_text_sh(render3d_st: mut Render3dState, x: int, y: int, size: int, s: string, r: float, g: float, b: float, a: float) -> int {
|
||||
let d = size / 18 + 1
|
||||
ov_text(x + d, y + d, size, s, 0.0, 0.0, 0.0, a * 0.7)
|
||||
return ov_text(x, y, size, s, r, g, b, a)
|
||||
ov_text(render3d_st, x + d, y + d, size, s, 0.0, 0.0, 0.0, a * 0.7)
|
||||
return ov_text(render3d_st, x, y, size, s, r, g, b, a)
|
||||
}
|
||||
function ov_text_center(cx: int, y: int, size: int, s: string, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_text(cx - ov_text_w(size, s) / 2, y, size, s, r, g, b, a)
|
||||
function ov_text_center(render3d_st: mut Render3dState, cx: int, y: int, size: int, s: string, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_text(render3d_st, cx - ov_text_w(render3d_st, size, s) / 2, y, size, s, r, g, b, a)
|
||||
}
|
||||
|
||||
# text wrapped at `maxw` pixels on spaces; returns the y after the last line
|
||||
function ov_text_wrap(x: int, y: int, size: int, maxw: int, s: string, r: float, g: float, b: float, a: float) -> int {
|
||||
function ov_text_wrap(render3d_st: mut Render3dState, x: int, y: int, size: int, maxw: int, s: string, r: float, g: float, b: float, a: float) -> int {
|
||||
let sp: pointer = s
|
||||
let n = len(sp)
|
||||
var first = 0
|
||||
|
|
@ -377,11 +339,11 @@ function ov_text_wrap(x: int, y: int, size: int, maxw: int, s: string, r: float,
|
|||
if sp[i] == 10 { stop = i; break }
|
||||
if sp[i] == 32 { last_space = i }
|
||||
let piece: string = s[first .. i + 1]
|
||||
if ov_text_w(size, piece) > maxw and last_space > first { stop = last_space; break }
|
||||
if ov_text_w(render3d_st, size, piece) > maxw and last_space > first { stop = last_space; break }
|
||||
i += 1
|
||||
}
|
||||
let line: string = s[first .. stop]
|
||||
ov_text(x, ly, size, line, r, g, b, a)
|
||||
ov_text(render3d_st, x, ly, size, line, r, g, b, a)
|
||||
ly += size * 13 / 10
|
||||
first = stop
|
||||
while first < n and (sp[first] == 32 or sp[first] == 10) { first += 1 }
|
||||
|
|
@ -391,14 +353,14 @@ function ov_text_wrap(x: int, y: int, size: int, maxw: int, s: string, r: float,
|
|||
|
||||
# ---- more shapes for a game's interface -------------------------------------------------
|
||||
# a sub-rectangle of a texture (uv corners as float bits) tinted, at integer pixels
|
||||
function ov_sub(tex: int, x: int, y: int, w: int, h: int, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(tex)
|
||||
ov_quad(float(x), float(y), float(x + w), float(y + h), u0, v0, u1, v1, r, g, b, a)
|
||||
function ov_sub(render3d_st: mut Render3dState, tex: int, x: int, y: int, w: int, h: int, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(render3d_st, tex)
|
||||
ov_quad(render3d_st, float(x), float(y), float(x + w), float(y + h), u0, v0, u1, v1, r, g, b, a)
|
||||
}
|
||||
# a whole texture stretched by nine slices: corners `src` texels wide in a `tw` px square
|
||||
# texture, drawn `dst` pixels wide, so rounded corners keep their shape at any size
|
||||
function ov_nine(tex: int, tw: int, src: int, x: int, y: int, w: int, h: int, dst: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(tex)
|
||||
function ov_nine(render3d_st: mut Render3dState, tex: int, tw: int, src: int, x: int, y: int, w: int, h: int, dst: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(render3d_st, tex)
|
||||
let s = float(src) / float(tw)
|
||||
let xs = floats(4); let ys = floats(4); let us = floats(4); let vs = floats(4)
|
||||
xs[0] = float(x); xs[1] = float(x + dst); xs[2] = float(x + w - dst); xs[3] = float(x + w)
|
||||
|
|
@ -406,33 +368,33 @@ function ov_nine(tex: int, tw: int, src: int, x: int, y: int, w: int, h: int, ds
|
|||
us[0] = 0.0; us[1] = s; us[2] = 1.0 - s; us[3] = 1.0
|
||||
vs[0] = 0.0; vs[1] = s; vs[2] = 1.0 - s; vs[3] = 1.0
|
||||
for j in 0 .. 3 {
|
||||
for i in 0 .. 3 { ov_quad(xs[i], ys[j], xs[i + 1], ys[j + 1], us[i], vs[j], us[i + 1], vs[j + 1], r, g, b, a) }
|
||||
for i in 0 .. 3 { ov_quad(render3d_st, xs[i], ys[j], xs[i + 1], ys[j + 1], us[i], vs[j], us[i + 1], vs[j + 1], r, g, b, a) }
|
||||
}
|
||||
free(xs); free(ys); free(us); free(vs)
|
||||
}
|
||||
# an arbitrary quad (float-bit pixel corners, clockwise from top-left) of a texture
|
||||
function ov_quad4(x0: float, y0: float, x1: float, y1: float, x2: float, y2: float, x3: float, y3: float, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
if ov_n >= OV_MAX_QUADS { ov_flush() }
|
||||
let k = ov_n * 6
|
||||
ov_vert(k, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(k + 1, x1, y1, u1, v0, r, g, b, a)
|
||||
ov_vert(k + 2, x2, y2, u1, v1, r, g, b, a)
|
||||
ov_vert(k + 3, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(k + 4, x2, y2, u1, v1, r, g, b, a)
|
||||
ov_vert(k + 5, x3, y3, u0, v1, r, g, b, a)
|
||||
ov_n += 1
|
||||
function ov_quad4(render3d_st: mut Render3dState, x0: float, y0: float, x1: float, y1: float, x2: float, y2: float, x3: float, y3: float, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
if render3d_st.ov_n >= OV_MAX_QUADS { ov_flush(render3d_st) }
|
||||
let k = render3d_st.ov_n * 6
|
||||
ov_vert(render3d_st, k, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 1, x1, y1, u1, v0, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 2, x2, y2, u1, v1, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 3, x0, y0, u0, v0, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 4, x2, y2, u1, v1, r, g, b, a)
|
||||
ov_vert(render3d_st, k + 5, x3, y3, u0, v1, r, g, b, a)
|
||||
render3d_st.ov_n += 1
|
||||
}
|
||||
# a line of thickness `t` pixels between two points (float-bit pixels)
|
||||
function ov_line(x0: float, y0: float, x1: float, y1: float, t: float, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(ov_white)
|
||||
function ov_line(render3d_st: mut Render3dState, x0: float, y0: float, x1: float, y1: float, t: float, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(render3d_st, render3d_st.ov_white)
|
||||
let dx = x1 - x0; let dy = y1 - y0
|
||||
let l = Math.max(Math.sqrt(dx * dx + dy * dy), 0.001)
|
||||
let nx = -dy / l * (t * 0.5); let ny = dx / l * (t * 0.5)
|
||||
ov_quad4(x0 + nx, y0 + ny, x1 + nx, y1 + ny, x1 - nx, y1 - ny, x0 - nx, y0 - ny, 0.0, 0.0, 1.0, 1.0, r, g, b, a)
|
||||
ov_quad4(render3d_st, x0 + nx, y0 + ny, x1 + nx, y1 + ny, x1 - nx, y1 - ny, x0 - nx, y0 - ny, 0.0, 0.0, 1.0, 1.0, r, g, b, a)
|
||||
}
|
||||
# a sub-rectangle of a texture rotated by `ang` radians about its centre (cx, cy), `w` x `h` pixels
|
||||
function ov_sub_rot(tex: int, cx: int, cy: int, w: int, h: int, ang: float, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(tex)
|
||||
function ov_sub_rot(render3d_st: mut Render3dState, tex: int, cx: int, cy: int, w: int, h: int, ang: float, u0: float, v0: float, u1: float, v1: float, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(render3d_st, tex)
|
||||
let c = Math.cos(ang); let s = Math.sin(ang)
|
||||
let hw = float(w) * 0.5; let hh = float(h) * 0.5
|
||||
let fx = float(cx); let fy = float(cy)
|
||||
|
|
@ -441,24 +403,24 @@ function ov_sub_rot(tex: int, cx: int, cy: int, w: int, h: int, ang: float, u0:
|
|||
let x1 = fx + (hw * c - -hh * s); let y1 = fy + (hw * s + -hh * c)
|
||||
let x2 = fx + (hw * c - hh * s); let y2 = fy + (hw * s + hh * c)
|
||||
let x3 = fx + (-hw * c - hh * s); let y3 = fy + (-hw * s + hh * c)
|
||||
ov_quad4(x0, y0, x1, y1, x2, y2, x3, y3, u0, v0, u1, v1, r, g, b, a)
|
||||
ov_quad4(render3d_st, x0, y0, x1, y1, x2, y2, x3, y3, u0, v0, u1, v1, r, g, b, a)
|
||||
}
|
||||
# a filled circle approximated by `n` wedges (float-bit centre and radius)
|
||||
function ov_disc(cx: float, cy: float, rad: float, n: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(ov_white)
|
||||
function ov_disc(render3d_st: mut Render3dState, cx: float, cy: float, rad: float, n: int, r: float, g: float, b: float, a: float) -> void {
|
||||
ov_use_tex(render3d_st, render3d_st.ov_white)
|
||||
let step = 2.0 * PI / float(n)
|
||||
for i in 0 .. n {
|
||||
let a0 = float(i) * step; let a1 = a0 + step
|
||||
let ax = cx + Math.cos(a0) * rad; let ay = cy + Math.sin(a0) * rad
|
||||
let bx = cx + Math.cos(a1) * rad; let by = cy + Math.sin(a1) * rad
|
||||
ov_quad4(cx, cy, ax, ay, bx, by, cx, cy, 0.0, 0.0, 1.0, 1.0, r, g, b, a)
|
||||
ov_quad4(render3d_st, cx, cy, ax, ay, bx, by, cx, cy, 0.0, 0.0, 1.0, 1.0, r, g, b, a)
|
||||
}
|
||||
}
|
||||
# a ring: `n` segments of thickness `t`, from angle a0 for `span` radians (float bits)
|
||||
function ov_arc(cx: float, cy: float, rad: float, t: float, a0: float, span: float, n: int, r: float, g: float, b: float, a: float) -> void {
|
||||
function ov_arc(render3d_st: mut Render3dState, cx: float, cy: float, rad: float, t: float, a0: float, span: float, n: int, r: float, g: float, b: float, a: float) -> void {
|
||||
let step = span / float(n)
|
||||
for i in 0 .. n {
|
||||
let b0 = a0 + float(i) * step; let b1 = b0 + step
|
||||
ov_line(cx + Math.cos(b0) * rad, cy + Math.sin(b0) * rad, cx + Math.cos(b1) * rad, cy + Math.sin(b1) * rad, t, r, g, b, a)
|
||||
ov_line(render3d_st, cx + Math.cos(b0) * rad, cy + Math.sin(b0) * rad, cx + Math.cos(b1) * rad, cy + Math.sin(b1) * rad, t, r, g, b, a)
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -6,27 +6,6 @@
|
|||
|
||||
const BLOOM_LEVELS: int = 6
|
||||
|
||||
var post_hdr: Target = null
|
||||
var post_ms_fbo: int = 0 # 4x multisampled scene target, resolved into post_hdr
|
||||
var post_ms_samples: int = 1 # temporal AA carries the edges; R3D_MSAA=n to compare
|
||||
var post_bloom: []Target = null
|
||||
var post_p_down: int = 0
|
||||
var post_p_up: int = 0
|
||||
var post_p_tone: int = 0
|
||||
var post_fs: Mesh = null
|
||||
var post_exposure: float = 0.0
|
||||
var post_bloom_strength: int = 0
|
||||
var post_vignette: float = 0.0
|
||||
var post_saturation: float = 0.0
|
||||
var post_contrast: float = 0.0
|
||||
var post_w: int = 0
|
||||
var post_h: int = 0
|
||||
var post_auto: bool = true
|
||||
var post_key: float = 0.0 # target mean luminance after exposure (float bits)
|
||||
var post_lum: words = null
|
||||
var post_mips: int = 0
|
||||
var post_adapt: float = 0.0 # smoothed exposure (float bits)
|
||||
var post_exposure_max: float = 20.0 # 20: the ceiling auto-exposure may reach (night lowers it)
|
||||
# ---- the grade ------------------------------------------------------------------------------
|
||||
# White balance, the shadows' floor and the highlights' gain. These were nine literals bound at
|
||||
# the draw, so the game had the same colour at seven in the morning as at one in the afternoon -
|
||||
|
|
@ -35,119 +14,72 @@ var post_exposure_max: float = 20.0 # 20: the ceiling auto-exposure may reach
|
|||
# program that never starts a day looks exactly as it did.
|
||||
# They are plain ints rather than a v3 on purpose: a grade is written by daylight_set, which can
|
||||
# run before post_init has allocated anything, and nine ints cannot be null.
|
||||
var post_wb_r: float = 1.02 # 1.02
|
||||
var post_wb_g: float = 1.0 # 1.0
|
||||
var post_wb_b: float = 0.97 # 0.97
|
||||
var post_lift_r: float = 0.004 # 0.004
|
||||
var post_lift_g: float = 0.004 # 0.004
|
||||
var post_lift_b: float = 0.012 # 0.012
|
||||
var post_gain_r: float = 0.99 # 0.99
|
||||
var post_gain_g: float = 0.995 # 0.995
|
||||
var post_gain_b: float = 1.0 # 1.0
|
||||
|
||||
var post_ao: Target = null
|
||||
var post_ao_blur: Target = null
|
||||
var post_p_ao: int = 0
|
||||
var post_p_ao_blur: int = 0
|
||||
var post_ao_radius: float = 0.0
|
||||
var post_contact: float = 1.25 # 1.25: how hard a thing is darkened where it meets the ground
|
||||
var post_ao_intensity: float = 0.0
|
||||
var post_ao_strength: float = 0.0
|
||||
var post_gi_strength: float = 0.4 # 0.4
|
||||
var post_no_gi: bool = false
|
||||
var post_ldr: Target = null
|
||||
var post_depth_copy: Target = null
|
||||
# ---- volumetric light ----------------------------------------------------------------------
|
||||
# Half resolution on purpose: in-scattered light is smooth, a shaft has no sharp edge, and the
|
||||
# march is the whole cost of the pass. post_vol_steps is the quality dial; 0 switches it off
|
||||
# and the pass is skipped entirely rather than run at one step.
|
||||
var post_vol: Target = null
|
||||
var post_p_vol: int = 0
|
||||
var post_vol_steps: float = 24.0 # 24
|
||||
var post_vol_density: float = 0.002 # 0.002
|
||||
var post_vol_falloff: float = 0.005 # 0.005
|
||||
var post_vol_far: float = 800.0 # 800 m
|
||||
var post_vol_g: float = 0.6 # 0.6: air throws light forward
|
||||
var post_vol_mist: float = 0.0 # the day sets these two
|
||||
var post_vol_mist_h: float = 40.0 # 40 m
|
||||
var post_prev: Target = null # last frame's scene colour, for the SSGI bounce only
|
||||
var post_scene: Target = null # this frame's scene colour before the water, for refraction
|
||||
var post_frame: int = 0
|
||||
var post_color_w: int = 0 # its size: the display's when DLSS upscaled it
|
||||
var post_color_h: int = 0
|
||||
var post_color: int = 0 # the HDR colour the rest of post reads # the resolved depth, copied so passes can read it while drawing into the frame
|
||||
var post_p_sharp: int = 0
|
||||
# Spatial anti-aliasing, in the sharpen pass because that pass already reads this pixel's
|
||||
# neighbourhood and runs last on the LDR image. 1 on, 0 off; the game's setting drives it.
|
||||
var post_fxaa: float = 1.0
|
||||
# ---- depth of field ------------------------------------------------------------------------
|
||||
# Off in ordinary play - the pass is skipped whole, not run at zero radius. The game turns it on
|
||||
# behind the viewfinder and says what to focus on.
|
||||
var post_p_dof: int = 0
|
||||
var post_dof: Target = null
|
||||
var post_dof_focus: float = 10.0 # 10 m
|
||||
var post_dof_aperture: float = 0.0 # 0 = no lens, and no pass
|
||||
var post_dof_max: float = 12.0 # 12 px
|
||||
var post_p_tone_hdr: int = 0 # the tonemap's HDR10 variant, made the first time HDR is on
|
||||
var post_ldr_hdr: bool = false # post_ldr was made for HDR10 output (10-bit)
|
||||
# the LDR image is 10-bit while the output is HDR10: PQ in 8 bits bands
|
||||
function post_ldr_fmt() -> int { if gpu_hdr_active() { return GL_RGB10_A2 }; return GL_RGBA8 }
|
||||
var post_sharpen: float = 0.0
|
||||
var post_grain: int = 0
|
||||
function post_ldr_fmt(render3d_st: Render3dState) -> int { if gpu_hdr_active(render3d_st) { return GL_RGB10_A2 }; return GL_RGBA8 }
|
||||
|
||||
# the screen-sized targets go away before post_init makes them at a new size
|
||||
function post_free() -> void {
|
||||
if post_hdr == null { return }
|
||||
if post_ms_fbo != 0 { gpu_fb_free(post_ms_fbo); post_ms_fbo = 0 }
|
||||
target_free(post_hdr); target_free(post_ao); target_free(post_ao_blur); target_free(post_ldr)
|
||||
target_free(post_depth_copy); target_free(post_prev); target_free(post_scene); target_free(post_vol); target_free(post_dof)
|
||||
for i in 0 .. len(post_bloom) { target_free(post_bloom[i]) }
|
||||
post_hdr = null
|
||||
function post_free(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.post_hdr == null { return }
|
||||
if render3d_st.post_ms_fbo != 0 { gpu_fb_free(render3d_st, render3d_st.post_ms_fbo); render3d_st.post_ms_fbo = 0 }
|
||||
target_free(render3d_st, render3d_st.post_hdr); target_free(render3d_st, render3d_st.post_ao); target_free(render3d_st, render3d_st.post_ao_blur); target_free(render3d_st, render3d_st.post_ldr)
|
||||
target_free(render3d_st, render3d_st.post_depth_copy); target_free(render3d_st, render3d_st.post_prev); target_free(render3d_st, render3d_st.post_scene); target_free(render3d_st, render3d_st.post_vol); target_free(render3d_st, render3d_st.post_dof)
|
||||
for i in 0 .. len(render3d_st.post_bloom) { target_free(render3d_st, render3d_st.post_bloom[i]) }
|
||||
render3d_st.post_hdr = null
|
||||
}
|
||||
# Multisampled scene: 1 (temporal AA alone), 2 or 4, remade at once. Vulkan draws it into
|
||||
# multisampled renderbuffers and resolves them in a pass; a device that cannot take the count asked
|
||||
# for gets the most it can (gpu_msaa_max).
|
||||
function post_msaa_live() -> bool { return gpu_is_gl() or gpu_msaa_max() > 1 }
|
||||
function post_set_msaa(n: int) -> void {
|
||||
function post_msaa_live(render3d_st: Render3dState) -> bool { return gpu_is_gl(render3d_st) or gpu_msaa_max(render3d_st) > 1 }
|
||||
function post_set_msaa(render3d_st: mut Render3dState, n: int) -> void {
|
||||
var want = n
|
||||
if want < 1 { want = 1 }
|
||||
if not post_msaa_live() { want = 1 }
|
||||
if want > gpu_msaa_max() and gpu_msaa_max() >= 1 { want = gpu_msaa_max() }
|
||||
if want == post_ms_samples { return }
|
||||
post_ms_samples = want
|
||||
if post_hdr != null {
|
||||
let w = post_w; let h = post_h
|
||||
post_free()
|
||||
post_init(w, h)
|
||||
if not post_msaa_live(render3d_st) { want = 1 }
|
||||
if want > gpu_msaa_max(render3d_st) and gpu_msaa_max(render3d_st) >= 1 { want = gpu_msaa_max(render3d_st) }
|
||||
if want == render3d_st.post_ms_samples { return }
|
||||
render3d_st.post_ms_samples = want
|
||||
if render3d_st.post_hdr != null {
|
||||
let w = render3d_st.post_w; let h = render3d_st.post_h
|
||||
post_free(render3d_st)
|
||||
post_init(render3d_st, w, h)
|
||||
}
|
||||
}
|
||||
|
||||
function post_init(w: int, h: int) -> void {
|
||||
post_w = w; post_h = h
|
||||
post_hdr = target_new(w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, true, GL_LINEAR)
|
||||
if post_ms_samples > 1 {
|
||||
post_ms_fbo = gpu_fb_new()
|
||||
gpu_fb_bind(post_ms_fbo)
|
||||
let rbc = gpu_rb_new()
|
||||
gpu_rb_storage(rbc, GL_RGBA16F, w, h, post_ms_samples)
|
||||
gpu_fb_color_rb(0, rbc)
|
||||
let rbd = gpu_rb_new()
|
||||
gpu_rb_storage(rbd, GL_DEPTH_COMPONENT32F, w, h, post_ms_samples)
|
||||
gpu_fb_depth_rb(rbd)
|
||||
let st = gpu_fb_status()
|
||||
if st != GL_FRAMEBUFFER_COMPLETE { print(`r3d: msaa framebuffer incomplete {st}`); post_ms_fbo = 0 }
|
||||
gpu_fb_bind(0)
|
||||
function post_init(render3d_st: mut Render3dState, w: int, h: int) -> void {
|
||||
render3d_st.post_w = w; render3d_st.post_h = h
|
||||
render3d_st.post_hdr = target_new(render3d_st, w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, true, GL_LINEAR)
|
||||
if render3d_st.post_ms_samples > 1 {
|
||||
render3d_st.post_ms_fbo = gpu_fb_new(render3d_st)
|
||||
gpu_fb_bind(render3d_st, render3d_st.post_ms_fbo)
|
||||
let rbc = gpu_rb_new(render3d_st)
|
||||
gpu_rb_storage(render3d_st, rbc, GL_RGBA16F, w, h, render3d_st.post_ms_samples)
|
||||
gpu_fb_color_rb(render3d_st, 0, rbc)
|
||||
let rbd = gpu_rb_new(render3d_st)
|
||||
gpu_rb_storage(render3d_st, rbd, GL_DEPTH_COMPONENT32F, w, h, render3d_st.post_ms_samples)
|
||||
gpu_fb_depth_rb(render3d_st, rbd)
|
||||
let st = gpu_fb_status(render3d_st)
|
||||
if st != GL_FRAMEBUFFER_COMPLETE { print(`r3d: msaa framebuffer incomplete {st}`); render3d_st.post_ms_fbo = 0 }
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
}
|
||||
post_bloom = new []Target
|
||||
render3d_st.post_bloom = new []Target
|
||||
var bw = w / 2; var bh = h / 2
|
||||
for i in 0 .. BLOOM_LEVELS {
|
||||
push(post_bloom, target_new(max(bw, 1), max(bh, 1), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR))
|
||||
push(render3d_st.post_bloom, target_new(render3d_st, max(bw, 1), max(bh, 1), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR))
|
||||
bw = bw / 2; bh = bh / 2
|
||||
}
|
||||
if post_p_down == 0 {
|
||||
post_p_down = r3d_program("fullscreen.vert", "bloom_down.frag", "")
|
||||
post_p_up = r3d_program("fullscreen.vert", "bloom_up.frag", "")
|
||||
post_p_tone = r3d_program("fullscreen.vert", "tonemap.frag", "")
|
||||
if render3d_st.post_p_down == 0 {
|
||||
render3d_st.post_p_down = r3d_program(render3d_st, "fullscreen.vert", "bloom_down.frag", "")
|
||||
render3d_st.post_p_up = r3d_program(render3d_st, "fullscreen.vert", "bloom_up.frag", "")
|
||||
render3d_st.post_p_tone = r3d_program(render3d_st, "fullscreen.vert", "tonemap.frag", "")
|
||||
}
|
||||
# Full resolution, not half. The occlusion is reconstructed from depth differences,
|
||||
# so on a surface seen at a grazing angle its gradient is steep in screen space; at
|
||||
|
|
@ -155,36 +87,36 @@ function post_init(w: int, h: int) -> void {
|
|||
# upsample in the tonemapper then stretched over the whole ground. They read as thin
|
||||
# transparent black bars, appear only where there is depth (never on the sky), and
|
||||
# are nothing to do with the shadow map or the reflection.
|
||||
post_ao = target_new(w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
post_ao_blur = target_new(w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if post_p_ao == 0 { post_p_ao = r3d_program("fullscreen.vert", "ssgi.frag", ""); post_p_ao_blur = r3d_program("fullscreen.vert", "ssao_blur.frag", "") }
|
||||
post_ldr = target_new(w, h, post_ldr_fmt(), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
||||
post_ldr_hdr = gpu_hdr_active()
|
||||
post_depth_copy = target_new(w, h, GL_R8, GL_RED, GL_UNSIGNED_BYTE, true, GL_NEAREST)
|
||||
post_dof = target_new(w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if post_p_dof == 0 { post_p_dof = r3d_program("fullscreen.vert", "dof.frag", "") }
|
||||
post_vol = target_new(max(w / 2, 1), max(h / 2, 1), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if post_p_vol == 0 { post_p_vol = r3d_program("fullscreen.vert", "volumetric.frag", "") }
|
||||
post_prev = target_new(w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
post_scene = target_new(w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if post_p_sharp == 0 { post_p_sharp = r3d_program("fullscreen.vert", "sharpen.frag", "") }
|
||||
post_sharpen = 1.2
|
||||
post_grain = float_bits(0.025)
|
||||
post_ao_radius = 0.7
|
||||
post_ao_intensity = 1.4
|
||||
post_ao_strength = 0.8
|
||||
post_fs = mesh_fullscreen()
|
||||
post_exposure = 0.36
|
||||
post_bloom_strength = float_bits(0.06)
|
||||
post_vignette = 0.35
|
||||
post_saturation = 1.04
|
||||
post_contrast = 1.12
|
||||
post_key = 0.19
|
||||
post_lum = words(4)
|
||||
render3d_st.post_ao = target_new(render3d_st, w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
render3d_st.post_ao_blur = target_new(render3d_st, w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if render3d_st.post_p_ao == 0 { render3d_st.post_p_ao = r3d_program(render3d_st, "fullscreen.vert", "ssgi.frag", ""); render3d_st.post_p_ao_blur = r3d_program(render3d_st, "fullscreen.vert", "ssao_blur.frag", "") }
|
||||
render3d_st.post_ldr = target_new(render3d_st, w, h, post_ldr_fmt(render3d_st), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
||||
render3d_st.post_ldr_hdr = gpu_hdr_active(render3d_st)
|
||||
render3d_st.post_depth_copy = target_new(render3d_st, w, h, GL_R8, GL_RED, GL_UNSIGNED_BYTE, true, GL_NEAREST)
|
||||
render3d_st.post_dof = target_new(render3d_st, w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if render3d_st.post_p_dof == 0 { render3d_st.post_p_dof = r3d_program(render3d_st, "fullscreen.vert", "dof.frag", "") }
|
||||
render3d_st.post_vol = target_new(render3d_st, max(w / 2, 1), max(h / 2, 1), GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if render3d_st.post_p_vol == 0 { render3d_st.post_p_vol = r3d_program(render3d_st, "fullscreen.vert", "volumetric.frag", "") }
|
||||
render3d_st.post_prev = target_new(render3d_st, w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
render3d_st.post_scene = target_new(render3d_st, w, h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
if render3d_st.post_p_sharp == 0 { render3d_st.post_p_sharp = r3d_program(render3d_st, "fullscreen.vert", "sharpen.frag", "") }
|
||||
render3d_st.post_sharpen = 1.2
|
||||
render3d_st.post_grain = float_bits(0.025)
|
||||
render3d_st.post_ao_radius = 0.7
|
||||
render3d_st.post_ao_intensity = 1.4
|
||||
render3d_st.post_ao_strength = 0.8
|
||||
render3d_st.post_fs = mesh_fullscreen(render3d_st)
|
||||
render3d_st.post_exposure = 0.36
|
||||
render3d_st.post_bloom_strength = float_bits(0.06)
|
||||
render3d_st.post_vignette = 0.35
|
||||
render3d_st.post_saturation = 1.04
|
||||
render3d_st.post_contrast = 1.12
|
||||
render3d_st.post_key = 0.19
|
||||
render3d_st.post_lum = words(4)
|
||||
var m = 1; var sz = max(w, h)
|
||||
while sz > 1 { sz = sz / 2; m += 1 }
|
||||
post_mips = m
|
||||
post_adapt = 0.0
|
||||
render3d_st.post_mips = m
|
||||
render3d_st.post_adapt = 0.0
|
||||
}
|
||||
|
||||
# Mean scene luminance from the HDR mip chain -> exposure = key / mean, eased over
|
||||
|
|
@ -197,63 +129,59 @@ function post_init(w: int, h: int) -> void {
|
|||
# buffer still synchronised the texture: 50% of the CPU's frame waiting, sampled), so
|
||||
# the adaptation now stays on the GPU: a 1x1 pass (adapt.frag) eases last frame's value
|
||||
# toward key / mean and the tonemapper samples it. The CPU never waits for the picture.
|
||||
var post_adapt_t: []Target = null
|
||||
var post_adapt_i: int = 0
|
||||
var post_p_adapt: int = 0
|
||||
var post_adapt_reset: bool = true
|
||||
function post_measure() -> void {
|
||||
gpu_tex_bind(GPU_TEX2D, post_hdr.color)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(GPU_TEX2D)
|
||||
if post_adapt_t == null {
|
||||
post_adapt_t = new []Target
|
||||
for k in 0 .. 2 { push(post_adapt_t, target_new(1, 1, GL_R32F, GL_RED, GL_FLOAT, false, GL_NEAREST)) }
|
||||
post_adapt_reset = true
|
||||
function post_measure(render3d_st: mut Render3dState) -> void {
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, render3d_st.post_hdr.color)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
||||
if render3d_st.post_adapt_t == null {
|
||||
render3d_st.post_adapt_t = new []Target
|
||||
for k in 0 .. 2 { push(render3d_st.post_adapt_t, target_new(render3d_st, 1, 1, GL_R32F, GL_RED, GL_FLOAT, false, GL_NEAREST)) }
|
||||
render3d_st.post_adapt_reset = true
|
||||
}
|
||||
if post_p_adapt == 0 { post_p_adapt = r3d_program("fullscreen.vert", "adapt.frag", "") }
|
||||
let next = 1 - post_adapt_i
|
||||
target_bind(post_adapt_t[next])
|
||||
gpu_depth_test(false)
|
||||
gpu_use_program(post_p_adapt)
|
||||
r3d_bind_2d(post_p_adapt, "u_scene", 0, post_hdr.color)
|
||||
r3d_bind_2d(post_p_adapt, "u_prev", 1, post_adapt_t[post_adapt_i].color)
|
||||
u_f(gpu_uniform(post_p_adapt, "u_lod"), float(post_mips - 1))
|
||||
u_f(gpu_uniform(post_p_adapt, "u_key"), post_key)
|
||||
u_f(gpu_uniform(post_p_adapt, "u_max"), post_exposure_max)
|
||||
u_f(gpu_uniform(post_p_adapt, "u_rate"), 0.08)
|
||||
if render3d_st.post_p_adapt == 0 { render3d_st.post_p_adapt = r3d_program(render3d_st, "fullscreen.vert", "adapt.frag", "") }
|
||||
let next = 1 - render3d_st.post_adapt_i
|
||||
target_bind(render3d_st, render3d_st.post_adapt_t[next])
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_adapt)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_adapt, "u_scene", 0, render3d_st.post_hdr.color)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_adapt, "u_prev", 1, render3d_st.post_adapt_t[render3d_st.post_adapt_i].color)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_adapt, "u_lod"), float(render3d_st.post_mips - 1))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_adapt, "u_key"), render3d_st.post_key)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_adapt, "u_max"), render3d_st.post_exposure_max)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_adapt, "u_rate"), 0.08)
|
||||
var reset = 0.0
|
||||
if post_adapt_reset { reset = 1.0; post_adapt_reset = false }
|
||||
u_f(gpu_uniform(post_p_adapt, "u_reset"), reset)
|
||||
mesh_draw(post_fs)
|
||||
post_adapt_i = next
|
||||
gpu_tex_bind(GPU_TEX2D, post_hdr.color)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
if render3d_st.post_adapt_reset { reset = 1.0; render3d_st.post_adapt_reset = false }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_adapt, "u_reset"), reset)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
render3d_st.post_adapt_i = next
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, render3d_st.post_hdr.color)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
}
|
||||
|
||||
function post_begin_scene() -> void {
|
||||
target_bind(post_hdr)
|
||||
if post_ms_fbo != 0 { gpu_fb_bind(post_ms_fbo); gpu_multisample(true) }
|
||||
gpu_depth_test(true)
|
||||
gpu_depth_func(GL_LESS)
|
||||
gpu_depth_write(true)
|
||||
gpu_cull(true)
|
||||
gpu_cull_face(GL_BACK)
|
||||
gpu_clear_color(0.0, 0.0, 0.0, 1.0)
|
||||
gpu_clear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
||||
function post_begin_scene(render3d_st: mut Render3dState) -> void {
|
||||
target_bind(render3d_st, render3d_st.post_hdr)
|
||||
if render3d_st.post_ms_fbo != 0 { gpu_fb_bind(render3d_st, render3d_st.post_ms_fbo); gpu_multisample(render3d_st, true) }
|
||||
gpu_depth_test(render3d_st, true)
|
||||
gpu_depth_func(render3d_st, GL_LESS)
|
||||
gpu_depth_write(render3d_st, true)
|
||||
gpu_cull(render3d_st, true)
|
||||
gpu_cull_face(render3d_st, GL_BACK)
|
||||
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 1.0)
|
||||
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
||||
}
|
||||
|
||||
# resolve the multisampled scene into the plain HDR target (colour + depth)
|
||||
function post_resolve() -> void {
|
||||
if post_ms_fbo != 0 {
|
||||
gpu_fb_bind_read(post_ms_fbo)
|
||||
gpu_fb_bind_draw(post_hdr.fbo)
|
||||
gpu_blit(post_w, post_h, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
||||
function post_resolve(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.post_ms_fbo != 0 {
|
||||
gpu_fb_bind_read(render3d_st, render3d_st.post_ms_fbo)
|
||||
gpu_fb_bind_draw(render3d_st, render3d_st.post_hdr.fbo)
|
||||
gpu_blit(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
||||
}
|
||||
# the depth copy every pass after this may read while the frame is still being drawn into
|
||||
gpu_fb_bind_read(post_hdr.fbo)
|
||||
gpu_fb_bind_draw(post_depth_copy.fbo)
|
||||
gpu_blit(post_w, post_h, GL_DEPTH_BUFFER_BIT)
|
||||
gpu_fb_bind(0)
|
||||
gpu_fb_bind_read(render3d_st, render3d_st.post_hdr.fbo)
|
||||
gpu_fb_bind_draw(render3d_st, render3d_st.post_depth_copy.fbo)
|
||||
gpu_blit(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_DEPTH_BUFFER_BIT)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
}
|
||||
|
||||
# There is no temporal anti-aliasing. It was reprojecting every pixel through the
|
||||
|
|
@ -268,194 +196,194 @@ function post_resolve() -> void {
|
|||
# then absorb it, which is what makes the surface read as a body of water rather than a
|
||||
# sheet laid over the ground: the bottom is seen THROUGH the water, tinted and dimmed by
|
||||
# how far the light travelled, instead of being the dry terrain showing through an alpha.
|
||||
function post_capture_scene() -> void {
|
||||
gpu_fb_bind_read(post_hdr.fbo)
|
||||
gpu_fb_bind_draw(post_scene.fbo)
|
||||
gpu_blit(post_w, post_h, GL_COLOR_BUFFER_BIT)
|
||||
gpu_fb_bind(post_hdr.fbo)
|
||||
gpu_viewport(0, 0, post_w, post_h)
|
||||
function post_capture_scene(render3d_st: mut Render3dState) -> void {
|
||||
gpu_fb_bind_read(render3d_st, render3d_st.post_hdr.fbo)
|
||||
gpu_fb_bind_draw(render3d_st, render3d_st.post_scene.fbo)
|
||||
gpu_blit(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_COLOR_BUFFER_BIT)
|
||||
gpu_fb_bind(render3d_st, render3d_st.post_hdr.fbo)
|
||||
gpu_viewport(render3d_st, 0, 0, render3d_st.post_w, render3d_st.post_h)
|
||||
}
|
||||
|
||||
# Keep a copy of the finished scene colour: the SSGI bounce reads last frame's colour.
|
||||
function post_capture_prev() -> void {
|
||||
gpu_fb_bind_read(post_hdr.fbo)
|
||||
gpu_fb_bind_draw(post_prev.fbo)
|
||||
gpu_blit(post_w, post_h, GL_COLOR_BUFFER_BIT)
|
||||
gpu_fb_bind(0)
|
||||
post_frame += 1
|
||||
function post_capture_prev(render3d_st: mut Render3dState) -> void {
|
||||
gpu_fb_bind_read(render3d_st, render3d_st.post_hdr.fbo)
|
||||
gpu_fb_bind_draw(render3d_st, render3d_st.post_prev.fbo)
|
||||
gpu_blit(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_COLOR_BUFFER_BIT)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
render3d_st.post_frame += 1
|
||||
}
|
||||
|
||||
function post_ssao_pass() -> void {
|
||||
gpu_depth_test(false)
|
||||
gpu_blend(false)
|
||||
target_bind(post_ao)
|
||||
gpu_use_program(post_p_ao)
|
||||
r3d_bind_2d(post_p_ao, "u_depth", 0, post_hdr.depth)
|
||||
r3d_bind_2d(post_p_ao, "u_prev_color", 1, post_prev.color)
|
||||
u_f(gpu_uniform(post_p_ao, "u_frame"), float(post_frame % 64))
|
||||
u_mat4(gpu_uniform(post_p_ao, "u_inv_proj"), cam_inv_proj)
|
||||
u_mat4(gpu_uniform(post_p_ao, "u_proj"), cam_proj)
|
||||
u_f2(gpu_uniform(post_p_ao, "u_texel"), 1.0 / float(post_w), 1.0 / float(post_h))
|
||||
u_f(gpu_uniform(post_p_ao, "u_radius"), post_ao_radius)
|
||||
u_f(gpu_uniform(post_p_ao, "u_intensity"), post_ao_intensity)
|
||||
var contact = post_contact
|
||||
if r3d_env_has("R3D_NOCONTACT") { contact = 0.0 }
|
||||
u_f(gpu_uniform(post_p_ao, "u_contact"), contact)
|
||||
mesh_draw(post_fs)
|
||||
target_bind(post_ao_blur)
|
||||
gpu_use_program(post_p_ao_blur)
|
||||
r3d_bind_2d(post_p_ao_blur, "u_ao", 0, post_ao.color)
|
||||
r3d_bind_2d(post_p_ao_blur, "u_depth", 1, post_hdr.depth)
|
||||
u_f2(gpu_uniform(post_p_ao_blur, "u_texel"), 1.0 / float(post_ao.w), 1.0 / float(post_ao.h))
|
||||
mesh_draw(post_fs)
|
||||
function post_ssao_pass(render3d_st: mut Render3dState) -> void {
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_blend(render3d_st, false)
|
||||
target_bind(render3d_st, render3d_st.post_ao)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_ao)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_ao, "u_depth", 0, render3d_st.post_hdr.depth)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_ao, "u_prev_color", 1, render3d_st.post_prev.color)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_frame"), float(render3d_st.post_frame % 64))
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_inv_proj"), render3d_st.cam_inv_proj)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_proj"), render3d_st.cam_proj)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_texel"), 1.0 / float(render3d_st.post_w), 1.0 / float(render3d_st.post_h))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_radius"), render3d_st.post_ao_radius)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_intensity"), render3d_st.post_ao_intensity)
|
||||
var contact = render3d_st.post_contact
|
||||
if r3d_env_has(render3d_st, "R3D_NOCONTACT") { contact = 0.0 }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao, "u_contact"), contact)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
target_bind(render3d_st, render3d_st.post_ao_blur)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_ao_blur)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_ao_blur, "u_ao", 0, render3d_st.post_ao.color)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_ao_blur, "u_depth", 1, render3d_st.post_hdr.depth)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_ao_blur, "u_texel"), 1.0 / float(render3d_st.post_ao.w), 1.0 / float(render3d_st.post_ao.h))
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
}
|
||||
|
||||
# The march, then the composite. It is added to the scene BEFORE bloom on purpose: a shaft of
|
||||
# light is a bright thing in the air and should bloom like one, and compositing it after the
|
||||
# bloom pyramid would give hard-edged rays with no glow at all.
|
||||
function post_volumetric_pass() -> void {
|
||||
if post_vol_steps <= 0.0 { return }
|
||||
if r3d_env_has("R3D_NOVOL") { return }
|
||||
gpu_depth_test(false)
|
||||
gpu_blend(false)
|
||||
target_bind(post_vol)
|
||||
gpu_use_program(post_p_vol)
|
||||
r3d_bind_2d(post_p_vol, "u_depth", 0, post_hdr.depth)
|
||||
shadow_bind(post_p_vol)
|
||||
sky_bind_lighting(post_p_vol)
|
||||
sky_bind_rot(post_p_vol)
|
||||
fog_bind(post_p_vol)
|
||||
u_mat4(gpu_uniform(post_p_vol, "u_inv_vp"), cam_inv_vp)
|
||||
u_v3(gpu_uniform(post_p_vol, "u_cam_pos"), cam_pos)
|
||||
u_v3(gpu_uniform(post_p_vol, "u_sun_dir"), sun_dir)
|
||||
u_v3(gpu_uniform(post_p_vol, "u_sun_color"), sun_color)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_steps"), post_vol_steps)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_density"), post_vol_density)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_falloff"), post_vol_falloff)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_far"), post_vol_far)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_g"), post_vol_g)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_mist"), post_vol_mist)
|
||||
u_f(gpu_uniform(post_p_vol, "u_vol_mist_h"), post_vol_mist_h)
|
||||
mesh_draw(post_fs)
|
||||
function post_volumetric_pass(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.post_vol_steps <= 0.0 { return }
|
||||
if r3d_env_has(render3d_st, "R3D_NOVOL") { return }
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_blend(render3d_st, false)
|
||||
target_bind(render3d_st, render3d_st.post_vol)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_vol)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_vol, "u_depth", 0, render3d_st.post_hdr.depth)
|
||||
shadow_bind(render3d_st, render3d_st.post_p_vol)
|
||||
sky_bind_lighting(render3d_st, render3d_st.post_p_vol)
|
||||
sky_bind_rot(render3d_st, render3d_st.post_p_vol)
|
||||
fog_bind(render3d_st, render3d_st.post_p_vol)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_inv_vp"), render3d_st.cam_inv_vp)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_cam_pos"), render3d_st.cam_pos)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_sun_dir"), render3d_st.sun_dir)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_sun_color"), render3d_st.sun_color)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_steps"), render3d_st.post_vol_steps)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_density"), render3d_st.post_vol_density)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_falloff"), render3d_st.post_vol_falloff)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_far"), render3d_st.post_vol_far)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_g"), render3d_st.post_vol_g)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_mist"), render3d_st.post_vol_mist)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_vol, "u_vol_mist_h"), render3d_st.post_vol_mist_h)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
# Composite into the HDR scene with the bloom pyramid's own upsample - a 3x3 tent under
|
||||
# ONE/ONE blending, which is exactly what is wanted here and already exists, rather than a
|
||||
# blit shader written for one caller. Into post_hdr, not post_scene: post_scene is the copy
|
||||
# the water refracts, so adding shafts there would put them UNDER the lake.
|
||||
target_bind(post_hdr)
|
||||
gpu_blend(true)
|
||||
gpu_blend_func(GL_ONE, GL_ONE)
|
||||
gpu_use_program(post_p_up)
|
||||
r3d_bind_2d(post_p_up, "u_src", 0, post_vol.color)
|
||||
u_f2(gpu_uniform(post_p_up, "u_texel"), 1.0 / float(post_vol.w), 1.0 / float(post_vol.h))
|
||||
u_f(gpu_uniform(post_p_up, "u_radius"), 1.0)
|
||||
mesh_draw(post_fs)
|
||||
gpu_blend(false)
|
||||
target_bind(render3d_st, render3d_st.post_hdr)
|
||||
gpu_blend(render3d_st, true)
|
||||
gpu_blend_func(render3d_st, GL_ONE, GL_ONE)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_up)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_up, "u_src", 0, render3d_st.post_vol.color)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_up, "u_texel"), 1.0 / float(render3d_st.post_vol.w), 1.0 / float(render3d_st.post_vol.h))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_up, "u_radius"), 1.0)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
gpu_blend(render3d_st, false)
|
||||
}
|
||||
|
||||
# The lens, between the scene and the bloom: a blurred highlight should still bloom, and a
|
||||
# bloom smeared by the lens afterwards would be a halo round nothing.
|
||||
function post_dof_pass() -> void {
|
||||
if post_dof_aperture == 0.0 { return }
|
||||
gpu_depth_test(false)
|
||||
gpu_blend(false)
|
||||
target_bind(post_dof)
|
||||
gpu_use_program(post_p_dof)
|
||||
r3d_bind_2d(post_p_dof, "u_src", 0, post_color)
|
||||
r3d_bind_2d(post_p_dof, "u_depth", 1, post_hdr.depth)
|
||||
u_mat4(gpu_uniform(post_p_dof, "u_inv_proj"), cam_inv_proj)
|
||||
u_f2(gpu_uniform(post_p_dof, "u_texel"), 1.0 / float(post_w), 1.0 / float(post_h))
|
||||
u_f(gpu_uniform(post_p_dof, "u_focus"), post_dof_focus)
|
||||
u_f(gpu_uniform(post_p_dof, "u_aperture"), post_dof_aperture)
|
||||
u_f(gpu_uniform(post_p_dof, "u_max_coc"), post_dof_max)
|
||||
mesh_draw(post_fs)
|
||||
post_color = post_dof.color
|
||||
function post_dof_pass(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.post_dof_aperture == 0.0 { return }
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_blend(render3d_st, false)
|
||||
target_bind(render3d_st, render3d_st.post_dof)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_dof)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_dof, "u_src", 0, render3d_st.post_color)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_dof, "u_depth", 1, render3d_st.post_hdr.depth)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_dof, "u_inv_proj"), render3d_st.cam_inv_proj)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_dof, "u_texel"), 1.0 / float(render3d_st.post_w), 1.0 / float(render3d_st.post_h))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_dof, "u_focus"), render3d_st.post_dof_focus)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_dof, "u_aperture"), render3d_st.post_dof_aperture)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_dof, "u_max_coc"), render3d_st.post_dof_max)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
render3d_st.post_color = render3d_st.post_dof.color
|
||||
}
|
||||
|
||||
function post_bloom_pass() -> void {
|
||||
gpu_depth_test(false)
|
||||
gpu_blend(false)
|
||||
var src = post_color
|
||||
var sw = post_color_w; var sh = post_color_h
|
||||
gpu_use_program(post_p_down)
|
||||
function post_bloom_pass(render3d_st: mut Render3dState) -> void {
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_blend(render3d_st, false)
|
||||
var src = render3d_st.post_color
|
||||
var sw = render3d_st.post_color_w; var sh = render3d_st.post_color_h
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_down)
|
||||
for i in 0 .. BLOOM_LEVELS {
|
||||
let t = post_bloom[i]
|
||||
target_bind(t)
|
||||
r3d_bind_2d(post_p_down, "u_src", 0, src)
|
||||
u_f2(gpu_uniform(post_p_down, "u_texel"), 1.0 / float(sw), 1.0 / float(sh))
|
||||
let t = render3d_st.post_bloom[i]
|
||||
target_bind(render3d_st, t)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_down, "u_src", 0, src)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_down, "u_texel"), 1.0 / float(sw), 1.0 / float(sh))
|
||||
var th = -1.0
|
||||
if i == 0 { th = 1.2 }
|
||||
u_f(gpu_uniform(post_p_down, "u_threshold"), th)
|
||||
mesh_draw(post_fs)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_down, "u_threshold"), th)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
src = t.color; sw = t.w; sh = t.h
|
||||
}
|
||||
gpu_use_program(post_p_up)
|
||||
gpu_blend(true)
|
||||
gpu_blend_func(GL_ONE, GL_ONE)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_up)
|
||||
gpu_blend(render3d_st, true)
|
||||
gpu_blend_func(render3d_st, GL_ONE, GL_ONE)
|
||||
var i = BLOOM_LEVELS - 1
|
||||
while i > 0 {
|
||||
let from = post_bloom[i]
|
||||
let to = post_bloom[i - 1]
|
||||
target_bind(to)
|
||||
r3d_bind_2d(post_p_up, "u_src", 0, from.color)
|
||||
u_f2(gpu_uniform(post_p_up, "u_texel"), 1.0 / float(from.w), 1.0 / float(from.h))
|
||||
u_f(gpu_uniform(post_p_up, "u_radius"), 1.0)
|
||||
mesh_draw(post_fs)
|
||||
let from = render3d_st.post_bloom[i]
|
||||
let to = render3d_st.post_bloom[i - 1]
|
||||
target_bind(render3d_st, to)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_up, "u_src", 0, from.color)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_up, "u_texel"), 1.0 / float(from.w), 1.0 / float(from.h))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_up, "u_radius"), 1.0)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
i -= 1
|
||||
}
|
||||
gpu_blend(false)
|
||||
gpu_blend(render3d_st, false)
|
||||
}
|
||||
|
||||
function post_tonemap(color_tex: int) -> void {
|
||||
var prog = post_p_tone
|
||||
if gpu_hdr_active() {
|
||||
if post_p_tone_hdr == 0 { post_p_tone_hdr = r3d_program("fullscreen.vert", "tonemap.frag", "#define HDR10\n") }
|
||||
prog = post_p_tone_hdr
|
||||
function post_tonemap(render3d_st: mut Render3dState, color_tex: int) -> void {
|
||||
var prog = render3d_st.post_p_tone
|
||||
if gpu_hdr_active(render3d_st) {
|
||||
if render3d_st.post_p_tone_hdr == 0 { render3d_st.post_p_tone_hdr = r3d_program(render3d_st, "fullscreen.vert", "tonemap.frag", "#define HDR10\n") }
|
||||
prog = render3d_st.post_p_tone_hdr
|
||||
}
|
||||
if post_ldr_hdr != gpu_hdr_active() {
|
||||
let w = post_ldr.w; let h = post_ldr.h
|
||||
target_free(post_ldr)
|
||||
post_ldr = target_new(w, h, post_ldr_fmt(), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
||||
post_ldr_hdr = gpu_hdr_active()
|
||||
if render3d_st.post_ldr_hdr != gpu_hdr_active(render3d_st) {
|
||||
let w = render3d_st.post_ldr.w; let h = render3d_st.post_ldr.h
|
||||
target_free(render3d_st, render3d_st.post_ldr)
|
||||
render3d_st.post_ldr = target_new(render3d_st, w, h, post_ldr_fmt(render3d_st), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
||||
render3d_st.post_ldr_hdr = gpu_hdr_active(render3d_st)
|
||||
}
|
||||
if post_auto { post_measure() }
|
||||
target_bind(post_ldr)
|
||||
gpu_depth_test(false)
|
||||
gpu_use_program(prog)
|
||||
r3d_bind_2d(prog, "u_hdr", 0, color_tex)
|
||||
r3d_bind_2d(prog, "u_bloom", 1, post_bloom[0].color)
|
||||
r3d_bind_2d(prog, "u_ao", 2, post_ao_blur.color)
|
||||
u_f(gpu_uniform(prog, "u_ao_strength"), post_ao_strength)
|
||||
u_f(gpu_uniform(prog, "u_gi_strength"), post_gi_strength)
|
||||
u_f(gpu_uniform(prog, "u_exposure"), post_exposure)
|
||||
if render3d_st.post_auto { post_measure(render3d_st) }
|
||||
target_bind(render3d_st, render3d_st.post_ldr)
|
||||
gpu_depth_test(render3d_st, false)
|
||||
gpu_use_program(render3d_st, prog)
|
||||
r3d_bind_2d(render3d_st, prog, "u_hdr", 0, color_tex)
|
||||
r3d_bind_2d(render3d_st, prog, "u_bloom", 1, render3d_st.post_bloom[0].color)
|
||||
r3d_bind_2d(render3d_st, prog, "u_ao", 2, render3d_st.post_ao_blur.color)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_ao_strength"), render3d_st.post_ao_strength)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_gi_strength"), render3d_st.post_gi_strength)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_exposure"), render3d_st.post_exposure)
|
||||
var auto = 0.0
|
||||
if post_auto and post_adapt_t != null { auto = 1.0; r3d_bind_2d(prog, "u_adapt", 3, post_adapt_t[post_adapt_i].color) }
|
||||
u_f(gpu_uniform(prog, "u_auto"), auto)
|
||||
u_f(gpu_uniform(prog, "u_bloom_strength"), float_from_bits(post_bloom_strength))
|
||||
u_f(gpu_uniform(prog, "u_vignette"), post_vignette)
|
||||
u_f(gpu_uniform(prog, "u_saturation"), post_saturation)
|
||||
u_f(gpu_uniform(prog, "u_contrast"), post_contrast)
|
||||
u_f3(gpu_uniform(prog, "u_wb"), post_wb_r, post_wb_g, post_wb_b)
|
||||
u_f3(gpu_uniform(prog, "u_lift"), post_lift_r, post_lift_g, post_lift_b)
|
||||
u_f3(gpu_uniform(prog, "u_gain"), post_gain_r, post_gain_g, post_gain_b)
|
||||
if render3d_st.post_auto and render3d_st.post_adapt_t != null { auto = 1.0; r3d_bind_2d(render3d_st, prog, "u_adapt", 3, render3d_st.post_adapt_t[render3d_st.post_adapt_i].color) }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_auto"), auto)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_bloom_strength"), float_from_bits(render3d_st.post_bloom_strength))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_vignette"), render3d_st.post_vignette)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_saturation"), render3d_st.post_saturation)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_contrast"), render3d_st.post_contrast)
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, prog, "u_wb"), render3d_st.post_wb_r, render3d_st.post_wb_g, render3d_st.post_wb_b)
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, prog, "u_lift"), render3d_st.post_lift_r, render3d_st.post_lift_g, render3d_st.post_lift_b)
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, prog, "u_gain"), render3d_st.post_gain_r, render3d_st.post_gain_g, render3d_st.post_gain_b)
|
||||
# the HDR10 variant's display calibration (the SDR program has none of these, and -1 sets nothing)
|
||||
u_f(gpu_uniform(prog, "u_hdr_peak"), r3d_hdr_peak_nits())
|
||||
u_f(gpu_uniform(prog, "u_hdr_paper"), r3d_hdr_paper_nits())
|
||||
u_f(gpu_uniform(prog, "u_hdr_black"), r3d_hdr_black_nits())
|
||||
mesh_draw(post_fs)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_hdr_peak"), r3d_hdr_peak_nits(render3d_st))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_hdr_paper"), r3d_hdr_paper_nits(render3d_st))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_hdr_black"), r3d_hdr_black_nits(render3d_st))
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
# sharpen + grain onto the screen
|
||||
gpu_fb_bind(gpu_screen_fb())
|
||||
gpu_viewport(0, 0, gl_w, gl_h)
|
||||
gpu_use_program(post_p_sharp)
|
||||
r3d_bind_2d(post_p_sharp, "u_src", 0, post_ldr.color)
|
||||
u_f2(gpu_uniform(post_p_sharp, "u_texel"), 1.0 / float(post_ldr.w), 1.0 / float(post_ldr.h))
|
||||
u_f(gpu_uniform(post_p_sharp, "u_amount"), post_sharpen)
|
||||
var fx = post_fxaa
|
||||
if r3d_env_has("R3D_NOFXAA") { fx = 0.0 }
|
||||
u_f(gpu_uniform(post_p_sharp, "u_fxaa"), fx)
|
||||
gpu_fb_bind(render3d_st, gpu_screen_fb(render3d_st))
|
||||
gpu_viewport(render3d_st, 0, 0, gl_w, gl_h)
|
||||
gpu_use_program(render3d_st, render3d_st.post_p_sharp)
|
||||
r3d_bind_2d(render3d_st, render3d_st.post_p_sharp, "u_src", 0, render3d_st.post_ldr.color)
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_sharp, "u_texel"), 1.0 / float(render3d_st.post_ldr.w), 1.0 / float(render3d_st.post_ldr.h))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_sharp, "u_amount"), render3d_st.post_sharpen)
|
||||
var fx = render3d_st.post_fxaa
|
||||
if r3d_env_has(render3d_st, "R3D_NOFXAA") { fx = 0.0 }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_sharp, "u_fxaa"), fx)
|
||||
# R3D_NOGRAIN=1: no film grain, so two frames of a still camera can be compared for what else moves
|
||||
var grain = post_grain
|
||||
if r3d_env_has("R3D_NOGRAIN") { grain = float_bits(0.0) }
|
||||
u_f(gpu_uniform(post_p_sharp, "u_grain"), float_from_bits(grain))
|
||||
u_f(gpu_uniform(post_p_sharp, "u_time"), r3d_time)
|
||||
mesh_draw(post_fs)
|
||||
var grain = render3d_st.post_grain
|
||||
if r3d_env_has(render3d_st, "R3D_NOGRAIN") { grain = float_bits(0.0) }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_sharp, "u_grain"), float_from_bits(grain))
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, render3d_st.post_p_sharp, "u_time"), render3d_st.r3d_time)
|
||||
mesh_draw(render3d_st, render3d_st.post_fs)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -14,98 +14,89 @@
|
|||
const PROF_SLOTS: int = 16
|
||||
const PROF_RING: int = 4 # frames of latency before a result is read
|
||||
|
||||
var prof_on: bool = false
|
||||
var prof_names: []pointer = null
|
||||
var prof_ids: words = null # PROF_SLOTS * PROF_RING query objects
|
||||
var prof_ns: []long = null # accumulated nanoseconds per slot
|
||||
var prof_hits: []long = null # samples accumulated per slot
|
||||
var prof_n: int = 0 # slots in use
|
||||
var prof_frame: int = 0
|
||||
var prof_active: int = -1 # slot whose query is currently open
|
||||
var prof_scratch: words = null
|
||||
|
||||
function prof_init() -> void {
|
||||
prof_on = r3d_env_has("R3D_PROF")
|
||||
if not prof_on { return }
|
||||
prof_names = new []pointer
|
||||
prof_ns = new []long
|
||||
prof_hits = new []long
|
||||
prof_ids = words(PROF_SLOTS * PROF_RING)
|
||||
prof_scratch = words(4)
|
||||
gpu_query_new(PROF_SLOTS * PROF_RING, prof_ids)
|
||||
prof_n = 0
|
||||
prof_frame = 0
|
||||
prof_active = -1
|
||||
function prof_init(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.prof_on = r3d_env_has(render3d_st, "R3D_PROF")
|
||||
if not render3d_st.prof_on { return }
|
||||
render3d_st.prof_names = new []pointer
|
||||
render3d_st.prof_ns = new []long
|
||||
render3d_st.prof_hits = new []long
|
||||
render3d_st.prof_ids = words(PROF_SLOTS * PROF_RING)
|
||||
render3d_st.prof_scratch = words(4)
|
||||
gpu_query_new(render3d_st, PROF_SLOTS * PROF_RING, render3d_st.prof_ids)
|
||||
render3d_st.prof_n = 0
|
||||
render3d_st.prof_frame = 0
|
||||
render3d_st.prof_active = -1
|
||||
}
|
||||
|
||||
# the slot index for `name`, registering it on first sight (order = frame order)
|
||||
function prof_slot(name: pointer) -> int {
|
||||
function prof_slot(render3d_st: mut Render3dState, name: pointer) -> int {
|
||||
var i = 0
|
||||
while i < prof_n {
|
||||
if prof_names[i] == name { return i }
|
||||
while i < render3d_st.prof_n {
|
||||
if render3d_st.prof_names[i] == name { return i }
|
||||
i += 1
|
||||
}
|
||||
if prof_n >= PROF_SLOTS { return -1 }
|
||||
push(prof_names, name)
|
||||
push(prof_ns, 0)
|
||||
push(prof_hits, 0)
|
||||
prof_n += 1
|
||||
return prof_n - 1
|
||||
if render3d_st.prof_n >= PROF_SLOTS { return -1 }
|
||||
push(render3d_st.prof_names, name)
|
||||
push(render3d_st.prof_ns, 0)
|
||||
push(render3d_st.prof_hits, 0)
|
||||
render3d_st.prof_n += 1
|
||||
return render3d_st.prof_n - 1
|
||||
}
|
||||
|
||||
function prof_begin(name: pointer) -> void {
|
||||
ds_set_pass(name)
|
||||
if not prof_on { return }
|
||||
if prof_active >= 0 { return } # GL_TIME_ELAPSED queries cannot nest
|
||||
let s = prof_slot(name)
|
||||
function prof_begin(render3d_st: mut Render3dState, name: pointer) -> void {
|
||||
ds_set_pass(render3d_st, name)
|
||||
if not render3d_st.prof_on { return }
|
||||
if render3d_st.prof_active >= 0 { return } # GL_TIME_ELAPSED queries cannot nest
|
||||
let s = prof_slot(render3d_st, name)
|
||||
if s < 0 { return }
|
||||
prof_active = s
|
||||
gpu_query_begin(prof_ids[s * PROF_RING + (prof_frame % PROF_RING)])
|
||||
render3d_st.prof_active = s
|
||||
gpu_query_begin(render3d_st, render3d_st.prof_ids[s * PROF_RING + (render3d_st.prof_frame % PROF_RING)])
|
||||
}
|
||||
|
||||
function prof_end() -> void {
|
||||
ds_set_pass(ds_outside)
|
||||
if not prof_on { return }
|
||||
if prof_active < 0 { return }
|
||||
gpu_query_end()
|
||||
prof_active = -1
|
||||
function prof_end(render3d_st: mut Render3dState) -> void {
|
||||
ds_set_pass(render3d_st, render3d_st.ds_outside)
|
||||
if not render3d_st.prof_on { return }
|
||||
if render3d_st.prof_active < 0 { return }
|
||||
gpu_query_end(render3d_st)
|
||||
render3d_st.prof_active = -1
|
||||
}
|
||||
|
||||
# Collect the queries issued PROF_RING-1 frames ago — long finished, so no stall.
|
||||
function prof_collect() -> void {
|
||||
if not prof_on { return }
|
||||
prof_frame += 1
|
||||
if prof_frame < PROF_RING { return }
|
||||
let slot_frame = (prof_frame + 1) % PROF_RING
|
||||
function prof_collect(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
render3d_st.prof_frame += 1
|
||||
if render3d_st.prof_frame < PROF_RING { return }
|
||||
let slot_frame = (render3d_st.prof_frame + 1) % PROF_RING
|
||||
var s = 0
|
||||
while s < prof_n {
|
||||
let q = prof_ids[s * PROF_RING + slot_frame]
|
||||
if gpu_query_result(q, prof_scratch) {
|
||||
while s < render3d_st.prof_n {
|
||||
let q = render3d_st.prof_ids[s * PROF_RING + slot_frame]
|
||||
if gpu_query_result(render3d_st, q, render3d_st.prof_scratch) {
|
||||
# the low 32 bits are ample: a pass is far below 4 seconds
|
||||
prof_ns[s] = prof_ns[s] + prof_scratch[0]
|
||||
prof_hits[s] = prof_hits[s] + 1
|
||||
render3d_st.prof_ns[s] = render3d_st.prof_ns[s] + render3d_st.prof_scratch[0]
|
||||
render3d_st.prof_hits[s] = render3d_st.prof_hits[s] + 1
|
||||
}
|
||||
s += 1
|
||||
}
|
||||
}
|
||||
|
||||
function prof_report() -> void {
|
||||
if not prof_on { return }
|
||||
function prof_report(render3d_st: Render3dState) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
print("")
|
||||
print("GPU time per pass (mean over the run):")
|
||||
var total = 0
|
||||
var s = 0
|
||||
while s < prof_n {
|
||||
if prof_hits[s] > 0 { total = total + prof_ns[s] / prof_hits[s] }
|
||||
while s < render3d_st.prof_n {
|
||||
if render3d_st.prof_hits[s] > 0 { total = total + render3d_st.prof_ns[s] / render3d_st.prof_hits[s] }
|
||||
s += 1
|
||||
}
|
||||
s = 0
|
||||
while s < prof_n {
|
||||
if prof_hits[s] > 0 {
|
||||
let us = prof_ns[s] / prof_hits[s] / 1000
|
||||
while s < render3d_st.prof_n {
|
||||
if render3d_st.prof_hits[s] > 0 {
|
||||
let us = render3d_st.prof_ns[s] / render3d_st.prof_hits[s] / 1000
|
||||
var pct = 0
|
||||
if total > 0 { pct = (prof_ns[s] / prof_hits[s]) * 100 / total }
|
||||
print(` {Text.pad_right(prof_names[s], 22)} {Text.pad_left(string(us), 7)} us {string(pct)}%`)
|
||||
if total > 0 { pct = (render3d_st.prof_ns[s] / render3d_st.prof_hits[s]) * 100 / total }
|
||||
print(` {Text.pad_right(render3d_st.prof_names[s], 22)} {Text.pad_left(string(us), 7)} us {string(pct)}%`)
|
||||
}
|
||||
s += 1
|
||||
}
|
||||
|
|
@ -118,99 +109,66 @@ function prof_report() -> void {
|
|||
# in fifty does all the work. `Time.delta` here is a fixed 60 Hz timestep and there is
|
||||
# no finer wall clock in the runtime, so measure the WORK instead — instances generated
|
||||
# per frame is exactly what the hitch is made of, and it needs no clock at all.
|
||||
var prof_gen: []int = null
|
||||
var prof_gen_cur: int = 0
|
||||
var prof_ft: []long = null # real frame times, microseconds
|
||||
var prof_last_us: long = 0
|
||||
|
||||
function prof_gen_add(n: int) -> void {
|
||||
if not prof_on { return }
|
||||
prof_gen_cur += n
|
||||
function prof_gen_add(render3d_st: mut Render3dState, n: int) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
render3d_st.prof_gen_cur += n
|
||||
}
|
||||
|
||||
# Per-chunk generation, so a hitch can be pinned on a stream and a band rather than on
|
||||
# "streaming". One chunk is generated atomically, so the worst chunk is the worst frame.
|
||||
var prof_chunk_us: []long = null
|
||||
var prof_chunk_kind: []int = null
|
||||
var prof_chunk_band: []int = null
|
||||
var prof_chunk_n: []int = null
|
||||
var prof_gen_us_cur: long = 0
|
||||
var prof_gen_us: []long = null # generation microseconds per frame
|
||||
# CPU work that only happens on some frames — which is what a hitch is made of.
|
||||
var prof_layer_us_cur: long = 0 # rebuilding and uploading instance buffers
|
||||
var prof_bake_us_cur: long = 0 # the height-field shadow rebake
|
||||
var prof_layer_us: []long = null
|
||||
var prof_bake_us: []long = null
|
||||
var prof_dt: []long = null # frame time, aligned with the arrays above
|
||||
var prof_up_cur: long = 0 # instance bytes uploaded this frame
|
||||
var prof_up: []long = null
|
||||
# Where a frame's time went at the coarsest useful split: work this process did, and
|
||||
# time spent waiting for the GPU to finish it. Headless drains the GPU inside swap, so
|
||||
# the two are cleanly separable there.
|
||||
var prof_pre_swap: long = 0
|
||||
var prof_cpu_us: []long = null
|
||||
var prof_sum_dt: long = 0 # sums over the frames past the first eight, for the mean split
|
||||
var prof_sum_cpu: long = 0
|
||||
var prof_sum_game: long = 0
|
||||
var prof_sum_n: int = 0
|
||||
function prof_before_swap() -> void { if prof_on { prof_pre_swap = gl_now_us() } }
|
||||
function prof_before_swap(render3d_st: mut Render3dState) -> void { if render3d_st.prof_on { render3d_st.prof_pre_swap = gl_now_us() } }
|
||||
|
||||
# The single most expensive CPU phase of each frame, and what it was. A hitch is one
|
||||
# phase running long on one frame, so recording the worst one per frame is enough to
|
||||
# name it without keeping a timeline.
|
||||
var prof_mark_last: long = 0
|
||||
var prof_mark_best: long = 0
|
||||
var prof_mark_name: pointer = null
|
||||
var prof_mark_us: []long = null
|
||||
var prof_mark_who: []pointer = null
|
||||
function prof_mark_start() -> void { if prof_on { prof_mark_last = gl_now_us(); prof_mark_best = 0; prof_mark_name = null } }
|
||||
function prof_mark_start(render3d_st: mut Render3dState) -> void { if render3d_st.prof_on { render3d_st.prof_mark_last = gl_now_us(); render3d_st.prof_mark_best = 0; render3d_st.prof_mark_name = null } }
|
||||
# ... and every phase's share of the average frame, past the first eight frames (loading, first fills)
|
||||
var prof_mk_names: []pointer = null
|
||||
var prof_mk_us: []long = null
|
||||
var prof_mk_n: int = 0
|
||||
function prof_cpu_mark(name: pointer) -> void {
|
||||
if not prof_on { return }
|
||||
function prof_cpu_mark(render3d_st: mut Render3dState, name: pointer) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
let now = gl_now_us()
|
||||
let d = now - prof_mark_last
|
||||
prof_mark_last = now
|
||||
if d > prof_mark_best { prof_mark_best = d; prof_mark_name = name }
|
||||
let d = now - render3d_st.prof_mark_last
|
||||
render3d_st.prof_mark_last = now
|
||||
if d > render3d_st.prof_mark_best { render3d_st.prof_mark_best = d; render3d_st.prof_mark_name = name }
|
||||
var k = -1
|
||||
var i = 0
|
||||
while i < prof_mk_n { if prof_mk_names[i] == name { k = i }; i += 1 }
|
||||
while i < render3d_st.prof_mk_n { if render3d_st.prof_mk_names[i] == name { k = i }; i += 1 }
|
||||
if k < 0 {
|
||||
if prof_mk_names == null { prof_mk_names = new []pointer; prof_mk_us = new []long }
|
||||
push(prof_mk_names, name); push(prof_mk_us, 0)
|
||||
prof_mk_n += 1
|
||||
k = prof_mk_n - 1
|
||||
if render3d_st.prof_mk_names == null { render3d_st.prof_mk_names = new []pointer; render3d_st.prof_mk_us = new []long }
|
||||
push(render3d_st.prof_mk_names, name); push(render3d_st.prof_mk_us, 0)
|
||||
render3d_st.prof_mk_n += 1
|
||||
k = render3d_st.prof_mk_n - 1
|
||||
}
|
||||
if prof_gen != null and len(prof_gen) > 8 { prof_mk_us[k] = prof_mk_us[k] + d }
|
||||
if render3d_st.prof_gen != null and len(render3d_st.prof_gen) > 8 { render3d_st.prof_mk_us[k] = render3d_st.prof_mk_us[k] + d }
|
||||
}
|
||||
|
||||
# the game's own per-frame work, kept apart from the renderer's
|
||||
var prof_game_us_cur: long = 0
|
||||
var prof_game_us: []long = null
|
||||
function prof_game_add(us: long) -> void { if prof_on { prof_game_us_cur = prof_game_us_cur + us } }
|
||||
function prof_game_add(render3d_st: mut Render3dState, us: long) -> void { if render3d_st.prof_on { render3d_st.prof_game_us_cur = render3d_st.prof_game_us_cur + us } }
|
||||
|
||||
function prof_layer_add(us: long, bytes: long) -> void {
|
||||
if not prof_on { return }
|
||||
prof_layer_us_cur = prof_layer_us_cur + us
|
||||
prof_up_cur = prof_up_cur + bytes
|
||||
function prof_layer_add(render3d_st: mut Render3dState, us: long, bytes: long) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
render3d_st.prof_layer_us_cur = render3d_st.prof_layer_us_cur + us
|
||||
render3d_st.prof_up_cur = render3d_st.prof_up_cur + bytes
|
||||
}
|
||||
function prof_bake_add(us: long) -> void { if prof_on { prof_bake_us_cur = prof_bake_us_cur + us } }
|
||||
function prof_bake_add(render3d_st: mut Render3dState, us: long) -> void { if render3d_st.prof_on { render3d_st.prof_bake_us_cur = render3d_st.prof_bake_us_cur + us } }
|
||||
|
||||
function prof_chunk(kind: int, band: int, count: int, us: long) -> void {
|
||||
if not prof_on { return }
|
||||
if prof_chunk_us == null {
|
||||
prof_chunk_us = new []long; prof_chunk_kind = new []int
|
||||
prof_chunk_band = new []int; prof_chunk_n = new []int
|
||||
function prof_chunk(render3d_st: mut Render3dState, kind: int, band: int, count: int, us: long) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
if render3d_st.prof_chunk_us == null {
|
||||
render3d_st.prof_chunk_us = new []long; render3d_st.prof_chunk_kind = new []int
|
||||
render3d_st.prof_chunk_band = new []int; render3d_st.prof_chunk_n = new []int
|
||||
}
|
||||
push(prof_chunk_us, us); push(prof_chunk_kind, kind)
|
||||
push(prof_chunk_band, band); push(prof_chunk_n, count)
|
||||
prof_gen_us_cur += us
|
||||
push(render3d_st.prof_chunk_us, us); push(render3d_st.prof_chunk_kind, kind)
|
||||
push(render3d_st.prof_chunk_band, band); push(render3d_st.prof_chunk_n, count)
|
||||
render3d_st.prof_gen_us_cur += us
|
||||
}
|
||||
|
||||
function prof_chunk_report() -> void {
|
||||
if not prof_on or prof_chunk_us == null { return }
|
||||
function prof_chunk_report(render3d_st: Render3dState) -> void {
|
||||
if not render3d_st.prof_on or render3d_st.prof_chunk_us == null { return }
|
||||
print("")
|
||||
print("chunk generation (one chunk is atomic, so the worst chunk is the worst frame):")
|
||||
# totals per (kind, band)
|
||||
|
|
@ -220,18 +178,18 @@ function prof_chunk_report() -> void {
|
|||
var cnt = new []int
|
||||
var mx = new []long
|
||||
var i = 0
|
||||
while i < len(prof_chunk_us) {
|
||||
while i < len(render3d_st.prof_chunk_us) {
|
||||
var f = -1
|
||||
var j = 0
|
||||
while j < len(kinds) { if kinds[j] == prof_chunk_kind[i] and bands[j] == prof_chunk_band[i] { f = j }; j += 1 }
|
||||
while j < len(kinds) { if kinds[j] == render3d_st.prof_chunk_kind[i] and bands[j] == render3d_st.prof_chunk_band[i] { f = j }; j += 1 }
|
||||
if f < 0 {
|
||||
push(kinds, prof_chunk_kind[i]); push(bands, prof_chunk_band[i])
|
||||
push(kinds, render3d_st.prof_chunk_kind[i]); push(bands, render3d_st.prof_chunk_band[i])
|
||||
push(tot, 0); push(cnt, 0); push(mx, 0)
|
||||
f = len(kinds) - 1
|
||||
}
|
||||
tot[f] = tot[f] + prof_chunk_us[i]
|
||||
tot[f] = tot[f] + render3d_st.prof_chunk_us[i]
|
||||
cnt[f] = cnt[f] + 1
|
||||
if prof_chunk_us[i] > mx[f] { mx[f] = prof_chunk_us[i] }
|
||||
if render3d_st.prof_chunk_us[i] > mx[f] { mx[f] = render3d_st.prof_chunk_us[i] }
|
||||
i += 1
|
||||
}
|
||||
var k = 0
|
||||
|
|
@ -240,10 +198,10 @@ function prof_chunk_report() -> void {
|
|||
k += 1
|
||||
}
|
||||
# the per-frame distribution of generation time: this is the hitch itself
|
||||
if prof_gen_us == null or len(prof_gen_us) < 16 { return }
|
||||
if render3d_st.prof_gen_us == null or len(render3d_st.prof_gen_us) < 16 { return }
|
||||
let sorted = new []long
|
||||
var a = 8
|
||||
while a < len(prof_gen_us) { push(sorted, prof_gen_us[a]); a += 1 }
|
||||
while a < len(render3d_st.prof_gen_us) { push(sorted, render3d_st.prof_gen_us[a]); a += 1 }
|
||||
var x = 1
|
||||
while x < len(sorted) {
|
||||
let v = sorted[x]
|
||||
|
|
@ -260,48 +218,48 @@ function prof_chunk_report() -> void {
|
|||
print(` generation per frame (us): p95 {string(sorted[(n * 95) / 100])} p99 {string(sorted[(n * 99) / 100])} worst {string(sorted[n - 1])} frames that generated: {string(busy)} of {string(n)} total {string(t / 1000)} ms`)
|
||||
}
|
||||
|
||||
function prof_gen_frame() -> void {
|
||||
ds_frame()
|
||||
if not prof_on { return }
|
||||
if prof_gen == null {
|
||||
prof_gen = new []int; prof_ft = new []long
|
||||
prof_gen_us = new []long; prof_layer_us = new []long
|
||||
prof_bake_us = new []long; prof_dt = new []long; prof_up = new []long
|
||||
prof_cpu_us = new []long; prof_game_us = new []long
|
||||
prof_mark_us = new []long; prof_mark_who = new []pointer
|
||||
function prof_gen_frame(render3d_st: mut Render3dState) -> void {
|
||||
ds_frame(render3d_st)
|
||||
if not render3d_st.prof_on { return }
|
||||
if render3d_st.prof_gen == null {
|
||||
render3d_st.prof_gen = new []int; render3d_st.prof_ft = new []long
|
||||
render3d_st.prof_gen_us = new []long; render3d_st.prof_layer_us = new []long
|
||||
render3d_st.prof_bake_us = new []long; render3d_st.prof_dt = new []long; render3d_st.prof_up = new []long
|
||||
render3d_st.prof_cpu_us = new []long; render3d_st.prof_game_us = new []long
|
||||
render3d_st.prof_mark_us = new []long; render3d_st.prof_mark_who = new []pointer
|
||||
}
|
||||
push(prof_gen, prof_gen_cur)
|
||||
prof_gen_cur = 0
|
||||
push(render3d_st.prof_gen, render3d_st.prof_gen_cur)
|
||||
render3d_st.prof_gen_cur = 0
|
||||
let now = gl_now_us()
|
||||
var dtf: long = 0
|
||||
if prof_last_us != 0 { push(prof_ft, now - prof_last_us); dtf = now - prof_last_us }
|
||||
prof_last_us = now
|
||||
if render3d_st.prof_last_us != 0 { push(render3d_st.prof_ft, now - render3d_st.prof_last_us); dtf = now - render3d_st.prof_last_us }
|
||||
render3d_st.prof_last_us = now
|
||||
# everything the frame just ended spent on work it only does sometimes
|
||||
push(prof_dt, dtf)
|
||||
push(prof_gen_us, prof_gen_us_cur)
|
||||
push(prof_layer_us, prof_layer_us_cur)
|
||||
push(prof_bake_us, prof_bake_us_cur)
|
||||
push(prof_up, prof_up_cur)
|
||||
push(render3d_st.prof_dt, dtf)
|
||||
push(render3d_st.prof_gen_us, render3d_st.prof_gen_us_cur)
|
||||
push(render3d_st.prof_layer_us, render3d_st.prof_layer_us_cur)
|
||||
push(render3d_st.prof_bake_us, render3d_st.prof_bake_us_cur)
|
||||
push(render3d_st.prof_up, render3d_st.prof_up_cur)
|
||||
var cpu: long = 0
|
||||
if prof_pre_swap != 0 and dtf != 0 { cpu = prof_pre_swap - (now - dtf) }
|
||||
push(prof_cpu_us, cpu)
|
||||
push(prof_game_us, prof_game_us_cur)
|
||||
if len(prof_gen) > 9 and dtf != 0 {
|
||||
prof_sum_dt = prof_sum_dt + dtf; prof_sum_cpu = prof_sum_cpu + cpu; prof_sum_game = prof_sum_game + prof_game_us_cur
|
||||
prof_sum_n += 1
|
||||
if render3d_st.prof_pre_swap != 0 and dtf != 0 { cpu = render3d_st.prof_pre_swap - (now - dtf) }
|
||||
push(render3d_st.prof_cpu_us, cpu)
|
||||
push(render3d_st.prof_game_us, render3d_st.prof_game_us_cur)
|
||||
if len(render3d_st.prof_gen) > 9 and dtf != 0 {
|
||||
render3d_st.prof_sum_dt = render3d_st.prof_sum_dt + dtf; render3d_st.prof_sum_cpu = render3d_st.prof_sum_cpu + cpu; render3d_st.prof_sum_game = render3d_st.prof_sum_game + render3d_st.prof_game_us_cur
|
||||
render3d_st.prof_sum_n += 1
|
||||
}
|
||||
push(prof_mark_us, prof_mark_best)
|
||||
if prof_mark_name == null { push(prof_mark_who, "-") } else { push(prof_mark_who, prof_mark_name) }
|
||||
prof_gen_us_cur = 0; prof_layer_us_cur = 0; prof_bake_us_cur = 0; prof_up_cur = 0; prof_game_us_cur = 0
|
||||
push(render3d_st.prof_mark_us, render3d_st.prof_mark_best)
|
||||
if render3d_st.prof_mark_name == null { push(render3d_st.prof_mark_who, "-") } else { push(render3d_st.prof_mark_who, render3d_st.prof_mark_name) }
|
||||
render3d_st.prof_gen_us_cur = 0; render3d_st.prof_layer_us_cur = 0; render3d_st.prof_bake_us_cur = 0; render3d_st.prof_up_cur = 0; render3d_st.prof_game_us_cur = 0
|
||||
}
|
||||
|
||||
# the distribution of REAL frame times: stutter lives in the tail, not the mean
|
||||
function prof_ft_report() -> void {
|
||||
if not prof_on { return }
|
||||
if prof_ft == null or len(prof_ft) < 16 { return }
|
||||
function prof_ft_report(render3d_st: Render3dState) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
if render3d_st.prof_ft == null or len(render3d_st.prof_ft) < 16 { return }
|
||||
let sorted = new []long
|
||||
var i = 8
|
||||
while i < len(prof_ft) { push(sorted, prof_ft[i]); i += 1 }
|
||||
while i < len(render3d_st.prof_ft) { push(sorted, render3d_st.prof_ft[i]); i += 1 }
|
||||
var a = 1
|
||||
while a < len(sorted) {
|
||||
let v = sorted[a]
|
||||
|
|
@ -340,19 +298,19 @@ function prof_ft_report() -> void {
|
|||
}
|
||||
print(` fps at median {string(1000000 / med)} frames over 1.3x median: {string(o13)}, over 1.5x: {string(o15)}, over 2x: {string(over)} — of {string(n)}`)
|
||||
print(` stutter: {string(excess / 1000)} ms of frame time beyond 1.2x median over the run`)
|
||||
print(` streaming totals over the run: walk {string(stream_us_walk / 1000)} ms (generate {string(stream_us_gen / 1000)} ms, gather {string(stream_us_gather / 1000)} ms), {string(stream_walks)} stream-walks`)
|
||||
print(` streaming totals over the run: walk {string(render3d_st.stream_us_walk / 1000)} ms (generate {string(render3d_st.stream_us_gen / 1000)} ms, gather {string(render3d_st.stream_us_gather / 1000)} ms), {string(render3d_st.stream_walks)} stream-walks`)
|
||||
# the average frame, split: what the process did before the swap (the game's share of it apart),
|
||||
# the rest waiting on the GPU and the swap, and the renderer's CPU phases in frame order
|
||||
if prof_sum_n > 0 {
|
||||
let dt = prof_sum_dt / prof_sum_n
|
||||
let cpu = prof_sum_cpu / prof_sum_n
|
||||
let game = prof_sum_game / prof_sum_n
|
||||
if render3d_st.prof_sum_n > 0 {
|
||||
let dt = render3d_st.prof_sum_dt / render3d_st.prof_sum_n
|
||||
let cpu = render3d_st.prof_sum_cpu / render3d_st.prof_sum_n
|
||||
let game = render3d_st.prof_sum_game / render3d_st.prof_sum_n
|
||||
print("")
|
||||
print(`MEAN frame {string(dt)} us over {string(prof_sum_n)} frames: CPU before the swap {string(cpu)} us (the game's own {string(game)} us), GPU and swap {string(dt - cpu)} us`)
|
||||
print(`MEAN frame {string(dt)} us over {string(render3d_st.prof_sum_n)} frames: CPU before the swap {string(cpu)} us (the game's own {string(game)} us), GPU and swap {string(dt - cpu)} us`)
|
||||
print(" renderer CPU by phase, mean per frame:")
|
||||
var i = 0
|
||||
while i < prof_mk_n {
|
||||
print(` {Text.pad_right(prof_mk_names[i], 20)} {Text.pad_left(string(prof_mk_us[i] / prof_sum_n), 7)} us`)
|
||||
while i < render3d_st.prof_mk_n {
|
||||
print(` {Text.pad_right(render3d_st.prof_mk_names[i], 20)} {Text.pad_left(string(render3d_st.prof_mk_us[i] / render3d_st.prof_sum_n), 7)} us`)
|
||||
i += 1
|
||||
}
|
||||
}
|
||||
|
|
@ -360,13 +318,13 @@ function prof_ft_report() -> void {
|
|||
|
||||
# The slowest frames of the run, with the once-in-a-while CPU work that landed in them.
|
||||
# An average never shows a hitch; this is the list of the frames you actually felt.
|
||||
function prof_hitch_report() -> void {
|
||||
if not prof_on or prof_dt == null or len(prof_dt) < 32 { return }
|
||||
let n = len(prof_dt)
|
||||
function prof_hitch_report(render3d_st: Render3dState) -> void {
|
||||
if not render3d_st.prof_on or render3d_st.prof_dt == null or len(render3d_st.prof_dt) < 32 { return }
|
||||
let n = len(render3d_st.prof_dt)
|
||||
# median, for a sense of what "slow" means here
|
||||
let sorted = new []long
|
||||
var i = 8
|
||||
while i < n { push(sorted, prof_dt[i]); i += 1 }
|
||||
while i < n { push(sorted, render3d_st.prof_dt[i]); i += 1 }
|
||||
var a = 1
|
||||
while a < len(sorted) {
|
||||
let v = sorted[a]
|
||||
|
|
@ -387,22 +345,22 @@ function prof_hitch_report() -> void {
|
|||
var bestv: long = -1
|
||||
var k = 8
|
||||
while k < n {
|
||||
if prof_dt[k] <= cut and prof_dt[k] > bestv { bestv = prof_dt[k]; best = k }
|
||||
if render3d_st.prof_dt[k] <= cut and render3d_st.prof_dt[k] > bestv { bestv = render3d_st.prof_dt[k]; best = k }
|
||||
k += 1
|
||||
}
|
||||
if best < 0 { return }
|
||||
print(` {Text.pad_left(string(best), 6)} {Text.pad_left(string(prof_dt[best]), 6)}us {Text.pad_left(string(prof_cpu_us[best]), 7)}us {Text.pad_left(string(prof_dt[best] - prof_cpu_us[best]), 8)}us {Text.pad_left(string(prof_gen_us[best]), 7)}us {Text.pad_left(string(prof_game_us[best]), 7)}us {Text.pad_left(string(prof_up[best] / 1024), 7)}KB {Text.pad_right(prof_mark_who[best], 18)} {Text.pad_left(string(prof_mark_us[best]), 7)}us`)
|
||||
print(` {Text.pad_left(string(best), 6)} {Text.pad_left(string(render3d_st.prof_dt[best]), 6)}us {Text.pad_left(string(render3d_st.prof_cpu_us[best]), 7)}us {Text.pad_left(string(render3d_st.prof_dt[best] - render3d_st.prof_cpu_us[best]), 8)}us {Text.pad_left(string(render3d_st.prof_gen_us[best]), 7)}us {Text.pad_left(string(render3d_st.prof_game_us[best]), 7)}us {Text.pad_left(string(render3d_st.prof_up[best] / 1024), 7)}KB {Text.pad_right(render3d_st.prof_mark_who[best], 18)} {Text.pad_left(string(render3d_st.prof_mark_us[best]), 7)}us`)
|
||||
cut = bestv - 1
|
||||
shown += 1
|
||||
}
|
||||
}
|
||||
|
||||
function prof_gen_report() -> void {
|
||||
if not prof_on { return }
|
||||
if prof_gen == null or len(prof_gen) < 8 { return }
|
||||
function prof_gen_report(render3d_st: Render3dState) -> void {
|
||||
if not render3d_st.prof_on { return }
|
||||
if render3d_st.prof_gen == null or len(render3d_st.prof_gen) < 8 { return }
|
||||
let sorted = new []int
|
||||
var i = 8 # skip the first frames: one-off initial fill
|
||||
while i < len(prof_gen) { push(sorted, prof_gen[i]); i += 1 }
|
||||
while i < len(render3d_st.prof_gen) { push(sorted, render3d_st.prof_gen[i]); i += 1 }
|
||||
var a = 1
|
||||
while a < len(sorted) {
|
||||
let v = sorted[a]
|
||||
|
|
|
|||
|
|
@ -10,16 +10,9 @@
|
|||
# copy them in. The scanned CC0 materials are the other half: too large to ship with a
|
||||
# toolchain and redistributable from their origin, so they are fetched into the project
|
||||
# (`ludic assets`) and read from there.
|
||||
var r3d_root: string = "packages/ludic.render3d" # where shaders/ lives
|
||||
var r3d_assets: string = "assets/polyhaven" # where the CC0 assets live
|
||||
var r3d_root_found: bool = false
|
||||
# defines prepended to EVERY program (set before any is built): renderer-wide switches
|
||||
var r3d_global_defs: string = ""
|
||||
var r3d_noise_src: string = null
|
||||
var r3d_lighting_src: string = null
|
||||
var r3d_wind_src: string = null
|
||||
|
||||
function r3d_set_paths(root: string, assets: string) -> void { r3d_root = root; r3d_assets = assets }
|
||||
function r3d_set_paths(render3d_st: mut Render3dState, root: string, assets: string) -> void { render3d_st.r3d_root = root; render3d_st.r3d_assets = assets }
|
||||
|
||||
# The install root, as the compiler computes it: $LUDIC_HOME, else the directory of the
|
||||
# `ludic` on PATH. A game running from its own tree finds this package there.
|
||||
|
|
@ -31,61 +24,60 @@ function r3d_home() -> string {
|
|||
|
||||
# Settle r3d_root on first use: the package as checked out beside the project, else the
|
||||
# copy that ships with the toolchain.
|
||||
function r3d_find_root() -> void {
|
||||
if r3d_root_found { return }
|
||||
r3d_root_found = true
|
||||
if Fs.exists(`{r3d_root}/shaders/lighting.glsl`) { return }
|
||||
function r3d_find_root(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.r3d_root_found { return }
|
||||
render3d_st.r3d_root_found = true
|
||||
if Fs.exists(`{render3d_st.r3d_root}/shaders/lighting.glsl`) { return }
|
||||
let home = r3d_home()
|
||||
let alt = `{home}/packages/ludic.render3d`
|
||||
if Fs.exists(`{alt}/shaders/lighting.glsl`) { r3d_root = alt; return }
|
||||
print(`r3d: cannot find the renderer's shaders (looked in {r3d_root}/shaders and {alt}/shaders)`)
|
||||
if Fs.exists(`{alt}/shaders/lighting.glsl`) { render3d_st.r3d_root = alt; return }
|
||||
print(`r3d: cannot find the renderer's shaders (looked in {render3d_st.r3d_root}/shaders and {alt}/shaders)`)
|
||||
}
|
||||
|
||||
function r3d_shader_file(name: string) -> string {
|
||||
r3d_find_root()
|
||||
let path = `{r3d_root}/shaders/{name}`
|
||||
function r3d_shader_file(render3d_st: mut Render3dState, name: string) -> string {
|
||||
r3d_find_root(render3d_st)
|
||||
let path = `{render3d_st.r3d_root}/shaders/{name}`
|
||||
let s = Fs.read_text(path)
|
||||
if s == null { print(`r3d: missing shader {path}`); return "" }
|
||||
return s
|
||||
}
|
||||
|
||||
function r3d_shader_src(name: string, defines: string, is_frag: bool) -> string {
|
||||
if r3d_noise_src == null { r3d_noise_src = r3d_shader_file("noise.glsl") }
|
||||
if r3d_lighting_src == null { r3d_lighting_src = r3d_shader_file("lighting.glsl") }
|
||||
function r3d_shader_src(render3d_st: mut Render3dState, name: string, defines: string, is_frag: bool) -> string {
|
||||
if render3d_st.r3d_noise_src == null { render3d_st.r3d_noise_src = r3d_shader_file(render3d_st, "noise.glsl") }
|
||||
if render3d_st.r3d_lighting_src == null { render3d_st.r3d_lighting_src = r3d_shader_file(render3d_st, "lighting.glsl") }
|
||||
# the wind goes to BOTH stages: the grass and the crowns move in the vertex shader, the water
|
||||
# ripples in the fragment one, and they have to be reading the same field or the meadow and
|
||||
# the lake disagree about the weather
|
||||
if r3d_wind_src == null { r3d_wind_src = r3d_shader_file("wind.glsl") }
|
||||
var s = "#version 410 core\n" + r3d_global_defs + defines + r3d_wind_src
|
||||
if is_frag { s = s + r3d_noise_src + r3d_lighting_src }
|
||||
return s + r3d_shader_file(name)
|
||||
if render3d_st.r3d_wind_src == null { render3d_st.r3d_wind_src = r3d_shader_file(render3d_st, "wind.glsl") }
|
||||
var s = "#version 410 core\n" + render3d_st.r3d_global_defs + defines + render3d_st.r3d_wind_src
|
||||
if is_frag { s = s + render3d_st.r3d_noise_src + render3d_st.r3d_lighting_src }
|
||||
return s + r3d_shader_file(render3d_st, name)
|
||||
}
|
||||
|
||||
# Build a program from a vertex + fragment file pair (defines apply to both).
|
||||
# R3D_PROGRAMS_LOG=<file>: append every program built (vertex, fragment, defines) - the list a
|
||||
# backend that cannot compile shaders at run time (Vulkan: SPIR-V is built ahead) must cover
|
||||
var r3d_prog_log: int = -1
|
||||
function r3d_program_log(vs: string, fs: string, defines: string) -> void {
|
||||
if r3d_prog_log < 0 { r3d_prog_log = 0; if r3d_env_has("R3D_PROGRAMS_LOG") { r3d_prog_log = 1 } }
|
||||
if r3d_prog_log == 0 { return }
|
||||
let f = file_open(r3d_env("R3D_PROGRAMS_LOG"), "ab")
|
||||
function r3d_program_log(render3d_st: mut Render3dState, vs: string, fs: string, defines: string) -> void {
|
||||
if render3d_st.r3d_prog_log < 0 { render3d_st.r3d_prog_log = 0; if r3d_env_has(render3d_st, "R3D_PROGRAMS_LOG") { render3d_st.r3d_prog_log = 1 } }
|
||||
if render3d_st.r3d_prog_log == 0 { return }
|
||||
let f = file_open(r3d_env(render3d_st, "R3D_PROGRAMS_LOG"), "ab")
|
||||
if f == null { return }
|
||||
# one line per program: the defines' newlines become ';' so a variant is one line
|
||||
let defs = Text.replace(`{r3d_global_defs}{defines}`, "\n", ";")
|
||||
let defs = Text.replace(`{render3d_st.r3d_global_defs}{defines}`, "\n", ";")
|
||||
let line = `{vs}|{fs}|{defs}\n`
|
||||
file_write(f, line, len(line))
|
||||
file_close(f)
|
||||
}
|
||||
function r3d_program(vs: string, fs: string, defines: string) -> int {
|
||||
r3d_program_log(vs, fs, defines)
|
||||
let p = gpu_program(r3d_shader_src(vs, defines, false), r3d_shader_src(fs, defines, true), vs, fs, `{r3d_global_defs}{defines}`)
|
||||
function r3d_program(render3d_st: mut Render3dState, vs: string, fs: string, defines: string) -> int {
|
||||
r3d_program_log(render3d_st, vs, fs, defines)
|
||||
let p = gpu_program(render3d_st, r3d_shader_src(render3d_st, vs, defines, false), r3d_shader_src(render3d_st, fs, defines, true), vs, fs, `{render3d_st.r3d_global_defs}{defines}`)
|
||||
if p == 0 { print(`r3d: program failed: {vs} + {fs}`) }
|
||||
return p
|
||||
}
|
||||
|
||||
# Bind a texture to a unit and point a sampler uniform at it.
|
||||
function r3d_bind_tex(prog: int, name: string, unit: int, kind: int, tex: int) -> void { gpu_bind_sampler(prog, name, unit, kind, tex) }
|
||||
function r3d_bind_2d(prog: int, name: string, unit: int, tex: int) -> void { gpu_bind_sampler(prog, name, unit, GPU_TEX2D, tex) }
|
||||
function r3d_bind_tex(render3d_st: mut Render3dState, prog: int, name: string, unit: int, kind: int, tex: int) -> void { gpu_bind_sampler(render3d_st, prog, name, unit, kind, tex) }
|
||||
function r3d_bind_2d(render3d_st: mut Render3dState, prog: int, name: string, unit: int, tex: int) -> void { gpu_bind_sampler(render3d_st, prog, name, unit, GPU_TEX2D, tex) }
|
||||
|
||||
# A framebuffer with one colour texture (and optionally a depth texture).
|
||||
property Target {
|
||||
|
|
@ -95,29 +87,29 @@ property Target {
|
|||
w: int = 0,
|
||||
h: int = 0
|
||||
}
|
||||
function target_new(w: int, h: int, ifmt: int, fmt: int, ty: int, with_depth: bool, filter: int) -> Target {
|
||||
function target_new(render3d_st: mut Render3dState, w: int, h: int, ifmt: int, fmt: int, ty: int, with_depth: bool, filter: int) -> Target {
|
||||
let t = new Target
|
||||
t.w = w; t.h = h
|
||||
t.fbo = gpu_fb_new()
|
||||
gpu_fb_bind(t.fbo)
|
||||
t.color = tex_target(w, h, ifmt, fmt, ty, filter)
|
||||
gpu_fb_color(0, t.color)
|
||||
t.fbo = gpu_fb_new(render3d_st)
|
||||
gpu_fb_bind(render3d_st, t.fbo)
|
||||
t.color = tex_target(render3d_st, w, h, ifmt, fmt, ty, filter)
|
||||
gpu_fb_color(render3d_st, 0, t.color)
|
||||
if with_depth {
|
||||
t.depth = tex_target(w, h, GL_DEPTH_COMPONENT32F, GL_DEPTH_COMPONENT, GL_FLOAT, GL_NEAREST)
|
||||
gpu_fb_depth(t.depth)
|
||||
t.depth = tex_target(render3d_st, w, h, GL_DEPTH_COMPONENT32F, GL_DEPTH_COMPONENT, GL_FLOAT, GL_NEAREST)
|
||||
gpu_fb_depth(render3d_st, t.depth)
|
||||
}
|
||||
let st = gpu_fb_status()
|
||||
let st = gpu_fb_status(render3d_st)
|
||||
if st != GL_FRAMEBUFFER_COMPLETE { print(`r3d: framebuffer incomplete {st} ({w}x{h})`) }
|
||||
gpu_fb_bind(0)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
return t
|
||||
}
|
||||
function target_free(t: Target) -> void {
|
||||
function target_free(render3d_st: mut Render3dState, t: Target) -> void {
|
||||
if t == null { return }
|
||||
gpu_fb_free(t.fbo)
|
||||
gpu_tex_free(t.color)
|
||||
if t.depth != 0 { gpu_tex_free(t.depth) }
|
||||
gpu_fb_free(render3d_st, t.fbo)
|
||||
gpu_tex_free(render3d_st, t.color)
|
||||
if t.depth != 0 { gpu_tex_free(render3d_st, t.depth) }
|
||||
}
|
||||
function target_bind(t: Target) -> void {
|
||||
gpu_fb_bind(t.fbo)
|
||||
gpu_viewport(0, 0, t.w, t.h)
|
||||
function target_bind(render3d_st: mut Render3dState, t: Target) -> void {
|
||||
gpu_fb_bind(render3d_st, t.fbo)
|
||||
gpu_viewport(render3d_st, 0, 0, t.w, t.h)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -57,18 +57,14 @@ function q_rotate(o: floats, q: floats, v: floats) -> void {
|
|||
o[0] = x; o[1] = y; o[2] = z
|
||||
}
|
||||
# pitch about X, yaw about Y, roll about Z, composed as yaw * pitch * roll
|
||||
var q_scratch: floats = null
|
||||
var q_sy: floats = null # views of q_scratch's second, third and fourth
|
||||
var q_sz: floats = null
|
||||
var q_st: floats = null
|
||||
function q_euler(o: floats, pitch: float, yaw: float, roll: float) -> void {
|
||||
if q_scratch == null {
|
||||
q_scratch = floats(16)
|
||||
q_sy = view(q_scratch, 4, 4)
|
||||
q_sz = view(q_scratch, 8, 4)
|
||||
q_st = view(q_scratch, 12, 4)
|
||||
function q_euler(render3d_st: mut Render3dState, o: floats, pitch: float, yaw: float, roll: float) -> void {
|
||||
if render3d_st.q_scratch == null {
|
||||
render3d_st.q_scratch = floats(16)
|
||||
render3d_st.q_sy = view(render3d_st.q_scratch, 4, 4)
|
||||
render3d_st.q_sz = view(render3d_st.q_scratch, 8, 4)
|
||||
render3d_st.q_st = view(render3d_st.q_scratch, 12, 4)
|
||||
}
|
||||
let qx = q_scratch; let qy = q_sy; let qz = q_sz; let t = q_st
|
||||
let qx = render3d_st.q_scratch; let qy = render3d_st.q_sy; let qz = render3d_st.q_sz; let t = render3d_st.q_st
|
||||
q_axis_angle(qx, 1.0, 0.0, 0.0, pitch)
|
||||
q_axis_angle(qy, 0.0, 1.0, 0.0, yaw)
|
||||
q_axis_angle(qz, 0.0, 0.0, 1.0, roll)
|
||||
|
|
|
|||
|
|
@ -5,120 +5,86 @@
|
|||
# scene_draw_casters, which the demo defines.
|
||||
# ============================================================================
|
||||
|
||||
var r3d_sky_prog: int = 0
|
||||
var r3d_fog_density: float = 0.0
|
||||
var r3d_fog_scale: float = 1.0 # float bits: a setting's multiplier over the density the day sets
|
||||
var r3d_fog_falloff: float = 0.0
|
||||
var r3d_fog_base: float = 0.0 # the height fog is measured from (float bits); 0 = y = 0
|
||||
# Both of these are the DAY's, set by daylight_set from the sun's own elevation. The inscatter
|
||||
# is how hard the air scatters the sun forward - nearly nothing at noon, and the glow a ridge
|
||||
# is silhouetted against at dusk. The desaturation is how fast distance takes a surface's own
|
||||
# colour away, which is the term that was missing entirely and is most of why a midday frame
|
||||
# had no depth in it at all.
|
||||
var r3d_fog_inscatter: float = 0.02 # 0.02, what the shader used to hard-code
|
||||
var r3d_fog_desat: float = 1.35 # 1.35
|
||||
# How much of the visible sky's colour comes from the analytic model rather than from the
|
||||
# photograph (sky.frag). The day fades it in: at night the sky has its own tint, a star field
|
||||
# and a moon, and relighting on top of those would only wash them out.
|
||||
var r3d_sky_relight: float = 0.0
|
||||
# What the ground reflects back up, low and high, and the height they cross at. The defaults
|
||||
# are Maroon Lake's - a green basin under a grey-rock treeline at about 480 m over the datum.
|
||||
var r3d_ground_alb_r: float = 0.3; var r3d_ground_alb_g: float = 0.34; var r3d_ground_alb_b: float = 0.14
|
||||
var r3d_ground_hi_r: float = 0.289; var r3d_ground_hi_g: float = 0.28; var r3d_ground_hi_b: float = 0.26
|
||||
var r3d_ground_hi_y: float = 480.0 # 480
|
||||
var r3d_ground_hi_w: float = 150.0 # 150
|
||||
var r3d_time: float = 0.0
|
||||
var r3d_ready: bool = false
|
||||
# set before r3d_init to build the landscape from a real height map
|
||||
var r3d_dem_path: string = null
|
||||
var r3d_dem_min: float = 0.0
|
||||
var r3d_dem_max: float = 0.0
|
||||
var r3d_dem_base: float = 0.0
|
||||
var r3d_dem_ox: float = 0.0
|
||||
var r3d_dem_oz: float = 0.0
|
||||
var r3d_ortho_path: string = null
|
||||
# A plate: no terrain, no grass, no water - the sky, the sun, the shadows and whatever the scene
|
||||
# draws (a lab's stage, ludic.lab). Set before the load; nothing of the landscape is made, so
|
||||
# its hundreds of megabytes of height field, materials and grass never exist.
|
||||
var r3d_plate: bool = false
|
||||
function r3d_plate_mode(on: bool) -> void { r3d_plate = on }
|
||||
function r3d_plate_mode(render3d_st: mut Render3dState, on: bool) -> void { render3d_st.r3d_plate = on }
|
||||
# the sky's HDR image, when it is not the default photograph under r3d_assets
|
||||
var r3d_sky_path: string = null
|
||||
var r3d_debug: bool = false
|
||||
var r3d_debug_shadow: bool = false
|
||||
var r3d_debug_max: bool = false
|
||||
|
||||
var r3d_cloud_shadow: float = 0.5 # 0.5
|
||||
|
||||
var r3d_clip_y: float = -2147483600.0 # -2^31: no clipping
|
||||
# R3D_NOPREPASS=1: light the foliage the old way, every card behind the front one included
|
||||
var r3d_prepass_env: int = -1
|
||||
function r3d_prepass_off() -> bool {
|
||||
if r3d_prepass_env < 0 { r3d_prepass_env = 0; if r3d_env_has("R3D_NOPREPASS") { r3d_prepass_env = 1 } }
|
||||
return r3d_prepass_env == 1
|
||||
function r3d_prepass_off(render3d_st: mut Render3dState) -> bool {
|
||||
if render3d_st.r3d_prepass_env < 0 { render3d_st.r3d_prepass_env = 0; if r3d_env_has(render3d_st, "R3D_NOPREPASS") { render3d_st.r3d_prepass_env = 1 } }
|
||||
return render3d_st.r3d_prepass_env == 1
|
||||
}
|
||||
function fog_bind(prog: int) -> void {
|
||||
u_f(gpu_uniform(prog, "u_clip_y"), r3d_clip_y)
|
||||
u_f(gpu_uniform(prog, "u_spec_scale"), 1.0)
|
||||
u_f(gpu_uniform(prog, "u_fog_density"), r3d_fog_density)
|
||||
u_f(gpu_uniform(prog, "u_fog_height_falloff"), r3d_fog_falloff)
|
||||
u_f(gpu_uniform(prog, "u_fog_base"), r3d_fog_base)
|
||||
u_f(gpu_uniform(prog, "u_fog_inscatter"), r3d_fog_inscatter)
|
||||
u_f(gpu_uniform(prog, "u_fog_desat"), r3d_fog_desat)
|
||||
u_f3(gpu_uniform(prog, "u_ground_alb"), r3d_ground_alb_r, r3d_ground_alb_g, r3d_ground_alb_b)
|
||||
u_f3(gpu_uniform(prog, "u_ground_alb_hi"), r3d_ground_hi_r, r3d_ground_hi_g, r3d_ground_hi_b)
|
||||
u_f(gpu_uniform(prog, "u_ground_hi_y"), r3d_ground_hi_y)
|
||||
u_f(gpu_uniform(prog, "u_ground_hi_w"), r3d_ground_hi_w)
|
||||
var cs = r3d_cloud_shadow
|
||||
if r3d_env_has("R3D_NOCLOUD") { cs = 0.0 }
|
||||
u_f(gpu_uniform(prog, "u_cloud_shadow"), cs)
|
||||
u_f(gpu_uniform(prog, "u_time"), r3d_time)
|
||||
function fog_bind(render3d_st: mut Render3dState, prog: int) -> void {
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_clip_y"), render3d_st.r3d_clip_y)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_spec_scale"), 1.0)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_fog_density"), render3d_st.r3d_fog_density)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_fog_height_falloff"), render3d_st.r3d_fog_falloff)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_fog_base"), render3d_st.r3d_fog_base)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_fog_inscatter"), render3d_st.r3d_fog_inscatter)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_fog_desat"), render3d_st.r3d_fog_desat)
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, prog, "u_ground_alb"), render3d_st.r3d_ground_alb_r, render3d_st.r3d_ground_alb_g, render3d_st.r3d_ground_alb_b)
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, prog, "u_ground_alb_hi"), render3d_st.r3d_ground_hi_r, render3d_st.r3d_ground_hi_g, render3d_st.r3d_ground_hi_b)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_ground_hi_y"), render3d_st.r3d_ground_hi_y)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_ground_hi_w"), render3d_st.r3d_ground_hi_w)
|
||||
var cs = render3d_st.r3d_cloud_shadow
|
||||
if r3d_env_has(render3d_st, "R3D_NOCLOUD") { cs = 0.0 }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_cloud_shadow"), cs)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_time"), render3d_st.r3d_time)
|
||||
}
|
||||
|
||||
# profiling switches (environment): R3D_NOSHADOW R3D_NOGI R3D_MSAA=n R3D_NOBLADES R3D_NOCARDS R3D_NOTREES R3D_NEAR=m
|
||||
var r3d_test_frame: int = 0
|
||||
var r3d_test_resize: int = 0
|
||||
var r3d_no_shadow: bool = false
|
||||
var r3d_no_trees: bool = false
|
||||
var r3d_no_refl: bool = false
|
||||
function r3d_env_flags() -> void {
|
||||
r3d_no_shadow = r3d_env_has("R3D_NOSHADOW")
|
||||
r3d_no_trees = r3d_env_has("R3D_NOTREES")
|
||||
r3d_no_refl = r3d_env_has("R3D_NOREFL")
|
||||
if r3d_env_has("R3D_DEBUG") { r3d_debug = true }
|
||||
if r3d_env_has("R3D_DBGSHADOW") { r3d_debug_shadow = true }
|
||||
if r3d_env_has("R3D_NOGI") { post_gi_strength = 0.0; post_ao_strength = 0.0; post_no_gi = true }
|
||||
if r3d_env_has("R3D_MSAA") { post_ms_samples = Text.to_int(r3d_env("R3D_MSAA")) }
|
||||
sc_skip_blade = r3d_env_has("R3D_NOBLADES")
|
||||
sc_skip_card = r3d_env_has("R3D_NOCARDS")
|
||||
sc_dbg_lod = r3d_env_has("R3D_LODDBG")
|
||||
if r3d_env_has("R3D_ANISO") {
|
||||
let a = Text.to_int(r3d_env("R3D_ANISO"))
|
||||
tex_anisotropy = 1.0
|
||||
if a >= 2 { tex_anisotropy = 2.0 }
|
||||
if a >= 4 { tex_anisotropy = 4.0 }
|
||||
if a >= 8 { tex_anisotropy = 8.0 }
|
||||
if a >= 16 { tex_anisotropy = 16.0 }
|
||||
function r3d_env_flags(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.r3d_no_shadow = r3d_env_has(render3d_st, "R3D_NOSHADOW")
|
||||
render3d_st.r3d_no_trees = r3d_env_has(render3d_st, "R3D_NOTREES")
|
||||
render3d_st.r3d_no_refl = r3d_env_has(render3d_st, "R3D_NOREFL")
|
||||
if r3d_env_has(render3d_st, "R3D_DEBUG") { render3d_st.r3d_debug = true }
|
||||
if r3d_env_has(render3d_st, "R3D_DBGSHADOW") { render3d_st.r3d_debug_shadow = true }
|
||||
if r3d_env_has(render3d_st, "R3D_NOGI") { render3d_st.post_gi_strength = 0.0; render3d_st.post_ao_strength = 0.0; render3d_st.post_no_gi = true }
|
||||
if r3d_env_has(render3d_st, "R3D_MSAA") { render3d_st.post_ms_samples = Text.to_int(r3d_env(render3d_st, "R3D_MSAA")) }
|
||||
render3d_st.sc_skip_blade = r3d_env_has(render3d_st, "R3D_NOBLADES")
|
||||
render3d_st.sc_skip_card = r3d_env_has(render3d_st, "R3D_NOCARDS")
|
||||
render3d_st.sc_dbg_lod = r3d_env_has(render3d_st, "R3D_LODDBG")
|
||||
if r3d_env_has(render3d_st, "R3D_ANISO") {
|
||||
let a = Text.to_int(r3d_env(render3d_st, "R3D_ANISO"))
|
||||
render3d_st.tex_anisotropy = 1.0
|
||||
if a >= 2 { render3d_st.tex_anisotropy = 2.0 }
|
||||
if a >= 4 { render3d_st.tex_anisotropy = 4.0 }
|
||||
if a >= 8 { render3d_st.tex_anisotropy = 8.0 }
|
||||
if a >= 16 { render3d_st.tex_anisotropy = 16.0 }
|
||||
}
|
||||
}
|
||||
# Start the renderer in one call: the window, then every load step in order. A game that shows
|
||||
# a loading screen calls r3d_open, draws its screen, and runs r3d_load_step itself between frames.
|
||||
function r3d_init(w: int, h: int, title: string) -> bool {
|
||||
if not r3d_open(w, h, title) { return false }
|
||||
for i in 0 .. r3d_load_count() { if not r3d_load_step(i) { return false } }
|
||||
function r3d_init(render3d_st: mut Render3dState, w: int, h: int, title: string) -> bool {
|
||||
if not r3d_open(render3d_st, w, h, title) { return false }
|
||||
for i in 0 .. r3d_load_count() { if not r3d_load_step(render3d_st, i) { return false } }
|
||||
return true
|
||||
}
|
||||
|
||||
# the window and the graphics backend; nothing is baked yet, but a frame can be presented
|
||||
function r3d_open(w: int, h: int, title: string) -> bool {
|
||||
r3d_env_flags()
|
||||
gpu_select()
|
||||
if not gpu_open(w, h, title) { print("r3d: no OpenGL context"); return false }
|
||||
if r3d_env_has("R3D_NOVSYNC") { gpu_vsync(0) }
|
||||
var renderer: string = gpu_renderer_name()
|
||||
function r3d_open(render3d_st: mut Render3dState, w: int, h: int, title: string) -> bool {
|
||||
r3d_env_flags(render3d_st)
|
||||
gpu_select(render3d_st)
|
||||
if not gpu_open(render3d_st, w, h, title) { print("r3d: no OpenGL context"); return false }
|
||||
if r3d_env_has(render3d_st, "R3D_NOVSYNC") { gpu_vsync(render3d_st, 0) }
|
||||
var renderer: string = gpu_renderer_name(render3d_st)
|
||||
print(`r3d: {gl_w}x{gl_h} on {renderer}`)
|
||||
prof_init()
|
||||
cam_init(float(gl_w) / float(gl_h))
|
||||
prof_init(render3d_st)
|
||||
cam_init(render3d_st, float(gl_w) / float(gl_h))
|
||||
return true
|
||||
}
|
||||
|
||||
|
|
@ -127,192 +93,191 @@ function r3d_open(w: int, h: int, title: string) -> bool {
|
|||
# programs), 5 the shadow maps and the screen targets, 6 the scattered cover and the actors, 7 the
|
||||
# grass and the sky pass. The order is r3d_init's.
|
||||
function r3d_load_count() -> int { return 4 + TERRAIN_INIT_STEPS }
|
||||
function r3d_load_step(i: int) -> bool {
|
||||
function r3d_load_step(render3d_st: mut Render3dState, i: int) -> bool {
|
||||
let t = i - 1
|
||||
if i == 0 {
|
||||
var sky = r3d_assets + "/hdri/kloofendal_48d_partly_cloudy_puresky_4k.hdr"
|
||||
if r3d_sky_path != null { sky = r3d_sky_path }
|
||||
if not sky_load(sky) { return false }
|
||||
daylight_init()
|
||||
var sky = render3d_st.r3d_assets + "/hdri/kloofendal_48d_partly_cloudy_puresky_4k.hdr"
|
||||
if render3d_st.r3d_sky_path != null { sky = render3d_st.r3d_sky_path }
|
||||
if not sky_load(render3d_st, sky) { return false }
|
||||
daylight_init(render3d_st)
|
||||
} else if t >= 0 and t < TERRAIN_INIT_STEPS {
|
||||
if r3d_plate { return true }
|
||||
if render3d_st.r3d_plate { return true }
|
||||
if t == 0 {
|
||||
if r3d_dem_path != null { terrain_use_dem(r3d_dem_path, r3d_dem_min, r3d_dem_max, r3d_dem_base, r3d_dem_ox, r3d_dem_oz) }
|
||||
if r3d_ortho_path != null { terrain_use_ortho(r3d_ortho_path) }
|
||||
if render3d_st.r3d_dem_path != null { terrain_use_dem(render3d_st, render3d_st.r3d_dem_path, render3d_st.r3d_dem_min, render3d_st.r3d_dem_max, render3d_st.r3d_dem_base, render3d_st.r3d_dem_ox, render3d_st.r3d_dem_oz) }
|
||||
if render3d_st.r3d_ortho_path != null { terrain_use_ortho(render3d_st, render3d_st.r3d_ortho_path) }
|
||||
}
|
||||
terrain_init_step(t)
|
||||
terrain_init_step(render3d_st, t)
|
||||
} else if i == 1 + TERRAIN_INIT_STEPS {
|
||||
shadow_init()
|
||||
post_init(gl_w, gl_h)
|
||||
shadow_init(render3d_st)
|
||||
post_init(render3d_st, gl_w, gl_h)
|
||||
} else if i == 2 + TERRAIN_INIT_STEPS {
|
||||
scatter_init()
|
||||
actor_init()
|
||||
scatter_init(render3d_st)
|
||||
actor_init(render3d_st)
|
||||
} else if i == 3 + TERRAIN_INIT_STEPS {
|
||||
if not r3d_plate { grass_init() }
|
||||
r3d_sky_prog = r3d_program("fullscreen.vert", "sky.frag", "")
|
||||
if not render3d_st.r3d_plate { grass_init(render3d_st) }
|
||||
render3d_st.r3d_sky_prog = r3d_program(render3d_st, "fullscreen.vert", "sky.frag", "")
|
||||
# The base is NOON's air, and it was tuned when nothing desaturated with distance - so it
|
||||
# was doing its whole job through colour and could not be raised without the valley going
|
||||
# blue. With the desaturation term carrying the depth, the air can be as thick as real air
|
||||
# at 2900 m and the far ridge recedes instead of tinting.
|
||||
r3d_fog_density = 0.00024
|
||||
r3d_fog_falloff = 0.002
|
||||
r3d_ready = true
|
||||
gpu_check("r3d init")
|
||||
render3d_st.r3d_fog_density = 0.00024
|
||||
render3d_st.r3d_fog_falloff = 0.002
|
||||
render3d_st.r3d_ready = true
|
||||
gpu_check(render3d_st, "r3d init")
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
function r3d_draw_sky() -> void {
|
||||
gpu_depth_func(GL_LEQUAL)
|
||||
gpu_depth_write(false)
|
||||
gpu_cull(false)
|
||||
let p = r3d_sky_prog
|
||||
gpu_use_program(p)
|
||||
r3d_bind_2d(p, "u_sky", 0, sky_tex)
|
||||
sky_bind_rot(p)
|
||||
sky_bind_lighting(p)
|
||||
u_mat4(gpu_uniform(p, "u_inv_vp"), cam_inv_vp)
|
||||
u_f(gpu_uniform(p, "u_sky_gain"), 0.95)
|
||||
u_f(gpu_uniform(p, "u_sky_relight"), r3d_sky_relight)
|
||||
u_f(gpu_uniform(p, "u_sky_sat"), 1.35)
|
||||
u_f(gpu_uniform(p, "u_time"), r3d_time)
|
||||
mesh_draw(sky_fullscreen)
|
||||
gpu_depth_write(true)
|
||||
gpu_depth_func(GL_LESS)
|
||||
function r3d_draw_sky(render3d_st: mut Render3dState) -> void {
|
||||
gpu_depth_func(render3d_st, GL_LEQUAL)
|
||||
gpu_depth_write(render3d_st, false)
|
||||
gpu_cull(render3d_st, false)
|
||||
let p = render3d_st.r3d_sky_prog
|
||||
gpu_use_program(render3d_st, p)
|
||||
r3d_bind_2d(render3d_st, p, "u_sky", 0, render3d_st.sky_tex)
|
||||
sky_bind_rot(render3d_st, p)
|
||||
sky_bind_lighting(render3d_st, p)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_inv_vp"), render3d_st.cam_inv_vp)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_sky_gain"), 0.95)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_sky_relight"), render3d_st.r3d_sky_relight)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_sky_sat"), 1.35)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_time"), render3d_st.r3d_time)
|
||||
mesh_draw(render3d_st, render3d_st.sky_fullscreen)
|
||||
gpu_depth_write(render3d_st, true)
|
||||
gpu_depth_func(render3d_st, GL_LESS)
|
||||
}
|
||||
|
||||
# The end of the game's frame, whichever backend draws it: a screenshot first if one is wanted,
|
||||
# then the present. A game calls these rather than Gl.swap / Gl.screenshot.
|
||||
function r3d_screenshot(path: string) -> bool { return gpu_screenshot(path) }
|
||||
function r3d_present() -> void { gpu_present() }
|
||||
function r3d_screenshot(render3d_st: mut Render3dState, path: string) -> bool { return gpu_screenshot(render3d_st, path) }
|
||||
function r3d_present(render3d_st: mut Render3dState) -> void { gpu_present(render3d_st) }
|
||||
|
||||
# the drawable changed size: the camera's aspect and every screen-sized target follow
|
||||
function r3d_resize() -> void {
|
||||
cam_aspect = float(gl_w) / float(gl_h)
|
||||
cam_update()
|
||||
post_free()
|
||||
post_init(gl_w, gl_h)
|
||||
if water_refl != null { target_free(water_refl); water_refl = null }
|
||||
function r3d_resize(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.cam_aspect = float(gl_w) / float(gl_h)
|
||||
cam_update(render3d_st)
|
||||
post_free(render3d_st)
|
||||
post_init(render3d_st, gl_w, gl_h)
|
||||
if render3d_st.water_refl != null { target_free(render3d_st, render3d_st.water_refl); render3d_st.water_refl = null }
|
||||
print(`r3d: resized to {gl_w}x{gl_h}`)
|
||||
}
|
||||
var r3d_cam_log: int = -1
|
||||
function r3d_frame(time: float) -> void {
|
||||
gpu_glcheck_after("the time between frames")
|
||||
outline_frame()
|
||||
if not r3d_ready { return }
|
||||
if gpu_resize_check() { r3d_resize() }
|
||||
function r3d_frame(render3d_st: mut Render3dState, time: float) -> void {
|
||||
gpu_glcheck_after(render3d_st, "the time between frames")
|
||||
outline_frame(render3d_st)
|
||||
if not render3d_st.r3d_ready { return }
|
||||
if gpu_resize_check(render3d_st) { r3d_resize(render3d_st) }
|
||||
# R3D_RESIZE_AT=<frame>: rebuild every screen-sized buffer mid-run, as a window resize
|
||||
# or a fullscreen change does. Headless has no window to resize, and this path is where
|
||||
# a stale attachment or a texture freed twice shows up.
|
||||
r3d_test_frame += 1
|
||||
if r3d_test_resize == 0 and r3d_env_has("R3D_RESIZE_AT") { r3d_test_resize = Text.to_int(r3d_env("R3D_RESIZE_AT")) }
|
||||
render3d_st.r3d_test_frame += 1
|
||||
if render3d_st.r3d_test_resize == 0 and r3d_env_has(render3d_st, "R3D_RESIZE_AT") { render3d_st.r3d_test_resize = Text.to_int(r3d_env(render3d_st, "R3D_RESIZE_AT")) }
|
||||
# R3D_RESIZE_AT=<n>: from frame n on, rebuild every screen-sized buffer every few
|
||||
# frames at a different size, as dragging a window edge or entering fullscreen does.
|
||||
if r3d_test_resize > 0 and r3d_test_frame >= r3d_test_resize and r3d_test_frame % 4 == 0 {
|
||||
let step = (r3d_test_frame / 4) % 4
|
||||
if render3d_st.r3d_test_resize > 0 and render3d_st.r3d_test_frame >= render3d_st.r3d_test_resize and render3d_st.r3d_test_frame % 4 == 0 {
|
||||
let step = (render3d_st.r3d_test_frame / 4) % 4
|
||||
var w = 1920; var h = 1080
|
||||
if step == 1 { w = 1440; h = 810 }
|
||||
if step == 2 { w = 2560; h = 1440 }
|
||||
if step == 3 { w = 1281; h = 721 }
|
||||
gl_w = w; gl_h = h
|
||||
r3d_resize()
|
||||
r3d_resize(render3d_st)
|
||||
}
|
||||
r3d_time = time
|
||||
gsl_frame_start()
|
||||
render3d_st.r3d_time = time
|
||||
gsl_frame_start(render3d_st)
|
||||
# the frame's counters close here, before any of its own work: the window each of them
|
||||
# covers is exactly one frame, from this point to the same point next time
|
||||
prof_gen_frame()
|
||||
prof_mark_start()
|
||||
cam_begin_frame(post_frame, gl_w, gl_h)
|
||||
prof_gen_frame(render3d_st)
|
||||
prof_mark_start(render3d_st)
|
||||
cam_begin_frame(render3d_st, render3d_st.post_frame, gl_w, gl_h)
|
||||
# R3D_CAM_LOG=1: the camera each frame, in millimetres and thousandths of a radian - to tell a
|
||||
# camera that moves while the hiker stands still from a picture that shakes on its own
|
||||
if r3d_cam_log < 0 { r3d_cam_log = 0; if r3d_env_has("R3D_CAM_LOG") { r3d_cam_log = 1 } }
|
||||
if r3d_cam_log == 1 {
|
||||
print(`cam {r3d_test_frame} pos {int(cam_pos[0] * 1000.0)} {int(cam_pos[1] * 1000.0)} {int(cam_pos[2] * 1000.0)} yaw {int(cam_yaw * 1000.0)} pitch {int(cam_pitch * 1000.0)} jitter {int(gsl_jitter_x * 1000000.0)} {int(gsl_jitter_y * 1000000.0)} render {post_w}x{post_h} reset {gsl_reset} evalok {gsl_eval_ok} fresh {gsl_fresh}`)
|
||||
if render3d_st.r3d_cam_log < 0 { render3d_st.r3d_cam_log = 0; if r3d_env_has(render3d_st, "R3D_CAM_LOG") { render3d_st.r3d_cam_log = 1 } }
|
||||
if render3d_st.r3d_cam_log == 1 {
|
||||
print(`cam {render3d_st.r3d_test_frame} pos {int(render3d_st.cam_pos[0] * 1000.0)} {int(render3d_st.cam_pos[1] * 1000.0)} {int(render3d_st.cam_pos[2] * 1000.0)} yaw {int(render3d_st.cam_yaw * 1000.0)} pitch {int(render3d_st.cam_pitch * 1000.0)} jitter {int(render3d_st.gsl_jitter_x * 1000000.0)} {int(render3d_st.gsl_jitter_y * 1000000.0)} render {render3d_st.post_w}x{render3d_st.post_h} reset {render3d_st.gsl_reset} evalok {render3d_st.gsl_eval_ok} fresh {render3d_st.gsl_fresh}`)
|
||||
}
|
||||
# the height-field shadow rebakes as the light moves in steps (daylight), or with the sky yaw when there is no clock
|
||||
if not r3d_plate and ((not day_on and ter_shadow_yaw != sky_yaw) or ter_shadow_gen != day_gen) {
|
||||
ter_shadow_gen = day_gen
|
||||
if not render3d_st.r3d_plate and ((not render3d_st.day_on and render3d_st.ter_shadow_yaw != render3d_st.sky_yaw) or render3d_st.ter_shadow_gen != render3d_st.day_gen) {
|
||||
render3d_st.ter_shadow_gen = render3d_st.day_gen
|
||||
let t_bk = gl_now_us()
|
||||
terrain_bake_shadow()
|
||||
prof_bake_add(gl_now_us() - t_bk)
|
||||
terrain_bake_shadow(render3d_st)
|
||||
prof_bake_add(render3d_st, gl_now_us() - t_bk)
|
||||
}
|
||||
scatter_begin_frame()
|
||||
prof_cpu_mark("shadow rebake")
|
||||
stream_update_all()
|
||||
prof_cpu_mark("streaming")
|
||||
if not r3d_no_shadow { prof_begin("shadow"); shadow_pass(); prof_end() }
|
||||
prof_cpu_mark("shadow pass")
|
||||
if water_on and not r3d_no_refl and water_reflect_visible() { prof_begin("water reflection"); water_reflection_pass(); prof_end() }
|
||||
prof_cpu_mark("reflection")
|
||||
post_begin_scene()
|
||||
scatter_begin_frame(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "shadow rebake")
|
||||
stream_update_all(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "streaming")
|
||||
if not render3d_st.r3d_no_shadow { prof_begin(render3d_st, "shadow"); shadow_pass(render3d_st); prof_end(render3d_st) }
|
||||
prof_cpu_mark(render3d_st, "shadow pass")
|
||||
if render3d_st.water_on and not render3d_st.r3d_no_refl and water_reflect_visible(render3d_st) { prof_begin(render3d_st, "water reflection"); water_reflection_pass(render3d_st); prof_end(render3d_st) }
|
||||
prof_cpu_mark(render3d_st, "reflection")
|
||||
post_begin_scene(render3d_st)
|
||||
# the near tree foliage lays its depth down before anything is shaded, so the terrain
|
||||
# under the stands and the cards behind the front ones are rejected before lighting.
|
||||
# It has to come straight after post_begin_scene: terrain_sun_prepare binds its own
|
||||
# target, and depth drawn after it lands in the sun buffer, not the scene's.
|
||||
if not r3d_prepass_off() {
|
||||
prof_begin("foliage prepass")
|
||||
gpu_color_write(false)
|
||||
scatter_draw_depth()
|
||||
gpu_color_write(true)
|
||||
prof_end()
|
||||
sc_prepass = true
|
||||
if not r3d_prepass_off(render3d_st) {
|
||||
prof_begin(render3d_st, "foliage prepass")
|
||||
gpu_color_write(render3d_st, false)
|
||||
scatter_draw_depth(render3d_st)
|
||||
gpu_color_write(render3d_st, true)
|
||||
prof_end(render3d_st)
|
||||
render3d_st.sc_prepass = true
|
||||
}
|
||||
prof_cpu_mark("foliage prepass")
|
||||
if not r3d_plate {
|
||||
prof_begin("terrain sun")
|
||||
terrain_sun_prepare()
|
||||
prof_end()
|
||||
prof_begin("terrain")
|
||||
terrain_draw()
|
||||
prof_end()
|
||||
prof_cpu_mark(render3d_st, "foliage prepass")
|
||||
if not render3d_st.r3d_plate {
|
||||
prof_begin(render3d_st, "terrain sun")
|
||||
terrain_sun_prepare(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
prof_begin(render3d_st, "terrain")
|
||||
terrain_draw(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
}
|
||||
prof_cpu_mark("terrain")
|
||||
prof_begin("scene (vegetation)")
|
||||
r3d_scene_draw()
|
||||
prof_end()
|
||||
sc_prepass = false
|
||||
prof_cpu_mark("vegetation")
|
||||
prof_begin("grass")
|
||||
grass_draw()
|
||||
prof_end()
|
||||
prof_cpu_mark("grass")
|
||||
prof_begin("sky")
|
||||
r3d_draw_sky()
|
||||
prof_end()
|
||||
prof_begin("resolve MSAA")
|
||||
post_resolve()
|
||||
prof_end()
|
||||
prof_cpu_mark(render3d_st, "terrain")
|
||||
prof_begin(render3d_st, "scene (vegetation)")
|
||||
r3d_scene_draw(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
render3d_st.sc_prepass = false
|
||||
prof_cpu_mark(render3d_st, "vegetation")
|
||||
prof_begin(render3d_st, "grass")
|
||||
grass_draw(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "grass")
|
||||
prof_begin(render3d_st, "sky")
|
||||
r3d_draw_sky(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
prof_begin(render3d_st, "resolve MSAA")
|
||||
post_resolve(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
# transparent water over the resolved frame: it tests against the frame's own
|
||||
# depth and reads a copy of it for the depth tint and soft shores
|
||||
if water_on {
|
||||
prof_begin("water surface")
|
||||
post_capture_scene()
|
||||
target_bind(post_hdr)
|
||||
gpu_depth_test(true)
|
||||
gpu_depth_func(GL_LESS)
|
||||
water_draw(post_depth_copy.depth)
|
||||
prof_end()
|
||||
if render3d_st.water_on {
|
||||
prof_begin(render3d_st, "water surface")
|
||||
post_capture_scene(render3d_st)
|
||||
target_bind(render3d_st, render3d_st.post_hdr)
|
||||
gpu_depth_test(render3d_st, true)
|
||||
gpu_depth_func(render3d_st, GL_LESS)
|
||||
water_draw(render3d_st, render3d_st.post_depth_copy.depth)
|
||||
prof_end(render3d_st)
|
||||
}
|
||||
post_color = post_hdr.color; post_color_w = post_w; post_color_h = post_h
|
||||
render3d_st.post_color = render3d_st.post_hdr.color; render3d_st.post_color_w = render3d_st.post_w; render3d_st.post_color_h = render3d_st.post_h
|
||||
# DLSS super resolution: the lit frame up to the display's size, before anything reads it
|
||||
if gsl_dlss_live() { prof_begin("DLSS"); post_color = gsl_dlss_eval(); prof_end() }
|
||||
if gsl_dlss_live(render3d_st) { prof_begin(render3d_st, "DLSS"); render3d_st.post_color = gsl_dlss_eval(render3d_st); prof_end(render3d_st) }
|
||||
# Before bloom and before the tonemap: a shaft of light is a bright thing in the air and
|
||||
# should bloom like one.
|
||||
prof_begin("volumetric"); post_volumetric_pass(); prof_end()
|
||||
prof_begin("dof"); post_dof_pass(); prof_end()
|
||||
if not post_no_gi { prof_begin("SSAO/GI"); post_ssao_pass(); prof_end() }
|
||||
if r3d_debug_max { tex_max(post_hdr.color, post_hdr.w, post_hdr.h, "hdr") }
|
||||
prof_begin("bloom")
|
||||
post_bloom_pass()
|
||||
prof_end()
|
||||
prof_begin("tonemap+exposure")
|
||||
post_tonemap(post_color)
|
||||
prof_end()
|
||||
prof_cpu_mark("post")
|
||||
prof_begin("prev-colour copy")
|
||||
post_capture_prev()
|
||||
prof_end()
|
||||
prof_collect()
|
||||
gpu_check("frame")
|
||||
prof_begin(render3d_st, "volumetric"); post_volumetric_pass(render3d_st); prof_end(render3d_st)
|
||||
prof_begin(render3d_st, "dof"); post_dof_pass(render3d_st); prof_end(render3d_st)
|
||||
if not render3d_st.post_no_gi { prof_begin(render3d_st, "SSAO/GI"); post_ssao_pass(render3d_st); prof_end(render3d_st) }
|
||||
if render3d_st.r3d_debug_max { tex_max(render3d_st, render3d_st.post_hdr.color, render3d_st.post_hdr.w, render3d_st.post_hdr.h, "hdr") }
|
||||
prof_begin(render3d_st, "bloom")
|
||||
post_bloom_pass(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
prof_begin(render3d_st, "tonemap+exposure")
|
||||
post_tonemap(render3d_st, render3d_st.post_color)
|
||||
prof_end(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "post")
|
||||
prof_begin(render3d_st, "prev-colour copy")
|
||||
post_capture_prev(render3d_st)
|
||||
prof_end(render3d_st)
|
||||
prof_collect(render3d_st)
|
||||
gpu_check(render3d_st, "frame")
|
||||
}
|
||||
|
|
|
|||
|
|
@ -4,31 +4,31 @@
|
|||
# at the boundary, so nothing raw crosses it.
|
||||
|
||||
# an accessor's float components (gltf_count elements of gltf_comps floats), as floats
|
||||
function gltf_accessor_floats(idx: int) -> floats {
|
||||
let p = gltf_accessor(idx)
|
||||
let n = gltf_count * gltf_comps
|
||||
function gltf_accessor_floats(render3d_st: mut Render3dState, idx: int) -> floats {
|
||||
let p = gltf_accessor(render3d_st, idx)
|
||||
let n = render3d_st.gltf_count * render3d_st.gltf_comps
|
||||
let out = floats(n)
|
||||
for i in 0 .. n { out[i] = float_from_bits(mem_get_f32_bits(p, i)) }
|
||||
free(p)
|
||||
return out
|
||||
}
|
||||
# how many elements an accessor holds, without reading them
|
||||
function gltf_accessor_count(idx: int) -> int {
|
||||
let acc = value_at(value_get(gltf_doc, "accessors"), idx)
|
||||
function gltf_accessor_count(render3d_st: Render3dState, idx: int) -> int {
|
||||
let acc = value_at(value_get(render3d_st.gltf_doc, "accessors"), idx)
|
||||
return jint(acc, "count", 0)
|
||||
}
|
||||
# the presented frame as RGB8, bottom row first, into `out` (at least w * h * 3 bytes)
|
||||
function gpu_read_screen_bytes(w: int, h: int, out: []byte) -> bool {
|
||||
function gpu_read_screen_bytes(render3d_st: mut Render3dState, w: int, h: int, out: []byte) -> bool {
|
||||
if out == null or len(out) < w * h * 3 { return false }
|
||||
gpu_read_screen(w, h, data_of(out))
|
||||
gpu_read_screen(render3d_st, w, h, data_of(out))
|
||||
return true
|
||||
}
|
||||
# a PNG's samples (tex_w x tex_h x tex_channels, 8 or 16 bits): into `reuse` when it is large
|
||||
# enough - a map swap decodes the same size again - else into a new buffer. null if unreadable.
|
||||
function png_decode_bytes(path: string, reuse: []byte) -> []byte {
|
||||
let p = png_decode(path)
|
||||
function png_decode_bytes(render3d_st: mut Render3dState, path: string, reuse: []byte) -> []byte {
|
||||
let p = png_decode(render3d_st, path)
|
||||
if p == null { return null }
|
||||
let n = tex_w * tex_h * tex_channels * (tex_depth / 8)
|
||||
let n = render3d_st.tex_w * render3d_st.tex_h * render3d_st.tex_channels * (render3d_st.tex_depth / 8)
|
||||
var out = reuse
|
||||
if out == null or len(out) < n { out = buffer(n) }
|
||||
for i in 0 .. n { out[i] = p[i] }
|
||||
|
|
@ -36,6 +36,6 @@ function png_decode_bytes(path: string, reuse: []byte) -> []byte {
|
|||
return out
|
||||
}
|
||||
# upload samples laid out as the last decode left them (tex_w, tex_h, tex_channels, tex_depth)
|
||||
function tex_upload_bytes(px: []byte, srgb: bool, mips: bool) -> int {
|
||||
return tex_upload(data_of(px), srgb, mips)
|
||||
function tex_upload_bytes(render3d_st: mut Render3dState, px: []byte, srgb: bool, mips: bool) -> int {
|
||||
return tex_upload(render3d_st, data_of(px), srgb, mips)
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -5,95 +5,80 @@
|
|||
# ============================================================================
|
||||
|
||||
# the size of each cascade's depth layer; shadow_set_res changes it at run time
|
||||
var shadow_res: int = 2048
|
||||
var shadow_refused: int = 0 # the last size the graphics card had no memory for (0: none)
|
||||
const SHADOW_CASCADES: int = 5
|
||||
|
||||
var sh_tex: int = 0
|
||||
var sh_fbo: int = 0
|
||||
var sh_vp: floats = null # the cascades' view-projections, 16 floats each
|
||||
var sh_vp_v: [][]float = null # a view of each
|
||||
var sh_split: floats = null # view-space far distance of each cascade
|
||||
var sh_range: floats = null # 4 light-frustum depth extents (metres)
|
||||
var sh_texel: floats = null # 4 shadow texel sizes (metres)
|
||||
var sh_tmp_proj: floats = null
|
||||
var sh_tmp_vp: floats = null
|
||||
var sh_tmp_inv: floats = null
|
||||
var sh_tmp_view: floats = null
|
||||
var sh_corner: floats = null
|
||||
var sh_cascade: int = 0 # the cascade being rendered (for casters that skip far ones)
|
||||
|
||||
function shadow_init() -> void {
|
||||
shadow_make_tex()
|
||||
sh_fbo = gpu_fb_new()
|
||||
gpu_fb_bind(sh_fbo)
|
||||
gpu_fb_no_color()
|
||||
gpu_fb_bind(0)
|
||||
sh_vp = floats(16 * SHADOW_CASCADES)
|
||||
sh_vp_v = m4_views(sh_vp, SHADOW_CASCADES)
|
||||
sh_split = floats(SHADOW_CASCADES)
|
||||
sh_range = floats(SHADOW_CASCADES)
|
||||
sh_texel = floats(SHADOW_CASCADES)
|
||||
function shadow_init(render3d_st: mut Render3dState) -> void {
|
||||
shadow_make_tex(render3d_st)
|
||||
render3d_st.sh_fbo = gpu_fb_new(render3d_st)
|
||||
gpu_fb_bind(render3d_st, render3d_st.sh_fbo)
|
||||
gpu_fb_no_color(render3d_st)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
render3d_st.sh_vp = floats(16 * SHADOW_CASCADES)
|
||||
render3d_st.sh_vp_v = m4_views(render3d_st.sh_vp, SHADOW_CASCADES)
|
||||
render3d_st.sh_split = floats(SHADOW_CASCADES)
|
||||
render3d_st.sh_range = floats(SHADOW_CASCADES)
|
||||
render3d_st.sh_texel = floats(SHADOW_CASCADES)
|
||||
# the fourth slice keeps a tree-sized texel out to a kilometre; only the massif uses the last
|
||||
sh_split[0] = 16.0; sh_split[1] = 60.0; sh_split[2] = 250.0; sh_split[3] = 1100.0; sh_split[4] = 6000.0
|
||||
sh_tmp_proj = m4_new(); sh_tmp_vp = m4_new(); sh_tmp_inv = m4_new(); sh_tmp_view = m4_new()
|
||||
sh_corner = floats(3)
|
||||
render3d_st.sh_split[0] = 16.0; render3d_st.sh_split[1] = 60.0; render3d_st.sh_split[2] = 250.0; render3d_st.sh_split[3] = 1100.0; render3d_st.sh_split[4] = 6000.0
|
||||
render3d_st.sh_tmp_proj = m4_new(); render3d_st.sh_tmp_vp = m4_new(); render3d_st.sh_tmp_inv = m4_new(); render3d_st.sh_tmp_view = m4_new()
|
||||
render3d_st.sh_corner = floats(3)
|
||||
}
|
||||
|
||||
# A shadow resolution setting: 1024, 2048 or 4096 per cascade. The depth layers are made again at
|
||||
# the new size; the pass attaches a layer per cascade every frame, and the lighting reads the
|
||||
# texel size from the map itself, so nothing else has to follow.
|
||||
function shadow_set_res(r: int) -> void {
|
||||
if r < 256 or r == shadow_res { return }
|
||||
let was = shadow_res
|
||||
shadow_res = r
|
||||
if sh_tex == 0 { return }
|
||||
gpu_tex_free(sh_tex)
|
||||
shadow_make_tex()
|
||||
function shadow_set_res(render3d_st: mut Render3dState, r: int) -> void {
|
||||
if r < 256 or r == render3d_st.shadow_res { return }
|
||||
let was = render3d_st.shadow_res
|
||||
render3d_st.shadow_res = r
|
||||
if render3d_st.sh_tex == 0 { return }
|
||||
gpu_tex_free(render3d_st, render3d_st.sh_tex)
|
||||
shadow_make_tex(render3d_st)
|
||||
# not enough video memory for that size: go back to the one that worked, and smaller again if even
|
||||
# that is refused now, rather than ending with no shadow map at all
|
||||
if not gpu_tex_ok(sh_tex) {
|
||||
shadow_refused = r
|
||||
if not gpu_tex_ok(render3d_st, render3d_st.sh_tex) {
|
||||
render3d_st.shadow_refused = r
|
||||
print(`r3d: shadows: no memory for {r} x {r} cascades; keeping {was}`)
|
||||
var size = was
|
||||
shadow_res = size
|
||||
gpu_tex_free(sh_tex)
|
||||
shadow_make_tex()
|
||||
while not gpu_tex_ok(sh_tex) and size > 512 {
|
||||
render3d_st.shadow_res = size
|
||||
gpu_tex_free(render3d_st, render3d_st.sh_tex)
|
||||
shadow_make_tex(render3d_st)
|
||||
while not gpu_tex_ok(render3d_st, render3d_st.sh_tex) and size > 512 {
|
||||
size = size / 2
|
||||
shadow_res = size
|
||||
gpu_tex_free(sh_tex)
|
||||
shadow_make_tex()
|
||||
render3d_st.shadow_res = size
|
||||
gpu_tex_free(render3d_st, render3d_st.sh_tex)
|
||||
shadow_make_tex(render3d_st)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function shadow_make_tex() -> void {
|
||||
sh_tex = gpu_tex_new()
|
||||
gpu_tex_bind(GPU_TEX2D_ARRAY, sh_tex)
|
||||
gpu_tex_image3d(GL_DEPTH_COMPONENT32F, shadow_res, shadow_res, SHADOW_CASCADES, GL_DEPTH_COMPONENT, GL_FLOAT, null)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_BORDER)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_BORDER)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_MODE, GL_COMPARE_REF_TO_TEXTURE)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_FUNC, GL_LEQUAL)
|
||||
function shadow_make_tex(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.sh_tex = gpu_tex_new(render3d_st)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D_ARRAY, render3d_st.sh_tex)
|
||||
gpu_tex_image3d(render3d_st, GL_DEPTH_COMPONENT32F, render3d_st.shadow_res, render3d_st.shadow_res, SHADOW_CASCADES, GL_DEPTH_COMPONENT, GL_FLOAT, null)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_BORDER)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_BORDER)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_MODE, GL_COMPARE_REF_TO_TEXTURE)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_FUNC, GL_LEQUAL)
|
||||
let border = gl_floats(4)
|
||||
gl_put(border, 0, 1.0); gl_put(border, 1, 1.0); gl_put(border, 2, 1.0); gl_put(border, 3, 1.0)
|
||||
gpu_tex_border(GPU_TEX2D_ARRAY, border)
|
||||
gpu_tex_border(render3d_st, GPU_TEX2D_ARRAY, border)
|
||||
free(border)
|
||||
}
|
||||
|
||||
# light view-projection for the camera-frustum slice [near, far]
|
||||
function shadow_fit(c: int, near: float, far: float) -> void {
|
||||
function shadow_fit(render3d_st: mut Render3dState, c: int, near: float, far: float) -> void {
|
||||
# Fit the slice in VIEW space, not world space. The bounding sphere of a frustum
|
||||
# slice depends only on near/far/fov/aspect — never on where the camera is pointing —
|
||||
# so computing it here makes the radius a constant per cascade. Doing it in world
|
||||
# space (as this did) let the radius wobble as the camera turned, which changed the
|
||||
# texel size, which moved the grid the projection is snapped to, so the whole shadow
|
||||
# map resampled every frame: that is the crawl and flicker seen while moving.
|
||||
m4_perspective(sh_tmp_proj, cam_fov, cam_aspect, near, far)
|
||||
m4_inverse(sh_tmp_inv, sh_tmp_proj) # NDC -> view space
|
||||
m4_perspective(render3d_st.sh_tmp_proj, render3d_st.cam_fov, render3d_st.cam_aspect, near, far)
|
||||
m4_inverse(render3d_st.sh_tmp_inv, render3d_st.sh_tmp_proj) # NDC -> view space
|
||||
let cview = v3_new(0.0, 0.0, 0.0)
|
||||
let corners = floats(24)
|
||||
for i in 0 .. 8 {
|
||||
|
|
@ -101,131 +86,123 @@ function shadow_fit(c: int, near: float, far: float) -> void {
|
|||
if (i & 1) != 0 { x = 1.0 }
|
||||
if (i & 2) != 0 { y = 1.0 }
|
||||
if (i & 4) != 0 { z = 1.0 }
|
||||
let w = m4_xform_point(sh_corner, sh_tmp_inv, x, y, z)
|
||||
let w = m4_xform_point(render3d_st.sh_corner, render3d_st.sh_tmp_inv, x, y, z)
|
||||
let iw = 1.0 / w
|
||||
corners[i * 3] = sh_corner[0] * iw; corners[i * 3 + 1] = sh_corner[1] * iw; corners[i * 3 + 2] = sh_corner[2] * iw
|
||||
corners[i * 3] = render3d_st.sh_corner[0] * iw; corners[i * 3 + 1] = render3d_st.sh_corner[1] * iw; corners[i * 3 + 2] = render3d_st.sh_corner[2] * iw
|
||||
cview[0] = cview[0] + corners[i * 3]; cview[1] = cview[1] + corners[i * 3 + 1]; cview[2] = cview[2] + corners[i * 3 + 2]
|
||||
}
|
||||
v3_scale(cview, cview, 1.0 / 8.0)
|
||||
var radius = 0.0
|
||||
for i in 0 .. 8 {
|
||||
v3_set(sh_corner, corners[i * 3], corners[i * 3 + 1], corners[i * 3 + 2])
|
||||
let d = v3_dist(sh_corner, cview)
|
||||
v3_set(render3d_st.sh_corner, corners[i * 3], corners[i * 3 + 1], corners[i * 3 + 2])
|
||||
let d = v3_dist(render3d_st.sh_corner, cview)
|
||||
if d > radius { radius = d }
|
||||
}
|
||||
radius = radius * 1.05
|
||||
# the slice centre back into world space
|
||||
m4_inverse(sh_tmp_vp, cam_view)
|
||||
m4_inverse(render3d_st.sh_tmp_vp, render3d_st.cam_view)
|
||||
let center = floats(3)
|
||||
m4_xform_point(center, sh_tmp_vp, cview[0], cview[1], cview[2])
|
||||
m4_xform_point(center, render3d_st.sh_tmp_vp, cview[0], cview[1], cview[2])
|
||||
free(cview)
|
||||
# light view: from far along the sun direction, looking at the centre
|
||||
let eye = floats(3)
|
||||
# casters up to ~900 m toward the sun (a mountain across the valley), and the
|
||||
# slice itself behind the centre: a tight depth range keeps the bias small
|
||||
let back = radius + 900.0
|
||||
v3_madd(eye, center, sun_dir, back)
|
||||
v3_madd(eye, center, render3d_st.sun_dir, back)
|
||||
let up = v3_new(0.0, 1.0, 0.0)
|
||||
m4_look_at(sh_tmp_view, eye, center, up)
|
||||
m4_look_at(render3d_st.sh_tmp_view, eye, center, up)
|
||||
# snap the ortho window to the shadow texel grid
|
||||
let texel = radius * 2.0 / float(shadow_res)
|
||||
m4_xform_point(sh_corner, sh_tmp_view, center[0], center[1], center[2])
|
||||
let ox = Math.floor(sh_corner[0] / texel) * texel - sh_corner[0]
|
||||
let oy = Math.floor(sh_corner[1] / texel) * texel - sh_corner[1]
|
||||
let texel = radius * 2.0 / float(render3d_st.shadow_res)
|
||||
m4_xform_point(render3d_st.sh_corner, render3d_st.sh_tmp_view, center[0], center[1], center[2])
|
||||
let ox = Math.floor(render3d_st.sh_corner[0] / texel) * texel - render3d_st.sh_corner[0]
|
||||
let oy = Math.floor(render3d_st.sh_corner[1] / texel) * texel - render3d_st.sh_corner[1]
|
||||
let nr = -radius
|
||||
let zfar = back + radius + 100.0
|
||||
m4_ortho(sh_tmp_proj, nr + ox, radius + ox, nr + oy, radius + oy, 1.0, zfar)
|
||||
sh_range[c] = zfar - 1.0
|
||||
sh_texel[c] = texel
|
||||
m4_ortho(render3d_st.sh_tmp_proj, nr + ox, radius + ox, nr + oy, radius + oy, 1.0, zfar)
|
||||
render3d_st.sh_range[c] = zfar - 1.0
|
||||
render3d_st.sh_texel[c] = texel
|
||||
let out = floats(16)
|
||||
m4_mul(out, sh_tmp_proj, sh_tmp_view)
|
||||
for i in 0 .. 16 { sh_vp[c * 16 + i] = out[i] }
|
||||
m4_mul(out, render3d_st.sh_tmp_proj, render3d_st.sh_tmp_view)
|
||||
for i in 0 .. 16 { render3d_st.sh_vp[c * 16 + i] = out[i] }
|
||||
free(out); free(eye); free(up); free(center); free(corners)
|
||||
}
|
||||
|
||||
function shadow_cascade_vp(c: int) -> floats { return sh_vp_v[c] }
|
||||
function shadow_cascade_vp(render3d_st: Render3dState, c: int) -> floats { return render3d_st.sh_vp_v[c] }
|
||||
|
||||
|
||||
# render every cascade; `draw` happens through terrain_draw_shadow + the scene's casters
|
||||
function shadow_pass() -> void {
|
||||
var near = cam_near
|
||||
gpu_fb_bind(sh_fbo)
|
||||
gpu_viewport(0, 0, shadow_res, shadow_res)
|
||||
gpu_depth_test(true)
|
||||
gpu_depth_func(GL_LESS)
|
||||
gpu_depth_bias(2.0, 4.0)
|
||||
gpu_cull(false)
|
||||
function shadow_pass(render3d_st: mut Render3dState) -> void {
|
||||
var near = render3d_st.cam_near
|
||||
gpu_fb_bind(render3d_st, render3d_st.sh_fbo)
|
||||
gpu_viewport(render3d_st, 0, 0, render3d_st.shadow_res, render3d_st.shadow_res)
|
||||
gpu_depth_test(render3d_st, true)
|
||||
gpu_depth_func(render3d_st, GL_LESS)
|
||||
gpu_depth_bias(render3d_st, 2.0, 4.0)
|
||||
gpu_cull(render3d_st, false)
|
||||
for c in 0 .. SHADOW_CASCADES {
|
||||
sh_cascade = c
|
||||
shadow_fit(c, near, sh_split[c])
|
||||
prof_cpu_mark("shadow fit")
|
||||
near = sh_split[c]
|
||||
gpu_fb_depth_layer(sh_tex, c)
|
||||
if r3d_debug and c == 0 { let st = gpu_fb_status(); print(`shadow fbo status {st}`) }
|
||||
gpu_clear(GL_DEPTH_BUFFER_BIT)
|
||||
let vp = shadow_cascade_vp(c)
|
||||
render3d_st.sh_cascade = c
|
||||
shadow_fit(render3d_st, c, near, render3d_st.sh_split[c])
|
||||
prof_cpu_mark(render3d_st, "shadow fit")
|
||||
near = render3d_st.sh_split[c]
|
||||
gpu_fb_depth_layer(render3d_st, render3d_st.sh_tex, c)
|
||||
if render3d_st.r3d_debug and c == 0 { let st = gpu_fb_status(render3d_st); print(`shadow fbo status {st}`) }
|
||||
gpu_clear(render3d_st, GL_DEPTH_BUFFER_BIT)
|
||||
let vp = shadow_cascade_vp(render3d_st, c)
|
||||
# shadows off (a video setting): the cascades stay cleared, so everything reads lit
|
||||
if sh_enabled {
|
||||
if not sh_skip_terrain { terrain_draw_shadow(vp) }
|
||||
r3d_scene_casters(vp)
|
||||
if render3d_st.sh_enabled {
|
||||
if not render3d_st.sh_skip_terrain { terrain_draw_shadow(vp) }
|
||||
r3d_scene_casters(render3d_st, vp)
|
||||
}
|
||||
}
|
||||
gpu_depth_bias(0.0, 0.0)
|
||||
gpu_fb_bind(0)
|
||||
if r3d_debug_shadow { shadow_dump() }
|
||||
if r3d_debug_shadow and not sh_printed2 {
|
||||
sh_printed2 = true
|
||||
gpu_depth_bias(render3d_st, 0.0, 0.0)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
if render3d_st.r3d_debug_shadow { shadow_dump(render3d_st) }
|
||||
if render3d_st.r3d_debug_shadow and not render3d_st.sh_printed2 {
|
||||
render3d_st.sh_printed2 = true
|
||||
let q = floats(3)
|
||||
if sh_probe_x != 0.0 {
|
||||
let vp = shadow_cascade_vp(2)
|
||||
m4_xform_point(q, vp, sh_probe_x, sh_probe_y, sh_probe_z)
|
||||
if render3d_st.sh_probe_x != 0.0 {
|
||||
let vp = shadow_cascade_vp(render3d_st, 2)
|
||||
m4_xform_point(q, vp, render3d_st.sh_probe_x, render3d_st.sh_probe_y, render3d_st.sh_probe_z)
|
||||
print(`probe base ndc {fixed(q[0])} {fixed(q[1])} {fixed(q[2])}`)
|
||||
m4_xform_point(q, vp, sh_probe_x, sh_probe_y + 15.0, sh_probe_z)
|
||||
print(`probe top ndc {fixed(q[0])} {fixed(q[1])} {fixed(q[2])} -> map texel {int((q[0] * 0.5 + 0.5) * float(shadow_res))} {int((q[1] * 0.5 + 0.5) * float(shadow_res))}`)
|
||||
m4_xform_point(q, vp, render3d_st.sh_probe_x, render3d_st.sh_probe_y + 15.0, render3d_st.sh_probe_z)
|
||||
print(`probe top ndc {fixed(q[0])} {fixed(q[1])} {fixed(q[2])} -> map texel {int((q[0] * 0.5 + 0.5) * float(render3d_st.shadow_res))} {int((q[1] * 0.5 + 0.5) * float(render3d_st.shadow_res))}`)
|
||||
# where the top's shadow lands on the ground: walk down the sun ray
|
||||
let gx = sh_probe_x + 0.0 - sun_dir[0] * (15.0 / sun_dir[1])
|
||||
let gz = sh_probe_z - sun_dir[2] * (15.0 / sun_dir[1])
|
||||
m4_xform_point(q, vp, gx, terrain_height(gx, gz), gz)
|
||||
let gx = render3d_st.sh_probe_x + 0.0 - render3d_st.sun_dir[0] * (15.0 / render3d_st.sun_dir[1])
|
||||
let gz = render3d_st.sh_probe_z - render3d_st.sun_dir[2] * (15.0 / render3d_st.sun_dir[1])
|
||||
m4_xform_point(q, vp, gx, terrain_height(render3d_st, gx, gz), gz)
|
||||
print(`shadow-of-top ground ndc {fixed(q[0])} {fixed(q[1])} {fixed(q[2])} at {fixed(gx)} {fixed(gz)}`)
|
||||
}
|
||||
for c in 0 .. SHADOW_CASCADES {
|
||||
let vp = shadow_cascade_vp(c)
|
||||
let vp = shadow_cascade_vp(render3d_st, c)
|
||||
# a point 5 m ahead of the camera on the ground
|
||||
let px = cam_pos[0] + cam_fwd[0] * 5.0; let pz = cam_pos[2] + cam_fwd[2] * 5.0
|
||||
let w = m4_xform_point(q, vp, px, terrain_height(px, pz), pz)
|
||||
let px = render3d_st.cam_pos[0] + render3d_st.cam_fwd[0] * 5.0; let pz = render3d_st.cam_pos[2] + render3d_st.cam_fwd[2] * 5.0
|
||||
let w = m4_xform_point(q, vp, px, terrain_height(render3d_st, px, pz), pz)
|
||||
print(`cascade {c}: ndc {fixed(q[0])} {fixed(q[1])} {fixed(q[2])} w {fixed(w)} m0 {fixed(vp[0])} m5 {fixed(vp[5])} m14 {fixed(vp[14])}`)
|
||||
}
|
||||
free(q)
|
||||
}
|
||||
}
|
||||
var sh_printed2: bool = false
|
||||
var sh_printed3: bool = false
|
||||
var sh_enabled: bool = true
|
||||
var sh_force: int = -1 # R3D_FORCE=<c> pins every pixel to cascade c (debug)
|
||||
var sh_skip_terrain: bool = false
|
||||
var sh_probe_x: float = 0.0
|
||||
var sh_probe_y: float = 0.0
|
||||
var sh_probe_z: float = 0.0
|
||||
|
||||
# Debug: cascade depths as grey PPMs (build/dbg_shadow_<c>.ppm)
|
||||
function shadow_dump() -> void {
|
||||
let n = shadow_res * shadow_res
|
||||
function shadow_dump(render3d_st: mut Render3dState) -> void {
|
||||
let n = render3d_st.shadow_res * render3d_st.shadow_res
|
||||
let buf = floats(n * SHADOW_CASCADES)
|
||||
gpu_tex_bind(GPU_TEX2D_ARRAY, sh_tex)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_MODE, GL_NONE)
|
||||
gpu_tex_read(GPU_TEX2D_ARRAY, GL_DEPTH_COMPONENT, GL_FLOAT, data_of(buf))
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_MODE, GL_COMPARE_REF_TO_TEXTURE)
|
||||
if sh_probe_x != 0.0 {
|
||||
let vp = shadow_cascade_vp(2)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D_ARRAY, render3d_st.sh_tex)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_MODE, GL_NONE)
|
||||
gpu_tex_read(render3d_st, GPU_TEX2D_ARRAY, GL_DEPTH_COMPONENT, GL_FLOAT, data_of(buf))
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_COMPARE_MODE, GL_COMPARE_REF_TO_TEXTURE)
|
||||
if render3d_st.sh_probe_x != 0.0 {
|
||||
let vp = shadow_cascade_vp(render3d_st, 2)
|
||||
let q = floats(3)
|
||||
m4_xform_point(q, vp, sh_probe_x, sh_probe_y + 12.0, sh_probe_z)
|
||||
let tx = int((q[0] * 0.5 + 0.5) * float(shadow_res))
|
||||
let ty = int((q[1] * 0.5 + 0.5) * float(shadow_res))
|
||||
m4_xform_point(q, vp, render3d_st.sh_probe_x, render3d_st.sh_probe_y + 12.0, render3d_st.sh_probe_z)
|
||||
let tx = int((q[0] * 0.5 + 0.5) * float(render3d_st.shadow_res))
|
||||
let ty = int((q[1] * 0.5 + 0.5) * float(render3d_st.shadow_res))
|
||||
let want = q[2] * 0.5 + 0.5
|
||||
print(`probe (12 m up) texel {tx} {ty} card depth {fixed(want * 1000.0)}/1000`)
|
||||
for dy in 0 .. 5 {
|
||||
let yy = ty - 40 + dy * 20
|
||||
print(` row {yy}: {fixed(buf[2 * n + yy * shadow_res + tx - 20] * 1000.0)} {fixed(buf[2 * n + yy * shadow_res + tx] * 1000.0)} {fixed(buf[2 * n + yy * shadow_res + tx + 20] * 1000.0)} /1000`)
|
||||
print(` row {yy}: {fixed(buf[2 * n + yy * render3d_st.shadow_res + tx - 20] * 1000.0)} {fixed(buf[2 * n + yy * render3d_st.shadow_res + tx] * 1000.0)} {fixed(buf[2 * n + yy * render3d_st.shadow_res + tx + 20] * 1000.0)} /1000`)
|
||||
}
|
||||
free(q)
|
||||
}
|
||||
|
|
@ -240,10 +217,10 @@ function shadow_dump() -> void {
|
|||
let f = file_open(`build/dbg_shadow_{c}.ppm`, "wb")
|
||||
let hdr = `P6\n{sm} {sm}\n255\n`
|
||||
file_write(f, hdr, len(hdr))
|
||||
let st = shadow_res / sm
|
||||
let st = render3d_st.shadow_res / sm
|
||||
for y in 0 .. sm {
|
||||
for x in 0 .. sm {
|
||||
let d = buf[c * n + (y * st) * shadow_res + x * st]
|
||||
let d = buf[c * n + (y * st) * render3d_st.shadow_res + x * st]
|
||||
let g = int(Math.clamp((d - lo) / Math.max(hi - lo, 0.0001), 0.0, 1.0) * 255.0)
|
||||
row[x * 3] = g; row[x * 3 + 1] = g; row[x * 3 + 2] = g
|
||||
}
|
||||
|
|
@ -254,32 +231,31 @@ function shadow_dump() -> void {
|
|||
free(buf); free(row)
|
||||
}
|
||||
|
||||
var sh_printed: bool = false
|
||||
# a uniform array's location: some drivers only answer to the "[0]" spelling
|
||||
function sh_loc(prog: int, name: string) -> int {
|
||||
var loc = gpu_uniform(prog, name + "[0]")
|
||||
if loc < 0 { loc = gpu_uniform(prog, name) }
|
||||
function sh_loc(render3d_st: Render3dState, prog: int, name: string) -> int {
|
||||
var loc = gpu_uniform(render3d_st, prog, name + "[0]")
|
||||
if loc < 0 { loc = gpu_uniform(render3d_st, prog, name) }
|
||||
return loc
|
||||
}
|
||||
function shadow_bind(prog: int) -> void {
|
||||
r3d_bind_tex(prog, "u_shadow", 15, GPU_TEX2D_ARRAY, sh_tex)
|
||||
function shadow_bind(render3d_st: mut Render3dState, prog: int) -> void {
|
||||
r3d_bind_tex(render3d_st, prog, "u_shadow", 15, GPU_TEX2D_ARRAY, render3d_st.sh_tex)
|
||||
# the height-field shadow (terrain.ludic); a stand-in texture keeps the unit valid before the bake
|
||||
var ts = ter_shadow_tex
|
||||
var ts = render3d_st.ter_shadow_tex
|
||||
var ts_on = 1.0
|
||||
if ts == 0 { ts = ter_height_tex; ts_on = 0.0 }
|
||||
r3d_bind_2d(prog, "u_tershadow", 6, ts)
|
||||
terrain_bind_height(prog)
|
||||
u_f(gpu_uniform(prog, "u_ts_on"), ts_on)
|
||||
u_f(gpu_uniform(prog, "u_ts_half"), float(TERRAIN_HALF))
|
||||
u_f2(gpu_uniform(prog, "u_ts_origin"), ter_ox, ter_oz)
|
||||
var loc = gpu_uniform(prog, "u_cascade_vp[0]")
|
||||
if loc < 0 { loc = gpu_uniform(prog, "u_cascade_vp") }
|
||||
if r3d_debug_shadow and not sh_printed { sh_printed = true; print(`cascade vp loc {loc} / {gpu_uniform(prog, "u_cascade_vp")} split loc {gpu_uniform(prog, "u_cascade_split")} shadow loc {gpu_uniform(prog, "u_shadow")}`) }
|
||||
u_mat4n(loc, SHADOW_CASCADES, sh_vp)
|
||||
u_fv(sh_loc(prog, "u_cascade_split"), SHADOW_CASCADES, sh_split)
|
||||
if r3d_debug_shadow and not sh_printed3 { sh_printed3 = true; print(`range {fixed(sh_range[0])} {fixed(sh_range[1])} {fixed(sh_range[2])} {fixed(sh_range[3])} texel*1000 {fixed(sh_texel[0] * 1000.0)} {fixed(sh_texel[1] * 1000.0)} {fixed(sh_texel[2] * 1000.0)} {fixed(sh_texel[3] * 1000.0)} locs {gpu_uniform(prog, "u_cascade_range")} {gpu_uniform(prog, "u_cascade_texel")}`) }
|
||||
u_fv(sh_loc(prog, "u_cascade_range"), SHADOW_CASCADES, sh_range)
|
||||
if r3d_env_has("R3D_FORCE") { sh_force = Text.to_int(r3d_env("R3D_FORCE")) }
|
||||
u_i(gpu_uniform(prog, "u_force_cascade"), sh_force)
|
||||
u_fv(sh_loc(prog, "u_cascade_texel"), SHADOW_CASCADES, sh_texel)
|
||||
if ts == 0 { ts = render3d_st.ter_height_tex; ts_on = 0.0 }
|
||||
r3d_bind_2d(render3d_st, prog, "u_tershadow", 6, ts)
|
||||
terrain_bind_height(render3d_st, prog)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_ts_on"), ts_on)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_ts_half"), float(render3d_st.TERRAIN_HALF))
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, prog, "u_ts_origin"), render3d_st.ter_ox, render3d_st.ter_oz)
|
||||
var loc = gpu_uniform(render3d_st, prog, "u_cascade_vp[0]")
|
||||
if loc < 0 { loc = gpu_uniform(render3d_st, prog, "u_cascade_vp") }
|
||||
if render3d_st.r3d_debug_shadow and not render3d_st.sh_printed { render3d_st.sh_printed = true; print(`cascade vp loc {loc} / {gpu_uniform(render3d_st, prog, "u_cascade_vp")} split loc {gpu_uniform(render3d_st, prog, "u_cascade_split")} shadow loc {gpu_uniform(render3d_st, prog, "u_shadow")}`) }
|
||||
u_mat4n(render3d_st, loc, SHADOW_CASCADES, render3d_st.sh_vp)
|
||||
u_fv(render3d_st, sh_loc(render3d_st, prog, "u_cascade_split"), SHADOW_CASCADES, render3d_st.sh_split)
|
||||
if render3d_st.r3d_debug_shadow and not render3d_st.sh_printed3 { render3d_st.sh_printed3 = true; print(`range {fixed(render3d_st.sh_range[0])} {fixed(render3d_st.sh_range[1])} {fixed(render3d_st.sh_range[2])} {fixed(render3d_st.sh_range[3])} texel*1000 {fixed(render3d_st.sh_texel[0] * 1000.0)} {fixed(render3d_st.sh_texel[1] * 1000.0)} {fixed(render3d_st.sh_texel[2] * 1000.0)} {fixed(render3d_st.sh_texel[3] * 1000.0)} locs {gpu_uniform(render3d_st, prog, "u_cascade_range")} {gpu_uniform(render3d_st, prog, "u_cascade_texel")}`) }
|
||||
u_fv(render3d_st, sh_loc(render3d_st, prog, "u_cascade_range"), SHADOW_CASCADES, render3d_st.sh_range)
|
||||
if r3d_env_has(render3d_st, "R3D_FORCE") { render3d_st.sh_force = Text.to_int(r3d_env(render3d_st, "R3D_FORCE")) }
|
||||
u_i(render3d_st, gpu_uniform(render3d_st, prog, "u_force_cascade"), render3d_st.sh_force)
|
||||
u_fv(render3d_st, sh_loc(render3d_st, prog, "u_cascade_texel"), SHADOW_CASCADES, render3d_st.sh_texel)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -50,31 +50,31 @@ function skin_jv3(o: floats, at: int, nd: Val, key: pointer, dx: float, dy: floa
|
|||
}
|
||||
|
||||
# JOINTS_0 / WEIGHTS_0 onto attributes 5 and 6 of the VAO being built (gltf_prim)
|
||||
function skin_attribs(m: Mesh, attrs: Val) -> bool {
|
||||
function skin_attribs(render3d_st: mut Render3dState, m: Mesh, attrs: Val) -> bool {
|
||||
if value_has(attrs, "JOINTS_0") == 0 or value_has(attrs, "WEIGHTS_0") == 0 { return false }
|
||||
let jd = gltf_accessor(value_as_int(value_get(attrs, "JOINTS_0")))
|
||||
let jd = gltf_accessor(render3d_st, value_as_int(value_get(attrs, "JOINTS_0")))
|
||||
var jsz = 1
|
||||
var jtype = GPU_U8
|
||||
if gltf_ctype == 5123 { jsz = 2; jtype = GPU_U16 }
|
||||
gpu_mesh_vertices(m, jd, gltf_count * gltf_comps * jsz, GPU_STATIC)
|
||||
gpu_mesh_attr(m, 5, gltf_comps, jtype, 0, 0, false) # integers, read as floats
|
||||
if render3d_st.gltf_ctype == 5123 { jsz = 2; jtype = GPU_U16 }
|
||||
gpu_mesh_vertices(render3d_st, m, jd, render3d_st.gltf_count * render3d_st.gltf_comps * jsz, GPU_STATIC)
|
||||
gpu_mesh_attr(render3d_st, m, 5, render3d_st.gltf_comps, jtype, 0, 0, false) # integers, read as floats
|
||||
free(jd)
|
||||
let wd = gltf_accessor(value_as_int(value_get(attrs, "WEIGHTS_0")))
|
||||
let wd = gltf_accessor(render3d_st, value_as_int(value_get(attrs, "WEIGHTS_0")))
|
||||
var wsz = 4
|
||||
var wtype = GPU_F32
|
||||
var norm = false
|
||||
if gltf_ctype == 5123 { wsz = 2; wtype = GPU_U16; norm = true }
|
||||
if gltf_ctype == 5121 { wsz = 1; wtype = GPU_U8; norm = true }
|
||||
gpu_mesh_vertices(m, wd, gltf_count * gltf_comps * wsz, GPU_STATIC)
|
||||
gpu_mesh_attr(m, 6, gltf_comps, wtype, 0, 0, norm)
|
||||
if render3d_st.gltf_ctype == 5123 { wsz = 2; wtype = GPU_U16; norm = true }
|
||||
if render3d_st.gltf_ctype == 5121 { wsz = 1; wtype = GPU_U8; norm = true }
|
||||
gpu_mesh_vertices(render3d_st, m, wd, render3d_st.gltf_count * render3d_st.gltf_comps * wsz, GPU_STATIC)
|
||||
gpu_mesh_attr(render3d_st, m, 6, render3d_st.gltf_comps, wtype, 0, 0, norm)
|
||||
free(wd)
|
||||
return true
|
||||
}
|
||||
|
||||
# the skin `idx` of the document being loaded (gltf_load holds gltf_doc / gltf_bin open)
|
||||
function skin_load(idx: int) -> Skin {
|
||||
function skin_load(render3d_st: mut Render3dState, idx: int) -> Skin {
|
||||
let sk = new Skin
|
||||
let nodes = value_get(gltf_doc, "nodes")
|
||||
let nodes = value_get(render3d_st.gltf_doc, "nodes")
|
||||
let n = value_count(nodes)
|
||||
sk.n_nodes = n
|
||||
sk.par = words(n); sk.walk = words(n)
|
||||
|
|
@ -121,7 +121,7 @@ function skin_load(idx: int) -> Skin {
|
|||
else { q_store(sk.rest_g, i, sk.tmp_a) }
|
||||
}
|
||||
# the skin: joints and inverse bind matrices
|
||||
let skv = value_at(value_get(gltf_doc, "skins"), idx)
|
||||
let skv = value_at(value_get(render3d_st.gltf_doc, "skins"), idx)
|
||||
let jl = value_get(skv, "joints")
|
||||
var nj = value_count(jl)
|
||||
if nj > SKIN_MAX_JOINTS { print(`skin: {nj} joints, only the first {SKIN_MAX_JOINTS} are used`); nj = SKIN_MAX_JOINTS }
|
||||
|
|
@ -133,7 +133,7 @@ function skin_load(idx: int) -> Skin {
|
|||
sk.bones_v = m4_views(sk.bones, nj)
|
||||
for j in 0 .. nj { sk.joints[j] = value_as_int(value_at(jl, j)) }
|
||||
if value_has(skv, "inverseBindMatrices") != 0 {
|
||||
let ib = gltf_accessor(value_as_int(value_get(skv, "inverseBindMatrices")))
|
||||
let ib = gltf_accessor(render3d_st, value_as_int(value_get(skv, "inverseBindMatrices")))
|
||||
for i in 0 .. nj * 16 { sk.inv_bind[i] = float_from_bits(mem_get_f32_bits(ib, i)) }
|
||||
free(ib)
|
||||
} else {
|
||||
|
|
@ -160,9 +160,9 @@ function skin_reset(sk: Skin) -> void {
|
|||
}
|
||||
}
|
||||
# a node's pose rotation in the model frame: pitch about X, yaw about Y, roll about Z (radians)
|
||||
function skin_set_rot(sk: Skin, node: int, pitch: float, yaw: float, roll: float) -> void {
|
||||
function skin_set_rot(render3d_st: mut Render3dState, sk: Skin, node: int, pitch: float, yaw: float, roll: float) -> void {
|
||||
if node < 0 { return }
|
||||
q_euler(sk.tmp_q, pitch, yaw, roll)
|
||||
q_euler(render3d_st, sk.tmp_q, pitch, yaw, roll)
|
||||
q_store(sk.pose_r, node, sk.tmp_q)
|
||||
}
|
||||
function skin_set_quat(sk: Skin, node: int, q: floats) -> void { if node >= 0 { q_store(sk.pose_r, node, q) } }
|
||||
|
|
@ -203,10 +203,10 @@ function skin_pose(sk: Skin) -> void {
|
|||
}
|
||||
|
||||
# the joint matrices onto a program's u_bones[]
|
||||
function skin_bind(sk: Skin, prog: int) -> void {
|
||||
var loc = gpu_uniform(prog, "u_bones[0]")
|
||||
if loc < 0 { loc = gpu_uniform(prog, "u_bones") }
|
||||
u_mat4n(loc, sk.n_joints, sk.bones)
|
||||
function skin_bind(render3d_st: mut Render3dState, sk: Skin, prog: int) -> void {
|
||||
var loc = gpu_uniform(render3d_st, prog, "u_bones[0]")
|
||||
if loc < 0 { loc = gpu_uniform(render3d_st, prog, "u_bones") }
|
||||
u_mat4n(render3d_st, loc, sk.n_joints, sk.bones)
|
||||
}
|
||||
|
||||
# the same skeleton posed on its own: shares the rest data, owns the pose and the matrices
|
||||
|
|
|
|||
|
|
@ -7,47 +7,27 @@
|
|||
|
||||
const SKY_PREFILTER_LEVELS: int = 6
|
||||
|
||||
var sky_tex: int = 0
|
||||
var sky_w: int = 0
|
||||
var sky_h: int = 0
|
||||
var sky_irradiance: int = 0
|
||||
var sky_prefilter: int = 0 # GL_TEXTURE_2D_ARRAY
|
||||
var sky_brdf: int = 0
|
||||
var sun_dir: floats = null # toward the sun (float bits)
|
||||
var sun_color: floats = null # radiance (float bits)
|
||||
var sky_fullscreen: Mesh = null
|
||||
var sky_yaw: float = 0.0 # radians: the HDRI is turned by this about y
|
||||
var sky_sun_boost: float = 2.3 # 2.3: the photograph's thin cloud dims its sun; a crisper day wants more
|
||||
var sky_rot_s: float = 0.0
|
||||
var sky_rot_c: float = 0.0
|
||||
var sun_hdri: floats = null # the sun direction as found in the file
|
||||
# The yaw to bake the light at the first time (radians, float bits). A game that turns the sky at
|
||||
# start sets it before r3d_init, so the image-based light is baked once at that yaw rather than at
|
||||
# 0 and then again - which is what sky_set_yaw at boot used to cost.
|
||||
var sky_start_yaw: float = 0.0
|
||||
var sky_baked: bool = false # the IBL set exists, baked at sky_baked_yaw
|
||||
var sky_baked_yaw: float = 0.0
|
||||
var sky_p_irr: int = 0
|
||||
var sky_p_pre: int = 0
|
||||
var sky_p_brdf: int = 0
|
||||
|
||||
# turn the HDRI so its sun sits at world azimuth `yaw` (radians, 0 = toward -z)
|
||||
function sky_set_yaw(yaw: float) -> void {
|
||||
sky_set_rot(yaw)
|
||||
function sky_set_yaw(render3d_st: mut Render3dState, yaw: float) -> void {
|
||||
sky_set_rot(render3d_st, yaw)
|
||||
# world sun = rotY(sun_hdri, -yaw): the lookup rotates a world direction by +yaw
|
||||
let s = sun_hdri
|
||||
v3_set(sun_dir, sky_rot_c * s[0] - sky_rot_s * s[2], s[1], sky_rot_s * s[0] + sky_rot_c * s[2])
|
||||
let s = render3d_st.sun_hdri
|
||||
v3_set(render3d_st.sun_dir, render3d_st.sky_rot_c * s[0] - render3d_st.sky_rot_s * s[2], s[1], render3d_st.sky_rot_s * s[0] + render3d_st.sky_rot_c * s[2])
|
||||
# the light is already baked at this yaw: turning to it again bakes nothing
|
||||
if sky_baked and yaw == sky_baked_yaw { return }
|
||||
sky_precompute()
|
||||
if render3d_st.sky_baked and yaw == render3d_st.sky_baked_yaw { return }
|
||||
sky_precompute(render3d_st)
|
||||
}
|
||||
# turn only the visible sky image (cheap, per frame): the light and the convolved
|
||||
# maps stay where they are — daylight.ludic moves those on its own terms
|
||||
function sky_set_rot(yaw: float) -> void {
|
||||
sky_yaw = yaw
|
||||
sky_rot_s = Math.sin(yaw); sky_rot_c = Math.cos(yaw)
|
||||
function sky_set_rot(render3d_st: mut Render3dState, yaw: float) -> void {
|
||||
render3d_st.sky_yaw = yaw
|
||||
render3d_st.sky_rot_s = Math.sin(yaw); render3d_st.sky_rot_c = Math.cos(yaw)
|
||||
}
|
||||
function sky_bind_rot(prog: int) -> void { u_f2(gpu_uniform(prog, "u_sky_rot"), sky_rot_s, sky_rot_c) }
|
||||
function sky_bind_rot(render3d_st: mut Render3dState, prog: int) -> void { u_f2(render3d_st, gpu_uniform(render3d_st, prog, "u_sky_rot"), render3d_st.sky_rot_s, render3d_st.sky_rot_c) }
|
||||
|
||||
# direction for an equirect uv (matches equirectUV in lighting.glsl)
|
||||
function sky_dir_from_uv(o: floats, u: float, v: float) -> void {
|
||||
|
|
@ -57,94 +37,93 @@ function sky_dir_from_uv(o: floats, u: float, v: float) -> void {
|
|||
v3_set(o, st * Math.sin(phi), Math.cos(theta), -(st * Math.cos(phi)))
|
||||
}
|
||||
|
||||
function sky_load(path: string) -> bool {
|
||||
sky_tex = tex_load_hdr(path)
|
||||
if sky_tex == 0 { return false }
|
||||
sky_w = tex_w; sky_h = tex_h
|
||||
sun_dir = floats(3)
|
||||
sun_hdri = floats(3)
|
||||
sky_dir_from_uv(sun_hdri, float(hdr_max_x * 2 + 1) / float(sky_w * 2), float(hdr_max_y * 2 + 1) / float(sky_h * 2))
|
||||
v3_copy(sun_dir, sun_hdri)
|
||||
sky_rot_c = 1.0
|
||||
function sky_load(render3d_st: mut Render3dState, path: string) -> bool {
|
||||
render3d_st.sky_tex = tex_load_hdr(render3d_st, path)
|
||||
if render3d_st.sky_tex == 0 { return false }
|
||||
render3d_st.sky_w = render3d_st.tex_w; render3d_st.sky_h = render3d_st.tex_h
|
||||
render3d_st.sun_dir = floats(3)
|
||||
render3d_st.sun_hdri = floats(3)
|
||||
sky_dir_from_uv(render3d_st.sun_hdri, float(render3d_st.hdr_max_x * 2 + 1) / float(render3d_st.sky_w * 2), float(render3d_st.hdr_max_y * 2 + 1) / float(render3d_st.sky_h * 2))
|
||||
v3_copy(render3d_st.sun_dir, render3d_st.sun_hdri)
|
||||
render3d_st.sky_rot_c = 1.0
|
||||
# the sun's irradiance is what the IBL clip leaves out of the map; lighting it
|
||||
# directly with that keeps sun and sky in the photograph's own proportion
|
||||
sun_color = v3_new(hdr_sun_r * sky_sun_boost, hdr_sun_g * sky_sun_boost, hdr_sun_b * sky_sun_boost)
|
||||
print(`sun irradiance: {fixed(hdr_sun_r)} {fixed(hdr_sun_g)} {fixed(hdr_sun_b)} (Q16.16), clip {fixed(hdr_clip)}`)
|
||||
print(`sky: {sky_w}x{sky_h}, sun at texel {hdr_max_x},{hdr_max_y}`)
|
||||
sky_fullscreen = mesh_fullscreen()
|
||||
if sky_start_yaw != 0.0 { sky_set_yaw(sky_start_yaw) } else { sky_precompute() }
|
||||
render3d_st.sun_color = v3_new(render3d_st.hdr_sun_r * render3d_st.sky_sun_boost, render3d_st.hdr_sun_g * render3d_st.sky_sun_boost, render3d_st.hdr_sun_b * render3d_st.sky_sun_boost)
|
||||
print(`sun irradiance: {fixed(render3d_st.hdr_sun_r)} {fixed(render3d_st.hdr_sun_g)} {fixed(render3d_st.hdr_sun_b)} (Q16.16), clip {fixed(render3d_st.hdr_clip)}`)
|
||||
print(`sky: {render3d_st.sky_w}x{render3d_st.sky_h}, sun at texel {render3d_st.hdr_max_x},{render3d_st.hdr_max_y}`)
|
||||
render3d_st.sky_fullscreen = mesh_fullscreen(render3d_st)
|
||||
if render3d_st.sky_start_yaw != 0.0 { sky_set_yaw(render3d_st, render3d_st.sky_start_yaw) } else { sky_precompute(render3d_st) }
|
||||
return true
|
||||
}
|
||||
|
||||
function sky_convolve(prog: int, target_tex: int, layer: int, w: int, h: int, rough: float) -> void {
|
||||
let fbo = gpu_fb_new()
|
||||
gpu_fb_bind(fbo)
|
||||
if layer < 0 { gpu_fb_color(0, target_tex) }
|
||||
else { gpu_fb_color_layer(0, target_tex, layer) }
|
||||
gpu_viewport(0, 0, w, h)
|
||||
gpu_use_program(prog)
|
||||
r3d_bind_2d(prog, "u_sky", 0, sky_tex)
|
||||
sky_bind_rot(prog)
|
||||
u_f(gpu_uniform(prog, "u_sun_clip"), hdr_clip)
|
||||
u_f(gpu_uniform(prog, "u_rough"), rough)
|
||||
u_f(gpu_uniform(prog, "u_sky_w"), float(sky_w))
|
||||
mesh_draw(sky_fullscreen)
|
||||
gpu_fb_bind(0)
|
||||
gpu_fb_free(fbo)
|
||||
function sky_convolve(render3d_st: mut Render3dState, prog: int, target_tex: int, layer: int, w: int, h: int, rough: float) -> void {
|
||||
let fbo = gpu_fb_new(render3d_st)
|
||||
gpu_fb_bind(render3d_st, fbo)
|
||||
if layer < 0 { gpu_fb_color(render3d_st, 0, target_tex) }
|
||||
else { gpu_fb_color_layer(render3d_st, 0, target_tex, layer) }
|
||||
gpu_viewport(render3d_st, 0, 0, w, h)
|
||||
gpu_use_program(render3d_st, prog)
|
||||
r3d_bind_2d(render3d_st, prog, "u_sky", 0, render3d_st.sky_tex)
|
||||
sky_bind_rot(render3d_st, prog)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_sun_clip"), render3d_st.hdr_clip)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_rough"), rough)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_sky_w"), float(render3d_st.sky_w))
|
||||
mesh_draw(render3d_st, render3d_st.sky_fullscreen)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
gpu_fb_free(render3d_st, fbo)
|
||||
}
|
||||
|
||||
# the prefiltered specular's width per level (its height is half): 256, 512 or 1024. The image
|
||||
# the reflections and the rough sheen are lit from; sky_set_quality bakes it again at a new size.
|
||||
var sky_prefilter_w: int = 512
|
||||
function sky_set_quality(w: int) -> void {
|
||||
if w < 64 or w == sky_prefilter_w { return }
|
||||
sky_prefilter_w = w
|
||||
if sky_irradiance != 0 { sky_precompute() }
|
||||
function sky_set_quality(render3d_st: mut Render3dState, w: int) -> void {
|
||||
if w < 64 or w == render3d_st.sky_prefilter_w { return }
|
||||
render3d_st.sky_prefilter_w = w
|
||||
if render3d_st.sky_irradiance != 0 { sky_precompute(render3d_st) }
|
||||
}
|
||||
|
||||
function sky_precompute() -> void {
|
||||
gpu_depth_test(false)
|
||||
if sky_irradiance != 0 {
|
||||
gpu_tex_free(sky_irradiance)
|
||||
gpu_tex_free(sky_prefilter)
|
||||
gpu_tex_free(sky_brdf)
|
||||
function sky_precompute(render3d_st: mut Render3dState) -> void {
|
||||
gpu_depth_test(render3d_st, false)
|
||||
if render3d_st.sky_irradiance != 0 {
|
||||
gpu_tex_free(render3d_st, render3d_st.sky_irradiance)
|
||||
gpu_tex_free(render3d_st, render3d_st.sky_prefilter)
|
||||
gpu_tex_free(render3d_st, render3d_st.sky_brdf)
|
||||
}
|
||||
# irradiance: 128 x 64 equirect
|
||||
if sky_p_irr == 0 { sky_p_irr = r3d_program("fullscreen.vert", "ibl_irradiance.frag", ""); sky_p_pre = r3d_program("fullscreen.vert", "ibl_prefilter.frag", ""); sky_p_brdf = r3d_program("fullscreen.vert", "ibl_brdf.frag", "") }
|
||||
let p_irr = sky_p_irr
|
||||
sky_irradiance = tex_target(128, 64, GL_RGB16F, GL_RGB, GL_FLOAT, GL_LINEAR)
|
||||
gpu_tex_bind(GPU_TEX2D, sky_irradiance)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
sky_convolve(p_irr, sky_irradiance, -1, 128, 64, 0.0)
|
||||
if render3d_st.sky_p_irr == 0 { render3d_st.sky_p_irr = r3d_program(render3d_st, "fullscreen.vert", "ibl_irradiance.frag", ""); render3d_st.sky_p_pre = r3d_program(render3d_st, "fullscreen.vert", "ibl_prefilter.frag", ""); render3d_st.sky_p_brdf = r3d_program(render3d_st, "fullscreen.vert", "ibl_brdf.frag", "") }
|
||||
let p_irr = render3d_st.sky_p_irr
|
||||
render3d_st.sky_irradiance = tex_target(render3d_st, 128, 64, GL_RGB16F, GL_RGB, GL_FLOAT, GL_LINEAR)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, render3d_st.sky_irradiance)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
sky_convolve(render3d_st, p_irr, render3d_st.sky_irradiance, -1, 128, 64, 0.0)
|
||||
# prefiltered specular: 6 roughness levels, sky_prefilter_w x half each, as a 2D array
|
||||
let p_pre = sky_p_pre
|
||||
sky_prefilter = gpu_tex_new()
|
||||
gpu_tex_bind(GPU_TEX2D_ARRAY, sky_prefilter)
|
||||
gpu_tex_image3d(GL_RGB16F, sky_prefilter_w, sky_prefilter_w / 2, SKY_PREFILTER_LEVELS, GL_RGB, GL_FLOAT, null)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(GPU_TEX2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
let p_pre = render3d_st.sky_p_pre
|
||||
render3d_st.sky_prefilter = gpu_tex_new(render3d_st)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D_ARRAY, render3d_st.sky_prefilter)
|
||||
gpu_tex_image3d(render3d_st, GL_RGB16F, render3d_st.sky_prefilter_w, render3d_st.sky_prefilter_w / 2, SKY_PREFILTER_LEVELS, GL_RGB, GL_FLOAT, null)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
for l in 0 .. SKY_PREFILTER_LEVELS {
|
||||
sky_convolve(p_pre, sky_prefilter, l, sky_prefilter_w, sky_prefilter_w / 2, float(l) / float(SKY_PREFILTER_LEVELS - 1))
|
||||
sky_convolve(render3d_st, p_pre, render3d_st.sky_prefilter, l, render3d_st.sky_prefilter_w, render3d_st.sky_prefilter_w / 2, float(l) / float(SKY_PREFILTER_LEVELS - 1))
|
||||
}
|
||||
# BRDF LUT
|
||||
let p_brdf = sky_p_brdf
|
||||
sky_brdf = tex_target(256, 256, GL_RG16F, GL_RG, GL_FLOAT, GL_LINEAR)
|
||||
sky_convolve(p_brdf, sky_brdf, -1, 256, 256, 0.0)
|
||||
sky_baked = true
|
||||
sky_baked_yaw = sky_yaw
|
||||
gpu_check("sky precompute")
|
||||
let p_brdf = render3d_st.sky_p_brdf
|
||||
render3d_st.sky_brdf = tex_target(render3d_st, 256, 256, GL_RG16F, GL_RG, GL_FLOAT, GL_LINEAR)
|
||||
sky_convolve(render3d_st, p_brdf, render3d_st.sky_brdf, -1, 256, 256, 0.0)
|
||||
render3d_st.sky_baked = true
|
||||
render3d_st.sky_baked_yaw = render3d_st.sky_yaw
|
||||
gpu_check(render3d_st, "sky precompute")
|
||||
}
|
||||
|
||||
# bind the IBL set + sun for a lit program (units 12..14)
|
||||
function sky_bind_lighting(prog: int) -> void {
|
||||
r3d_bind_2d(prog, "u_irradiance", 12, sky_irradiance)
|
||||
r3d_bind_tex(prog, "u_prefilter", 13, GPU_TEX2D_ARRAY, sky_prefilter)
|
||||
r3d_bind_2d(prog, "u_brdf", 14, sky_brdf)
|
||||
u_v3(gpu_uniform(prog, "u_sun_dir"), sun_dir)
|
||||
u_v3(gpu_uniform(prog, "u_sun_color"), sun_color)
|
||||
u_v3(gpu_uniform(prog, "u_cam_pos"), cam_pos)
|
||||
u_f(gpu_uniform(prog, "u_prefilter_levels"), float(SKY_PREFILTER_LEVELS))
|
||||
daylight_bind(prog)
|
||||
function sky_bind_lighting(render3d_st: mut Render3dState, prog: int) -> void {
|
||||
r3d_bind_2d(render3d_st, prog, "u_irradiance", 12, render3d_st.sky_irradiance)
|
||||
r3d_bind_tex(render3d_st, prog, "u_prefilter", 13, GPU_TEX2D_ARRAY, render3d_st.sky_prefilter)
|
||||
r3d_bind_2d(render3d_st, prog, "u_brdf", 14, render3d_st.sky_brdf)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_sun_dir"), render3d_st.sun_dir)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_sun_color"), render3d_st.sun_color)
|
||||
u_v3(render3d_st, gpu_uniform(render3d_st, prog, "u_cam_pos"), render3d_st.cam_pos)
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, prog, "u_prefilter_levels"), float(SKY_PREFILTER_LEVELS))
|
||||
daylight_bind(render3d_st, prog)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -12,7 +12,6 @@
|
|||
# stream_emit(...) per instance. One Stream drives one Layer.
|
||||
# ============================================================================
|
||||
|
||||
var STREAM_MAX_CHUNKS: int = 4096
|
||||
|
||||
property Chunk {
|
||||
key: int = 0, # packed (cx, cz, band)
|
||||
|
|
@ -40,22 +39,13 @@ property Stream {
|
|||
htab: words # open-addressed key -> chunk index + 1 (0 = empty)
|
||||
}
|
||||
|
||||
var stream_all: []Stream = null
|
||||
var stream_cap_read: bool = false
|
||||
var stream_no_evict: bool = false # R3D_NOEVICT: the old behaviour, for comparison
|
||||
var stream_walk_no: int = 0 # counts ring walks; a chunk's age is measured in these
|
||||
var stream_evictions: int = 0
|
||||
# microseconds spent per frame, split so the hitch can be attributed (R3D_PROF=1)
|
||||
var stream_us_gen: long = 0 # generating new chunks (stream_fill)
|
||||
var stream_us_gather: long = 0 # copying cached chunks into the layer buffer
|
||||
var stream_us_walk: long = 0 # the ring walk itself
|
||||
var stream_walks: int = 0 # streams that walked their whole ring this frame
|
||||
|
||||
function stream_new(layer: Layer, size: float, reach: float, b0: float, b1: float, b2: float, b3: float) -> Stream {
|
||||
if not stream_cap_read {
|
||||
stream_cap_read = true
|
||||
if r3d_env_has("R3D_STREAM_CAP") { STREAM_MAX_CHUNKS = Text.to_int(r3d_env("R3D_STREAM_CAP")) }
|
||||
stream_no_evict = r3d_env_has("R3D_NOEVICT")
|
||||
function stream_new(render3d_st: mut Render3dState, layer: Layer, size: float, reach: float, b0: float, b1: float, b2: float, b3: float) -> Stream {
|
||||
if not render3d_st.stream_cap_read {
|
||||
render3d_st.stream_cap_read = true
|
||||
if r3d_env_has(render3d_st, "R3D_STREAM_CAP") { render3d_st.STREAM_MAX_CHUNKS = Text.to_int(r3d_env(render3d_st, "R3D_STREAM_CAP")) }
|
||||
render3d_st.stream_no_evict = r3d_env_has(render3d_st, "R3D_NOEVICT")
|
||||
}
|
||||
let s = new Stream
|
||||
s.layer = layer; s.size = size; s.reach = reach
|
||||
|
|
@ -64,11 +54,11 @@ function stream_new(layer: Layer, size: float, reach: float, b0: float, b1: floa
|
|||
s.bands = floats(4)
|
||||
s.bands[0] = b0; s.bands[1] = b1; s.bands[2] = b2; s.bands[3] = b3
|
||||
s.chunks = new []Chunk
|
||||
s.keys = words(STREAM_MAX_CHUNKS)
|
||||
s.keys = words(render3d_st.STREAM_MAX_CHUNKS)
|
||||
s.htab = words(STREAM_HASH)
|
||||
for i in 0 .. STREAM_HASH { s.htab[i] = 0 }
|
||||
if stream_all == null { stream_all = new []Stream }
|
||||
push(stream_all, s)
|
||||
if render3d_st.stream_all == null { render3d_st.stream_all = new []Stream }
|
||||
push(render3d_st.stream_all, s)
|
||||
return s
|
||||
}
|
||||
|
||||
|
|
@ -107,16 +97,14 @@ function stream_remember(s: Stream, key: int, idx: int) -> void {
|
|||
# the generator adds instances to the chunk being filled (into a shared scratch; the
|
||||
# chunk gets an exactly-sized copy when the fill ends)
|
||||
const STREAM_CHUNK_MAX: int = 262144
|
||||
var stream_debug_n: int = 0
|
||||
var stream_scratch: floats = null
|
||||
function stream_emit(s: Stream, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void {
|
||||
function stream_emit(render3d_st: mut Render3dState, s: Stream, x: float, y: float, z: float, scale: float, yaw: float, seed: float, wind: float) -> void {
|
||||
let c = s.cur
|
||||
if c.count >= STREAM_CHUNK_MAX { return }
|
||||
if stream_scratch == null { stream_scratch = floats(STREAM_CHUNK_MAX * INST_FLOATS) }
|
||||
if render3d_st.stream_scratch == null { render3d_st.stream_scratch = floats(STREAM_CHUNK_MAX * INST_FLOATS) }
|
||||
if c.count == 0 { c.ymin = y; c.ymax = y } else { c.ymin = Math.min(c.ymin, y); c.ymax = Math.max(c.ymax, y) }
|
||||
let o = c.count * INST_FLOATS
|
||||
stream_scratch[o] = x; stream_scratch[o + 1] = y; stream_scratch[o + 2] = z; stream_scratch[o + 3] = scale
|
||||
stream_scratch[o + 4] = Math.sin(yaw); stream_scratch[o + 5] = Math.cos(yaw); stream_scratch[o + 6] = seed; stream_scratch[o + 7] = wind
|
||||
render3d_st.stream_scratch[o] = x; render3d_st.stream_scratch[o + 1] = y; render3d_st.stream_scratch[o + 2] = z; render3d_st.stream_scratch[o + 3] = scale
|
||||
render3d_st.stream_scratch[o + 4] = Math.sin(yaw); render3d_st.stream_scratch[o + 5] = Math.cos(yaw); render3d_st.stream_scratch[o + 6] = seed; render3d_st.stream_scratch[o + 7] = wind
|
||||
c.count += 1
|
||||
}
|
||||
|
||||
|
|
@ -143,14 +131,11 @@ function stream_band(s: Stream, d: float) -> int {
|
|||
# by band and kind. With a real microsecond clock the budget can just be the thing we
|
||||
# actually care about — how long this frame is allowed to spend growing ground cover.
|
||||
# Overshoot is bounded by one chunk, so keep chunks small on the dense near streams.
|
||||
var STREAM_BUDGET_US: int = 2500 # microseconds of generation per frame; a setting may move it
|
||||
var stream_deadline: long = 0
|
||||
const STREAM_BUDGET: int = 8000 # kept for the work counter only
|
||||
# The worst frame is now bounded by one chunk, not by the budget: stream_fill emits a
|
||||
# whole chunk in one call, and the densest band-0 chunk is ~114k instances. Splitting a
|
||||
# chunk's generation across frames would need a resumable generator contract; that is
|
||||
# the next step if the residual hitch ever matters.
|
||||
var stream_budget_left: int = 0
|
||||
|
||||
# Drop the half of the cache nobody has asked for in the longest time, and rebuild the
|
||||
# index over what is left.
|
||||
|
|
@ -160,10 +145,10 @@ var stream_budget_left: int = 0
|
|||
# — it is a cliff. Past it every frame pays the whole generation budget and the ground
|
||||
# visibly re-grows as you turn, and it arrives after enough of the map has been walked,
|
||||
# which is exactly when a player is least likely to connect it to anything.
|
||||
function stream_evict(s: Stream) -> void {
|
||||
function stream_evict(render3d_st: mut Render3dState, s: Stream) -> void {
|
||||
# the age threshold that keeps about half, found by bisection on the count (no sort)
|
||||
var lo = 0
|
||||
var hi = stream_walk_no
|
||||
var hi = render3d_st.stream_walk_no
|
||||
var keep = s.n / 2
|
||||
var t = 0
|
||||
var it = 0
|
||||
|
|
@ -181,7 +166,7 @@ function stream_evict(s: Stream) -> void {
|
|||
var i = 0
|
||||
while i < s.n {
|
||||
let c = s.chunks[i]
|
||||
if c.used >= t or c.used == stream_walk_no { push(kept, c) }
|
||||
if c.used >= t or c.used == render3d_st.stream_walk_no { push(kept, c) }
|
||||
else { if c.data != null { free(c.data) } }
|
||||
i += 1
|
||||
}
|
||||
|
|
@ -190,21 +175,21 @@ function stream_evict(s: Stream) -> void {
|
|||
for h in 0 .. STREAM_HASH { s.htab[h] = 0 }
|
||||
i = 0
|
||||
while i < s.n { s.keys[i] = s.chunks[i].key; stream_remember(s, s.chunks[i].key, i); i += 1 }
|
||||
stream_evictions += 1
|
||||
render3d_st.stream_evictions += 1
|
||||
}
|
||||
|
||||
function stream_update(s: Stream, cam_x: float, cam_z: float) -> void {
|
||||
function stream_update(render3d_st: mut Render3dState, s: Stream, cam_x: float, cam_z: float) -> void {
|
||||
let ccx = int(Math.floor(cam_x / s.size))
|
||||
let ccz = int(Math.floor(cam_z / s.size))
|
||||
if ccx == s.last_cx and ccz == s.last_cz and not s.pending and s.view_gen == sc_view_gen { return }
|
||||
if ccx == s.last_cx and ccz == s.last_cz and not s.pending and s.view_gen == render3d_st.sc_view_gen { return }
|
||||
let first = s.last_cx == 999999
|
||||
s.view_gen = sc_view_gen
|
||||
s.view_gen = render3d_st.sc_view_gen
|
||||
s.last_cx = ccx; s.last_cz = ccz
|
||||
let l = s.layer
|
||||
l.count = 0
|
||||
var missing = false
|
||||
stream_walks += 1
|
||||
stream_walk_no += 1
|
||||
render3d_st.stream_walks += 1
|
||||
render3d_st.stream_walk_no += 1
|
||||
let tw = gl_now_us()
|
||||
let r = int(s.reach / s.size) + 1
|
||||
# rings outward from the camera's cell: the nearest chunks are generated first
|
||||
|
|
@ -224,31 +209,31 @@ function stream_update(s: Stream, cam_x: float, cam_z: float) -> void {
|
|||
if band < 4 and band >= s.min_band and d < s.reach + s.size {
|
||||
let key = stream_key(cx, cz, band)
|
||||
var c = stream_find(s, key)
|
||||
if c != null { c.used = stream_walk_no }
|
||||
if c != null { c.used = render3d_st.stream_walk_no }
|
||||
# The cell underfoot and its neighbours are never deferred: they are what you
|
||||
# are looking at, and a hole there is the grass vanishing as you walk into it.
|
||||
let urgent = band == 0 and ring <= 1
|
||||
if c == null and (first or urgent or gl_now_us() < stream_deadline) {
|
||||
if c == null and (first or urgent or gl_now_us() < render3d_st.stream_deadline) {
|
||||
c = new Chunk
|
||||
c.key = key
|
||||
s.cur = c
|
||||
let t0 = gl_now_us()
|
||||
r3d_stream_fill(s, cx, cz, band)
|
||||
r3d_stream_fill(render3d_st, s, cx, cz, band)
|
||||
let dt = gl_now_us() - t0
|
||||
stream_us_gen = stream_us_gen + dt
|
||||
prof_chunk(s.kind, band, c.count, dt)
|
||||
if r3d_debug and band == 0 and stream_debug_n < 40 { stream_debug_n += 1; print(`stream kind {s.kind} band {band} chunk {cx},{cz}: {c.count} instances`) }
|
||||
if c.count > 0 { c.data = words(c.count * INST_FLOATS); mem_copy(c.data, stream_scratch, c.count * INST_FLOATS * 4) }
|
||||
if s.n >= STREAM_MAX_CHUNKS and not stream_no_evict { stream_evict(s) }
|
||||
render3d_st.stream_us_gen = render3d_st.stream_us_gen + dt
|
||||
prof_chunk(render3d_st, s.kind, band, c.count, dt)
|
||||
if render3d_st.r3d_debug and band == 0 and render3d_st.stream_debug_n < 40 { render3d_st.stream_debug_n += 1; print(`stream kind {s.kind} band {band} chunk {cx},{cz}: {c.count} instances`) }
|
||||
if c.count > 0 { c.data = words(c.count * INST_FLOATS); mem_copy(c.data, render3d_st.stream_scratch, c.count * INST_FLOATS * 4) }
|
||||
if s.n >= render3d_st.STREAM_MAX_CHUNKS and not render3d_st.stream_no_evict { stream_evict(render3d_st, s) }
|
||||
# If the walk in progress wants more chunks than the cache can hold, there
|
||||
# is nothing to evict and this one is used and dropped, as every chunk used
|
||||
# to be. The cap has to exceed one walk's ring for the cache to work at all.
|
||||
if s.n < STREAM_MAX_CHUNKS {
|
||||
if s.n < render3d_st.STREAM_MAX_CHUNKS {
|
||||
push(s.chunks, c); s.keys[s.n] = key; stream_remember(s, key, s.n); s.n += 1
|
||||
c.used = stream_walk_no
|
||||
c.used = render3d_st.stream_walk_no
|
||||
}
|
||||
|
||||
prof_gen_add(c.count + 512)
|
||||
prof_gen_add(render3d_st, c.count + 512)
|
||||
}
|
||||
if c == null {
|
||||
missing = true
|
||||
|
|
@ -257,13 +242,13 @@ function stream_update(s: Stream, cam_x: float, cam_z: float) -> void {
|
|||
# few frames instead of disappearing.
|
||||
var b2 = band + 1
|
||||
while c == null and b2 < 4 { c = stream_find(s, stream_key(cx, cz, b2)); b2 += 1 }
|
||||
if c != null { c.used = stream_walk_no }
|
||||
if c != null { c.used = render3d_st.stream_walk_no }
|
||||
}
|
||||
if c != null and c.count > 0 and l.count + c.count <= l.cap and stream_chunk_visible(s, cx, cz, c) {
|
||||
if c != null and c.count > 0 and l.count + c.count <= l.cap and stream_chunk_visible(render3d_st, s, cx, cz, c) {
|
||||
let tg = gl_now_us()
|
||||
mem_copy(mem_off(l.inst, l.count * INST_FLOATS * 4), data_of(c.data), c.count * INST_FLOATS * 4)
|
||||
l.count += c.count
|
||||
stream_us_gather = stream_us_gather + (gl_now_us() - tg)
|
||||
render3d_st.stream_us_gather = render3d_st.stream_us_gather + (gl_now_us() - tg)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -274,7 +259,7 @@ function stream_update(s: Stream, cam_x: float, cam_z: float) -> void {
|
|||
ring += 1
|
||||
}
|
||||
s.pending = missing
|
||||
stream_us_walk = stream_us_walk + (gl_now_us() - tw)
|
||||
render3d_st.stream_us_walk = render3d_st.stream_us_walk + (gl_now_us() - tw)
|
||||
# force the layer to re-partition its (new) instances
|
||||
l.view_gen = -1
|
||||
}
|
||||
|
|
@ -282,24 +267,24 @@ function stream_update(s: Stream, cam_x: float, cam_z: float) -> void {
|
|||
# Only chunks that can be seen are gathered: a sphere around the chunk's footprint and
|
||||
# height range, padded for the tallest cover and for casters just outside the frame
|
||||
# whose short shadows still fall inside it.
|
||||
function stream_chunk_visible(s: Stream, cx: int, cz: int, c: Chunk) -> bool {
|
||||
function stream_chunk_visible(render3d_st: Render3dState, s: Stream, cx: int, cz: int, c: Chunk) -> bool {
|
||||
let half = s.size * 0.5
|
||||
let wx = float(cx) * s.size + half
|
||||
let wz = float(cz) * s.size + half
|
||||
let hy = (c.ymax - c.ymin) * 0.5
|
||||
let cy = c.ymin + hy
|
||||
let r = Math.sqrt(half * half * 2.0 + hy * hy) + 8.0
|
||||
return cam_sphere_visible(wx, cy, wz, r)
|
||||
return cam_sphere_visible(render3d_st, wx, cy, wz, r)
|
||||
}
|
||||
|
||||
# what the caches hold, and whether they are being churned (R3D_PROF)
|
||||
function stream_census() -> void {
|
||||
if stream_all == null { return }
|
||||
function stream_census(render3d_st: Render3dState) -> void {
|
||||
if render3d_st.stream_all == null { return }
|
||||
print("")
|
||||
print(`ground-cover chunk caches (cap {string(STREAM_MAX_CHUNKS)} each, {string(stream_evictions)} evictions over the run):`)
|
||||
print(`ground-cover chunk caches (cap {string(render3d_st.STREAM_MAX_CHUNKS)} each, {string(render3d_st.stream_evictions)} evictions over the run):`)
|
||||
var inst = 0
|
||||
for i in 0 .. len(stream_all) {
|
||||
let s = stream_all[i]
|
||||
for i in 0 .. len(render3d_st.stream_all) {
|
||||
let s = render3d_st.stream_all[i]
|
||||
var n = 0
|
||||
for k in 0 .. s.n { n += s.chunks[k].count }
|
||||
inst += n
|
||||
|
|
@ -308,21 +293,21 @@ function stream_census() -> void {
|
|||
print(` {string(inst)} instances held, {string(inst * INST_FLOATS * 4 / 1024)} KB`)
|
||||
}
|
||||
|
||||
function stream_update_all() -> void {
|
||||
if stream_all == null { return }
|
||||
stream_deadline = gl_now_us() + STREAM_BUDGET_US
|
||||
for i in 0 .. len(stream_all) { stream_update(stream_all[i], cam_pos[0], cam_pos[2]) }
|
||||
function stream_update_all(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.stream_all == null { return }
|
||||
render3d_st.stream_deadline = gl_now_us() + render3d_st.STREAM_BUDGET_US
|
||||
for i in 0 .. len(render3d_st.stream_all) { stream_update(render3d_st, render3d_st.stream_all[i], render3d_st.cam_pos[0], render3d_st.cam_pos[2]) }
|
||||
}
|
||||
|
||||
# every stream and its cached chunks, for a world being replaced (scatter_clear_all)
|
||||
function stream_clear_all() -> void {
|
||||
if stream_all == null { return }
|
||||
for i in 0 .. len(stream_all) {
|
||||
let s = stream_all[i]
|
||||
function stream_clear_all(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.stream_all == null { return }
|
||||
for i in 0 .. len(render3d_st.stream_all) {
|
||||
let s = render3d_st.stream_all[i]
|
||||
if s.chunks != null { for c in 0 .. len(s.chunks) { if s.chunks[c].data != null { free(s.chunks[c].data) } } }
|
||||
if s.keys != null { free(s.keys) }
|
||||
if s.htab != null { free(s.htab) }
|
||||
if s.bands != null { free(s.bands) }
|
||||
}
|
||||
stream_all = null
|
||||
render3d_st.stream_all = null
|
||||
}
|
||||
|
|
|
|||
|
|
@ -30,22 +30,7 @@ const GSL_PRESENT_END: int = 5
|
|||
|
||||
const GSL_LAYOUT_READ: int = 5 # VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL
|
||||
|
||||
var gsl_on: bool = false # slInit succeeded: the interposer is the loader, the plugins are in
|
||||
var gsl_dlss_ok: bool = false # what this adapter can run, from slIsFeatureSupported
|
||||
var gsl_rr_ok: bool = false
|
||||
var gsl_fg_ok: bool = false
|
||||
var gsl_reflex_ok: bool = false
|
||||
var gsl_pcl_ok: bool = false
|
||||
|
||||
var gsl_feats: bytes = null # kept alive: Preferences points at it
|
||||
var gsl_logdir: bytes = null # and at this
|
||||
var gsl_vp: bytes = null # sl::ViewportHandle 0, the one view
|
||||
var gsl_tok_buf: bytes = null
|
||||
var gsl_idx_buf: bytes = null
|
||||
var gsl_token: pointer = null # this frame's sl::FrameToken, owned by Streamline
|
||||
var gsl_frame_n: int = 0
|
||||
var gsl_said: bool = false # one failure message, not one a frame
|
||||
var gsl_fresh: bool = false # a token was taken since the last frame started (a present took it)
|
||||
|
||||
# ---- structure headers ------------------------------------------------------------------
|
||||
function gsl_le16(v: int) -> int { return ((v >> 8) & 255) | ((v & 255) << 8) }
|
||||
|
|
@ -74,36 +59,36 @@ function gsl_fn(feature: int, name: string) -> pointer {
|
|||
|
||||
# ---- start and stop ---------------------------------------------------------------------
|
||||
# Before Vk.open(): ask the loader for the interposer. R3D_NO_STREAMLINE=1 keeps plain Vulkan.
|
||||
function gsl_boot() -> void {
|
||||
function gsl_boot(render3d_st: mut Render3dState) -> void {
|
||||
if Os.platform() != "windows" { return }
|
||||
if r3d_env_has("R3D_NO_STREAMLINE") { return }
|
||||
if r3d_env_has(render3d_st, "R3D_NO_STREAMLINE") { return }
|
||||
Vk.sl_prefer(1)
|
||||
}
|
||||
|
||||
# After Vk.open(), before vkCreateInstance.
|
||||
function gsl_init() -> void {
|
||||
if gsl_on or Vk.sl_active() == 0 { return }
|
||||
gsl_feats = gsl_struct(20)
|
||||
Vk.put_i32(gsl_feats, 0, GSL_DLSS); Vk.put_i32(gsl_feats, 4, GSL_REFLEX); Vk.put_i32(gsl_feats, 8, GSL_PCL)
|
||||
Vk.put_i32(gsl_feats, 12, GSL_DLSS_RR); Vk.put_i32(gsl_feats, 16, GSL_DLSS_G)
|
||||
function gsl_init(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gsl_on or Vk.sl_active() == 0 { return }
|
||||
render3d_st.gsl_feats = gsl_struct(20)
|
||||
Vk.put_i32(render3d_st.gsl_feats, 0, GSL_DLSS); Vk.put_i32(render3d_st.gsl_feats, 4, GSL_REFLEX); Vk.put_i32(render3d_st.gsl_feats, 8, GSL_PCL)
|
||||
Vk.put_i32(render3d_st.gsl_feats, 12, GSL_DLSS_RR); Vk.put_i32(render3d_st.gsl_feats, 16, GSL_DLSS_G)
|
||||
# sl::Preferences
|
||||
let pref = gsl_struct(144)
|
||||
gsl_header(pref, 0x1ca1, 0x0965, 0xbf8e, 0x432b, 0x8da1, 0x6716, 0xd879, 0xfb14, 1)
|
||||
# logLevel eOff; R3D_SL_LOG=<directory> writes Streamline's own verbose log (sl.log) there
|
||||
var level = 0
|
||||
if r3d_env_has("R3D_SL_LOG") {
|
||||
if r3d_env_has(render3d_st, "R3D_SL_LOG") {
|
||||
level = 2
|
||||
let dir: pointer = r3d_env("R3D_SL_LOG")
|
||||
let n = Text.length(r3d_env("R3D_SL_LOG"))
|
||||
gsl_logdir = gsl_struct(n * 2 + 8) # pathToLogsAndData is a wide string
|
||||
for i in 0 .. n { Vk.put_i32(gsl_logdir, i * 2, dir[i] & 255) }
|
||||
Vk.put_ptr(pref, 56, gsl_logdir)
|
||||
let dir: pointer = r3d_env(render3d_st, "R3D_SL_LOG")
|
||||
let n = Text.length(r3d_env(render3d_st, "R3D_SL_LOG"))
|
||||
render3d_st.gsl_logdir = gsl_struct(n * 2 + 8) # pathToLogsAndData is a wide string
|
||||
for i in 0 .. n { Vk.put_i32(render3d_st.gsl_logdir, i * 2, dir[i] & 255) }
|
||||
Vk.put_ptr(pref, 56, render3d_st.gsl_logdir)
|
||||
}
|
||||
Vk.put_i32(pref, 36, level)
|
||||
# eDisableCLStateTracking | eAllowOTA | eLoadDownloadedPlugins | eUseFrameBasedResourceTagging
|
||||
let flags: long = 1 | 8 | 64 | 128
|
||||
Vk.put_i64(pref, 88, flags)
|
||||
Vk.put_ptr(pref, 96, gsl_feats)
|
||||
Vk.put_ptr(pref, 96, render3d_st.gsl_feats)
|
||||
Vk.put_i32(pref, 104, 5)
|
||||
Vk.put_i32(pref, 112, 0) # engine eCustom
|
||||
Vk.put_ptr(pref, 120, "ludic render3d")
|
||||
|
|
@ -111,11 +96,11 @@ function gsl_init() -> void {
|
|||
Vk.put_i32(pref, 136, 2) # renderAPI eVulkan
|
||||
let r = Vk.sl_init(pref, gsl_sdk_version())
|
||||
if r != 0 { print(`r3d: streamline: slInit failed ({r}); DLSS and Reflex are off`); return }
|
||||
gsl_on = true
|
||||
gsl_vp = gsl_struct(40)
|
||||
gsl_header(gsl_vp, 0x171b, 0x6435, 0x9b3c, 0x4fc8, 0x9994, 0xfbe5, 0x2569, 0xaaa4, 1)
|
||||
gsl_tok_buf = bytes(8)
|
||||
gsl_idx_buf = bytes(4)
|
||||
render3d_st.gsl_on = true
|
||||
render3d_st.gsl_vp = gsl_struct(40)
|
||||
gsl_header(render3d_st.gsl_vp, 0x171b, 0x6435, 0x9b3c, 0x4fc8, 0x9994, 0xfbe5, 0x2569, 0xaaa4, 1)
|
||||
render3d_st.gsl_tok_buf = bytes(8)
|
||||
render3d_st.gsl_idx_buf = bytes(4)
|
||||
}
|
||||
|
||||
# kSDKVersion, built from longs: (2 << 48) on ints is computed in 32 bits and arrives as garbage,
|
||||
|
|
@ -129,90 +114,86 @@ function gsl_sdk_version() -> long {
|
|||
}
|
||||
|
||||
# After vkCreateDevice: what this adapter supports.
|
||||
function gsl_probe_device(pd: pointer) -> void {
|
||||
if not gsl_on { return }
|
||||
function gsl_probe_device(render3d_st: mut Render3dState, pd: pointer) -> void {
|
||||
if not render3d_st.gsl_on { return }
|
||||
let ai = gsl_struct(56) # sl::AdapterInfo
|
||||
gsl_header(ai, 0x0677, 0x315f, 0xa746, 0x4492, 0x9f42, 0xcb61, 0x42c9, 0xc3d4, 1)
|
||||
Vk.put_ptr(ai, 48, pd)
|
||||
gsl_dlss_ok = Vk.sl_is_feature_supported(GSL_DLSS, ai) == 0
|
||||
gsl_rr_ok = Vk.sl_is_feature_supported(GSL_DLSS_RR, ai) == 0
|
||||
gsl_fg_ok = Vk.sl_is_feature_supported(GSL_DLSS_G, ai) == 0
|
||||
gsl_reflex_ok = Vk.sl_is_feature_supported(GSL_REFLEX, ai) == 0
|
||||
gsl_pcl_ok = Vk.sl_is_feature_supported(GSL_PCL, ai) == 0
|
||||
print(`r3d: streamline: DLSS {gsl_dlss_ok}, ray reconstruction {gsl_rr_ok}, frame generation {gsl_fg_ok}, Reflex {gsl_reflex_ok}`)
|
||||
render3d_st.gsl_dlss_ok = Vk.sl_is_feature_supported(GSL_DLSS, ai) == 0
|
||||
render3d_st.gsl_rr_ok = Vk.sl_is_feature_supported(GSL_DLSS_RR, ai) == 0
|
||||
render3d_st.gsl_fg_ok = Vk.sl_is_feature_supported(GSL_DLSS_G, ai) == 0
|
||||
render3d_st.gsl_reflex_ok = Vk.sl_is_feature_supported(GSL_REFLEX, ai) == 0
|
||||
render3d_st.gsl_pcl_ok = Vk.sl_is_feature_supported(GSL_PCL, ai) == 0
|
||||
print(`r3d: streamline: DLSS {render3d_st.gsl_dlss_ok}, ray reconstruction {render3d_st.gsl_rr_ok}, frame generation {render3d_st.gsl_fg_ok}, Reflex {render3d_st.gsl_reflex_ok}`)
|
||||
}
|
||||
|
||||
# Before the device goes.
|
||||
function gsl_shutdown() -> void {
|
||||
if not gsl_on { return }
|
||||
function gsl_shutdown(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.gsl_on { return }
|
||||
Vk.sl_shutdown()
|
||||
gsl_on = false
|
||||
render3d_st.gsl_on = false
|
||||
}
|
||||
|
||||
# ---- frames, Reflex and the latency markers ---------------------------------------------
|
||||
# A frame's token is taken straight after the previous present, where Reflex sleeps: the
|
||||
# wait lands before the game reads input and simulates, which is the latency it removes.
|
||||
var gsl_reflex_mode: int = 0 # 0 off, 1 low latency, 2 low latency + boost
|
||||
var gsl_reflex_applied: int = -1
|
||||
var gsl_f_sleep: pointer = null
|
||||
var gsl_f_marker: pointer = null
|
||||
|
||||
function r3d_reflex(mode: int) -> void {
|
||||
gsl_reflex_mode = mode
|
||||
if r3d_env_has("R3D_REFLEX") { gsl_reflex_mode = Text.to_int(r3d_env("R3D_REFLEX")) }
|
||||
function r3d_reflex(render3d_st: mut Render3dState, mode: int) -> void {
|
||||
render3d_st.gsl_reflex_mode = mode
|
||||
if r3d_env_has(render3d_st, "R3D_REFLEX") { render3d_st.gsl_reflex_mode = Text.to_int(r3d_env(render3d_st, "R3D_REFLEX")) }
|
||||
}
|
||||
function r3d_reflex_live() -> bool { return gsl_on and gsl_reflex_ok and gsl_reflex_mode > 0 }
|
||||
function r3d_reflex_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_reflex_ok and render3d_st.gsl_reflex_mode > 0 }
|
||||
|
||||
function gsl_marker(m: int) -> void {
|
||||
if not gsl_pcl_ok or gsl_token == null { return }
|
||||
if gsl_f_marker == null { gsl_f_marker = gsl_fn(GSL_PCL, "slPCLSetMarker") }
|
||||
if gsl_f_marker != null { Vk.sl_call_ip(gsl_f_marker, m, gsl_token) }
|
||||
function gsl_marker(render3d_st: mut Render3dState, m: int) -> void {
|
||||
if not render3d_st.gsl_pcl_ok or render3d_st.gsl_token == null { return }
|
||||
if render3d_st.gsl_f_marker == null { render3d_st.gsl_f_marker = gsl_fn(GSL_PCL, "slPCLSetMarker") }
|
||||
if render3d_st.gsl_f_marker != null { Vk.sl_call_ip(render3d_st.gsl_f_marker, m, render3d_st.gsl_token) }
|
||||
}
|
||||
|
||||
function gsl_reflex_apply() -> void {
|
||||
if not gsl_reflex_ok or gsl_reflex_applied == gsl_reflex_mode { return }
|
||||
function gsl_reflex_apply(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.gsl_reflex_ok or render3d_st.gsl_reflex_applied == render3d_st.gsl_reflex_mode { return }
|
||||
let f = gsl_fn(GSL_REFLEX, "slReflexSetOptions")
|
||||
if f == null { return }
|
||||
let o = gsl_struct(48) # sl::ReflexOptions
|
||||
gsl_header(o, 0xf03a, 0xf81a, 0x6d0b, 0x4902, 0xa651, 0xc496, 0x5e21, 0x5434, 1)
|
||||
Vk.put_i32(o, 32, gsl_reflex_mode)
|
||||
if gsl_pcl_ok { Vk.put_i32(o, 40, 1) } # useMarkersToOptimize
|
||||
if Vk.sl_call_p(f, o) == 0 { gsl_reflex_applied = gsl_reflex_mode }
|
||||
Vk.put_i32(o, 32, render3d_st.gsl_reflex_mode)
|
||||
if render3d_st.gsl_pcl_ok { Vk.put_i32(o, 40, 1) } # useMarkersToOptimize
|
||||
if Vk.sl_call_p(f, o) == 0 { render3d_st.gsl_reflex_applied = render3d_st.gsl_reflex_mode }
|
||||
}
|
||||
|
||||
function gsl_new_frame() -> void {
|
||||
Vk.put_i32(gsl_idx_buf, 0, gsl_frame_n)
|
||||
gsl_frame_n += 1
|
||||
gsl_token = null
|
||||
gsl_fresh = true
|
||||
if Vk.sl_get_new_frame_token(gsl_tok_buf, gsl_idx_buf) == 0 { gsl_token = Vk.get_ptr(gsl_tok_buf, 0) }
|
||||
if gsl_token == null { return }
|
||||
gsl_reflex_apply()
|
||||
if r3d_reflex_live() {
|
||||
if gsl_f_sleep == null { gsl_f_sleep = gsl_fn(GSL_REFLEX, "slReflexSleep") }
|
||||
if gsl_f_sleep != null { Vk.sl_call_p(gsl_f_sleep, gsl_token) }
|
||||
function gsl_new_frame(render3d_st: mut Render3dState) -> void {
|
||||
Vk.put_i32(render3d_st.gsl_idx_buf, 0, render3d_st.gsl_frame_n)
|
||||
render3d_st.gsl_frame_n += 1
|
||||
render3d_st.gsl_token = null
|
||||
render3d_st.gsl_fresh = true
|
||||
if Vk.sl_get_new_frame_token(render3d_st.gsl_tok_buf, render3d_st.gsl_idx_buf) == 0 { render3d_st.gsl_token = Vk.get_ptr(render3d_st.gsl_tok_buf, 0) }
|
||||
if render3d_st.gsl_token == null { return }
|
||||
gsl_reflex_apply(render3d_st)
|
||||
if r3d_reflex_live(render3d_st) {
|
||||
if render3d_st.gsl_f_sleep == null { render3d_st.gsl_f_sleep = gsl_fn(GSL_REFLEX, "slReflexSleep") }
|
||||
if render3d_st.gsl_f_sleep != null { Vk.sl_call_p(render3d_st.gsl_f_sleep, render3d_st.gsl_token) }
|
||||
}
|
||||
gsl_marker(GSL_SIM_START)
|
||||
gsl_marker(render3d_st, GSL_SIM_START)
|
||||
}
|
||||
|
||||
# r3d_frame's first line: the game has simulated, the renderer starts recording
|
||||
function gsl_frame_start() -> void {
|
||||
if not gsl_on or gpu_kind != GPU_VK { return }
|
||||
function gsl_frame_start(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.gsl_on or render3d_st.gpu_kind != GPU_VK { return }
|
||||
# a headless run never presents: each frame takes its own token here
|
||||
if not gsl_fresh { gsl_new_frame() }
|
||||
gsl_fresh = false
|
||||
gsl_marker(GSL_SIM_END)
|
||||
gsl_marker(GSL_SUBMIT_START)
|
||||
if not render3d_st.gsl_fresh { gsl_new_frame(render3d_st) }
|
||||
render3d_st.gsl_fresh = false
|
||||
gsl_marker(render3d_st, GSL_SIM_END)
|
||||
gsl_marker(render3d_st, GSL_SUBMIT_START)
|
||||
}
|
||||
function gsl_before_present() -> void {
|
||||
if not gsl_on { return }
|
||||
gsl_marker(GSL_SUBMIT_END)
|
||||
gsl_marker(GSL_PRESENT_START)
|
||||
function gsl_before_present(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.gsl_on { return }
|
||||
gsl_marker(render3d_st, GSL_SUBMIT_END)
|
||||
gsl_marker(render3d_st, GSL_PRESENT_START)
|
||||
}
|
||||
function gsl_after_present() -> void {
|
||||
if not gsl_on { return }
|
||||
gsl_marker(GSL_PRESENT_END)
|
||||
gsl_new_frame()
|
||||
function gsl_after_present(render3d_st: mut Render3dState) -> void {
|
||||
if not render3d_st.gsl_on { return }
|
||||
gsl_marker(render3d_st, GSL_PRESENT_END)
|
||||
gsl_new_frame(render3d_st)
|
||||
}
|
||||
|
||||
# ---- DLSS super resolution ----------------------------------------------------------------
|
||||
|
|
@ -221,39 +202,16 @@ function gsl_after_present() -> void {
|
|||
# sharpen pass change size. The projection is jittered on a Halton (2, 3) cycle while DLSS is on.
|
||||
# There are no per-object motion vectors yet: a zero target is tagged and Streamline adds the
|
||||
# camera's own motion from depth and clipToPrevClip.
|
||||
var gsl_dlss_mode: int = 0 # 0 off, 1 DLAA, 2 quality, 3 balanced, 4 performance
|
||||
var gsl_opts: bytes = null
|
||||
var gsl_opt_mode: int = -1
|
||||
var gsl_opt_w: int = 0
|
||||
var gsl_opt_h: int = 0
|
||||
var gsl_rw: int = 0 # the render size DLSS asked for
|
||||
var gsl_rh: int = 0
|
||||
var gsl_set_mode: int = -1
|
||||
var gsl_set_w: int = 0
|
||||
var gsl_set_h: int = 0
|
||||
var gsl_mv: Target = null
|
||||
var gsl_out: Target = null
|
||||
var gsl_consts: bytes = null
|
||||
var gsl_tags: bytes = null
|
||||
var gsl_res: bytes = null
|
||||
var gsl_inputs: bytes = null
|
||||
var gsl_prev_vp: floats = null
|
||||
var gsl_reset: bool = true
|
||||
var gsl_jitter_x: float = 0.0 # float bits, NDC offsets the projection carries this frame
|
||||
var gsl_jitter_y: float = 0.0
|
||||
var gsl_jpx: float = 0.0 # float bits, the same in pixels
|
||||
var gsl_jpy: float = 0.0
|
||||
var gsl_eval_ok: bool = true # the last evaluate worked: only then is the next frame jittered
|
||||
|
||||
# R3D_DLSS=0..4 overrides the setting, for a headless take
|
||||
function r3d_dlss(mode: int) -> void {
|
||||
function r3d_dlss(render3d_st: mut Render3dState, mode: int) -> void {
|
||||
var m = mode
|
||||
if r3d_env_has("R3D_DLSS") { m = Text.to_int(r3d_env("R3D_DLSS")) }
|
||||
if m != gsl_dlss_mode { gsl_reset = true }
|
||||
gsl_dlss_mode = m
|
||||
if r3d_env_has(render3d_st, "R3D_DLSS") { m = Text.to_int(r3d_env(render3d_st, "R3D_DLSS")) }
|
||||
if m != render3d_st.gsl_dlss_mode { render3d_st.gsl_reset = true }
|
||||
render3d_st.gsl_dlss_mode = m
|
||||
}
|
||||
function gsl_dlss_live() -> bool { return gsl_on and gsl_dlss_ok and gsl_dlss_mode > 0 and gpu_kind == GPU_VK and gsl_token != null }
|
||||
function r3d_dlss_live() -> bool { return gsl_on and gsl_dlss_ok and gsl_dlss_mode > 0 and gpu_kind == GPU_VK }
|
||||
function gsl_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK and render3d_st.gsl_token != null }
|
||||
function r3d_dlss_live(render3d_st: Render3dState) -> bool { return render3d_st.gsl_on and render3d_st.gsl_dlss_ok and render3d_st.gsl_dlss_mode > 0 and render3d_st.gpu_kind == GPU_VK }
|
||||
|
||||
# sl::DLSSMode from the setting
|
||||
function gsl_sl_mode(m: int) -> int {
|
||||
|
|
@ -264,43 +222,43 @@ function gsl_sl_mode(m: int) -> int {
|
|||
return 0
|
||||
}
|
||||
|
||||
function gsl_fill_options(w: int, h: int) -> void {
|
||||
if gsl_opts == null { gsl_opts = gsl_struct(88) }
|
||||
Vk.zero(gsl_opts, 88)
|
||||
gsl_header(gsl_opts, 0x6ac8, 0x26e4, 0x4c61, 0x4101, 0xa92d, 0x638d, 0x4210, 0x57b8, 3)
|
||||
Vk.put_i32(gsl_opts, 32, gsl_sl_mode(gsl_dlss_mode))
|
||||
Vk.put_i32(gsl_opts, 36, w)
|
||||
Vk.put_i32(gsl_opts, 40, h)
|
||||
Vk.put_i32(gsl_opts, 48, float_bits(1.0)) # preExposure
|
||||
Vk.put_i32(gsl_opts, 52, float_bits(1.0)) # exposureScale
|
||||
Vk.put_i32(gsl_opts, 56, 1) # colorBuffersHDR eTrue
|
||||
function gsl_fill_options(render3d_st: mut Render3dState, w: int, h: int) -> void {
|
||||
if render3d_st.gsl_opts == null { render3d_st.gsl_opts = gsl_struct(88) }
|
||||
Vk.zero(render3d_st.gsl_opts, 88)
|
||||
gsl_header(render3d_st.gsl_opts, 0x6ac8, 0x26e4, 0x4c61, 0x4101, 0xa92d, 0x638d, 0x4210, 0x57b8, 3)
|
||||
Vk.put_i32(render3d_st.gsl_opts, 32, gsl_sl_mode(render3d_st.gsl_dlss_mode))
|
||||
Vk.put_i32(render3d_st.gsl_opts, 36, w)
|
||||
Vk.put_i32(render3d_st.gsl_opts, 40, h)
|
||||
Vk.put_i32(render3d_st.gsl_opts, 48, float_bits(1.0)) # preExposure
|
||||
Vk.put_i32(render3d_st.gsl_opts, 52, float_bits(1.0)) # exposureScale
|
||||
Vk.put_i32(render3d_st.gsl_opts, 56, 1) # colorBuffersHDR eTrue
|
||||
# The model: preset K (the transformer NVIDIA calls its best image quality) in every mode. The
|
||||
# defaults put Performance on preset M, which on an RTX 3070 Ti at 4K evaluated in 18 ms against
|
||||
# K's 2.8 - slower than no DLSS at all (33 fps against 41; with K, 60). R3D_DLSS_PRESET=<n>
|
||||
# (sl::DLSSPreset: 11 K, 12 L, 13 M) sets every mode's, to measure them against each other.
|
||||
var preset = 11
|
||||
if r3d_env_has("R3D_DLSS_PRESET") { preset = Text.to_int(r3d_env("R3D_DLSS_PRESET")) }
|
||||
if preset > 0 { for f in 0 .. 6 { Vk.put_i32(gsl_opts, 60 + f * 4, preset) } } # dlaa, quality, balanced, performance, ultra performance, ultra quality
|
||||
if r3d_env_has(render3d_st, "R3D_DLSS_PRESET") { preset = Text.to_int(r3d_env(render3d_st, "R3D_DLSS_PRESET")) }
|
||||
if preset > 0 { for f in 0 .. 6 { Vk.put_i32(render3d_st.gsl_opts, 60 + f * 4, preset) } } # dlaa, quality, balanced, performance, ultra performance, ultra quality
|
||||
}
|
||||
|
||||
# the render size for the display's size and the mode, asked once per change
|
||||
function gsl_optimal() -> void {
|
||||
if gsl_opt_mode == gsl_dlss_mode and gsl_opt_w == gl_w and gsl_opt_h == gl_h { return }
|
||||
gsl_opt_mode = gsl_dlss_mode; gsl_opt_w = gl_w; gsl_opt_h = gl_h
|
||||
gsl_rw = gl_w; gsl_rh = gl_h
|
||||
function gsl_optimal(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gsl_opt_mode == render3d_st.gsl_dlss_mode and render3d_st.gsl_opt_w == gl_w and render3d_st.gsl_opt_h == gl_h { return }
|
||||
render3d_st.gsl_opt_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_opt_w = gl_w; render3d_st.gsl_opt_h = gl_h
|
||||
render3d_st.gsl_rw = gl_w; render3d_st.gsl_rh = gl_h
|
||||
let f = gsl_fn(GSL_DLSS, "slDLSSGetOptimalSettings")
|
||||
if f == null { return }
|
||||
gsl_fill_options(gl_w, gl_h)
|
||||
gsl_fill_options(render3d_st, gl_w, gl_h)
|
||||
let os = gsl_struct(64) # sl::DLSSOptimalSettings
|
||||
gsl_header(os, 0xef1d, 0x0957, 0xfd58, 0x4df7, 0xb504, 0x8b69, 0xd8aa, 0x6b76, 1)
|
||||
if Vk.sl_call_pp(f, gsl_opts, os) == 0 {
|
||||
if Vk.sl_call_pp(f, render3d_st.gsl_opts, os) == 0 {
|
||||
let w = Vk.get_i32(os, 32)
|
||||
let h = Vk.get_i32(os, 36)
|
||||
if w > 0 and h > 0 { gsl_rw = w; gsl_rh = h }
|
||||
if w > 0 and h > 0 { render3d_st.gsl_rw = w; render3d_st.gsl_rh = h }
|
||||
}
|
||||
}
|
||||
function r3d_dlss_render_w() -> int { if not r3d_dlss_live() { return gl_w }; gsl_optimal(); return gsl_rw }
|
||||
function r3d_dlss_render_h() -> int { if not r3d_dlss_live() { return gl_h }; gsl_optimal(); return gsl_rh }
|
||||
function r3d_dlss_render_w(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_w }; gsl_optimal(render3d_st); return render3d_st.gsl_rw }
|
||||
function r3d_dlss_render_h(render3d_st: mut Render3dState) -> int { if not r3d_dlss_live(render3d_st) { return gl_h }; gsl_optimal(render3d_st); return render3d_st.gsl_rh }
|
||||
|
||||
# a radical-inverse sample in [0, 1), float bits
|
||||
function gsl_halton(i: int, b: int) -> float {
|
||||
|
|
@ -317,52 +275,52 @@ function gsl_halton(i: int, b: int) -> float {
|
|||
}
|
||||
|
||||
# cam_begin_frame: this frame's sub-pixel offset, before the camera builds its matrices
|
||||
function gsl_jitter_frame() -> void {
|
||||
gsl_jitter_x = 0.0; gsl_jitter_y = 0.0; gsl_jpx = 0.0; gsl_jpy = 0.0
|
||||
function gsl_jitter_frame(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.gsl_jitter_x = 0.0; render3d_st.gsl_jitter_y = 0.0; render3d_st.gsl_jpx = 0.0; render3d_st.gsl_jpy = 0.0
|
||||
# a jittered frame nobody resolves shakes on screen however still the camera is: jitter only while
|
||||
# this frame holds a DLSS token and the last evaluate worked. (Not gsl_fresh: gsl_frame_start has
|
||||
# already taken the token and cleared it by the time the camera asks, which turned jitter off.)
|
||||
if not r3d_dlss_live() or not gsl_eval_ok or gsl_token == null or post_w <= 0 or post_h <= 0 { return }
|
||||
if not r3d_dlss_live(render3d_st) or not render3d_st.gsl_eval_ok or render3d_st.gsl_token == null or render3d_st.post_w <= 0 or render3d_st.post_h <= 0 { return }
|
||||
# DLSS wants at least 8 x (display / render)^2 phases; 32 covers performance mode
|
||||
let i = (gsl_frame_n % 32) + 1
|
||||
gsl_jpx = gsl_halton(i, 2) - 0.5
|
||||
gsl_jpy = gsl_halton(i, 3) - 0.5
|
||||
gsl_jitter_x = 2.0 * gsl_jpx / float(post_w)
|
||||
gsl_jitter_y = 2.0 * gsl_jpy / float(post_h)
|
||||
let i = (render3d_st.gsl_frame_n % 32) + 1
|
||||
render3d_st.gsl_jpx = gsl_halton(i, 2) - 0.5
|
||||
render3d_st.gsl_jpy = gsl_halton(i, 3) - 0.5
|
||||
render3d_st.gsl_jitter_x = 2.0 * render3d_st.gsl_jpx / float(render3d_st.post_w)
|
||||
render3d_st.gsl_jitter_y = 2.0 * render3d_st.gsl_jpy / float(render3d_st.post_h)
|
||||
}
|
||||
|
||||
# sl::Resource for one of the renderer's textures, in the layout every pass leaves them in
|
||||
function gsl_resource(at: int, tex: int) -> void {
|
||||
let p = Vk.at(gsl_res, at)
|
||||
function gsl_resource(render3d_st: Render3dState, at: int, tex: int) -> void {
|
||||
let p = Vk.at(render3d_st.gsl_res, at)
|
||||
Vk.zero(p, 112)
|
||||
gsl_header(p, 0x3a9d, 0x70cf, 0x2418, 0x4b72, 0x8391, 0x13f8, 0x721c, 0x7261, 1)
|
||||
Vk.put_i64(p, 40, gvk_tex_image[tex])
|
||||
Vk.put_i64(p, 48, gvk_mem_handle(gvk_mem_id(gvk_tex_mem[tex])))
|
||||
Vk.put_i64(p, 56, gvk_tex_view[tex])
|
||||
Vk.put_i64(p, 40, render3d_st.gvk_tex_image[tex])
|
||||
Vk.put_i64(p, 48, gvk_mem_handle(render3d_st, gvk_mem_id(render3d_st.gvk_tex_mem[tex])))
|
||||
Vk.put_i64(p, 56, render3d_st.gvk_tex_view[tex])
|
||||
Vk.put_i32(p, 64, GSL_LAYOUT_READ)
|
||||
Vk.put_i32(p, 68, gvk_tex_dims_w[tex])
|
||||
Vk.put_i32(p, 72, gvk_tex_dims_h[tex])
|
||||
Vk.put_i32(p, 76, gvk_tex_vkfmt[tex])
|
||||
Vk.put_i32(p, 80, gvk_tex_levels[tex])
|
||||
Vk.put_i32(p, 84, gvk_tex_layers[tex])
|
||||
Vk.put_i32(p, 100, gvk_tex_usage(tex))
|
||||
Vk.put_i32(p, 68, render3d_st.gvk_tex_dims_w[tex])
|
||||
Vk.put_i32(p, 72, render3d_st.gvk_tex_dims_h[tex])
|
||||
Vk.put_i32(p, 76, render3d_st.gvk_tex_vkfmt[tex])
|
||||
Vk.put_i32(p, 80, render3d_st.gvk_tex_levels[tex])
|
||||
Vk.put_i32(p, 84, render3d_st.gvk_tex_layers[tex])
|
||||
Vk.put_i32(p, 100, gvk_tex_usage(render3d_st, tex))
|
||||
}
|
||||
# the usage gvk_tex_storage gave the image
|
||||
function gvk_tex_usage(tex: int) -> int {
|
||||
let fmt = gvk_tex_vkfmt[tex]
|
||||
function gvk_tex_usage(render3d_st: Render3dState, tex: int) -> int {
|
||||
let fmt = render3d_st.gvk_tex_vkfmt[tex]
|
||||
var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT
|
||||
if fmt == VK_FORMAT_D32_SFLOAT { return usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT }
|
||||
usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT
|
||||
if gvk_tex_samples[tex] <= 1 and (fmt == VK_FORMAT_R16G16B16A16_SFLOAT or fmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }
|
||||
if render3d_st.gvk_tex_samples[tex] <= 1 and (fmt == VK_FORMAT_R16G16B16A16_SFLOAT or fmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }
|
||||
return usage
|
||||
}
|
||||
|
||||
# sl::ResourceTag i, pointing at resource i
|
||||
function gsl_tag(i: int, buffer: int, w: int, h: int) -> void {
|
||||
let p = Vk.at(gsl_tags, i * 64)
|
||||
function gsl_tag(render3d_st: Render3dState, i: int, buffer: int, w: int, h: int) -> void {
|
||||
let p = Vk.at(render3d_st.gsl_tags, i * 64)
|
||||
Vk.zero(p, 64)
|
||||
gsl_header(p, 0x4c6a, 0x5aad, 0xb445, 0x496c, 0x87ff, 0x1af3, 0x845b, 0xe653, 1)
|
||||
Vk.put_ptr(p, 32, Vk.at(gsl_res, i * 112))
|
||||
Vk.put_ptr(p, 32, Vk.at(render3d_st.gsl_res, i * 112))
|
||||
Vk.put_i32(p, 40, buffer)
|
||||
Vk.put_i32(p, 44, 2) # eValidUntilEvaluate
|
||||
Vk.put_i32(p, 56, w)
|
||||
|
|
@ -387,15 +345,15 @@ function gsl_clip_fix(m: floats) -> void {
|
|||
m[14] = 0.5
|
||||
}
|
||||
|
||||
function gsl_constants() -> void {
|
||||
if gsl_consts == null { gsl_consts = gsl_struct(456) }
|
||||
let k = gsl_consts
|
||||
function gsl_constants(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gsl_consts == null { render3d_st.gsl_consts = gsl_struct(456) }
|
||||
let k = render3d_st.gsl_consts
|
||||
Vk.zero(k, 456)
|
||||
gsl_header(k, 0xdcd3, 0x5ad7, 0x4e4a, 0x4bad, 0xa90c, 0xe0c4, 0x9eb2, 0x3afe, 2)
|
||||
let fix = m4_new(); let proj = m4_new(); let v2c = m4_new(); let c2v = m4_new()
|
||||
let cur = m4_new(); let prev = m4_new(); let inv_cur = m4_new(); let c2p = m4_new(); let p2c = m4_new()
|
||||
gsl_clip_fix(fix)
|
||||
m4_perspective(proj, cam_fov, cam_aspect, cam_near, cam_far)
|
||||
m4_perspective(proj, render3d_st.cam_fov, render3d_st.cam_aspect, render3d_st.cam_near, render3d_st.cam_far)
|
||||
m4_mul(v2c, fix, proj)
|
||||
m4_inverse(c2v, v2c)
|
||||
gsl_put_m4(k, 32, v2c) # cameraViewToClip (no jitter)
|
||||
|
|
@ -403,9 +361,9 @@ function gsl_constants() -> void {
|
|||
let ident = m4_new()
|
||||
m4_identity(ident)
|
||||
gsl_put_m4(k, 160, ident) # clipToLensClip
|
||||
if gsl_prev_vp == null { gsl_prev_vp = m4_new(); for i in 0 .. 16 { gsl_prev_vp[i] = cam_vp_clean[i] } }
|
||||
m4_mul(cur, fix, cam_vp_clean)
|
||||
m4_mul(prev, fix, gsl_prev_vp)
|
||||
if render3d_st.gsl_prev_vp == null { render3d_st.gsl_prev_vp = m4_new(); for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] } }
|
||||
m4_mul(cur, fix, render3d_st.cam_vp_clean)
|
||||
m4_mul(prev, fix, render3d_st.gsl_prev_vp)
|
||||
m4_inverse(inv_cur, cur)
|
||||
m4_mul(c2p, prev, inv_cur)
|
||||
m4_inverse(p2c, c2p)
|
||||
|
|
@ -414,87 +372,87 @@ function gsl_constants() -> void {
|
|||
# the sample's offset from the pixel centre, in the image Streamline sees: row 0 at the top, so the
|
||||
# vertical offset flips with it. With jy unflipped DLSS resolved the ground into concentric
|
||||
# rings; with jx flipped too, thin stems doubled sideways (PC shots, 2026-09-15).
|
||||
var jx = -gsl_jpx
|
||||
var jy = -gsl_jpy
|
||||
var jx = -render3d_st.gsl_jpx
|
||||
var jy = -render3d_st.gsl_jpy
|
||||
# R3D_DLSS_JX / R3D_DLSS_JY = -1 flip a sign, to check the convention against the picture
|
||||
if r3d_env_has("R3D_DLSS_JX") and Text.to_int(r3d_env("R3D_DLSS_JX")) < 0 { jx = -jx }
|
||||
if r3d_env_has("R3D_DLSS_JY") and Text.to_int(r3d_env("R3D_DLSS_JY")) < 0 { jy = -jy }
|
||||
if r3d_env_has(render3d_st, "R3D_DLSS_JX") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JX")) < 0 { jx = -jx }
|
||||
if r3d_env_has(render3d_st, "R3D_DLSS_JY") and Text.to_int(r3d_env(render3d_st, "R3D_DLSS_JY")) < 0 { jy = -jy }
|
||||
Vk.put_i32(k, 352, float_bits(jx))
|
||||
Vk.put_i32(k, 356, float_bits(jy))
|
||||
Vk.put_i32(k, 360, float_bits(1.0)) # mvecScale
|
||||
Vk.put_i32(k, 364, float_bits(1.0))
|
||||
gsl_put_v3(k, 376, cam_pos)
|
||||
gsl_put_v3(k, 376, render3d_st.cam_pos)
|
||||
let up = floats(3)
|
||||
v3_cross(up, cam_right, cam_fwd)
|
||||
v3_cross(up, render3d_st.cam_right, render3d_st.cam_fwd)
|
||||
gsl_put_v3(k, 388, up)
|
||||
gsl_put_v3(k, 400, cam_right)
|
||||
gsl_put_v3(k, 412, cam_fwd)
|
||||
Vk.put_i32(k, 424, float_bits(cam_near))
|
||||
Vk.put_i32(k, 428, float_bits(cam_far))
|
||||
Vk.put_i32(k, 432, float_bits(cam_fov))
|
||||
Vk.put_i32(k, 436, float_bits(cam_aspect))
|
||||
gsl_put_v3(k, 400, render3d_st.cam_right)
|
||||
gsl_put_v3(k, 412, render3d_st.cam_fwd)
|
||||
Vk.put_i32(k, 424, float_bits(render3d_st.cam_near))
|
||||
Vk.put_i32(k, 428, float_bits(render3d_st.cam_far))
|
||||
Vk.put_i32(k, 432, float_bits(render3d_st.cam_fov))
|
||||
Vk.put_i32(k, 436, float_bits(render3d_st.cam_aspect))
|
||||
Vk.put_i32(k, 440, float_bits(0.0)) # motionVectorsInvalidValue
|
||||
# depthInverted, cameraMotionIncluded, motionVectors3D false; reset on a cut; not orthographic,
|
||||
# not dilated, not jittered
|
||||
if gsl_reset { Vk.put_i32(k, 444, 256 * 256 * 256) }
|
||||
if render3d_st.gsl_reset { Vk.put_i32(k, 444, 256 * 256 * 256) }
|
||||
Vk.put_i32(k, 452, float_bits(40.0)) # minRelativeLinearDepthObjectSeparation
|
||||
free(fix); free(proj); free(v2c); free(c2v); free(cur); free(prev); free(inv_cur); free(c2p); free(p2c); free(ident); free(up)
|
||||
}
|
||||
|
||||
# make the DLSS targets at this frame's sizes
|
||||
function gsl_targets() -> void {
|
||||
if gsl_mv == null or gsl_mv.w != post_w or gsl_mv.h != post_h {
|
||||
if gsl_mv != null { target_free(gsl_mv) }
|
||||
gsl_mv = target_new(post_w, post_h, GL_RG16F, GL_RG, GL_HALF_FLOAT, false, GL_NEAREST)
|
||||
gsl_reset = true
|
||||
function gsl_targets(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.gsl_mv == null or render3d_st.gsl_mv.w != render3d_st.post_w or render3d_st.gsl_mv.h != render3d_st.post_h {
|
||||
if render3d_st.gsl_mv != null { target_free(render3d_st, render3d_st.gsl_mv) }
|
||||
render3d_st.gsl_mv = target_new(render3d_st, render3d_st.post_w, render3d_st.post_h, GL_RG16F, GL_RG, GL_HALF_FLOAT, false, GL_NEAREST)
|
||||
render3d_st.gsl_reset = true
|
||||
}
|
||||
if gsl_out == null or gsl_out.w != gl_w or gsl_out.h != gl_h {
|
||||
if gsl_out != null { target_free(gsl_out) }
|
||||
gsl_out = target_new(gl_w, gl_h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
gsl_reset = true
|
||||
if render3d_st.gsl_out == null or render3d_st.gsl_out.w != gl_w or render3d_st.gsl_out.h != gl_h {
|
||||
if render3d_st.gsl_out != null { target_free(render3d_st, render3d_st.gsl_out) }
|
||||
render3d_st.gsl_out = target_new(render3d_st, gl_w, gl_h, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, false, GL_LINEAR)
|
||||
render3d_st.gsl_reset = true
|
||||
}
|
||||
# the LDR image the tonemap writes follows the upscaled size
|
||||
if post_ldr.w != gl_w or post_ldr.h != gl_h {
|
||||
target_free(post_ldr)
|
||||
post_ldr = target_new(gl_w, gl_h, post_ldr_fmt(), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
||||
post_ldr_hdr = gpu_hdr_active()
|
||||
if render3d_st.post_ldr.w != gl_w or render3d_st.post_ldr.h != gl_h {
|
||||
target_free(render3d_st, render3d_st.post_ldr)
|
||||
render3d_st.post_ldr = target_new(render3d_st, gl_w, gl_h, post_ldr_fmt(render3d_st), GL_RGBA, GL_UNSIGNED_BYTE, false, GL_LINEAR)
|
||||
render3d_st.post_ldr_hdr = gpu_hdr_active(render3d_st)
|
||||
}
|
||||
if gsl_res == null { gsl_res = gsl_struct(4 * 112); gsl_tags = gsl_struct(4 * 64); gsl_inputs = bytes(8) }
|
||||
if render3d_st.gsl_res == null { render3d_st.gsl_res = gsl_struct(4 * 112); render3d_st.gsl_tags = gsl_struct(4 * 64); render3d_st.gsl_inputs = bytes(8) }
|
||||
}
|
||||
|
||||
# Upscale post_hdr into gsl_out; the colour the rest of post reads (post_hdr's own if it failed).
|
||||
function gsl_dlss_eval() -> int {
|
||||
gsl_targets()
|
||||
function gsl_dlss_eval(render3d_st: mut Render3dState) -> int {
|
||||
gsl_targets(render3d_st)
|
||||
# no motion of its own: the camera's comes from depth
|
||||
target_bind(gsl_mv)
|
||||
gpu_clear_color(0.0, 0.0, 0.0, 0.0)
|
||||
gpu_clear(GL_COLOR_BUFFER_BIT)
|
||||
gvk_pass_end()
|
||||
let cb = gvk_frame_cb()
|
||||
if gsl_set_mode != gsl_dlss_mode or gsl_set_w != gl_w or gsl_set_h != gl_h {
|
||||
target_bind(render3d_st, render3d_st.gsl_mv)
|
||||
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 0.0)
|
||||
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT)
|
||||
gvk_pass_end(render3d_st)
|
||||
let cb = gvk_frame_cb(render3d_st)
|
||||
if render3d_st.gsl_set_mode != render3d_st.gsl_dlss_mode or render3d_st.gsl_set_w != gl_w or render3d_st.gsl_set_h != gl_h {
|
||||
let f = gsl_fn(GSL_DLSS, "slDLSSSetOptions")
|
||||
gsl_fill_options(gl_w, gl_h)
|
||||
if f != null and Vk.sl_call_pp(f, gsl_vp, gsl_opts) == 0 { gsl_set_mode = gsl_dlss_mode; gsl_set_w = gl_w; gsl_set_h = gl_h }
|
||||
gsl_fill_options(render3d_st, gl_w, gl_h)
|
||||
if f != null and Vk.sl_call_pp(f, render3d_st.gsl_vp, render3d_st.gsl_opts) == 0 { render3d_st.gsl_set_mode = render3d_st.gsl_dlss_mode; render3d_st.gsl_set_w = gl_w; render3d_st.gsl_set_h = gl_h }
|
||||
}
|
||||
gsl_constants()
|
||||
Vk.sl_set_constants(gsl_consts, gsl_token, gsl_vp)
|
||||
gsl_resource(0, post_hdr.depth); gsl_tag(0, 0, post_w, post_h) # kBufferTypeDepth
|
||||
gsl_resource(112, gsl_mv.color); gsl_tag(1, 1, post_w, post_h) # kBufferTypeMotionVectors
|
||||
gsl_resource(224, post_hdr.color); gsl_tag(2, 3, post_w, post_h) # kBufferTypeScalingInputColor
|
||||
gsl_resource(336, gsl_out.color); gsl_tag(3, 4, gl_w, gl_h) # kBufferTypeScalingOutputColor
|
||||
Vk.sl_set_tag_for_frame(gsl_token, gsl_vp, gsl_tags, 4, cb)
|
||||
Vk.put_ptr(gsl_inputs, 0, gsl_vp)
|
||||
let r = Vk.sl_evaluate_feature(GSL_DLSS, gsl_token, gsl_inputs, 1, cb)
|
||||
for i in 0 .. 16 { gsl_prev_vp[i] = cam_vp_clean[i] }
|
||||
gsl_reset = false
|
||||
gsl_constants(render3d_st)
|
||||
Vk.sl_set_constants(render3d_st.gsl_consts, render3d_st.gsl_token, render3d_st.gsl_vp)
|
||||
gsl_resource(render3d_st, 0, render3d_st.post_hdr.depth); gsl_tag(render3d_st, 0, 0, render3d_st.post_w, render3d_st.post_h) # kBufferTypeDepth
|
||||
gsl_resource(render3d_st, 112, render3d_st.gsl_mv.color); gsl_tag(render3d_st, 1, 1, render3d_st.post_w, render3d_st.post_h) # kBufferTypeMotionVectors
|
||||
gsl_resource(render3d_st, 224, render3d_st.post_hdr.color); gsl_tag(render3d_st, 2, 3, render3d_st.post_w, render3d_st.post_h) # kBufferTypeScalingInputColor
|
||||
gsl_resource(render3d_st, 336, render3d_st.gsl_out.color); gsl_tag(render3d_st, 3, 4, gl_w, gl_h) # kBufferTypeScalingOutputColor
|
||||
Vk.sl_set_tag_for_frame(render3d_st.gsl_token, render3d_st.gsl_vp, render3d_st.gsl_tags, 4, cb)
|
||||
Vk.put_ptr(render3d_st.gsl_inputs, 0, render3d_st.gsl_vp)
|
||||
let r = Vk.sl_evaluate_feature(GSL_DLSS, render3d_st.gsl_token, render3d_st.gsl_inputs, 1, cb)
|
||||
for i in 0 .. 16 { render3d_st.gsl_prev_vp[i] = render3d_st.cam_vp_clean[i] }
|
||||
render3d_st.gsl_reset = false
|
||||
# Streamline records its own pipeline and descriptors into the command buffer; nothing needs
|
||||
# forgetting, because every gvk_draw binds its pipeline, view and set afresh
|
||||
gsl_eval_ok = r == 0
|
||||
render3d_st.gsl_eval_ok = r == 0
|
||||
if r != 0 {
|
||||
if not gsl_said { print(`r3d: streamline: DLSS evaluate failed ({r}); drawing without it`); gsl_said = true }
|
||||
post_color_w = post_w; post_color_h = post_h
|
||||
return post_hdr.color
|
||||
if not render3d_st.gsl_said { print(`r3d: streamline: DLSS evaluate failed ({r}); drawing without it`); render3d_st.gsl_said = true }
|
||||
render3d_st.post_color_w = render3d_st.post_w; render3d_st.post_color_h = render3d_st.post_h
|
||||
return render3d_st.post_hdr.color
|
||||
}
|
||||
post_color_w = gl_w; post_color_h = gl_h
|
||||
return gsl_out.color
|
||||
render3d_st.post_color_w = gl_w; render3d_st.post_color_h = gl_h
|
||||
return render3d_st.gsl_out.color
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -11,38 +11,32 @@
|
|||
|
||||
const GL_TEXTURE_MAX_ANISOTROPY_EXT: int = 0x84FE
|
||||
|
||||
var tex_w: int = 0 # the last decoded image
|
||||
var tex_h: int = 0
|
||||
var tex_channels: int = 0
|
||||
var tex_depth: int = 0 # bits per sample (8 or 16)
|
||||
var tex_file_len: int = 0
|
||||
var tex_anisotropy: fixed = 16.0
|
||||
|
||||
# The anisotropic filtering level for every mipmapped texture, loaded or not: 1 (off), 2, 4, 8
|
||||
# or 16. Textures uploaded before a change are updated in place - on OpenGL by setting the
|
||||
# parameter again, on Vulkan by rewriting the record the sampler cache reads - so a settings
|
||||
# screen can offer it live instead of on the next start.
|
||||
function r3d_set_anisotropy(level: int) -> void {
|
||||
function r3d_set_anisotropy(render3d_st: mut Render3dState, level: int) -> void {
|
||||
var a: fixed = 1.0
|
||||
if level >= 2 { a = 2.0 }
|
||||
if level >= 4 { a = 4.0 }
|
||||
if level >= 8 { a = 8.0 }
|
||||
if level >= 16 { a = 16.0 }
|
||||
if a == tex_anisotropy { return }
|
||||
tex_anisotropy = a
|
||||
if gpu_tx == null { return }
|
||||
let keep = gpu_bound_2d
|
||||
for t in 1 .. gpu_tx_cap {
|
||||
if a == render3d_st.tex_anisotropy { return }
|
||||
render3d_st.tex_anisotropy = a
|
||||
if render3d_st.gpu_tx == null { return }
|
||||
let keep = render3d_st.gpu_bound_2d
|
||||
for t in 1 .. render3d_st.gpu_tx_cap {
|
||||
let o = t * GPU_TX_W
|
||||
# a 2D texture with mipmaps that was given a level when it was uploaded
|
||||
if gpu_tx[o] != GPU_TEX2D or gpu_tx[o + 10] != 1 or gpu_tx[o + 11] == 0 { continue }
|
||||
gpu_tex_bind(GPU_TEX2D, t)
|
||||
gpu_tex_paramf(GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, a)
|
||||
if render3d_st.gpu_tx[o] != GPU_TEX2D or render3d_st.gpu_tx[o + 10] != 1 or render3d_st.gpu_tx[o + 11] == 0 { continue }
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, t)
|
||||
gpu_tex_paramf(render3d_st, GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, a)
|
||||
}
|
||||
if keep > 0 { gpu_tex_bind(GPU_TEX2D, keep) }
|
||||
if keep > 0 { gpu_tex_bind(render3d_st, GPU_TEX2D, keep) }
|
||||
}
|
||||
|
||||
function r3d_read_file(path: pointer) -> pointer {
|
||||
function r3d_read_file(render3d_st: mut Render3dState, path: pointer) -> pointer {
|
||||
let f = file_open(path, "rb")
|
||||
if f == null { return null }
|
||||
file_seek(f, 0, 2)
|
||||
|
|
@ -52,7 +46,7 @@ function r3d_read_file(path: pointer) -> pointer {
|
|||
let buf = bytes(n + 8)
|
||||
file_read(f, buf, n)
|
||||
file_close(f)
|
||||
tex_file_len = n
|
||||
render3d_st.tex_file_len = n
|
||||
return buf
|
||||
}
|
||||
|
||||
|
|
@ -118,10 +112,10 @@ function png_unfilter(raw: pointer, cur: int, prev: int, stride: int, fbpp: int,
|
|||
}
|
||||
}
|
||||
|
||||
function png_decode(path: pointer) -> pointer {
|
||||
let d = r3d_read_file(path)
|
||||
function png_decode(render3d_st: mut Render3dState, path: pointer) -> pointer {
|
||||
let d = r3d_read_file(render3d_st, path)
|
||||
if d == null { print(`png: cannot read {path}`); return null }
|
||||
let size = tex_file_len
|
||||
let size = render3d_st.tex_file_len
|
||||
if size < 8 or d[0] != 137 or d[1] != 80 { free(d); print(`png: not a png: {path}`); return null }
|
||||
var w = 0; var h = 0; var bd = 0; var ct = 0
|
||||
let plte = bytes(768)
|
||||
|
|
@ -188,16 +182,16 @@ function png_decode(path: pointer) -> pointer {
|
|||
free(raw)
|
||||
}
|
||||
free(plte)
|
||||
tex_w = w; tex_h = h; tex_channels = channels; tex_depth = bd
|
||||
render3d_st.tex_w = w; render3d_st.tex_h = h; render3d_st.tex_channels = channels; render3d_st.tex_depth = bd
|
||||
return out
|
||||
}
|
||||
|
||||
# Edge padding for cut-out atlases: pixels darker than `thresh` (the unused
|
||||
# background) take the mean of their lit neighbours, repeated `passes` times, so
|
||||
# mipmaps and bilinear taps never pull black into the blades. 8-bit RGB/RGBA only.
|
||||
function tex_dilate(px: pointer, thresh: int, passes: int) -> void {
|
||||
if tex_depth != 8 or tex_channels < 3 { return }
|
||||
let w = tex_w; let h = tex_h; let c = tex_channels
|
||||
function tex_dilate(render3d_st: Render3dState, px: pointer, thresh: int, passes: int) -> void {
|
||||
if render3d_st.tex_depth != 8 or render3d_st.tex_channels < 3 { return }
|
||||
let w = render3d_st.tex_w; let h = render3d_st.tex_h; let c = render3d_st.tex_channels
|
||||
let mask = bytes(w * h)
|
||||
var i = 0
|
||||
while i < w * h { let o = i * c; if px[o] + px[o + 1] + px[o + 2] < thresh { mask[i] = 1 } else { mask[i] = 0 }; i += 1 }
|
||||
|
|
@ -227,101 +221,91 @@ function tex_dilate(px: pointer, thresh: int, passes: int) -> void {
|
|||
}
|
||||
|
||||
# Upload the last-decoded samples as a 2D texture. srgb: colour data (8-bit only).
|
||||
function tex_upload(px: pointer, srgb: bool, mips: bool) -> int {
|
||||
let id = gpu_tex_new()
|
||||
gpu_tex_bind(GPU_TEX2D, id)
|
||||
function tex_upload(render3d_st: mut Render3dState, px: pointer, srgb: bool, mips: bool) -> int {
|
||||
let id = gpu_tex_new(render3d_st)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, id)
|
||||
var fmt = GL_RED
|
||||
if tex_channels == 2 { fmt = GL_RG }
|
||||
if tex_channels == 3 { fmt = GL_RGB }
|
||||
if tex_channels == 4 { fmt = GL_RGBA }
|
||||
if render3d_st.tex_channels == 2 { fmt = GL_RG }
|
||||
if render3d_st.tex_channels == 3 { fmt = GL_RGB }
|
||||
if render3d_st.tex_channels == 4 { fmt = GL_RGBA }
|
||||
var ifmt = GL_R8
|
||||
var ty = GL_UNSIGNED_BYTE
|
||||
if tex_depth == 16 {
|
||||
if render3d_st.tex_depth == 16 {
|
||||
ty = GL_UNSIGNED_SHORT
|
||||
ifmt = GL_R16
|
||||
if tex_channels == 2 { ifmt = GL_RG16 }
|
||||
if tex_channels == 3 { ifmt = GL_RGB16 }
|
||||
if tex_channels == 4 { ifmt = GL_RGBA16 }
|
||||
gpu_pixel_store(GL_UNPACK_SWAP_BYTES, 1)
|
||||
if render3d_st.tex_channels == 2 { ifmt = GL_RG16 }
|
||||
if render3d_st.tex_channels == 3 { ifmt = GL_RGB16 }
|
||||
if render3d_st.tex_channels == 4 { ifmt = GL_RGBA16 }
|
||||
gpu_pixel_store(render3d_st, GL_UNPACK_SWAP_BYTES, 1)
|
||||
} else {
|
||||
if tex_channels == 2 { ifmt = GL_RG8 }
|
||||
if tex_channels == 3 { ifmt = GL_RGB8; if srgb { ifmt = GL_SRGB8 } }
|
||||
if tex_channels == 4 { ifmt = GL_RGBA8; if srgb { ifmt = GL_SRGB8_ALPHA8 } }
|
||||
gpu_pixel_store(GL_UNPACK_SWAP_BYTES, 0)
|
||||
if render3d_st.tex_channels == 2 { ifmt = GL_RG8 }
|
||||
if render3d_st.tex_channels == 3 { ifmt = GL_RGB8; if srgb { ifmt = GL_SRGB8 } }
|
||||
if render3d_st.tex_channels == 4 { ifmt = GL_RGBA8; if srgb { ifmt = GL_SRGB8_ALPHA8 } }
|
||||
gpu_pixel_store(render3d_st, GL_UNPACK_SWAP_BYTES, 0)
|
||||
}
|
||||
gpu_pixel_store(GL_UNPACK_ALIGNMENT, 1)
|
||||
gpu_tex_image2d(ifmt, tex_w, tex_h, fmt, ty, px)
|
||||
gpu_pixel_store(GL_UNPACK_SWAP_BYTES, 0)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_REPEAT)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_pixel_store(render3d_st, GL_UNPACK_ALIGNMENT, 1)
|
||||
gpu_tex_image2d(render3d_st, ifmt, render3d_st.tex_w, render3d_st.tex_h, fmt, ty, px)
|
||||
gpu_pixel_store(render3d_st, GL_UNPACK_SWAP_BYTES, 0)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_REPEAT)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
if mips {
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(GPU_TEX2D)
|
||||
gpu_tex_paramf(GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, tex_anisotropy)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
||||
gpu_tex_paramf(render3d_st, GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, render3d_st.tex_anisotropy)
|
||||
} else {
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
# Load a PNG as a mipmapped, anisotropic texture (0 on failure). srgb for albedo.
|
||||
function tex_load(path: pointer, srgb: bool) -> int { return tex_load_ex(path, srgb, 0) }
|
||||
function tex_load(render3d_st: mut Render3dState, path: pointer, srgb: bool) -> int { return tex_load_ex(render3d_st, path, srgb, 0) }
|
||||
# ... with `dilate` passes of edge padding for a cut-out atlas (0 = none)
|
||||
function tex_load_ex(path: pointer, srgb: bool, dilate: int) -> int {
|
||||
let px = png_decode(path)
|
||||
function tex_load_ex(render3d_st: mut Render3dState, path: pointer, srgb: bool, dilate: int) -> int {
|
||||
let px = png_decode(render3d_st, path)
|
||||
if px == null { return 0 }
|
||||
if dilate > 0 { tex_dilate(px, 60, dilate) }
|
||||
let id = tex_upload(px, srgb, true)
|
||||
if dilate > 0 { tex_dilate(render3d_st, px, 60, dilate) }
|
||||
let id = tex_upload(render3d_st, px, srgb, true)
|
||||
free(px)
|
||||
tex_note_size(id)
|
||||
tex_note_size(render3d_st, id)
|
||||
return id
|
||||
}
|
||||
# the size each loaded texture was, by id: a nine-slice or a UI image asks
|
||||
var tex_size_ids: []int = new []int
|
||||
var tex_size_ws: []int = new []int
|
||||
var tex_size_hs: []int = new []int
|
||||
function tex_note_size(id: int) -> void {
|
||||
push(tex_size_ids, id)
|
||||
push(tex_size_ws, tex_w)
|
||||
push(tex_size_hs, tex_h)
|
||||
function tex_note_size(render3d_st: mut Render3dState, id: int) -> void {
|
||||
push(render3d_st.tex_size_ids, id)
|
||||
push(render3d_st.tex_size_ws, render3d_st.tex_w)
|
||||
push(render3d_st.tex_size_hs, render3d_st.tex_h)
|
||||
}
|
||||
function tex_width(id: int) -> int {
|
||||
for i in 0 .. len(tex_size_ids) {
|
||||
if tex_size_ids[i] == id { return tex_size_ws[i] }
|
||||
function tex_width(render3d_st: Render3dState, id: int) -> int {
|
||||
for i in 0 .. len(render3d_st.tex_size_ids) {
|
||||
if render3d_st.tex_size_ids[i] == id { return render3d_st.tex_size_ws[i] }
|
||||
}
|
||||
return 0
|
||||
}
|
||||
function tex_height(id: int) -> int {
|
||||
for i in 0 .. len(tex_size_ids) {
|
||||
if tex_size_ids[i] == id { return tex_size_hs[i] }
|
||||
function tex_height(render3d_st: Render3dState, id: int) -> int {
|
||||
for i in 0 .. len(render3d_st.tex_size_ids) {
|
||||
if render3d_st.tex_size_ids[i] == id { return render3d_st.tex_size_hs[i] }
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
# A small solid-colour fallback texture (linear rgb 0..255), for missing maps.
|
||||
function tex_solid(r: int, g: int, b: int, a: int) -> int {
|
||||
function tex_solid(render3d_st: mut Render3dState, r: int, g: int, b: int, a: int) -> int {
|
||||
let px = bytes(16)
|
||||
for i in 0 .. 4 { px[i * 4] = r; px[i * 4 + 1] = g; px[i * 4 + 2] = b; px[i * 4 + 3] = a }
|
||||
tex_w = 2; tex_h = 2; tex_channels = 4; tex_depth = 8
|
||||
let id = tex_upload(px, false, false)
|
||||
render3d_st.tex_w = 2; render3d_st.tex_h = 2; render3d_st.tex_channels = 4; render3d_st.tex_depth = 8
|
||||
let id = tex_upload(render3d_st, px, false, false)
|
||||
free(px)
|
||||
return id
|
||||
}
|
||||
|
||||
# ---- Radiance .hdr (RGBE, new-style RLE) -> RGB float bits -------------------------
|
||||
var hdr_max_lum: float = 0.0 # float bits of the brightest texel (sun finding)
|
||||
var hdr_max_x: int = 0
|
||||
var hdr_max_y: int = 0
|
||||
var hdr_sun_r: float = 0.0 # irradiance (float bits) of everything above the IBL clip: the sun
|
||||
var hdr_sun_g: float = 0.0
|
||||
var hdr_sun_b: float = 0.0
|
||||
var hdr_clip: float = 0.0 # float bits; texels above this (per channel) feed the sun, not the IBL
|
||||
|
||||
function hdr_decode(path: pointer) -> floats {
|
||||
let d = r3d_read_file(path)
|
||||
function hdr_decode(render3d_st: mut Render3dState, path: pointer) -> floats {
|
||||
let d = r3d_read_file(render3d_st, path)
|
||||
if d == null { print(`hdr: cannot read {path}`); return null }
|
||||
let size = tex_file_len
|
||||
let size = render3d_st.tex_file_len
|
||||
# header: lines until an empty line, then "-Y h +X w"
|
||||
var i = 0
|
||||
var blank = false
|
||||
|
|
@ -340,7 +324,7 @@ function hdr_decode(path: pointer) -> floats {
|
|||
let out = floats(w * h * 3)
|
||||
let line = bytes(w * 4)
|
||||
var maxl = 0.0
|
||||
if hdr_clip == 0.0 { hdr_clip = 20.0 }
|
||||
if render3d_st.hdr_clip == 0.0 { render3d_st.hdr_clip = 20.0 }
|
||||
var sr = 0.0; var sg = 0.0; var sb = 0.0
|
||||
var skye = 0.0 # sky irradiance on an upward face (clipped part only)
|
||||
let dphi = 2.0 * PI / float(w)
|
||||
|
|
@ -385,61 +369,60 @@ function hdr_decode(path: pointer) -> floats {
|
|||
out[o + 1] = Math.min(vg, 60000.0)
|
||||
out[o + 2] = Math.min(vb, 60000.0)
|
||||
let lum = vr + vg + vb
|
||||
if maxl < lum { maxl = lum; hdr_max_x = x; hdr_max_y = y }
|
||||
if y < h / 2 { skye = skye + Math.min(vg, hdr_clip) * Math.cos((float(y) + 0.5) * dth) * domega }
|
||||
if hdr_clip < vg or hdr_clip < vr {
|
||||
sr = sr + Math.max(vr - hdr_clip, 0.0) * domega
|
||||
sg = sg + Math.max(vg - hdr_clip, 0.0) * domega
|
||||
sb = sb + Math.max(vb - hdr_clip, 0.0) * domega
|
||||
if maxl < lum { maxl = lum; render3d_st.hdr_max_x = x; render3d_st.hdr_max_y = y }
|
||||
if y < h / 2 { skye = skye + Math.min(vg, render3d_st.hdr_clip) * Math.cos((float(y) + 0.5) * dth) * domega }
|
||||
if render3d_st.hdr_clip < vg or render3d_st.hdr_clip < vr {
|
||||
sr = sr + Math.max(vr - render3d_st.hdr_clip, 0.0) * domega
|
||||
sg = sg + Math.max(vg - render3d_st.hdr_clip, 0.0) * domega
|
||||
sb = sb + Math.max(vb - render3d_st.hdr_clip, 0.0) * domega
|
||||
}
|
||||
}
|
||||
}
|
||||
y += 1
|
||||
}
|
||||
free(line); free(d)
|
||||
tex_w = w; tex_h = h; tex_channels = 3; tex_depth = 32
|
||||
hdr_max_lum = maxl
|
||||
hdr_sun_r = sr; hdr_sun_g = sg; hdr_sun_b = sb
|
||||
render3d_st.tex_w = w; render3d_st.tex_h = h; render3d_st.tex_channels = 3; render3d_st.tex_depth = 32
|
||||
render3d_st.hdr_max_lum = maxl
|
||||
render3d_st.hdr_sun_r = sr; render3d_st.hdr_sun_g = sg; render3d_st.hdr_sun_b = sb
|
||||
print(`hdr: peak/1000 {fixed(maxl / 1000.0)} sky irradiance(up) {fixed(skye)} sun irradiance {fixed(sg)} (Q16.16 = /65536)`)
|
||||
return out
|
||||
}
|
||||
|
||||
# Load an equirectangular .hdr as an RGB16F texture with mips (clamped in v).
|
||||
function tex_load_hdr(path: pointer) -> int {
|
||||
let px = hdr_decode(path)
|
||||
function tex_load_hdr(render3d_st: mut Render3dState, path: pointer) -> int {
|
||||
let px = hdr_decode(render3d_st, path)
|
||||
if px == null { return 0 }
|
||||
let id = gpu_tex_new()
|
||||
gpu_tex_bind(GPU_TEX2D, id)
|
||||
gpu_pixel_store(GL_UNPACK_ALIGNMENT, 4)
|
||||
gpu_tex_image2d(GL_RGB16F, tex_w, tex_h, GL_RGB, GL_FLOAT, data_of(px))
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(GPU_TEX2D)
|
||||
let id = gpu_tex_new(render3d_st)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, id)
|
||||
gpu_pixel_store(render3d_st, GL_UNPACK_ALIGNMENT, 4)
|
||||
gpu_tex_image2d(render3d_st, GL_RGB16F, render3d_st.tex_w, render3d_st.tex_h, GL_RGB, GL_FLOAT, data_of(px))
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
||||
free(px)
|
||||
return id
|
||||
}
|
||||
|
||||
# An empty render-target texture of the given internal format (no mips, clamped).
|
||||
function tex_target(w: int, h: int, ifmt: int, fmt: int, ty: int, filter: int) -> int {
|
||||
let id = gpu_tex_new()
|
||||
gpu_tex_bind(GPU_TEX2D, id)
|
||||
gpu_tex_image2d(ifmt, w, h, fmt, ty, null)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MAG_FILTER, filter)
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, filter)
|
||||
function tex_target(render3d_st: mut Render3dState, w: int, h: int, ifmt: int, fmt: int, ty: int, filter: int) -> int {
|
||||
let id = gpu_tex_new(render3d_st)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, id)
|
||||
gpu_tex_image2d(render3d_st, ifmt, w, h, fmt, ty, null)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MAG_FILTER, filter)
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, filter)
|
||||
return id
|
||||
}
|
||||
|
||||
var tex_dump_alpha: bool = false
|
||||
# Debug: the brightest texel of an RGBA float texture and where it is.
|
||||
function tex_max(tex: int, w: int, h: int, tag: pointer) -> void {
|
||||
function tex_max(render3d_st: mut Render3dState, tex: int, w: int, h: int, tag: pointer) -> void {
|
||||
let buf = floats(w * h * 4)
|
||||
gpu_tex_bind(GPU_TEX2D, tex)
|
||||
gpu_pixel_store(GL_PACK_ALIGNMENT, 4)
|
||||
gpu_tex_read(GPU_TEX2D, GL_RGBA, GL_FLOAT, data_of(buf))
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, tex)
|
||||
gpu_pixel_store(render3d_st, GL_PACK_ALIGNMENT, 4)
|
||||
gpu_tex_read(render3d_st, GPU_TEX2D, GL_RGBA, GL_FLOAT, data_of(buf))
|
||||
var best = 0.0; var bx = 0; var by = 0
|
||||
var i = 0
|
||||
while i < w * h {
|
||||
|
|
@ -455,20 +438,20 @@ function tex_max(tex: int, w: int, h: int, tag: pointer) -> void {
|
|||
free(buf)
|
||||
}
|
||||
# Debug: write a 2D texture's level 0 (RGBA8, alpha dropped) as a binary PPM.
|
||||
function tex_dump(tex: int, w: int, h: int, path: pointer) -> void {
|
||||
function tex_dump(render3d_st: mut Render3dState, tex: int, w: int, h: int, path: pointer) -> void {
|
||||
let f = file_open(path, "wb")
|
||||
if f == null { return }
|
||||
let buf = bytes(w * h * 4)
|
||||
gpu_tex_bind(GPU_TEX2D, tex)
|
||||
gpu_pixel_store(GL_PACK_ALIGNMENT, 1)
|
||||
gpu_tex_read(GPU_TEX2D, GL_RGBA, GL_UNSIGNED_BYTE, buf)
|
||||
gpu_tex_bind(render3d_st, GPU_TEX2D, tex)
|
||||
gpu_pixel_store(render3d_st, GL_PACK_ALIGNMENT, 1)
|
||||
gpu_tex_read(render3d_st, GPU_TEX2D, GL_RGBA, GL_UNSIGNED_BYTE, buf)
|
||||
let hdr = `P6\n{w} {h}\n255\n`
|
||||
file_write(f, hdr, len(hdr))
|
||||
let row = bytes(w * 3)
|
||||
for y in 0 .. h {
|
||||
for x in 0 .. w {
|
||||
row[x * 3] = buf[(y * w + x) * 4]; row[x * 3 + 1] = buf[(y * w + x) * 4 + 1]; row[x * 3 + 2] = buf[(y * w + x) * 4 + 2]
|
||||
if tex_dump_alpha { let a = buf[(y * w + x) * 4 + 3]; row[x * 3] = a; row[x * 3 + 1] = a; row[x * 3 + 2] = a }
|
||||
if render3d_st.tex_dump_alpha { let a = buf[(y * w + x) * 4 + 3]; row[x * 3] = a; row[x * 3 + 1] = a; row[x * 3 + 2] = a }
|
||||
}
|
||||
file_write(f, row, w * 3)
|
||||
}
|
||||
|
|
@ -478,10 +461,10 @@ function tex_dump(tex: int, w: int, h: int, path: pointer) -> void {
|
|||
|
||||
# A binary PPM (P6, what Gl.screenshot writes) as an RGB8 texture, box-filtered down by
|
||||
# `shrink` (a photo thumbnail); 0 when the file is missing.
|
||||
function tex_load_ppm(path: pointer, shrink: int) -> int {
|
||||
let d = r3d_read_file(path)
|
||||
function tex_load_ppm(render3d_st: mut Render3dState, path: pointer, shrink: int) -> int {
|
||||
let d = r3d_read_file(render3d_st, path)
|
||||
if d == null { return 0 }
|
||||
let size = tex_file_len
|
||||
let size = render3d_st.tex_file_len
|
||||
var i = 2
|
||||
var w = 0; var h = 0; var mx = 0
|
||||
var field = 0
|
||||
|
|
@ -511,8 +494,8 @@ function tex_load_ppm(path: pointer, shrink: int) -> int {
|
|||
}
|
||||
}
|
||||
free(d)
|
||||
tex_w = ow; tex_h = oh; tex_channels = 3; tex_depth = 8
|
||||
let id = tex_upload(px, true, false)
|
||||
render3d_st.tex_w = ow; render3d_st.tex_h = oh; render3d_st.tex_channels = 3; render3d_st.tex_depth = 8
|
||||
let id = tex_upload(render3d_st, px, true, false)
|
||||
free(px)
|
||||
return id
|
||||
}
|
||||
|
|
|
|||
|
|
@ -5,108 +5,83 @@
|
|||
# tinted body read from the scene depth, and soft shores.
|
||||
# ============================================================================
|
||||
|
||||
var water_mesh: Mesh = null
|
||||
var water_prog: int = 0
|
||||
var water_level: float = 0.0
|
||||
var water_cx: float = 0.0
|
||||
var water_cz: float = 0.0
|
||||
var water_ex: float = 0.0
|
||||
var water_ez: float = 0.0
|
||||
var water_on: bool = false
|
||||
# Where a body is standing in the water and how hard it is disturbing it. The game sets it;
|
||||
# strength 0 means nobody is in the water and the whole term is skipped.
|
||||
var wt_wade_x: float = 0.0
|
||||
var wt_wade_z: float = 0.0
|
||||
var wt_wade_s: float = 0.0
|
||||
var water_refl: Target = null # the world mirrored in the surface, half resolution
|
||||
var water_refl_div: int = 2 # R3D_REFLDIV overrides: 2 = half res, 4 = quarter
|
||||
var water_saved: floats = null # the real camera's matrices, restored after the pass
|
||||
var water_saved_vp: floats = null # views of its second and third matrices
|
||||
var water_saved_ivp: floats = null
|
||||
var water_dumped: bool = false
|
||||
# Several still-water planes, each at its own level over its own bounds: the sea round an
|
||||
# island and a lake a hundred metres above it cannot be one surface. Each draws the same way;
|
||||
# only one - the first added with `reflect` - gets the planar reflection pass, because every
|
||||
# mirrored plane is another full scene pass. water_level / water_cx ... mirror that one, so
|
||||
# code written against a single plane still reads the surface that reflects.
|
||||
const WATER_MAX: int = 8
|
||||
var wb_n: int = 0
|
||||
var wb_level: floats = null
|
||||
var wb_cx: floats = null
|
||||
var wb_cz: floats = null
|
||||
var wb_ex: floats = null
|
||||
var wb_ez: floats = null
|
||||
var wb_reflect: words = null
|
||||
var wb_primary: int = -1 # the body the reflection pass mirrors, or -1
|
||||
|
||||
# Render the scene through a camera mirrored in the water plane into water_refl,
|
||||
# clipping everything below the surface; terrain, scattered layers and sky.
|
||||
function water_reflection_pass() -> void {
|
||||
if wb_primary < 0 { return } # no body reflects: nothing to mirror
|
||||
if water_refl == null {
|
||||
if r3d_env_has("R3D_REFLDIV") { water_refl_div = Text.to_int(r3d_env("R3D_REFLDIV")) }
|
||||
if water_refl != null { target_free(water_refl) }
|
||||
function water_reflection_pass(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.wb_primary < 0 { return } # no body reflects: nothing to mirror
|
||||
if render3d_st.water_refl == null {
|
||||
if r3d_env_has(render3d_st, "R3D_REFLDIV") { render3d_st.water_refl_div = Text.to_int(r3d_env(render3d_st, "R3D_REFLDIV")) }
|
||||
if render3d_st.water_refl != null { target_free(render3d_st, render3d_st.water_refl) }
|
||||
# sized from the scene target, not the window: with a render scale below 1 the frame
|
||||
# this reflection is composited into is smaller than the drawable, and a reflection
|
||||
# rendered at the window's size would be paying for pixels the water never samples
|
||||
water_refl = target_new(post_w / water_refl_div, post_h / water_refl_div, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, true, GL_LINEAR)
|
||||
water_saved = floats(16 * 4 + 3)
|
||||
water_saved_vp = view(water_saved, 16, 16)
|
||||
water_saved_ivp = view(water_saved, 32, 16)
|
||||
render3d_st.water_refl = target_new(render3d_st, render3d_st.post_w / render3d_st.water_refl_div, render3d_st.post_h / render3d_st.water_refl_div, GL_RGBA16F, GL_RGBA, GL_HALF_FLOAT, true, GL_LINEAR)
|
||||
render3d_st.water_saved = floats(16 * 4 + 3)
|
||||
render3d_st.water_saved_vp = view(render3d_st.water_saved, 16, 16)
|
||||
render3d_st.water_saved_ivp = view(render3d_st.water_saved, 32, 16)
|
||||
}
|
||||
# save the camera
|
||||
m4_copy(water_saved, cam_view)
|
||||
m4_copy(water_saved_vp, cam_vp)
|
||||
m4_copy(water_saved_ivp, cam_inv_vp)
|
||||
let sx = cam_pos[0]; let sy = cam_pos[1]; let sz = cam_pos[2]
|
||||
m4_copy(render3d_st.water_saved, render3d_st.cam_view)
|
||||
m4_copy(render3d_st.water_saved_vp, render3d_st.cam_vp)
|
||||
m4_copy(render3d_st.water_saved_ivp, render3d_st.cam_inv_vp)
|
||||
let sx = render3d_st.cam_pos[0]; let sy = render3d_st.cam_pos[1]; let sz = render3d_st.cam_pos[2]
|
||||
# the mirrored camera: view' = view * R, R reflecting y about the surface (y' = 2L - y).
|
||||
# R has determinant -1, so the winding flips (front faces culled below) and the image
|
||||
# lands exactly where the main camera's pixels expect the reflection.
|
||||
let eye = v3_new(sx, 2.0 * water_level - sy, sz)
|
||||
let eye = v3_new(sx, 2.0 * render3d_st.water_level - sy, sz)
|
||||
let refl = m4_new()
|
||||
refl[5] = -1.0
|
||||
refl[13] = 2.0 * water_level
|
||||
refl[13] = 2.0 * render3d_st.water_level
|
||||
let mv = floats(16)
|
||||
m4_mul(mv, water_saved, refl)
|
||||
m4_copy(cam_view, mv)
|
||||
m4_mul(mv, render3d_st.water_saved, refl)
|
||||
m4_copy(render3d_st.cam_view, mv)
|
||||
free(mv); free(refl)
|
||||
let fwd = words(3); let up = words(3); let at = words(3)
|
||||
m4_mul(cam_vp, cam_proj, cam_view)
|
||||
m4_inverse(cam_inv_vp, cam_vp)
|
||||
v3_copy(cam_pos, eye)
|
||||
r3d_clip_y = water_level - 0.05
|
||||
target_bind(water_refl)
|
||||
gpu_depth_test(true)
|
||||
gpu_depth_func(GL_LESS)
|
||||
gpu_depth_write(true)
|
||||
gpu_cull(true)
|
||||
gpu_cull_face(GL_FRONT) # the mirror flips the winding
|
||||
gpu_clear_color(0.0, 0.0, 0.0, 1.0)
|
||||
gpu_clear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
||||
sc_freeze = true
|
||||
let sb = sc_skip_blade
|
||||
sc_skip_blade = true # blades are invisible at this scale in a reflection
|
||||
ter_reflect = true
|
||||
prof_cpu_mark("reflection setup")
|
||||
terrain_draw()
|
||||
prof_cpu_mark("reflection terrain")
|
||||
r3d_scene_draw()
|
||||
prof_cpu_mark("reflection scene")
|
||||
r3d_draw_sky()
|
||||
prof_cpu_mark("reflection sky")
|
||||
ter_reflect = false
|
||||
sc_skip_blade = sb
|
||||
sc_freeze = false
|
||||
gpu_cull_face(GL_BACK)
|
||||
if r3d_env_has("R3D_DUMP_REFL") and not water_dumped { water_dumped = true; tex_dump(water_refl.color, water_refl.w, water_refl.h, "build/dbg_refl.ppm") }
|
||||
m4_mul(render3d_st.cam_vp, render3d_st.cam_proj, render3d_st.cam_view)
|
||||
m4_inverse(render3d_st.cam_inv_vp, render3d_st.cam_vp)
|
||||
v3_copy(render3d_st.cam_pos, eye)
|
||||
render3d_st.r3d_clip_y = render3d_st.water_level - 0.05
|
||||
target_bind(render3d_st, render3d_st.water_refl)
|
||||
gpu_depth_test(render3d_st, true)
|
||||
gpu_depth_func(render3d_st, GL_LESS)
|
||||
gpu_depth_write(render3d_st, true)
|
||||
gpu_cull(render3d_st, true)
|
||||
gpu_cull_face(render3d_st, GL_FRONT) # the mirror flips the winding
|
||||
gpu_clear_color(render3d_st, 0.0, 0.0, 0.0, 1.0)
|
||||
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT)
|
||||
render3d_st.sc_freeze = true
|
||||
let sb = render3d_st.sc_skip_blade
|
||||
render3d_st.sc_skip_blade = true # blades are invisible at this scale in a reflection
|
||||
render3d_st.ter_reflect = true
|
||||
prof_cpu_mark(render3d_st, "reflection setup")
|
||||
terrain_draw(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "reflection terrain")
|
||||
r3d_scene_draw(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "reflection scene")
|
||||
r3d_draw_sky(render3d_st)
|
||||
prof_cpu_mark(render3d_st, "reflection sky")
|
||||
render3d_st.ter_reflect = false
|
||||
render3d_st.sc_skip_blade = sb
|
||||
render3d_st.sc_freeze = false
|
||||
gpu_cull_face(render3d_st, GL_BACK)
|
||||
if r3d_env_has(render3d_st, "R3D_DUMP_REFL") and not render3d_st.water_dumped { render3d_st.water_dumped = true; tex_dump(render3d_st, render3d_st.water_refl.color, render3d_st.water_refl.w, render3d_st.water_refl.h, "build/dbg_refl.ppm") }
|
||||
# restore
|
||||
r3d_clip_y = -2147483600.0
|
||||
m4_copy(cam_view, water_saved)
|
||||
m4_copy(cam_vp, water_saved_vp)
|
||||
m4_copy(cam_inv_vp, water_saved_ivp)
|
||||
v3_set(cam_pos, sx, sy, sz)
|
||||
render3d_st.r3d_clip_y = -2147483600.0
|
||||
m4_copy(render3d_st.cam_view, render3d_st.water_saved)
|
||||
m4_copy(render3d_st.cam_vp, render3d_st.water_saved_vp)
|
||||
m4_copy(render3d_st.cam_inv_vp, render3d_st.water_saved_ivp)
|
||||
v3_set(render3d_st.cam_pos, sx, sy, sz)
|
||||
free(eye); free(fwd); free(up); free(at)
|
||||
gpu_fb_bind(0)
|
||||
gpu_fb_bind(render3d_st, 0)
|
||||
}
|
||||
|
||||
# Is any of the mirrored water in view? The reflection pass is a second copy of the terrain, the
|
||||
|
|
@ -118,39 +93,32 @@ function water_reflection_pass() -> void {
|
|||
# asking, so it errs on the side of running. R3D_REFL_ALWAYS=1 runs it on every frame, for comparing.
|
||||
# Mac, the five shot viewpoints: looking down at the meadow (c) 225 reflection draws -> none; every view
|
||||
# with the lake in it unchanged, byte-identical frames.
|
||||
var wb_cells: floats = null # x0, z0, x1, z1 of each rectangle with water showing
|
||||
var wb_ncells: int = -1 # -1: not built for the current mirrored body
|
||||
var wb_blocks: words = null # x0, z0, x1, z1, first cell, cells past the last - of each block
|
||||
var wb_nblocks: int = 0
|
||||
var wb_refl_always: int = -1
|
||||
var wb_dbg: int = -1
|
||||
var wb_build_us: long = 0 # how long the last water_cells_build took (R3D_REFL_DBG prints it)
|
||||
function water_reflect_visible() -> bool {
|
||||
if wb_primary < 0 { return false }
|
||||
if wb_refl_always < 0 { wb_refl_always = 0; if r3d_env_has("R3D_REFL_ALWAYS") { wb_refl_always = 1 } }
|
||||
if wb_refl_always == 1 or ter_heights == null { return true }
|
||||
if wb_ncells < 0 { let t0 = gl_now_us(); water_cells_build(); wb_build_us = gl_now_us() - t0 }
|
||||
if wb_dbg < 0 { wb_dbg = 0; if r3d_env_has("R3D_REFL_DBG") { wb_dbg = 1 } }
|
||||
if wb_dbg == 1 {
|
||||
function water_reflect_visible(render3d_st: mut Render3dState) -> bool {
|
||||
if render3d_st.wb_primary < 0 { return false }
|
||||
if render3d_st.wb_refl_always < 0 { render3d_st.wb_refl_always = 0; if r3d_env_has(render3d_st, "R3D_REFL_ALWAYS") { render3d_st.wb_refl_always = 1 } }
|
||||
if render3d_st.wb_refl_always == 1 or render3d_st.ter_heights == null { return true }
|
||||
if render3d_st.wb_ncells < 0 { let t0 = gl_now_us(); water_cells_build(render3d_st); render3d_st.wb_build_us = gl_now_us() - t0 }
|
||||
if render3d_st.wb_dbg < 0 { render3d_st.wb_dbg = 0; if r3d_env_has(render3d_st, "R3D_REFL_DBG") { render3d_st.wb_dbg = 1 } }
|
||||
if render3d_st.wb_dbg == 1 {
|
||||
# R3D_REFL_DBG: once, what the test is made of and how much of it is in view
|
||||
wb_dbg = 2
|
||||
render3d_st.wb_dbg = 2
|
||||
var seen = 0
|
||||
var hidden = 0
|
||||
for i in 0 .. wb_ncells { if wb_cell_visible(i) { seen += 1; if wb_cell_occluded(i) { hidden += 1 } } }
|
||||
print(`water reflection test: {wb_ncells} wet rectangles in {wb_nblocks} blocks (built in {wb_build_us} us), {seen} in view, {hidden} of them behind the ground`)
|
||||
for i in 0 .. render3d_st.wb_ncells { if wb_cell_visible(render3d_st, i) { seen += 1; if wb_cell_occluded(render3d_st, i) { hidden += 1 } } }
|
||||
print(`water reflection test: {render3d_st.wb_ncells} wet rectangles in {render3d_st.wb_nblocks} blocks (built in {render3d_st.wb_build_us} us), {seen} in view, {hidden} of them behind the ground`)
|
||||
}
|
||||
# A rectangle in the frustum can still be behind the ground or dry where it shows: standing on a shore
|
||||
# above the lake and looking down at your feet puts water inside the view and none of it on screen.
|
||||
# The first 16 in view are checked against the height field; past that the pass simply runs.
|
||||
var checked = 0
|
||||
for b in 0 .. wb_nblocks {
|
||||
for b in 0 .. render3d_st.wb_nblocks {
|
||||
let o = b * 6
|
||||
if not wb_rect_visible(float_from_bits(wb_blocks[o]), float_from_bits(wb_blocks[o + 1]), float_from_bits(wb_blocks[o + 2]), float_from_bits(wb_blocks[o + 3])) { continue }
|
||||
for i in wb_blocks[o + 4] .. wb_blocks[o + 5] {
|
||||
if not wb_cell_visible(i) { continue }
|
||||
if not wb_rect_visible(render3d_st, float_from_bits(render3d_st.wb_blocks[o]), float_from_bits(render3d_st.wb_blocks[o + 1]), float_from_bits(render3d_st.wb_blocks[o + 2]), float_from_bits(render3d_st.wb_blocks[o + 3])) { continue }
|
||||
for i in render3d_st.wb_blocks[o + 4] .. render3d_st.wb_blocks[o + 5] {
|
||||
if not wb_cell_visible(render3d_st, i) { continue }
|
||||
if checked >= 16 { return true }
|
||||
checked += 1
|
||||
if not wb_cell_occluded(i) { return true }
|
||||
if not wb_cell_occluded(render3d_st, i) { return true }
|
||||
}
|
||||
}
|
||||
return false
|
||||
|
|
@ -158,82 +126,82 @@ function water_reflect_visible() -> bool {
|
|||
# A flat rectangle at the water's level against the view's side planes: hidden when all four corners
|
||||
# lie outside one plane. (A bounding sphere was the first version: a 156 m cell's sphere reached into
|
||||
# the view from a lake the camera had its back to, and the pass never skipped.)
|
||||
function wb_rect_visible(x0: float, z0: float, x1: float, z1: float) -> bool {
|
||||
if cam_planes == null { return true }
|
||||
function wb_rect_visible(render3d_st: Render3dState, x0: float, z0: float, x1: float, z1: float) -> bool {
|
||||
if render3d_st.cam_planes == null { return true }
|
||||
for p in 0 .. 4 {
|
||||
let q = p * 4
|
||||
let a = cam_planes[q]; let c = cam_planes[q + 2]
|
||||
let by = cam_planes[q + 1] * water_level + cam_planes[q + 3]
|
||||
let a = render3d_st.cam_planes[q]; let c = render3d_st.cam_planes[q + 2]
|
||||
let by = render3d_st.cam_planes[q + 1] * render3d_st.water_level + render3d_st.cam_planes[q + 3]
|
||||
let ax0 = a * x0; let ax1 = a * x1; let cz0 = c * z0; let cz1 = c * z1
|
||||
if ax0 + cz0 + by < 0.0 and ax1 + cz0 + by < 0.0 and ax0 + cz1 + by < 0.0 and ax1 + cz1 + by < 0.0 { return false }
|
||||
}
|
||||
return true
|
||||
}
|
||||
function wb_cell_visible(i: int) -> bool {
|
||||
function wb_cell_visible(render3d_st: Render3dState, i: int) -> bool {
|
||||
let o = i * 4
|
||||
return wb_rect_visible(wb_cells[o], wb_cells[o + 1], wb_cells[o + 2], wb_cells[o + 3])
|
||||
return wb_rect_visible(render3d_st, render3d_st.wb_cells[o], render3d_st.wb_cells[o + 1], render3d_st.wb_cells[o + 2], render3d_st.wb_cells[o + 3])
|
||||
}
|
||||
# Is the water at (x, z) out of sight - dry ground there, outside the frame, or behind the ground? The
|
||||
# sight line is sampled at seven points short of both ends, with half a metre (and a little more with
|
||||
# distance) of margin, so a far ridge the drawn terrain rounds off never hides real water. A camera at
|
||||
# or under the surface hides nothing behind the ground.
|
||||
function wb_point_hidden(x: float, z: float) -> bool {
|
||||
if terrain_height(x, z) > water_level + 0.05 { return true }
|
||||
if not cam_sphere_visible(x, water_level, z, 0.5) { return true }
|
||||
let ey = cam_pos[1]
|
||||
if not (ey > water_level) { return false }
|
||||
let dx = x - cam_pos[0]; let dz = z - cam_pos[2]
|
||||
function wb_point_hidden(render3d_st: mut Render3dState, x: float, z: float) -> bool {
|
||||
if terrain_height(render3d_st, x, z) > render3d_st.water_level + 0.05 { return true }
|
||||
if not cam_sphere_visible(render3d_st, x, render3d_st.water_level, z, 0.5) { return true }
|
||||
let ey = render3d_st.cam_pos[1]
|
||||
if not (ey > render3d_st.water_level) { return false }
|
||||
let dx = x - render3d_st.cam_pos[0]; let dz = z - render3d_st.cam_pos[2]
|
||||
let margin = 0.5 + Math.sqrt(dx * dx + dz * dz) * 0.004
|
||||
for k in 1 .. 8 {
|
||||
let t = float(k) / 8.0
|
||||
let sy = ey + (water_level - ey) * t
|
||||
if terrain_height(cam_pos[0] + dx * t, cam_pos[2] + dz * t) > sy + margin { return true }
|
||||
let sy = ey + (render3d_st.water_level - ey) * t
|
||||
if terrain_height(render3d_st, render3d_st.cam_pos[0] + dx * t, render3d_st.cam_pos[2] + dz * t) > sy + margin { return true }
|
||||
}
|
||||
return false
|
||||
}
|
||||
# every point the wet test sampled (the centre and four near the corners) is out of sight
|
||||
function wb_cell_occluded(i: int) -> bool {
|
||||
function wb_cell_occluded(render3d_st: mut Render3dState, i: int) -> bool {
|
||||
let o = i * 4
|
||||
let x0 = wb_cells[o]; let z0 = wb_cells[o + 1]; let x1 = wb_cells[o + 2]; let z1 = wb_cells[o + 3]
|
||||
let x0 = render3d_st.wb_cells[o]; let z0 = render3d_st.wb_cells[o + 1]; let x1 = render3d_st.wb_cells[o + 2]; let z1 = render3d_st.wb_cells[o + 3]
|
||||
let cx = (x0 + x1) * 0.5; let cz = (z0 + z1) * 0.5
|
||||
let ox = (x1 - x0) * 0.45; let oz = (z1 - z0) * 0.45
|
||||
if not wb_point_hidden(cx, cz) { return false }
|
||||
if not wb_point_hidden(cx - ox, cz - oz) { return false }
|
||||
if not wb_point_hidden(cx + ox, cz - oz) { return false }
|
||||
if not wb_point_hidden(cx - ox, cz + oz) { return false }
|
||||
return wb_point_hidden(cx + ox, cz + oz)
|
||||
if not wb_point_hidden(render3d_st, cx, cz) { return false }
|
||||
if not wb_point_hidden(render3d_st, cx - ox, cz - oz) { return false }
|
||||
if not wb_point_hidden(render3d_st, cx + ox, cz - oz) { return false }
|
||||
if not wb_point_hidden(render3d_st, cx - ox, cz + oz) { return false }
|
||||
return wb_point_hidden(render3d_st, cx + ox, cz + oz)
|
||||
}
|
||||
function wb_cell_add(x0: float, z0: float, x1: float, z1: float) -> void {
|
||||
let o = wb_ncells * 4
|
||||
wb_cells[o] = x0; wb_cells[o + 1] = z0; wb_cells[o + 2] = x1; wb_cells[o + 3] = z1
|
||||
wb_ncells += 1
|
||||
function wb_cell_add(render3d_st: mut Render3dState, x0: float, z0: float, x1: float, z1: float) -> void {
|
||||
let o = render3d_st.wb_ncells * 4
|
||||
render3d_st.wb_cells[o] = x0; render3d_st.wb_cells[o + 1] = z0; render3d_st.wb_cells[o + 2] = x1; render3d_st.wb_cells[o + 3] = z1
|
||||
render3d_st.wb_ncells += 1
|
||||
}
|
||||
function wb_block_add(x0: float, z0: float, x1: float, z1: float, first: int) -> void {
|
||||
if wb_ncells <= first { return }
|
||||
let o = wb_nblocks * 6
|
||||
wb_blocks[o] = float_bits(x0); wb_blocks[o + 1] = float_bits(z0); wb_blocks[o + 2] = float_bits(x1); wb_blocks[o + 3] = float_bits(z1)
|
||||
wb_blocks[o + 4] = first; wb_blocks[o + 5] = wb_ncells
|
||||
wb_nblocks += 1
|
||||
function wb_block_add(render3d_st: mut Render3dState, x0: float, z0: float, x1: float, z1: float, first: int) -> void {
|
||||
if render3d_st.wb_ncells <= first { return }
|
||||
let o = render3d_st.wb_nblocks * 6
|
||||
render3d_st.wb_blocks[o] = float_bits(x0); render3d_st.wb_blocks[o + 1] = float_bits(z0); render3d_st.wb_blocks[o + 2] = float_bits(x1); render3d_st.wb_blocks[o + 3] = float_bits(z1)
|
||||
render3d_st.wb_blocks[o + 4] = first; render3d_st.wb_blocks[o + 5] = render3d_st.wb_ncells
|
||||
render3d_st.wb_nblocks += 1
|
||||
}
|
||||
function wb_wet(x: float, z: float, off: float) -> bool {
|
||||
if terrain_height(x, z) < water_level { return true }
|
||||
function wb_wet(render3d_st: mut Render3dState, x: float, z: float, off: float) -> bool {
|
||||
if terrain_height(render3d_st, x, z) < render3d_st.water_level { return true }
|
||||
if off == 0.0 { return false }
|
||||
if terrain_height(x - off, z - off) < water_level { return true }
|
||||
if terrain_height(x + off, z - off) < water_level { return true }
|
||||
if terrain_height(x - off, z + off) < water_level { return true }
|
||||
return terrain_height(x + off, z + off) < water_level
|
||||
if terrain_height(render3d_st, x - off, z - off) < render3d_st.water_level { return true }
|
||||
if terrain_height(render3d_st, x + off, z - off) < render3d_st.water_level { return true }
|
||||
if terrain_height(render3d_st, x - off, z + off) < render3d_st.water_level { return true }
|
||||
return terrain_height(render3d_st, x + off, z + off) < render3d_st.water_level
|
||||
}
|
||||
# Over the height map, 32 m rectangles (at most 512 a side) tested at five points and kept in blocks of
|
||||
# 8 x 8, so a block out of view skips its rectangles in one test - looking away from the water was a
|
||||
# walk of every rectangle, every frame. The body beyond the map (a sea runs far past it) is 64 x 64
|
||||
# coarse rectangles tested at their centre, in one last block, which only ever matter kilometres away.
|
||||
const WB_BLOCK: int = 8
|
||||
function water_cells_build() -> void {
|
||||
let bx0 = water_cx - water_ex; let bx1 = water_cx + water_ex
|
||||
let bz0 = water_cz - water_ez; let bz1 = water_cz + water_ez
|
||||
let th = float(TERRAIN_HALF)
|
||||
let tx0 = Math.max(bx0, ter_ox - th); let tx1 = Math.min(bx1, ter_ox + th)
|
||||
let tz0 = Math.max(bz0, ter_oz - th); let tz1 = Math.min(bz1, ter_oz + th)
|
||||
function water_cells_build(render3d_st: mut Render3dState) -> void {
|
||||
let bx0 = render3d_st.water_cx - render3d_st.water_ex; let bx1 = render3d_st.water_cx + render3d_st.water_ex
|
||||
let bz0 = render3d_st.water_cz - render3d_st.water_ez; let bz1 = render3d_st.water_cz + render3d_st.water_ez
|
||||
let th = float(render3d_st.TERRAIN_HALF)
|
||||
let tx0 = Math.max(bx0, render3d_st.ter_ox - th); let tx1 = Math.min(bx1, render3d_st.ter_ox + th)
|
||||
let tz0 = Math.max(bz0, render3d_st.ter_oz - th); let tz1 = Math.min(bz1, render3d_st.ter_oz + th)
|
||||
let inside = tx1 > tx0 and tz1 > tz0
|
||||
var cell = 32.0
|
||||
var nx = 0
|
||||
|
|
@ -244,35 +212,35 @@ function water_cells_build() -> void {
|
|||
nx = int((tx1 - tx0) / cell) + 1
|
||||
nz = int((tz1 - tz0) / cell) + 1
|
||||
}
|
||||
if wb_cells != null { free(wb_cells) }
|
||||
if wb_blocks != null { free(wb_blocks) }
|
||||
wb_cells = floats((nx * nz + 64 * 64) * 4)
|
||||
if render3d_st.wb_cells != null { free(render3d_st.wb_cells) }
|
||||
if render3d_st.wb_blocks != null { free(render3d_st.wb_blocks) }
|
||||
render3d_st.wb_cells = floats((nx * nz + 64 * 64) * 4)
|
||||
let nbx = (nx + WB_BLOCK - 1) / WB_BLOCK
|
||||
let nbz = (nz + WB_BLOCK - 1) / WB_BLOCK
|
||||
wb_blocks = words((nbx * nbz + 1) * 6)
|
||||
wb_ncells = 0
|
||||
wb_nblocks = 0
|
||||
render3d_st.wb_blocks = words((nbx * nbz + 1) * 6)
|
||||
render3d_st.wb_ncells = 0
|
||||
render3d_st.wb_nblocks = 0
|
||||
let off = cell * 0.45
|
||||
let half = cell * 0.5
|
||||
for bz in 0 .. nbz {
|
||||
for bx in 0 .. nbx {
|
||||
let first = wb_ncells
|
||||
let first = render3d_st.wb_ncells
|
||||
var iz = bz * WB_BLOCK
|
||||
while iz < (bz + 1) * WB_BLOCK and iz < nz {
|
||||
let z0 = tz0 + float(iz) * cell
|
||||
var ix = bx * WB_BLOCK
|
||||
while ix < (bx + 1) * WB_BLOCK and ix < nx {
|
||||
let x0 = tx0 + float(ix) * cell
|
||||
if wb_wet(x0 + half, z0 + half, off) { wb_cell_add(x0, z0, x0 + cell, z0 + cell) }
|
||||
if wb_wet(render3d_st, x0 + half, z0 + half, off) { wb_cell_add(render3d_st, x0, z0, x0 + cell, z0 + cell) }
|
||||
ix += 1
|
||||
}
|
||||
iz += 1
|
||||
}
|
||||
let x0 = tx0 + float(bx * WB_BLOCK) * cell; let z0 = tz0 + float(bz * WB_BLOCK) * cell
|
||||
wb_block_add(x0, z0, x0 + float(WB_BLOCK) * cell, z0 + float(WB_BLOCK) * cell, first)
|
||||
wb_block_add(render3d_st, x0, z0, x0 + float(WB_BLOCK) * cell, z0 + float(WB_BLOCK) * cell, first)
|
||||
}
|
||||
}
|
||||
let first = wb_ncells
|
||||
let first = render3d_st.wb_ncells
|
||||
let cw = (bx1 - bx0) / 64.0
|
||||
let ch = (bz1 - bz0) / 64.0
|
||||
for iz in 0 .. 64 {
|
||||
|
|
@ -281,64 +249,64 @@ function water_cells_build() -> void {
|
|||
let x0 = bx0 + float(ix) * cw; let x1 = x0 + cw
|
||||
# inside the height map's rectangle: the fine cells above cover it
|
||||
if inside and not (x0 < tx0) and not (x1 > tx1) and not (z0 < tz0) and not (z1 > tz1) { continue }
|
||||
if wb_wet(x0 + cw * 0.5, z0 + ch * 0.5, 0.0) { wb_cell_add(x0, z0, x1, z1) }
|
||||
if wb_wet(render3d_st, x0 + cw * 0.5, z0 + ch * 0.5, 0.0) { wb_cell_add(render3d_st, x0, z0, x1, z1) }
|
||||
}
|
||||
}
|
||||
wb_block_add(bx0, bz0, bx1, bz1, first)
|
||||
wb_block_add(render3d_st, bx0, bz0, bx1, bz1, first)
|
||||
}
|
||||
|
||||
# the program and the plane are the process's, built once; the bodies are the map's
|
||||
function water_setup() -> void {
|
||||
if water_prog != 0 { return }
|
||||
water_mesh = mesh_grid(2, 0.5)
|
||||
water_prog = r3d_program("water.vert", "water.frag", "")
|
||||
wb_level = floats(WATER_MAX); wb_cx = floats(WATER_MAX); wb_cz = floats(WATER_MAX)
|
||||
wb_ex = floats(WATER_MAX); wb_ez = floats(WATER_MAX); wb_reflect = words(WATER_MAX)
|
||||
function water_setup(render3d_st: mut Render3dState) -> void {
|
||||
if render3d_st.water_prog != 0 { return }
|
||||
render3d_st.water_mesh = mesh_grid(render3d_st, 2, 0.5)
|
||||
render3d_st.water_prog = r3d_program(render3d_st, "water.vert", "water.frag", "")
|
||||
render3d_st.wb_level = floats(WATER_MAX); render3d_st.wb_cx = floats(WATER_MAX); render3d_st.wb_cz = floats(WATER_MAX)
|
||||
render3d_st.wb_ex = floats(WATER_MAX); render3d_st.wb_ez = floats(WATER_MAX); render3d_st.wb_reflect = words(WATER_MAX)
|
||||
}
|
||||
function water_bodies_clear() -> void {
|
||||
wb_n = 0
|
||||
wb_primary = -1
|
||||
wb_ncells = -1
|
||||
water_on = false
|
||||
function water_bodies_clear(render3d_st: mut Render3dState) -> void {
|
||||
render3d_st.wb_n = 0
|
||||
render3d_st.wb_primary = -1
|
||||
render3d_st.wb_ncells = -1
|
||||
render3d_st.water_on = false
|
||||
}
|
||||
# Add a plane at `level` over (cx, cz) +- (ex, ez); true `reflect` makes it the mirrored one
|
||||
# if none is yet. Returns its index, or -1 once WATER_MAX are in use.
|
||||
function water_body_add(level: float, cx: float, cz: float, ex: float, ez: float, reflect: bool) -> int {
|
||||
water_setup()
|
||||
if wb_n >= WATER_MAX { return -1 }
|
||||
let i = wb_n
|
||||
wb_level[i] = level; wb_cx[i] = cx; wb_cz[i] = cz; wb_ex[i] = ex; wb_ez[i] = ez
|
||||
wb_reflect[i] = 0
|
||||
if reflect { wb_reflect[i] = 1 }
|
||||
wb_n += 1
|
||||
water_on = true
|
||||
if reflect and wb_primary < 0 {
|
||||
wb_primary = i
|
||||
wb_ncells = -1
|
||||
water_level = level; water_cx = cx; water_cz = cz; water_ex = ex; water_ez = ez
|
||||
function water_body_add(render3d_st: mut Render3dState, level: float, cx: float, cz: float, ex: float, ez: float, reflect: bool) -> int {
|
||||
water_setup(render3d_st)
|
||||
if render3d_st.wb_n >= WATER_MAX { return -1 }
|
||||
let i = render3d_st.wb_n
|
||||
render3d_st.wb_level[i] = level; render3d_st.wb_cx[i] = cx; render3d_st.wb_cz[i] = cz; render3d_st.wb_ex[i] = ex; render3d_st.wb_ez[i] = ez
|
||||
render3d_st.wb_reflect[i] = 0
|
||||
if reflect { render3d_st.wb_reflect[i] = 1 }
|
||||
render3d_st.wb_n += 1
|
||||
render3d_st.water_on = true
|
||||
if reflect and render3d_st.wb_primary < 0 {
|
||||
render3d_st.wb_primary = i
|
||||
render3d_st.wb_ncells = -1
|
||||
render3d_st.water_level = level; render3d_st.water_cx = cx; render3d_st.water_cz = cz; render3d_st.water_ex = ex; render3d_st.water_ez = ez
|
||||
}
|
||||
return i
|
||||
}
|
||||
# one reflecting plane: what this function always meant, without a new mesh and program
|
||||
# every time it is called
|
||||
function water_init(level: float, cx: float, cz: float, ex: float, ez: float) -> void {
|
||||
water_bodies_clear()
|
||||
water_body_add(level, cx, cz, ex, ez, true)
|
||||
function water_init(render3d_st: mut Render3dState, level: float, cx: float, cz: float, ex: float, ez: float) -> void {
|
||||
water_bodies_clear(render3d_st)
|
||||
water_body_add(render3d_st, level, cx, cz, ex, ez, true)
|
||||
}
|
||||
|
||||
# call after the opaque pass, before the sky: blends over the resolved depth
|
||||
function water_draw(depth_tex: int) -> void {
|
||||
if not water_on or wb_n == 0 { return }
|
||||
let p = water_prog
|
||||
gpu_use_program(p)
|
||||
u_mat4(gpu_uniform(p, "u_view"), cam_view)
|
||||
u_mat4(gpu_uniform(p, "u_proj"), cam_proj)
|
||||
u_mat4(gpu_uniform(p, "u_inv_vp"), cam_inv_vp)
|
||||
function water_draw(render3d_st: mut Render3dState, depth_tex: int) -> void {
|
||||
if not render3d_st.water_on or render3d_st.wb_n == 0 { return }
|
||||
let p = render3d_st.water_prog
|
||||
gpu_use_program(render3d_st, p)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_view"), render3d_st.cam_view)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_proj"), render3d_st.cam_proj)
|
||||
u_mat4(render3d_st, gpu_uniform(render3d_st, p, "u_inv_vp"), render3d_st.cam_inv_vp)
|
||||
# gl_FragCoord here runs over the scene target, which is post_w x post_h — not the
|
||||
# window. They are the same size only at a render scale of 1; at anything less, taking
|
||||
# the window's size sent the refraction and depth reads into the wrong corner of the
|
||||
# frame, and the lake showed a squashed copy of it instead of its own bed.
|
||||
u_f2(gpu_uniform(p, "u_screen"), float(post_w), float(post_h))
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, p, "u_screen"), float(render3d_st.post_w), float(render3d_st.post_h))
|
||||
# Read into locals first. Passing these three globals straight into the call gives the
|
||||
# shader wrong values - the whole lake churns instead of a patch of it - and a single dead
|
||||
# `let junk = wt_wade_x` above the same unchanged call is enough to make it correct again.
|
||||
|
|
@ -347,49 +315,49 @@ function water_draw(depth_tex: int) -> void {
|
|||
# right after gpu_use_program) nor the nested gpu_uniform call (hoisting that changes
|
||||
# nothing). Verified by picture, both backends. Read a global into a local before handing it
|
||||
# to a uniform call.
|
||||
let wx = wt_wade_x
|
||||
let wz = wt_wade_z
|
||||
let ws = wt_wade_s
|
||||
u_f3(gpu_uniform(p, "u_wade"), wx, wz, ws)
|
||||
let wx = render3d_st.wt_wade_x
|
||||
let wz = render3d_st.wt_wade_z
|
||||
let ws = render3d_st.wt_wade_s
|
||||
u_f3(render3d_st, gpu_uniform(render3d_st, p, "u_wade"), wx, wz, ws)
|
||||
var ron = 0.0
|
||||
if water_refl != null and wb_primary >= 0 {
|
||||
if render3d_st.water_refl != null and render3d_st.wb_primary >= 0 {
|
||||
# bind on its own unit first: generating the mip chain re-binds the texture on the active unit,
|
||||
# and it must not displace the depth texture the shader reads for the shore
|
||||
r3d_bind_2d(p, "u_refl", 1, water_refl.color); ron = 1.0
|
||||
gpu_tex_param(GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(GPU_TEX2D)
|
||||
r3d_bind_2d(render3d_st, p, "u_refl", 1, render3d_st.water_refl.color); ron = 1.0
|
||||
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
|
||||
gpu_tex_mips(render3d_st, GPU_TEX2D)
|
||||
}
|
||||
r3d_bind_2d(p, "u_depth", 0, depth_tex)
|
||||
r3d_bind_2d(p, "u_scene", 2, post_scene.color)
|
||||
sky_bind_lighting(p)
|
||||
shadow_bind(p)
|
||||
fog_bind(p)
|
||||
r3d_bind_2d(render3d_st, p, "u_depth", 0, depth_tex)
|
||||
r3d_bind_2d(render3d_st, p, "u_scene", 2, render3d_st.post_scene.color)
|
||||
sky_bind_lighting(render3d_st, p)
|
||||
shadow_bind(render3d_st, p)
|
||||
fog_bind(render3d_st, p)
|
||||
# Opaque. The surface composites the refracted bed itself, so there is nothing for
|
||||
# hardware blending to do — and an alpha was what left see-through gaps in the foam
|
||||
# and a clear band at the shore wide enough to give the plane away.
|
||||
gpu_blend(false)
|
||||
gpu_blend_func(GL_SRC_ALPHA, GL_ONE_MINUS_SRC_ALPHA)
|
||||
gpu_blend(render3d_st, false)
|
||||
gpu_blend_func(render3d_st, GL_SRC_ALPHA, GL_ONE_MINUS_SRC_ALPHA)
|
||||
# the surface writes depth: the ambient-occlusion and temporal passes read the frame's depth,
|
||||
# and the bed 9 m below the shore would otherwise darken a band along the water line
|
||||
gpu_depth_write(true)
|
||||
gpu_cull(false)
|
||||
gpu_depth_write(render3d_st, true)
|
||||
gpu_cull(render3d_st, false)
|
||||
# every body is the same plane at its own level and bounds; only the mirrored one samples
|
||||
# the reflection - another body reading it would show the wrong world upside down
|
||||
for i in 0 .. wb_n {
|
||||
u_f(gpu_uniform(p, "u_level"), wb_level[i])
|
||||
u_f2(gpu_uniform(p, "u_center"), wb_cx[i], wb_cz[i])
|
||||
u_f2(gpu_uniform(p, "u_extent"), wb_ex[i], wb_ez[i])
|
||||
for i in 0 .. render3d_st.wb_n {
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_level"), render3d_st.wb_level[i])
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, p, "u_center"), render3d_st.wb_cx[i], render3d_st.wb_cz[i])
|
||||
u_f2(render3d_st, gpu_uniform(render3d_st, p, "u_extent"), render3d_st.wb_ex[i], render3d_st.wb_ez[i])
|
||||
var r = 0.0
|
||||
if i == wb_primary { r = ron }
|
||||
u_f(gpu_uniform(p, "u_refl_on"), r)
|
||||
if i == render3d_st.wb_primary { r = ron }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_refl_on"), r)
|
||||
# Every body is clipped to the ellipse inside its bounds but an unbounded mirrored sea,
|
||||
# whose rectangle is the point. A mirrored LAKE is clipped like any other: a map whose
|
||||
# reflection belongs to its lake (the Bells in Maroon Lake) would otherwise draw that
|
||||
# lake's level over every hollow in the survey.
|
||||
var clip = 1.0
|
||||
if i == wb_primary and Math.max(wb_ex[i], wb_ez[i]) > 10000.0 { clip = 0.0 }
|
||||
u_f(gpu_uniform(p, "u_clip_ellipse"), clip)
|
||||
mesh_draw(water_mesh)
|
||||
if i == render3d_st.wb_primary and Math.max(render3d_st.wb_ex[i], render3d_st.wb_ez[i]) > 10000.0 { clip = 0.0 }
|
||||
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_clip_ellipse"), clip)
|
||||
mesh_draw(render3d_st, render3d_st.water_mesh)
|
||||
}
|
||||
gpu_blend(false)
|
||||
gpu_blend(render3d_st, false)
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue