render3d: the terrain's GPU maps paged round the camera while the tiles are on

At the cut the whole 4096 height, normal and photograph textures go, replaced by a coarse
2048 level (the CPU's tt_coarse uploaded; the normal and photograph blitted down on the GPU,
the photograph mipmapped) and a pool of fine tiles in three array textures - heights, normals,
photograph, each layer a tile with a one-texel border - addressed through a tt_n^2 page table
(u_tp_page: slot + 1, 0 = the coarse level). The pool holds the tiles within the reach (900 m,
or the fog wall's when nearer) and a ring, and is made again when the fog wall changes its size
(terrain_pages_fog). Once a frame (tp_frame, beside tt_frame) tiles past the reach and two tiles
go and the wanted ones come in nearest first, 8 a frame, written into a staging buffer kept for
the process (two halves, one per frame in flight) and copied into their layers inside the frame's
own command buffer - no submit of their own - with the page table re-uploaded only when it
changed. tp_bind binds the pool (or 1-layer stand-in arrays while paging is off, u_tp_on = 0) for
every program that reads the ground: terrain_bind_height (models, scatter, shadow_bind, grass),
terrain_bind_prog, the sun pass and the shadow bake. grass_cull.comp reads through the same page
table. R3D_VKMEM prints the pool's line. Tiles off, nothing changes. Compile-only: not run.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-29 17:13:11 +03:00
parent 329439cf61
commit 617ac10c5a
15 changed files with 571 additions and 12 deletions

View file

@ -848,6 +848,35 @@ export state Render3dState {
tt_reads: int = 0 # tiles read from the file, over the run and this frame
tt_frame_reads: int = 0
tt_warned: bool = false
tp_on: bool = false # the GPU pages the fine tiles round the camera (terrain_pages.ludic)
tp_page_tex: int = 0 # R32F tt_n a side: slot + 1 of each resident tile, 0 = the coarse level
tp_h_tex: int = 0 # the pool: heights, normals and photograph, a layer a slot
tp_n_tex: int = 0
tp_o_tex: int = 0
tp_dummy_page: int = 0 # bound while paging is off: an array sampler needs an array
tp_dummy_h: int = 0
tp_dummy_n: int = 0
tp_dummy_o: int = 0
tp_slots: int = 0
tp_reach: float = 0.0 # metres the pool was sized for
tp_page: floats = null # the page table's CPU mirror, uploaded whole when it changes
tp_tile_of: words = null # per slot: its tile, or -1
tp_free: words = null # the free slots, a stack
tp_nfree: int = 0
tp_dirty: bool = false
tp_cam_t: int = -1 # the camera's tile when residency last ran
tp_short: bool = false # a wanted tile was left for the next frame
tp_cand: words = null # this frame's loads, nearest first: their tiles, distances, slots
tp_cand_d: floats = null
tp_cand_s: words = null
tp_zero: words = null # the one layer a 2-D texture has, for gpu_layers_copy
tp_st_buf: long = 0 # the staging buffer the loads are written into, two halves
tp_st_mem: long = 0
tp_st_ptr: pointer = null
tp_st_bytes: int = 0
tp_loads: int = 0 # tiles loaded over the run, and this frame
tp_frame_loads: int = 0
tp_bytes: int = 0 # the pool's and the page table's GPU bytes
ter_shadow_prog_aux: int = 0 # the bake of the aux target (TS_AUX)
ter_shadow_aux: int = 0 # beside it, RG16F: the occluder distance and the cloud mask
ter_shadow_yaw: float = 1000000000.0 # the sky yaw it was baked for

View file

@ -815,7 +815,7 @@ function gvk_startup_state(render3d_st: mut Render3dState) -> void {
render3d_st.gvk_pc_last = words(4096)
for i in 0 .. 4096 { render3d_st.gvk_pc_last[i] = -1 }
render3d_st.gg_rb = words(4); render3d_st.gg_cb = words(1); render3d_st.gg_bufs = words(3)
render3d_st.gg_texs = words(3); render3d_st.gg_pr = words(48)
render3d_st.gg_texs = words(7); render3d_st.gg_pr = words(52)
if render3d_st.gpu_unit_2d == null {
render3d_st.gpu_unit_2d = words(32)
for i in 0 .. 32 { render3d_st.gpu_unit_2d[i] = -1 }

View file

@ -0,0 +1,94 @@
# gpu_vk_layers.ludic — what the terrain's page pool asks of Vulkan (terrain_pages.ludic): a staging
# buffer kept for the process, regions of it copied into chosen layers of an array texture inside the
# frame's own command buffer (no submit of their own), and a one-off linear blit for a coarse level.
# a host-visible staging buffer of n bytes kept for good (the pool's uploads reuse it every frame)
@alloc_ok("made once per resource and kept for its life: the page pool's staging buffer")
function gpu_staging_keep(render3d_st: mut Render3dState, n: int) -> pointer {
gvk_flush(render3d_st)
let p = gvk_staging(render3d_st, n, VK_BUFFER_USAGE_TRANSFER_SRC_BIT)
if p == null { return null }
render3d_st.tp_st_buf = render3d_st.gvk_st_buf
render3d_st.tp_st_mem = render3d_st.gvk_st_mem
render3d_st.tp_st_bytes = n
return p
}
# ... and let go (a map whose tiles need a larger one)
function gpu_staging_drop(render3d_st: mut Render3dState) -> void {
if render3d_st.tp_st_buf == 0 { return }
gvk_flush(render3d_st)
let buf = render3d_st.gvk_st_buf
let mem = render3d_st.gvk_st_mem
render3d_st.gvk_st_buf = render3d_st.tp_st_buf
render3d_st.gvk_st_mem = render3d_st.tp_st_mem
gvk_staging_free(render3d_st)
render3d_st.gvk_st_buf = buf
render3d_st.gvk_st_mem = mem
let zero: long = 0
render3d_st.tp_st_buf = zero; render3d_st.tp_st_mem = zero; render3d_st.tp_st_ptr = null; render3d_st.tp_st_bytes = 0
}
# the frame's command buffer, outside any pass, for copies recorded before the frame's draws
function gpu_upload_cb(render3d_st: mut Render3dState) -> pointer {
gvk_pass_end(render3d_st)
return gvk_frame_cb(render3d_st)
}
# n regions of the kept staging buffer, each w x h texels starting at base + k * stride, into layer
# layers[k] of tex's level 0: one barrier each way around them all, recorded into cb
function gpu_layers_copy(render3d_st: mut Render3dState, cb: pointer, tex: int, w: int, h: int, base: int, stride: int, layers: words, n: int) -> void {
if n <= 0 or tex <= 0 or cb == null { return }
let image = render3d_st.gvk_tex_image[tex]
let all = render3d_st.gvk_tex_layers[tex]
let levels = render3d_st.gvk_tex_levels[tex]
gvk_barrier(render3d_st, cb, image, false, 0, levels, all, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
let sz = VkBufferImageCopy_sizeof
let bic = gvk_tmp(render3d_st, sz * n)
Vk.zero(bic, sz * n)
let sub = VkBufferImageCopy_imageSubresource
for k in 0 .. n {
let at: long = base + k * stride
Vk.put_i64(bic, k * sz + VkBufferImageCopy_bufferOffset, at)
Vk.put_i32(bic, k * sz + sub + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
Vk.put_i32(bic, k * sz + sub + VkImageSubresourceLayers_baseArrayLayer, layers[k])
Vk.put_i32(bic, k * sz + sub + VkImageSubresourceLayers_layerCount, 1)
Vk.put_i32(bic, k * sz + VkBufferImageCopy_imageExtent + VkExtent3D_width, w)
Vk.put_i32(bic, k * sz + VkBufferImageCopy_imageExtent + VkExtent3D_height, h)
Vk.put_i32(bic, k * sz + VkBufferImageCopy_imageExtent + VkExtent3D_depth, 1)
}
Vk.cmd_copy_buffer_to_image(cb, render3d_st.tp_st_buf, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, n, bic)
gvk_barrier(render3d_st, cb, image, false, 0, levels, all, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
}
# level 0 of src (sw x sh) filtered into level 0 of dst (dw x dh), once: a coarse level made on the GPU
function gpu_tex_blit(render3d_st: mut Render3dState, src: int, sw: int, sh: int, dst: int, dw: int, dh: int) -> bool {
gvk_flush(render3d_st)
let si = render3d_st.gvk_tex_image[src]
let di = render3d_st.gvk_tex_image[dst]
if si == 0 or di == 0 { return false }
let cb = gvk_once_begin(render3d_st)
gvk_barrier(render3d_st, cb, si, false, 0, 1, 1, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)
gvk_barrier(render3d_st, cb, di, false, 0, 1, 1, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
let blit = gvk_tmp(render3d_st, VkImageBlit_sizeof)
Vk.zero(blit, VkImageBlit_sizeof)
Vk.put_i32(blit, VkImageBlit_srcSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
Vk.put_i32(blit, VkImageBlit_srcSubresource + VkImageSubresourceLayers_layerCount, 1)
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_x, sw)
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_y, sh)
Vk.put_i32(blit, VkImageBlit_srcOffsets + VkOffset3D_sizeof + VkOffset3D_z, 1)
Vk.put_i32(blit, VkImageBlit_dstSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
Vk.put_i32(blit, VkImageBlit_dstSubresource + VkImageSubresourceLayers_layerCount, 1)
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_x, dw)
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_y, dh)
Vk.put_i32(blit, VkImageBlit_dstOffsets + VkOffset3D_sizeof + VkOffset3D_z, 1)
Vk.cmd_blit_image(cb, si, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, di, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, blit, VK_FILTER_LINEAR)
gvk_barrier(render3d_st, cb, si, false, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
gvk_barrier(render3d_st, cb, di, false, 0, 1, 1, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
return gvk_once_end(render3d_st, cb)
}
# pixels into level 0 of a texture made without them (tex_target), laid out as gvk_tex_upload takes them
function gpu_tex_fill(render3d_st: mut Render3dState, tex: int, ifmt: int, w: int, h: int, fmt: int, ty: int, data: pointer) -> bool {
gvk_flush(render3d_st)
return gvk_tex_upload(render3d_st, tex, ifmt, w, h, 1, fmt, ty, data)
}

View file

@ -38,7 +38,7 @@ function gg_init__t(render3d_st: mut Render3dState) -> void {
if not gpu_has_compute(render3d_st) or render3d_st.grass_prog == 0 { return }
if r3d_env_has(render3d_st, "R3D_GRASS_GPU") and r3d_env(render3d_st, "R3D_GRASS_GPU") == "0" { return }
render3d_st.gg_reset = gpu_compute(render3d_st, "grass_reset", 1)
render3d_st.gg_cull = gpu_compute_tex(render3d_st, "grass_cull", 3, 3)
render3d_st.gg_cull = gpu_compute_tex(render3d_st, "grass_cull", 3, 7)
render3d_st.gg_prog = r3d_program(render3d_st, "grass_inst.vert", "model.frag", "#define FOLIAGE\n#define BLADE\n#define GBLADE\n#define GINST\n")
if render3d_st.gg_reset == 0 or render3d_st.gg_cull == 0 or render3d_st.gg_prog == 0 { return }
render3d_st.gg_tiles_buf = new []int
@ -125,13 +125,21 @@ function gg_cull_frame(render3d_st: mut Render3dState) -> void {
bufs[0] = tb; bufs[1] = render3d_st.gg_out; bufs[2] = render3d_st.gg_cmds
let texs = render3d_st.gg_texs
texs[0] = render3d_st.ter_height_tex; texs[1] = render3d_st.ter_ortho_tex; texs[2] = render3d_st.ter_normal_tex
gpu_dispatch_tex(render3d_st, render3d_st.gg_cull, data_of(pr), 192, bufs, texs, (render3d_st.gg_max + 63) / 64, render3d_st.gg_n)
gg_page_texs(render3d_st, texs)
gpu_dispatch_tex(render3d_st, render3d_st.gg_cull, data_of(pr), 208, bufs, texs, (render3d_st.gg_max + 63) / 64, render3d_st.gg_n)
}
# grass_cull.comp's Params, std140: 12 vec4s, into the block made once in gg_cull_frame
# the page pool's four, after the three whole-map ones (the stand-ins while paging is off)
function gg_page_texs(render3d_st: mut Render3dState, texs: words) -> void {
tp_dummies(render3d_st)
texs[3] = render3d_st.tp_dummy_page; texs[4] = render3d_st.tp_dummy_h; texs[5] = render3d_st.tp_dummy_n; texs[6] = render3d_st.tp_dummy_o
if render3d_st.tp_on { texs[3] = render3d_st.tp_page_tex; texs[4] = render3d_st.tp_h_tex; texs[5] = render3d_st.tp_n_tex; texs[6] = render3d_st.tp_o_tex }
}
# grass_cull.comp's Params, std140: 13 vec4s, into the block made once in gg_cull_frame
function gg_params(render3d_st: mut Render3dState) -> words {
let pr = render3d_st.gg_pr
for i in 0 .. 48 { pr[i] = 0 }
for i in 0 .. 52 { pr[i] = 0 }
if render3d_st.cam_planes != null { for i in 0 .. 16 { pr[i] = float_bits(render3d_st.cam_planes[i]) } }
pr[16] = float_bits(render3d_st.cam_pos[0]); pr[17] = float_bits(render3d_st.cam_pos[1]); pr[18] = float_bits(render3d_st.cam_pos[2]); pr[19] = float_bits(grass_reach(render3d_st))
var ph = render3d_st.post_h
@ -148,6 +156,9 @@ function gg_params(render3d_st: mut Render3dState) -> words {
pr[32] = float_bits(render3d_st.ter_ox); pr[33] = float_bits(render3d_st.ter_oz); pr[34] = float_bits(float(render3d_st.TERRAIN_HALF))
pr[36] = float_bits(GG_NEAR); pr[37] = float_bits(GG_MID)
for b in 0 .. GG_BANDS { pr[40 + b] = gg_base(b); pr[44 + b] = gg_cap(b) }
# the page pool (terrain_pages.ludic): on in lev.w, and its dims as the terrain's shaders take them
if render3d_st.tp_on { pr[31] = float_bits(1.0) }
pr[48] = float_bits(float(render3d_st.tt_n)); pr[49] = float_bits(float(TT_TEX)); pr[50] = float_bits(float(render3d_st.tt_oside)); pr[51] = float_bits(float(TERRAIN_RES))
return pr
}

View file

@ -12,6 +12,7 @@ import "gpu_manifest.ludic"
import "gpu_vk.ludic"
import "vkmem_report.ludic"
import "gpu_vk_res.ludic"
import "gpu_vk_layers.ludic"
import "gpu_vk_bc.ludic"
import "texture_dds.ludic"
import "prof.ludic"
@ -27,6 +28,10 @@ import "terrain.ludic"
import "terrain_chunks.ludic"
import "terrain_tiles.ludic"
import "terrain_tiles_cut.ludic"
import "terrain_pages.ludic"
import "terrain_pages_coarse.ludic"
import "terrain_pages_frame.ludic"
import "terrain_pages_fill.ludic"
import "overlay.ludic"
import "shadow.ludic"
import "post.ludic"

View file

@ -62,6 +62,7 @@ export function r3d_fog_wall(render3d_st: mut Render3dState, dist: float) -> voi
fog_impostors(render3d_st)
shadow_fog(render3d_st)
fog_streams(render3d_st)
terrain_pages_fog(render3d_st)
}
# how far anything is drawn: the fog wall and a margin, or `far` (0: no limit of its own) without one
function r3d_reach(render3d_st: Render3dState, far: float) -> float {
@ -227,6 +228,7 @@ function r3d_frame__t(render3d_st: mut Render3dState, time: float) -> void {
render3d_st.r3d_test_frame += 1
vkmem_tick(render3d_st)
tt_frame(render3d_st)
tp_frame(render3d_st)
fog_at_tick(render3d_st)
if render3d_st.r3d_test_resize == 0 and r3d_env_has(render3d_st, "R3D_RESIZE_AT") { render3d_st.r3d_test_resize = Text.to_int(r3d_env(render3d_st, "R3D_RESIZE_AT")) }
# R3D_RESIZE_AT=<n>: from frame n on, rebuild every screen-sized buffer every few

View file

@ -15,11 +15,12 @@ layout(set = 0, binding = 0) uniform Params {
vec4 cam; // xyz the camera, w no blades past this (u_radius)
vec4 dens; // x s0, y d0, z one pixel in radians, w the snow line
vec4 lake; // the carved lake: centre x/z, half extents (z = 0: none)
vec4 lev; // x the lake's level, y the sea's, z the photograph on (1), w unused
vec4 lev; // x the lake's level, y the sea's, z the photograph on (1), w the page pool on (1)
vec4 ts; // the height texture: origin x/z, half extent, unused
vec4 bands; // outer distance of bands 0 and 1 (x, y); band 2 is the rest
uvec4 base; // each band's first record in Out
uvec4 cap; // ... and how many it holds
vec4 tp; // the page pool (u_tp_dims): tiles a side, height texels a tile, photograph texels a tile, height texels a side
} pr;
layout(set = 0, binding = 1) readonly buffer Tiles { vec4 tiles[]; }; // corner x/z, indices per cell, cells per side
@ -28,6 +29,12 @@ layout(set = 0, binding = 3) buffer Cmds { uint cmds[]; }; // VkDra
layout(set = 0, binding = 4) uniform sampler2D u_height; // the height (terrain's u_ts_height)
layout(set = 0, binding = 5) uniform sampler2D u_ortho; // the photograph
layout(set = 0, binding = 6) uniform sampler2D u_ter_normal; // the normal: x and z, y rebuilt
// the page pool (terrain_pages.ludic): while it is on, the three above are the coarse whole-map level
// and the tiles round the camera are layers of these, addressed through the page table
layout(set = 0, binding = 7) uniform sampler2D u_tp_page; // slot + 1 per tile, 0 = coarse
layout(set = 0, binding = 8) uniform sampler2DArray u_tp_h; // (T + 2)^2 a layer, a texel of border
layout(set = 0, binding = 9) uniform sampler2DArray u_tp_nrm;
layout(set = 0, binding = 10) uniform sampler2DArray u_tp_ortho; // (S + 2)^2 a layer
const float CELL = 4.0; // grass.ludic GRASS_CELL
@ -36,8 +43,35 @@ float bladeHash(ivec2 cell, int j, int k) {
uint h = pcg(uint(cell.x + 32768) * 73856093u ^ uint(cell.y + 32768) * 19349663u ^ uint(j) * 83492791u ^ uint(k) * 2654435761u);
return float(h) * (1.0 / 4294967295.0);
}
// the resident tile's slot under a full-map uv, or -1 (then the coarse level answers)
float tpSlot(vec2 uv) {
if (pr.lev.w < 0.5) return -1.0;
ivec2 t = ivec2(floor(clamp(uv, 0.0, 0.999999) * pr.tp.x));
return texelFetch(u_tp_page, t, 0).r - 1.0;
}
// where a full-map uv falls in its slot's layer of side T (+ a border of one)
vec3 tpUv(vec2 uv, float side, float slot) {
vec2 f = fract(clamp(uv, 0.0, 0.999999) * pr.tp.x);
return vec3((1.0 + f * side) / (side + 2.0), slot);
}
float groundH(vec2 uv) {
float s = tpSlot(uv);
if (s >= 0.0) return textureLod(u_tp_h, tpUv(uv, pr.tp.y, s), 0.0).r;
return textureLod(u_height, uv, 0.0).r;
}
vec2 groundN(vec2 uv) {
float s = tpSlot(uv);
if (s >= 0.0) return textureLod(u_tp_nrm, tpUv(uv, pr.tp.y, s), 0.0).rg;
return textureLod(u_ter_normal, uv, 0.0).rg;
}
// the photograph: a resident tile's layer has no mips, so it is read at its own level
vec3 groundO(vec2 uv) {
float s = tpSlot(uv);
if (s >= 0.0) return textureLod(u_tp_ortho, tpUv(uv, pr.tp.z, s), 0.0).rgb;
return textureLod(u_ortho, uv, 1.5).rgb;
}
float heightSmooth(vec2 uv) {
vec2 res = vec2(textureSize(u_height, 0));
vec2 res = pr.lev.w > 0.5 ? vec2(pr.tp.w) : vec2(textureSize(u_height, 0));
vec2 t = uv * res - 0.5;
vec2 f = fract(t);
vec2 i = floor(t);
@ -48,8 +82,8 @@ float heightSmooth(vec2 uv) {
vec2 s0 = w0 + w1, s1 = w2 + w3;
vec2 o0 = (i - 1.0 + w1 / s0 + 0.5) / res;
vec2 o1 = (i + 1.0 + w3 / s1 + 0.5) / res;
return (textureLod(u_height, vec2(o0.x, o0.y), 0.0).r * s0.x + textureLod(u_height, vec2(o1.x, o0.y), 0.0).r * s1.x) * s0.y
+ (textureLod(u_height, vec2(o0.x, o1.y), 0.0).r * s0.x + textureLod(u_height, vec2(o1.x, o1.y), 0.0).r * s1.x) * s1.y;
return (groundH(vec2(o0.x, o0.y)) * s0.x + groundH(vec2(o1.x, o0.y)) * s1.x) * s0.y
+ (groundH(vec2(o0.x, o1.y)) * s0.x + groundH(vec2(o1.x, o1.y)) * s1.x) * s1.y;
}
// grass.vert's bladeField: the region, the patchiness, the dry patches and the tussocks' shade
vec3 bladeField(vec2 xz, float y) {
@ -89,18 +123,18 @@ void main() {
float life = 1.0 - smoothstep(keep * 0.75, keep, r);
vec2 huv = (xz - pr.ts.xy) / (2.0 * pr.ts.z) + 0.5;
if (huv.x < 0.0 || huv.x > 1.0 || huv.y < 0.0 || huv.y > 1.0) return;
vec4 ht = textureLod(u_height, huv, 0.0);
vec4 ht = vec4(groundH(huv));
// the frustum, on a sphere round the blade (it stands at most 0.9 m tall)
vec3 root = vec3(xz.x, ht.r, xz.y);
for (int k = 0; k < 4; k++) { if (dot(pr.planes[k].xyz, root) + pr.planes[k].w < -1.0) return; }
vec2 gxz = textureLod(u_ter_normal, huv, 0.0).rg;
vec2 gxz = groundN(huv);
vec3 gn = vec3(gxz.x, sqrt(max(1.0 - dot(gxz, gxz), 0.0)), gxz.y);
float h3 = bladeHash(ci, j, 2), h4 = bladeHash(ci, j, 3);
float wl = pr.lev.y;
if (pr.lake.z > 0.0) { vec2 q = (xz - pr.lake.xy) / pr.lake.zw; if (dot(q, q) < 1.0) wl = max(wl, pr.lev.x); }
float ok = (1.0 - smoothstep(0.30, 0.55, 1.0 - gn.y)) * smoothstep(0.0, 0.6, ht.r - wl - 0.15) * smoothstep(pr.dens.w - 80.0, pr.dens.w - 200.0, ht.r);
if (pr.lev.z > 0.5) {
vec3 oc = textureLod(u_ortho, huv, 1.5).rgb;
vec3 oc = groundO(huv);
ok *= 0.40 + 0.60 * smoothstep(0.0, 0.025, oc.g - oc.b);
}
if (h4 > ok) return;

View file

@ -253,6 +253,7 @@ function terrain_generate__t(render3d_st: mut Render3dState) -> void {
# The height field for shaders that place things on the ground (model.vert's u_ground)
function terrain_bind_height(render3d_st: mut Render3dState, p: int) -> void {
tp_bind(render3d_st, p)
r3d_bind_2d(render3d_st, p, "u_ts_height", 5, render3d_st.ter_height_tex)
r3d_bind_2d(render3d_st, p, "u_ter_normal", 6, render3d_st.ter_normal_tex)
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_ts_half"), float(render3d_st.TERRAIN_HALF))
@ -287,6 +288,7 @@ function terrain_bake_pass(render3d_st: mut Render3dState, p: int, target: int)
gpu_depth_test(render3d_st, false)
gpu_blend(render3d_st, false)
gpu_use_program(render3d_st, p)
tp_bind(render3d_st, p)
r3d_bind_2d(render3d_st, p, "u_height", 0, render3d_st.ter_height_tex)
r3d_bind_2d(render3d_st, p, "u_ter_normal", 7, render3d_st.ter_normal_tex)
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_half"), float(render3d_st.TERRAIN_HALF))
@ -400,6 +402,7 @@ function terrain_init(render3d_st: mut Render3dState) -> void {
render3d_st.gvk_tag = was
}
function terrain_init__t(render3d_st: mut Render3dState) -> void {
tp_dummies(render3d_st)
for i in 0 .. TERRAIN_INIT_STEPS { terrain_init_step(render3d_st, i) }
}
@ -473,6 +476,7 @@ function terrain_draw_shadow(light_vp: floats) -> void {
# selection can switch between them per patch without re-binding anything but the node.
function terrain_bind_prog(render3d_st: mut Render3dState, p: int) -> void {
gpu_use_program(render3d_st, p)
tp_bind(render3d_st, p)
r3d_bind_2d(render3d_st, p, "u_height", 0, render3d_st.ter_height_tex)
r3d_bind_2d(render3d_st, p, "u_ter_normal", 7, render3d_st.ter_normal_tex)
r3d_bind_2d(render3d_st, p, "u_grass_d", 1, render3d_st.ter_tex[0]); r3d_bind_2d(render3d_st, p, "u_grass_n", 2, render3d_st.ter_tex[1]); r3d_bind_2d(render3d_st, p, "u_grass_a", 3, render3d_st.ter_tex[2])
@ -582,6 +586,7 @@ function terrain_sun_pass(render3d_st: mut Render3dState, w: int, h: int, depth:
gpu_clear(render3d_st, GL_COLOR_BUFFER_BIT)
let p = render3d_st.ter_sun_prog
gpu_use_program(render3d_st, p)
tp_bind(render3d_st, p)
r3d_bind_2d(render3d_st, p, "u_height", 0, render3d_st.ter_height_tex)
r3d_bind_2d(render3d_st, p, "u_ter_normal", 7, render3d_st.ter_normal_tex)
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_half"), float(render3d_st.TERRAIN_HALF))
@ -822,6 +827,7 @@ function terrain_unload(render3d_st: mut Render3dState) -> void {
if render3d_st.ter_normal_tex != 0 { gpu_tex_free(render3d_st, render3d_st.ter_normal_tex); render3d_st.ter_normal_tex = 0 }
if render3d_st.ter_dem_tex != 0 { gpu_tex_free(render3d_st, render3d_st.ter_dem_tex); render3d_st.ter_dem_tex = 0 }
if render3d_st.ter_ortho_tex != 0 { gpu_tex_free(render3d_st, render3d_st.ter_ortho_tex); render3d_st.ter_ortho_tex = 0 }
tp_pool_free(render3d_st)
if render3d_st.ter_heights != null { free(render3d_st.ter_heights); render3d_st.ter_heights = null }
tt_close(render3d_st)
if render3d_st.ter_ortho_px != null { free(render3d_st.ter_ortho_px); render3d_st.ter_ortho_px = null }

View file

@ -0,0 +1,148 @@
# terrain_pages.ludic — the height field, its normals and the photograph on the GPU while the tiles are
# on (terrain_tiles.ludic): a coarse whole-map level (TT_COARSE a side) and, round the camera, a pool
# of fine tiles in three array textures addressed through a page table (u_tp_page: slot + 1 of a
# resident tile, 0 = read the coarse level). A shader reads a full-map uv as
# slot = texelFetch(u_tp_page, floor(uv * n)) - 1; slot uv = (1 + fract(uv * n) * T) / (T + 2)
# and each layer carries a one-texel border from its neighbours, so a bilinear tap never leaves it.
# Tiles off, nothing changes: the whole textures, u_tp_on = 0 and stand-in arrays bound.
const TP_REACH_MAX: float = 900.0 # metres of fine ground round the camera, fog or none
const TP_LAYERS_MAX: int = 2048 # what a desktop Vulkan device takes in an array (the floor is 256)
# the page pool's samplers and numbers for program p; every program that reads the ground calls it
function tp_bind(render3d_st: mut Render3dState, p: int) -> void {
tp_dummies(render3d_st)
var pg = render3d_st.tp_dummy_page
var h = render3d_st.tp_dummy_h
var nm = render3d_st.tp_dummy_n
var o = render3d_st.tp_dummy_o
var on = 0.0
if render3d_st.tp_on {
pg = render3d_st.tp_page_tex; h = render3d_st.tp_h_tex; nm = render3d_st.tp_n_tex; o = render3d_st.tp_o_tex; on = 1.0
}
r3d_bind_2d(render3d_st, p, "u_tp_page", 20, pg)
r3d_bind_tex(render3d_st, p, "u_tp_h", 21, GPU_TEX2D_ARRAY, h)
r3d_bind_tex(render3d_st, p, "u_tp_nrm", 22, GPU_TEX2D_ARRAY, nm)
r3d_bind_tex(render3d_st, p, "u_tp_ortho", 23, GPU_TEX2D_ARRAY, o)
u_f(render3d_st, gpu_uniform(render3d_st, p, "u_tp_on"), on)
u_f4(render3d_st, gpu_uniform(render3d_st, p, "u_tp_dims"), float(render3d_st.tt_n), float(TT_TEX), float(render3d_st.tt_oside), float(TERRAIN_RES))
}
# the stand-ins, once: an unbound sampler gets a white 2-D texture, which an array binding cannot take
@alloc_ok("made once: the stand-ins for the page pool's samplers")
function tp_dummies(render3d_st: mut Render3dState) -> void {
if render3d_st.tp_dummy_h != 0 { return }
let was = render3d_st.gvk_tag
render3d_st.gvk_tag = VKM_TERRAIN
let z = words(4)
for k in 0 .. 4 { z[k] = 0 }
render3d_st.tp_dummy_page = tex_target(render3d_st, 1, 1, GL_R32F, GL_RED, GL_FLOAT, GL_NEAREST)
gpu_tex_fill(render3d_st, render3d_st.tp_dummy_page, GL_R32F, 1, 1, GL_RED, GL_FLOAT, data_of(z))
free(z)
render3d_st.tp_dummy_h = tp_array(render3d_st, GL_R32F, 1, 1)
render3d_st.tp_dummy_n = tp_array(render3d_st, GL_RG16F, 1, 1)
render3d_st.tp_dummy_o = tp_array(render3d_st, GL_SRGB8_ALPHA8, 1, 1)
render3d_st.gvk_tag = was
}
# an array texture of `layers` side x side layers, no mips, read bilinear and clamped
function tp_array(render3d_st: mut Render3dState, ifmt: int, side: int, layers: int) -> int {
let t = gpu_tex_new(render3d_st)
gpu_tex_bind(render3d_st, GPU_TEX2D_ARRAY, t)
gpu_tex_image3d(render3d_st, ifmt, side, side, layers, GL_RED, GL_FLOAT, null)
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_MIN_FILTER, GL_LINEAR)
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_S, GL_CLAMP_TO_EDGE)
gpu_tex_param(render3d_st, GPU_TEX2D_ARRAY, GL_TEXTURE_WRAP_T, GL_CLAMP_TO_EDGE)
return t
}
# how far the fine tiles reach: the fog wall's reach when it is nearer than TP_REACH_MAX
function tp_reach_m(render3d_st: Render3dState) -> float {
if render3d_st.r3d_fog_wall > 0.0 { return Math.min(r3d_reach(render3d_st, 0.0), TP_REACH_MAX) }
return TP_REACH_MAX
}
function tp_tile_m(render3d_st: Render3dState) -> float { return float(TT_TEX) * terrain_texel(render3d_st) }
# the tiles within r and a one-tile ring, and a few spare
function tp_slots_for(render3d_st: Render3dState, r: float) -> int {
let a = r / tp_tile_m(render3d_st) + 1.5
let n = int(PI * a * a) + 9
return min(min(n, TP_LAYERS_MAX), render3d_st.tt_n * render3d_st.tt_n)
}
# the staging a frame's loads take: the page table, then TP_BUDGET tiles of each of the three
function tp_half_bytes(render3d_st: Render3dState) -> int {
let hs = TT_TEX + 2
let os = render3d_st.tt_oside + 2
return render3d_st.tt_n * render3d_st.tt_n * 4 + TP_BUDGET * (2 * hs * hs + os * os) * 4
}
# the pool, the page table (all 0) and the staging, (re)made for the reach the fog wall allows
@alloc_ok("a map made, or the fog wall moved: the page pool, once")
function tp_pool(render3d_st: mut Render3dState) -> void {
tp_pool_free(render3d_st)
let n = render3d_st.tt_n
let r = tp_reach_m(render3d_st)
let slots = tp_slots_for(render3d_st, r)
let hs = TT_TEX + 2
let os = render3d_st.tt_oside + 2
render3d_st.tp_h_tex = tp_array(render3d_st, GL_R32F, hs, slots)
render3d_st.tp_n_tex = tp_array(render3d_st, GL_RG16F, hs, slots)
render3d_st.tp_o_tex = tp_array(render3d_st, GL_SRGB8_ALPHA8, os, slots)
if render3d_st.tp_page == null or len(render3d_st.tp_page) != n * n {
if render3d_st.tp_page != null { free(render3d_st.tp_page) }
render3d_st.tp_page = floats(n * n)
}
for t in 0 .. n * n { render3d_st.tp_page[t] = 0.0 }
render3d_st.tp_page_tex = tex_target(render3d_st, n, n, GL_R32F, GL_RED, GL_FLOAT, GL_NEAREST)
gpu_tex_fill(render3d_st, render3d_st.tp_page_tex, GL_R32F, n, n, GL_RED, GL_FLOAT, data_of(render3d_st.tp_page))
tp_slot_lists(render3d_st, slots)
if render3d_st.tp_st_bytes < 2 * tp_half_bytes(render3d_st) {
gpu_staging_drop(render3d_st)
render3d_st.tp_st_ptr = gpu_staging_keep(render3d_st, 2 * tp_half_bytes(render3d_st))
}
render3d_st.tp_slots = slots; render3d_st.tp_reach = r
render3d_st.tp_cam_t = -1; render3d_st.tp_short = false; render3d_st.tp_dirty = false
render3d_st.tp_bytes = slots * (2 * hs * hs + os * os) * 4 + n * n * 4
render3d_st.tp_on = render3d_st.tp_st_ptr != null and gpu_tex_ok(render3d_st, render3d_st.tp_h_tex) and gpu_tex_ok(render3d_st, render3d_st.tp_o_tex)
tp_say_pool(slots, r, render3d_st.tp_bytes, render3d_st.tp_on)
}
# every slot free, and this frame's candidate lists (made once)
@alloc_ok("a map made, or the fog wall moved: the page pool's lists, once")
function tp_slot_lists(render3d_st: mut Render3dState, slots: int) -> void {
if render3d_st.tp_tile_of != null { free(render3d_st.tp_tile_of); free(render3d_st.tp_free) }
render3d_st.tp_tile_of = words(slots)
render3d_st.tp_free = words(slots)
for s in 0 .. slots { render3d_st.tp_tile_of[s] = -1; render3d_st.tp_free[s] = slots - 1 - s }
render3d_st.tp_nfree = slots
if render3d_st.tp_cand == null {
render3d_st.tp_cand = words(TP_BUDGET); render3d_st.tp_cand_s = words(TP_BUDGET); render3d_st.tp_cand_d = floats(TP_BUDGET)
render3d_st.tp_zero = words(1); render3d_st.tp_zero[0] = 0
}
}
# the pool and its page table let go (a world swap, or a remake); the coarse level is terrain_unload's
function tp_pool_free(render3d_st: mut Render3dState) -> void {
render3d_st.tp_on = false
if render3d_st.tp_page_tex != 0 { gpu_tex_free(render3d_st, render3d_st.tp_page_tex); render3d_st.tp_page_tex = 0 }
if render3d_st.tp_h_tex != 0 { gpu_tex_free(render3d_st, render3d_st.tp_h_tex); render3d_st.tp_h_tex = 0 }
if render3d_st.tp_n_tex != 0 { gpu_tex_free(render3d_st, render3d_st.tp_n_tex); render3d_st.tp_n_tex = 0 }
if render3d_st.tp_o_tex != 0 { gpu_tex_free(render3d_st, render3d_st.tp_o_tex); render3d_st.tp_o_tex = 0 }
render3d_st.tp_slots = 0; render3d_st.tp_bytes = 0
}
# r3d_fog_wall moved: a pool sized for another reach is made again (the page table starts empty)
function terrain_pages_fog(render3d_st: mut Render3dState) -> void {
if not render3d_st.tp_on { return }
let r = tp_reach_m(render3d_st)
render3d_st.tp_reach = r
render3d_st.tp_cam_t = -1
if tp_slots_for(render3d_st, r) == render3d_st.tp_slots { return }
let was = render3d_st.gvk_tag
render3d_st.gvk_tag = VKM_TERRAIN
tp_pool(render3d_st)
render3d_st.gvk_tag = was
}
@alloc_ok("a message, built only when it is said: a map made or the fog wall moved")
function tp_say_pool(slots: int, r: float, bytes: int, on: bool) -> void { print(`r3d: terrain: {slots} page slots for {int(r)} m round the camera ({bytes / 1024} KB on the GPU), paging {on}`) }

View file

@ -0,0 +1,30 @@
# terrain_pages_coarse.ludic — the whole-map level the page pool falls back to (terrain_pages.ludic),
# made on the GPU from the whole textures at the cut, before they go.
# at the cut (terrain_tiles_cut): the coarse level made from the whole textures, which then go - the
# patch bounds and the sun's bake have already read them - and the pool made
function tp_cut(render3d_st: mut Render3dState) -> void {
let was = render3d_st.gvk_tag
render3d_st.gvk_tag = VKM_TERRAIN
tp_cut__t(render3d_st)
render3d_st.gvk_tag = was
}
function tp_cut__t(render3d_st: mut Render3dState) -> void {
if render3d_st.tt_file == null or render3d_st.tt_coarse == null or render3d_st.ter_ortho_tex == 0 { return }
let c = TT_COARSE
let h = tex_target(render3d_st, c, c, GL_R32F, GL_RED, GL_FLOAT, GL_LINEAR)
gpu_tex_fill(render3d_st, h, GL_R32F, c, c, GL_RED, GL_FLOAT, data_of(render3d_st.tt_coarse))
let nm = tex_target(render3d_st, c, c, GL_RG16F, GL_RG, GL_FLOAT, GL_LINEAR)
gpu_tex_blit(render3d_st, render3d_st.ter_normal_tex, TERRAIN_RES, TERRAIN_RES, nm, c, c)
let o = tex_target(render3d_st, c, c, GL_SRGB8_ALPHA8, GL_RGBA, GL_UNSIGNED_BYTE, GL_LINEAR)
gpu_tex_blit(render3d_st, render3d_st.ter_ortho_tex, render3d_st.ter_ortho_w, render3d_st.ter_ortho_w, o, c, c)
gpu_tex_bind(render3d_st, GPU_TEX2D, o)
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
gpu_tex_mips(render3d_st, GPU_TEX2D)
gpu_tex_paramf(render3d_st, GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, render3d_st.tex_anisotropy)
gpu_tex_free(render3d_st, render3d_st.ter_height_tex)
gpu_tex_free(render3d_st, render3d_st.ter_normal_tex)
gpu_tex_free(render3d_st, render3d_st.ter_ortho_tex)
render3d_st.ter_height_tex = h; render3d_st.ter_normal_tex = nm; render3d_st.ter_ortho_tex = o
tp_pool(render3d_st)
}

View file

@ -0,0 +1,74 @@
# terrain_pages_fill.ludic — a tile's three layers written into the staging, border and all, and the
# frame's copies recorded (terrain_pages_frame.ludic). The staging half is laid out as the page table,
# then TP_BUDGET heights, TP_BUDGET normals and TP_BUDGET photographs.
# the k-th load of the frame: tile t's heights and normals, (TT_TEX + 2) a side, each texel the one
# the whole copy held at that place, clamped at the map's edge
function tp_fill(render3d_st: mut Render3dState, t: int, base: int, k: int) -> void {
let n = render3d_st.tt_n
let hs = TT_TEX + 2
let hb = hs * hs * 4
let ho = base + n * n * 4 + k * hb
let no = base + n * n * 4 + TP_BUDGET * hb + k * hb
let i0 = (t % n) * TT_TEX - 1
let j0 = (t / n) * TT_TEX - 1
let p = render3d_st.tp_st_ptr
for b in 0 .. hs {
let tz = min(max(j0 + b, 0), TERRAIN_RES - 1)
for a in 0 .. hs {
let tx = min(max(i0 + a, 0), TERRAIN_RES - 1)
let o = (b * hs + a) * 4
Vk.put_i32(p, ho + o, float_bits(ter_h(render3d_st, tx, tz)))
Vk.put_i32(p, no + o, ter_n(render3d_st, tx, tz))
}
}
tp_fill_ortho(render3d_st, t, base + n * n * 4 + 2 * TP_BUDGET * hb, k)
}
# ... and its photograph, (tt_oside + 2) a side, as sRGB RGBA8 bytes with alpha 255
function tp_fill_ortho(render3d_st: mut Render3dState, t: int, at: int, k: int) -> void {
let n = render3d_st.tt_n
let side = render3d_st.tt_oside
let os = side + 2
let top = n * side - 1
let o0 = at + k * os * os * 4
let i0 = (t % n) * side - 1
let j0 = (t / n) * side - 1
let p: pointer = render3d_st.tp_st_ptr
for b in 0 .. os {
let pz = min(max(j0 + b, 0), top)
for a in 0 .. os {
let v = ter_o(render3d_st, min(max(i0 + a, 0), top), pz)
let o = o0 + (b * os + a) * 4
p[o] = (v >> 16) & 255
p[o + 1] = (v >> 8) & 255
p[o + 2] = v & 255
p[o + 3] = 255
}
}
}
# the frame's k loads into their slots, then the page table if it changed: all in cb, no submit
function tp_upload(render3d_st: mut Render3dState, cb: pointer, base: int, k: int) -> void {
let n = render3d_st.tt_n
let hs = TT_TEX + 2
let hb = hs * hs * 4
let os = render3d_st.tt_oside + 2
let p0 = base + n * n * 4
let slots = render3d_st.tp_cand_s
gpu_layers_copy(render3d_st, cb, render3d_st.tp_h_tex, hs, hs, p0, hb, slots, k)
gpu_layers_copy(render3d_st, cb, render3d_st.tp_n_tex, hs, hs, p0 + TP_BUDGET * hb, hb, slots, k)
gpu_layers_copy(render3d_st, cb, render3d_st.tp_o_tex, os, os, p0 + 2 * TP_BUDGET * hb, os * os * 4, slots, k)
if not render3d_st.tp_dirty { return }
mem_copy(mem_off(render3d_st.tp_st_ptr, base), data_of(render3d_st.tp_page), n * n * 4)
gpu_layers_copy(render3d_st, cb, render3d_st.tp_page_tex, n, n, base, 0, render3d_st.tp_zero, 1)
render3d_st.tp_dirty = false
}
# R3D_VKMEM: one line on the pool, next to the rest of the report
function tp_report(render3d_st: Render3dState) -> void {
if not render3d_st.tp_on { return }
tp_say_report(render3d_st.tp_slots - render3d_st.tp_nfree, render3d_st.tp_slots, render3d_st.tp_frame_loads, render3d_st.tp_loads, render3d_st.tp_bytes, render3d_st.tp_reach)
}
@alloc_ok("a report printed once, under R3D_VKMEM")
function tp_say_report(res: int, slots: int, now: int, all: int, bytes: int, r: float) -> void { print(`vkmem: terrain pages: {res} of {slots} slots resident, {now} loaded this frame, {all} over the run, {bytes / 1024} KB on the GPU, {int(r)} m round the camera`) }

View file

@ -0,0 +1,124 @@
# terrain_pages_frame.ludic — which fine tiles the GPU holds, once a frame (terrain_pages.ludic). Tiles
# within the reach and a tile of the camera are wanted, nearest first, at most TP_BUDGET loaded a frame;
# a resident tile goes only past the reach and two tiles, so walking a tile edge back and forth loads
# nothing. A load is written into the kept staging (the half the frame in flight is not reading) and
# copied into its slot inside the frame's own command buffer. Nothing here allocates.
const TP_BUDGET: int = 8
function tp_frame(render3d_st: mut Render3dState) -> void {
render3d_st.tp_frame_loads = 0
if not render3d_st.tp_on or render3d_st.cam_pos == null { return }
let tile_m = tp_tile_m(render3d_st)
let fx = (render3d_st.cam_pos[0] - render3d_st.ter_ox + float(render3d_st.TERRAIN_HALF)) / tile_m
let fz = (render3d_st.cam_pos[2] - render3d_st.ter_oz + float(render3d_st.TERRAIN_HALF)) / tile_m
let n = render3d_st.tt_n
let ct = min(max(int(Math.floor(fz)), 0), n - 1) * n + min(max(int(Math.floor(fx)), 0), n - 1)
# the same tile as last time with nothing left over: every wanted tile is in already
if ct == render3d_st.tp_cam_t and not render3d_st.tp_short { return }
render3d_st.tp_cam_t = ct
let rt = render3d_st.tp_reach / tile_m
tp_evict(render3d_st, fx, fz, rt + 2.0)
let m = tp_pick(render3d_st, fx, fz, rt + 1.0)
if m > 0 or render3d_st.tp_dirty { tp_load(render3d_st, m, fx, fz, rt + 1.0) }
}
# tile t's centre from (fx, fz), in tiles
function tp_dist(t: int, n: int, fx: float, fz: float) -> float {
let dx = float(t % n) + 0.5 - fx
let dz = float(t / n) + 0.5 - fz
return Math.sqrt(dx * dx + dz * dz)
}
# every resident tile farther than `out` let go
function tp_evict(render3d_st: mut Render3dState, fx: float, fz: float, out: float) -> void {
for s in 0 .. render3d_st.tp_slots {
let t = render3d_st.tp_tile_of[s]
if t >= 0 and tp_dist(t, render3d_st.tt_n, fx, fz) > out { tp_drop(render3d_st, s) }
}
}
function tp_drop(render3d_st: mut Render3dState, s: int) -> void {
render3d_st.tp_page[render3d_st.tp_tile_of[s]] = 0.0
render3d_st.tp_tile_of[s] = -1
render3d_st.tp_free[render3d_st.tp_nfree] = s
render3d_st.tp_nfree += 1
render3d_st.tp_dirty = true
}
# the nearest TP_BUDGET wanted tiles not in, into tp_cand; whether more were left is tp_short
function tp_pick(render3d_st: mut Render3dState, fx: float, fz: float, want: float) -> int {
let n = render3d_st.tt_n
var m = 0
var missing = 0
for j in max(int(Math.floor(fz - want)), 0) .. min(int(fz + want) + 1, n) {
for i in max(int(Math.floor(fx - want)), 0) .. min(int(fx + want) + 1, n) {
let t = j * n + i
if render3d_st.tp_page[t] != 0.0 { continue }
let d = tp_dist(t, n, fx, fz)
if d > want { continue }
missing += 1
m = tp_cand_put(render3d_st, m, t, d)
}
}
render3d_st.tp_short = missing > m
return m
}
# tile t at distance d into the sorted candidates, the farthest falling off the end
function tp_cand_put(render3d_st: mut Render3dState, m: int, t: int, d: float) -> int {
var k = m
if k == TP_BUDGET {
if d >= render3d_st.tp_cand_d[k - 1] { return m }
k = TP_BUDGET - 1
}
while k > 0 and render3d_st.tp_cand_d[k - 1] > d {
render3d_st.tp_cand[k] = render3d_st.tp_cand[k - 1]
render3d_st.tp_cand_d[k] = render3d_st.tp_cand_d[k - 1]
k -= 1
}
render3d_st.tp_cand[k] = t
render3d_st.tp_cand_d[k] = d
return min(m + 1, TP_BUDGET)
}
# a free slot, or the farthest resident tile past `keep` given up for one; -1 when every slot is wanted
function tp_take(render3d_st: mut Render3dState, fx: float, fz: float, keep: float) -> int {
if render3d_st.tp_nfree == 0 {
var far = -1
var fd = keep
for s in 0 .. render3d_st.tp_slots {
let t = render3d_st.tp_tile_of[s]
if t < 0 { continue }
let d = tp_dist(t, render3d_st.tt_n, fx, fz)
if d > fd { fd = d; far = s }
}
if far < 0 { return -1 }
tp_drop(render3d_st, far)
}
render3d_st.tp_nfree -= 1
return render3d_st.tp_free[render3d_st.tp_nfree]
}
# the frame's loads written and copied, and the page table after them when it changed
function tp_load(render3d_st: mut Render3dState, m: int, fx: float, fz: float, keep: float) -> void {
# the frame's command buffer first: opening it waits for the frame that read this half last
let cb = gpu_upload_cb(render3d_st)
let base = (render3d_st.gvk_frame_no & 1) * tp_half_bytes(render3d_st)
# the tiles' reads from the file are the pool's own, not a frame asking too much (tt_frame)
let reads = render3d_st.tt_frame_reads
var k = 0
for c in 0 .. m {
let s = tp_take(render3d_st, fx, fz, keep)
if s < 0 { render3d_st.tp_short = true; break }
let t = render3d_st.tp_cand[c]
tp_fill(render3d_st, t, base, k)
render3d_st.tp_cand_s[k] = s
render3d_st.tp_tile_of[s] = t
render3d_st.tp_page[t] = float(s + 1)
render3d_st.tp_dirty = true
k += 1
}
render3d_st.tt_frame_reads = reads
render3d_st.tp_frame_loads = k
render3d_st.tp_loads += k
tp_upload(render3d_st, cb, base, k)
}

View file

@ -35,6 +35,7 @@ function terrain_tiles_cut(render3d_st: mut Render3dState) -> void {
if r3d_env_has(render3d_st, "R3D_TT_CHECK") { tt_check(render3d_st) }
free(render3d_st.ter_heights); render3d_st.ter_heights = null
free(render3d_st.ter_ortho_px); render3d_st.ter_ortho_px = null
tp_cut(render3d_st)
tt_say_cut(path, TT_HEAD + n * n * (TT_TEX * TT_TEX * 2 + side * side) * 4, (gl_now_us() - t0) / 1000)
}

View file

@ -50,6 +50,7 @@ function vkmem_report(render3d_st: Render3dState) -> void {
vkmem_layers(render3d_st)
vkmem_big_bufs(render3d_st)
vkmem_say_ring(render3d_st.gvk_ring_peak, gvk_ring_bytes(render3d_st))
tp_report(render3d_st)
}
# the largest images, with their owner, size and format