diff --git a/packages/ludic.render3d/gpu.ludic b/packages/ludic.render3d/gpu.ludic index 6015b0fe..bf707f47 100644 --- a/packages/ludic.render3d/gpu.ludic +++ b/packages/ludic.render3d/gpu.ludic @@ -819,7 +819,13 @@ function gpu_rb_new() -> int { } # storage for a renderbuffer: samples > 0 makes it multisampled function gpu_rb_storage(rb: int, ifmt: int, w: int, h: int, samples: int) -> void { - if gpu_kind == GPU_VK { gvk_flush(); gpu_rb_samples = samples; gvk_tex_storage(rb, false, ifmt, w, h, 1, false); return } + if gpu_kind == GPU_VK { + gvk_flush(); gpu_rb_samples = samples + gvk_storage_samples = samples + gvk_tex_storage(rb, false, ifmt, w, h, 1, false) + gvk_storage_samples = 1 + return + } gl_bind_renderbuffer(GL_RENDERBUFFER, rb) if samples > 0 { gl_renderbuffer_storage_multisample(GL_RENDERBUFFER, samples, ifmt, w, h) } else { gl_renderbuffer_storage(GL_RENDERBUFFER, ifmt, w, h) } @@ -928,6 +934,9 @@ function gpu_blit(w: int, h: int, mask: int) -> void { # the framebuffer the finished frame is presented from (an offscreen one, headless) function gpu_screen_fb() -> int { if gpu_kind == GPU_VK { return 0 }; return gl_screen } function gpu_multisample(on: bool) -> void { if gpu_kind == GPU_VK { return }; gpu_gl_cap(GL_MULTISAMPLE, gpu_b(on)); gpu_glcheck_after("multisample") } +# the most samples the scene may be drawn with: the device's colour-and-depth limit on Vulkan (0 before +# it is open), 4 on OpenGL, which has always asked for up to that +function gpu_msaa_max() -> int { if gpu_kind == GPU_VK { return gvk_msaa_max }; return 4 } function gpu_wireframe(on: bool) -> void { if gpu_kind == GPU_VK { gvk_wireframe = gpu_b(on); return } if on { gl_polygon_mode(GL_FRONT_AND_BACK, GL_LINE) } else { gl_polygon_mode(GL_FRONT_AND_BACK, GL_FILL) } diff --git a/packages/ludic.render3d/gpu_vk_draw.ludic b/packages/ludic.render3d/gpu_vk_draw.ludic index 6660865d..7d53e9d3 100644 --- a/packages/ludic.render3d/gpu_vk_draw.ludic +++ b/packages/ludic.render3d/gpu_vk_draw.ludic @@ -654,6 +654,7 @@ var gvk_ring_off: int = 0 var gvk_ring_align: int = 256 var gvk_dpool: long = 0 var gvk_white: int = 0 # a 1x1 white texture for a sampler nothing was bound to +var gvk_msaa_max: int = 0 # samples the scene may use (gvk_frame_init reads the device's limits) function gvk_frame_init() -> bool { let props = bytes(VkPhysicalDeviceProperties_sizeof) @@ -661,6 +662,12 @@ function gvk_frame_init() -> bool { let al = Vk.get_i64(props, VkPhysicalDeviceProperties_limits + VkPhysicalDeviceLimits_minUniformBufferOffsetAlignment) gvk_ring_align = Text.to_int(string(al)) if gvk_ring_align < 16 { gvk_ring_align = 16 } + # MSAA: the most samples (up to the 4 the scene asks for) both a colour and a depth target can take + let lim = VkPhysicalDeviceProperties_limits + let counts = Vk.get_i32(props, lim + VkPhysicalDeviceLimits_framebufferColorSampleCounts) & Vk.get_i32(props, lim + VkPhysicalDeviceLimits_framebufferDepthSampleCounts) + gvk_msaa_max = 1 + if (counts & VK_SAMPLE_COUNT_2_BIT) != 0 { gvk_msaa_max = 2 } + if (counts & VK_SAMPLE_COUNT_4_BIT) != 0 { gvk_msaa_max = 4 } gvk_ring_buf = gvk_buf_new() if not gvk_buf_reserve(gvk_ring_buf, GVK_RING_BYTES) { return false } let sizes = bytes(VkDescriptorPoolSize_sizeof * 3) @@ -921,6 +928,7 @@ var gvk_pass_h: int = 0 var gvk_pass_ncolor: int = 0 var gvk_pass_cfmt: int = 0 var gvk_pass_dfmt: int = 0 +var gvk_pass_samples: int = 1 # the attachments' sample count, which every pipeline in the pass must match var gvk_pass_col: words = null # the pass's colour attachments: texture, layer + 1 var gvk_pass_dep: words = null # its depth attachment: texture, layer + 1 var gvk_clear_bits: int = 0 # GL_COLOR_BUFFER_BIT / GL_DEPTH_BUFFER_BIT waiting for the pass @@ -1043,6 +1051,7 @@ function gvk_pass_begin(rec: words, rec_o: int) -> void { gvk_pass_h = 0 gvk_pass_cfmt = VK_FORMAT_UNDEFINED gvk_pass_dfmt = VK_FORMAT_UNDEFINED + gvk_pass_samples = 1 for c in 0 .. gvk_pass_ncolor { let tex = gvk_pass_col[c * 2] let layer1 = gvk_pass_col[c * 2 + 1] @@ -1056,6 +1065,7 @@ function gvk_pass_begin(rec: words, rec_o: int) -> void { for k in 0 .. 4 { Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_clearValue + k * 4, gvk_clear_rgba[k]) } } else { Vk.put_i32(catt, c * aw + VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) } gvk_pass_cfmt = gvk_tex_vkfmt[tex] + gvk_pass_samples = gvk_tex_samples_of(tex) if gvk_pass_w == 0 { gvk_pass_w = gvk_tex_w(tex); gvk_pass_h = gvk_tex_h(tex) } } let datt = bytes(aw) @@ -1072,6 +1082,7 @@ function gvk_pass_begin(rec: words, rec_o: int) -> void { Vk.put_i32(datt, VkRenderingAttachmentInfo_clearValue, 0x3F800000) } else { Vk.put_i32(datt, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) } gvk_pass_dfmt = VK_FORMAT_D32_SFLOAT + gvk_pass_samples = gvk_tex_samples_of(dtex) if gvk_pass_w == 0 { gvk_pass_w = gvk_tex_w(dtex); gvk_pass_h = gvk_tex_h(dtex) } } gvk_clear_bits = 0 @@ -1168,7 +1179,7 @@ function gvk_draw(p: int, m: Mesh, st: GvkState, first: int, count: int, instanc if prof { t0 = gl_now_us() } gvk_pass_begin(rec, rec_o) let cb = gvk_cb - let samples = 1 + let samples = gvk_pass_samples let pipe = gvk_pipeline_fast(p, m, st, gvk_pass_ncolor, gvk_pass_cfmt, gvk_pass_dfmt, samples) if pipe == 0 { # Say so, once per program: a driver that refuses a pipeline other drivers accept (MoltenVK @@ -1431,8 +1442,45 @@ function gvk_fb_att(fb: int, depth: bool) -> int { if depth { return gpu_fb[o + 2] } return gpu_fb[o] } +function gvk_tex_samples_of(tex: int) -> int { + if tex <= 0 or gvk_tex_samples == null or tex >= len(gvk_tex_samples) or gvk_tex_samples[tex] < 1 { return 1 } + return gvk_tex_samples[tex] +} +# A multisampled image into a single-sampled one: a pass that draws nothing and resolves on its end - +# colour averaged, depth from sample zero. vkCmdResolveImage cannot resolve depth; a pass can. +function gvk_resolve(cb: pointer, src: int, dst: int, depth: bool, w: int, h: int) -> void { + var layout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL + var mode = VK_RESOLVE_MODE_AVERAGE_BIT + if depth { layout = VK_IMAGE_LAYOUT_DEPTH_ATTACHMENT_OPTIMAL; mode = VK_RESOLVE_MODE_SAMPLE_ZERO_BIT } + gvk_att_barrier(cb, src, 0, depth, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, layout) + gvk_att_barrier(cb, dst, 0, depth, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, layout) + let aw = VkRenderingAttachmentInfo_sizeof + let att = bytes(aw) + Vk.zero(att, aw) + Vk.put_i32(att, VkRenderingAttachmentInfo_sType, VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO) + Vk.put_i64(att, VkRenderingAttachmentInfo_imageView, gvk_view_of(src, 0)) + Vk.put_i32(att, VkRenderingAttachmentInfo_imageLayout, layout) + Vk.put_i32(att, VkRenderingAttachmentInfo_resolveMode, mode) + Vk.put_i64(att, VkRenderingAttachmentInfo_resolveImageView, gvk_view_of(dst, 0)) + Vk.put_i32(att, VkRenderingAttachmentInfo_resolveImageLayout, layout) + Vk.put_i32(att, VkRenderingAttachmentInfo_loadOp, VK_ATTACHMENT_LOAD_OP_LOAD) + Vk.put_i32(att, VkRenderingAttachmentInfo_storeOp, VK_ATTACHMENT_STORE_OP_STORE) + let ri = bytes(VkRenderingInfo_sizeof) + Vk.zero(ri, VkRenderingInfo_sizeof) + Vk.put_i32(ri, VkRenderingInfo_sType, VK_STRUCTURE_TYPE_RENDERING_INFO) + Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_width, w) + Vk.put_i32(ri, VkRenderingInfo_renderArea + VkRect2D_extent + VkExtent2D_height, h) + Vk.put_i32(ri, VkRenderingInfo_layerCount, 1) + if depth { Vk.put_ptr(ri, VkRenderingInfo_pDepthAttachment, att) } + else { Vk.put_i32(ri, VkRenderingInfo_colorAttachmentCount, 1); Vk.put_ptr(ri, VkRenderingInfo_pColorAttachments, att) } + Vk.cmd_begin_rendering(cb, ri) + Vk.cmd_end_rendering(cb) + gvk_att_barrier(cb, src, 0, depth, layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL) + gvk_att_barrier(cb, dst, 0, depth, layout, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL) +} function gvk_copy(cb: pointer, src: int, dst: int, depth: bool, w: int, h: int) -> void { if src <= 0 or dst <= 0 or gvk_tex_image[src] == 0 or gvk_tex_image[dst] == 0 { return } + if gvk_tex_samples_of(src) > 1 and gvk_tex_samples_of(dst) == 1 { gvk_resolve(cb, src, dst, depth, w, h); return } gvk_barrier(cb, gvk_tex_image[src], depth, 0, 1, gvk_tex_layers[src], VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL) gvk_barrier(cb, gvk_tex_image[dst], depth, 0, 1, gvk_tex_layers[dst], VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL) let ic = bytes(VkImageCopy_sizeof) @@ -1737,7 +1785,7 @@ function gvk_pipeline_fast(p: int, m: Mesh, st: GvkState, n_color: int, color_fm var state = st.depth_test | (st.depth_write << 1) | (st.blend << 2) | (st.cull << 3) | (st.color_write << 4) | (st.a2c << 5) | (st.bias << 6) | (st.wireframe << 7) state = state | ((st.depth_func & 15) << 8) | ((st.cull_face & 15) << 12) | (gvk_blend_index(st.blend_src) << 16) | (gvk_blend_index(st.blend_dst) << 20) let bias = st.bias_factor * 31 + st.bias_units - let pass = ((n_color * 1000 + color_fmt) * 1000 + depth_fmt) * 8 + samples + let pass = ((n_color * 1000 + color_fmt) * 1000 + depth_fmt) * 64 + samples if gvk_pc_prog == null { gvk_pc_prog = new []int; gvk_pc_layout = new []int; gvk_pc_state = new []int; gvk_pc_bias = new []int gvk_pc_pass = new []int; gvk_pc_pipe = new []long diff --git a/packages/ludic.render3d/gpu_vk_res.ludic b/packages/ludic.render3d/gpu_vk_res.ludic index c263a059..c50e169a 100644 --- a/packages/ludic.render3d/gpu_vk_res.ludic +++ b/packages/ludic.render3d/gpu_vk_res.ludic @@ -57,6 +57,8 @@ var gvk_tex_array: []int = null # 1 for an array image var gvk_tex_gen: []int = null # bumped whenever the image behind a handle is replaced var gvk_tex_smp_sig: []int = null # the sampling parameters the cached sampler was made for var gvk_tex_smp: []long = null +var gvk_tex_samples: []int = null # 1, or the sample count of a multisampled renderbuffer +var gvk_storage_samples: int = 1 # what the next gvk_tex_storage makes: gpu_rb_storage sets it around its call var gvk_unpack_swap: bool = false # GL_UNPACK_SWAP_BYTES: 16-bit PNG samples arrive big-endian function gvk_tex_new() -> int { @@ -67,6 +69,7 @@ function gvk_tex_new() -> int { gvk_tex_dims_w = new []int; gvk_tex_dims_h = new []int gvk_tex_glfmt = new []int; gvk_tex_array = new []int; gvk_tex_gen = new []int gvk_tex_smp_sig = new []int; gvk_tex_smp = new []long + gvk_tex_samples = new []int; push(gvk_tex_samples, 0) push(gvk_tex_dims_w, 0); push(gvk_tex_dims_h, 0) push(gvk_tex_glfmt, 0); push(gvk_tex_array, 0); push(gvk_tex_gen, 0) push(gvk_tex_smp_sig, 0); push(gvk_tex_smp, zero) @@ -79,6 +82,7 @@ function gvk_tex_new() -> int { push(gvk_tex_dims_w, 0); push(gvk_tex_dims_h, 0) push(gvk_tex_glfmt, 0); push(gvk_tex_array, 0); push(gvk_tex_gen, 0) push(gvk_tex_smp_sig, 0); push(gvk_tex_smp, zero) + push(gvk_tex_samples, 0) return len(gvk_tex_image) - 1 } @@ -162,6 +166,9 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer let depth = gvk_is_depth(ifmt) var levels = 1 if with_mips and not depth { levels = gvk_mip_levels(w, h) } + var samples = gvk_storage_samples + if samples < 1 { samples = 1 } + if samples > 1 { levels = 1 } var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT if depth { usage = usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } else { usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT } let ici = bytes(VkImageCreateInfo_sizeof) @@ -174,7 +181,7 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer Vk.put_i32(ici, VkImageCreateInfo_extent + VkExtent3D_depth, 1) Vk.put_i32(ici, VkImageCreateInfo_mipLevels, levels) Vk.put_i32(ici, VkImageCreateInfo_arrayLayers, layers) - Vk.put_i32(ici, VkImageCreateInfo_samples, VK_SAMPLE_COUNT_1_BIT) + Vk.put_i32(ici, VkImageCreateInfo_samples, samples) Vk.put_i32(ici, VkImageCreateInfo_tiling, VK_IMAGE_TILING_OPTIMAL) Vk.put_i32(ici, VkImageCreateInfo_usage, usage) Vk.put_i32(ici, VkImageCreateInfo_sharingMode, VK_SHARING_MODE_EXCLUSIVE) @@ -212,6 +219,7 @@ function gvk_tex_storage(tex: int, array: bool, ifmt: int, w: int, h: int, layer gvk_tex_levels[tex] = levels; gvk_tex_layers[tex] = layers; gvk_tex_vkfmt[tex] = vkfmt gvk_tex_dims_w[tex] = w; gvk_tex_dims_h[tex] = h gvk_tex_glfmt[tex] = ifmt; gvk_tex_gen[tex] = gvk_tex_gen[tex] + 1 + gvk_tex_samples[tex] = samples if array { gvk_tex_array[tex] = 1 } else { gvk_tex_array[tex] = 0 } # a target starts cleared-to-nothing but readable: every level in the layout samplers expect let cb = gvk_once_begin() diff --git a/packages/ludic.render3d/post.ludic b/packages/ludic.render3d/post.ludic index f5217f93..3093a81f 100644 --- a/packages/ludic.render3d/post.ludic +++ b/packages/ludic.render3d/post.ludic @@ -56,13 +56,15 @@ function post_free() -> void { for i in 0 .. len(post_bloom) { target_free(post_bloom[i]) } post_hdr = null } -# Multisampled scene: 1 (temporal AA alone), 2 or 4. OpenGL remakes the scene targets at once; -# the Vulkan backend has no sample counts yet and keeps drawing single-sampled (post_msaa_live says so). -function post_msaa_live() -> bool { return gpu_is_gl() } +# Multisampled scene: 1 (temporal AA alone), 2 or 4, remade at once. Vulkan draws it into +# multisampled renderbuffers and resolves them in a pass; a device that cannot take the count asked +# for gets the most it can (gpu_msaa_max). +function post_msaa_live() -> bool { return gpu_is_gl() or gpu_msaa_max() > 1 } function post_set_msaa(n: int) -> void { var want = n if want < 1 { want = 1 } if not post_msaa_live() { want = 1 } + if want > gpu_msaa_max() and gpu_msaa_max() >= 1 { want = gpu_msaa_max() } if want == post_ms_samples { return } post_ms_samples = want if post_hdr != null {