mirror of
https://github.com/blender/blender
synced 2026-09-29 04:37:17 +03:00
GPU: Mipmap filtering consistency
In OpenGL mipmap creation is a driver responsibility. In Metal and Vulkan this became an application responsibility. This results in a mismatch between OpenGL and the other backends. This PR will implement mipmap creation inside the GPU module. It is for now only enabled for Vulkan. OpenGL and Metal will be added in separately. This PR introduces a compute shader that will generate the mipmap chain. The compute shader is be based on https://github.com/nvpro-samples/vk_compute_mipmaps/tree/main/nvpro_pyramid general sharer. This shader can calculate 2 mipmap levels at a time. This PR adds the first couple of texture formats and falls back to the backend specific implementation for the rest. The supported texture formats are: - 2D texture/UNORM_8_8_8_8 - 2D arrayed texture/UNORM_8_8_8_8 - 2D texture/SFLOAT_16 - 2D arrayed texture/SFLOAT_16 - 2D texture/SFLOAT_16_16_16_16 - 2D arrayed texture/SFLOAT_16_16_16_16 Other texture formats will be added later on and OpenGL/Metal enablement will be added later. Pull Request: https://projects.blender.org/blender/blender/pulls/155463
This commit is contained in:
parent
1c7d9d159d
commit
2f657feb36
11 changed files with 867 additions and 7 deletions
|
|
@ -100,6 +100,7 @@ set(SRC
|
|||
intern/gpu_storage_buffer.cc
|
||||
intern/gpu_texture.cc
|
||||
intern/gpu_texture_pool.cc
|
||||
intern/gpu_texture_mipmap.cc
|
||||
intern/gpu_uniform_buffer.cc
|
||||
intern/gpu_vertex_buffer.cc
|
||||
intern/gpu_vertex_format.cc
|
||||
|
|
@ -545,6 +546,7 @@ set(GLSL_SRC
|
|||
shaders/gpu_shader_3D_smooth_color_vert.glsl
|
||||
shaders/gpu_shader_3D_smooth_color_frag.glsl
|
||||
shaders/gpu_shader_3D_clipped_uniform_color_vert.glsl
|
||||
shaders/gpu_shader_2D_update_mipmaps.bsl.hh
|
||||
|
||||
shaders/gpu_shader_point_uniform_color_aa_frag.glsl
|
||||
shaders/gpu_shader_point_uniform_color_outline_aa_frag.glsl
|
||||
|
|
|
|||
|
|
@ -97,6 +97,11 @@ enum GPUBuiltinShader {
|
|||
GPU_SHADER_INDEXBUF_LINES,
|
||||
GPU_SHADER_INDEXBUF_TRIS,
|
||||
|
||||
/** Compute shaders to generate mipmaps. */
|
||||
GPU_SHADER_2D_UPDATE_MIPMAPS_UNORM_8_8_8_8,
|
||||
GPU_SHADER_2D_UPDATE_MIPMAPS_SFLOAT_16,
|
||||
GPU_SHADER_2D_UPDATE_MIPMAPS_SFLOAT_16_16_16_16,
|
||||
|
||||
/**
|
||||
* ----------------------- Shaders exposed through pyGPU module -----------------------
|
||||
*
|
||||
|
|
|
|||
|
|
@ -1014,7 +1014,9 @@ void GPU_texture_copy(gpu::Texture *dst, gpu::Texture *src);
|
|||
|
||||
/**
|
||||
* Update the mip-map levels using the mip 0 data.
|
||||
*
|
||||
* \note this doesn't work on depth or compressed textures.
|
||||
* \note post-condition: All bound images could be unbound.
|
||||
*/
|
||||
void GPU_texture_update_mipmap_chain(gpu::Texture *texture);
|
||||
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@
|
|||
* \ingroup gpu
|
||||
*/
|
||||
|
||||
#include "GPU_shader_builtin.hh"
|
||||
#include "BKE_global.hh"
|
||||
#include "BLI_utildefines.h"
|
||||
|
||||
|
|
@ -125,6 +126,12 @@ static const char *builtin_shader_create_info_name(GPUBuiltinShader shader)
|
|||
return "gpu_shader_index_2d_array_lines";
|
||||
case GPU_SHADER_INDEXBUF_TRIS:
|
||||
return "gpu_shader_index_2d_array_tris";
|
||||
case GPU_SHADER_2D_UPDATE_MIPMAPS_UNORM_8_8_8_8:
|
||||
return "gpu_shader_2D_update_mipmaps_unorm_8_8_8_8";
|
||||
case GPU_SHADER_2D_UPDATE_MIPMAPS_SFLOAT_16:
|
||||
return "gpu_shader_2D_update_mipmaps_sfloat_16";
|
||||
case GPU_SHADER_2D_UPDATE_MIPMAPS_SFLOAT_16_16_16_16:
|
||||
return "gpu_shader_2D_update_mipmaps_sfloat_16_16_16_16";
|
||||
default:
|
||||
BLI_assert_unreachable();
|
||||
return "";
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@
|
|||
#include "gpu_context_private.hh"
|
||||
#include "gpu_framebuffer_private.hh"
|
||||
|
||||
#include "gpu_shader_private.hh"
|
||||
#include "gpu_texture_private.hh"
|
||||
|
||||
namespace blender {
|
||||
|
|
@ -639,11 +640,6 @@ void GPU_texture_image_unbind_all()
|
|||
Context::get()->state_manager->image_unbind_all();
|
||||
}
|
||||
|
||||
void GPU_texture_update_mipmap_chain(gpu::Texture *tex)
|
||||
{
|
||||
tex->generate_mipmap();
|
||||
}
|
||||
|
||||
void GPU_texture_copy(gpu::Texture *dst_, gpu::Texture *src_)
|
||||
{
|
||||
Texture *src = src_;
|
||||
|
|
|
|||
171
source/blender/gpu/intern/gpu_texture_mipmap.cc
Normal file
171
source/blender/gpu/intern/gpu_texture_mipmap.cc
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
/* SPDX-FileCopyrightText: 2026 Blender Authors
|
||||
*
|
||||
* SPDX-License-Identifier: GPL-2.0-or-later */
|
||||
|
||||
/** \file
|
||||
* \ingroup gpu
|
||||
*/
|
||||
|
||||
#include "BLI_index_range.hh"
|
||||
|
||||
#include "GPU_compute.hh"
|
||||
#include "GPU_debug.hh"
|
||||
#include "GPU_platform.hh"
|
||||
#include "GPU_platform_backend_enum.h"
|
||||
#include "GPU_shader.hh"
|
||||
#include "GPU_shader_builtin.hh"
|
||||
#include "GPU_state.hh"
|
||||
#include "GPU_texture.hh"
|
||||
|
||||
#include "gpu_context_private.hh"
|
||||
#include "gpu_shader_private.hh"
|
||||
#include "gpu_texture_private.hh"
|
||||
|
||||
#include "CLG_log.h"
|
||||
|
||||
static CLG_LogRef LOG = {"gpu.mipmap"};
|
||||
|
||||
namespace blender {
|
||||
|
||||
namespace gpu {
|
||||
|
||||
static Shader *get_update_mipmap_shader(TextureFormat texture_format)
|
||||
{
|
||||
switch (texture_format) {
|
||||
case TextureFormat::UNORM_8_8_8_8:
|
||||
return GPU_shader_get_builtin_shader(GPU_SHADER_2D_UPDATE_MIPMAPS_UNORM_8_8_8_8);
|
||||
case TextureFormat::SFLOAT_16:
|
||||
return GPU_shader_get_builtin_shader(GPU_SHADER_2D_UPDATE_MIPMAPS_SFLOAT_16);
|
||||
case TextureFormat::SFLOAT_16_16_16_16:
|
||||
return GPU_shader_get_builtin_shader(GPU_SHADER_2D_UPDATE_MIPMAPS_SFLOAT_16_16_16_16);
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
static TextureFormat get_view_format(TextureFormat texture_format)
|
||||
{
|
||||
switch (texture_format) {
|
||||
case TextureFormat::SRGBA_8_8_8_8:
|
||||
return TextureFormat::UNORM_8_8_8_8;
|
||||
|
||||
default:
|
||||
return texture_format;
|
||||
}
|
||||
|
||||
return texture_format;
|
||||
}
|
||||
|
||||
static void update_mipmaps(Texture &texture, Shader &shader, int layer)
|
||||
{
|
||||
const int num_mipmaps = texture.mip_count();
|
||||
const TextureFormat view_format = get_view_format(texture.format_get());
|
||||
Vector<Texture *, 16> views;
|
||||
for (int mipmap : IndexRange(num_mipmaps)) {
|
||||
views.append(GPU_texture_create_view(
|
||||
__func__, &texture, view_format, mipmap, 1, layer, 1, false, false));
|
||||
}
|
||||
|
||||
constexpr int max_levels_per_dispatch = 2;
|
||||
|
||||
for (int mip_start = 0; mip_start < num_mipmaps - 1; mip_start += max_levels_per_dispatch) {
|
||||
GPU_memory_barrier(GPU_BARRIER_SHADER_IMAGE_ACCESS);
|
||||
GPU_texture_image_bind(views[mip_start], 0);
|
||||
for (int mip_offset = 1; mip_offset <= max_levels_per_dispatch; mip_offset++) {
|
||||
GPU_texture_image_bind(views[min_ii(mip_start + mip_offset, views.size() - 1)], mip_offset);
|
||||
}
|
||||
int num_levels = min_ii(views.size() - mip_start - 1, max_levels_per_dispatch);
|
||||
GPU_shader_uniform_1i(&shader, "num_levels", num_levels);
|
||||
|
||||
int3 mip_size(1, 1, 1);
|
||||
texture.mip_size_get(mip_start + num_levels, mip_size);
|
||||
|
||||
if (num_levels == 1U) {
|
||||
/* Each thread writes one sample. */
|
||||
constexpr uint32_t warps = 4;
|
||||
const uint32_t samples = mip_size.x * mip_size.y;
|
||||
const uint32_t threads = warps * 32U;
|
||||
int group_len = divide_ceil_u(samples, threads);
|
||||
GPU_compute_dispatch(&shader, group_len, 1, 1);
|
||||
}
|
||||
else {
|
||||
/* Each workgroup handles a tile. */
|
||||
constexpr uint32_t TileWidth = 8;
|
||||
constexpr uint32_t TileHeight = 8;
|
||||
const uint32_t horizontalTiles = divide_ceil_u(mip_size.x, TileWidth);
|
||||
const uint32_t verticalTiles = divide_ceil_u(mip_size.y, TileHeight);
|
||||
int group_len = horizontalTiles * verticalTiles;
|
||||
GPU_compute_dispatch(&shader, group_len, 1, 1);
|
||||
}
|
||||
}
|
||||
|
||||
for (Texture *view : views) {
|
||||
GPU_texture_free(view);
|
||||
}
|
||||
GPU_memory_barrier(GPU_BARRIER_TEXTURE_FETCH | GPU_BARRIER_SHADER_IMAGE_ACCESS);
|
||||
}
|
||||
|
||||
static void update_mipmaps(Texture &texture, Shader &shader)
|
||||
{
|
||||
|
||||
Context &context = *Context::get();
|
||||
Shader *prev_shader = context.shader;
|
||||
|
||||
for (int layer : IndexRange(texture.layer_count())) {
|
||||
update_mipmaps(texture, shader, layer);
|
||||
}
|
||||
|
||||
/* Clear all bound images.
|
||||
*
|
||||
* Current OpenGL API doesn't have a way to rebind the previous state as it only keeps track of
|
||||
* handles. Using a temporary state manager doesn't fit with Metal as the state is stored in
|
||||
* multiple places.
|
||||
*
|
||||
* To not over complicate the implementation for something that is not likely to happen it was
|
||||
* decided to unbind all images. When artifacts happen the calling code must be fixed. */
|
||||
context.state_manager->image_unbind_all();
|
||||
|
||||
/* Reset original state. */
|
||||
if (prev_shader) {
|
||||
GPU_shader_bind(prev_shader);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace gpu
|
||||
|
||||
using namespace blender::gpu;
|
||||
|
||||
void GPU_texture_update_mipmap_chain(Texture *tex)
|
||||
{
|
||||
BLI_assert(tex);
|
||||
|
||||
const int num_mipmaps = tex->mip_count();
|
||||
/* Early exit - nothing to generate as texture only contains 1 mipmap level. */
|
||||
if (num_mipmaps == 1) {
|
||||
return;
|
||||
}
|
||||
|
||||
/* Currently only enabled for Vulkan. OpenGL and Metal have render issues that needs to be
|
||||
* inspected. */
|
||||
if (GPU_type_matches_ex(GPU_DEVICE_ANY, GPU_OS_ANY, GPU_DRIVER_ANY, GPU_BACKEND_VULKAN)) {
|
||||
const TextureFormat texture_format = tex->format_get();
|
||||
Shader *shader = get_update_mipmap_shader(texture_format);
|
||||
if (shader) {
|
||||
GPU_debug_group_begin("Update Mipmaps");
|
||||
update_mipmaps(*tex, *shader);
|
||||
GPU_debug_group_end();
|
||||
return;
|
||||
}
|
||||
CLOG_INFO(&LOG,
|
||||
"No shader exists for updating mipmaps (format=%s). Fallback to backend "
|
||||
"implementation, this could lead to different results between platforms.",
|
||||
GPU_texture_format_name(texture_format));
|
||||
}
|
||||
|
||||
/* No mipmap shader exists for this texture format. Fallback to backend implementation. */
|
||||
tex->generate_mipmap();
|
||||
}
|
||||
|
||||
} // namespace blender
|
||||
|
|
@ -6,7 +6,8 @@
|
|||
* Compile shader files as C++ inside one compilation unit to lint syntax and get IDE integration.
|
||||
*/
|
||||
|
||||
#include "gpu_shader_2D_nodelink.bsl.hh" /* IWYU pragma: export */
|
||||
#include "gpu_shader_2D_widget_base.bsl.hh" /* IWYU pragma: export */
|
||||
#include "gpu_shader_2D_nodelink.bsl.hh" /* IWYU pragma: export */
|
||||
#include "gpu_shader_2D_update_mipmaps.bsl.hh" /* IWYU pragma: export */
|
||||
#include "gpu_shader_2D_widget_base.bsl.hh" /* IWYU pragma: export */
|
||||
|
||||
void main() {}
|
||||
|
|
|
|||
598
source/blender/gpu/shaders/gpu_shader_2D_update_mipmaps.bsl.hh
Normal file
598
source/blender/gpu/shaders/gpu_shader_2D_update_mipmaps.bsl.hh
Normal file
|
|
@ -0,0 +1,598 @@
|
|||
/* SPDX-FileCopyrightText: 2021 NVIDIA Corporation
|
||||
* SPDX-FileCopyrightText: 2026 Blender Foundation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0
|
||||
*
|
||||
* Adapted code from NVIDIA Corporation. */
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "gpu_shader_compat.hh"
|
||||
|
||||
namespace builtin::mipmaps {
|
||||
|
||||
/* Conversion functions. */
|
||||
template<typename DstType, typename SrcType>
|
||||
void convert(DstType &dst_value, const SrcType src_value)
|
||||
{
|
||||
}
|
||||
|
||||
template<> void convert<float4, float>(float4 &dst_value, const float src_value)
|
||||
{
|
||||
dst_value.x = src_value;
|
||||
dst_value.y = 0.0;
|
||||
dst_value.z = 0.0;
|
||||
dst_value.w = 0.0;
|
||||
}
|
||||
template<> void convert<float, float4>(float &dst_value, const float4 src_value)
|
||||
{
|
||||
dst_value = src_value.x;
|
||||
}
|
||||
template<> void convert<float4, float4>(float4 &dst_value, const float4 src_value)
|
||||
{
|
||||
dst_value = src_value;
|
||||
}
|
||||
|
||||
/* Color transfer functions */
|
||||
/* TODO: should be moved to a library */
|
||||
float srgb_to_linearrgb(float c)
|
||||
{
|
||||
if (c < 0.04045f) {
|
||||
return (c < 0.0f) ? 0.0f : c * (1.0f / 12.92f);
|
||||
}
|
||||
|
||||
return pow((c + 0.055f) * (1.0f / 1.055f), 2.4f);
|
||||
}
|
||||
|
||||
float linearrgb_to_srgb(float c)
|
||||
{
|
||||
if (c < 0.0031308f) {
|
||||
return (c < 0.0f) ? 0.0f : c * 12.92f;
|
||||
}
|
||||
|
||||
return 1.055f * pow(c, 1.0f / 2.4f) - 0.055f;
|
||||
}
|
||||
|
||||
/**
|
||||
* General-case shader for generating 1 or 2 levels of the mip pyramid.
|
||||
* When generating 1 level, each workgroup handles up to 128 samples of the
|
||||
* output mip level. When generating 2 levels, each workgroup handles
|
||||
* a 8x8 tile of the last (2nd) output mip level, generating up to
|
||||
* 17x17 samples of the intermediate (1st) output mip level along the way.
|
||||
*
|
||||
* Dispatch with y, z = 1
|
||||
*/
|
||||
#define LOCAL_SIZE_X 128
|
||||
#define TILE_SIZE 8
|
||||
#define MAX_SHARED_SAMPLES (TILE_SIZE + TILE_SIZE + 1)
|
||||
#define INPUT_LEVEL 0
|
||||
|
||||
/** Shared storage that can store intermediate results using without encoding. */
|
||||
template<typename T> struct Shared {
|
||||
/**
|
||||
* When generating 2 levels, the results the first level are cached here; this is the input tile
|
||||
* needed to generate the 8x8 tile of the second level.
|
||||
*/
|
||||
[[shared]] T intermediate_level[MAX_SHARED_SAMPLES][MAX_SHARED_SAMPLES];
|
||||
|
||||
void store_sample(int2 dst_coord, T color)
|
||||
{
|
||||
intermediate_level[dst_coord.y][dst_coord.x] = color;
|
||||
}
|
||||
|
||||
T load_sample(int2 src_coord)
|
||||
{
|
||||
return intermediate_level[src_coord.y][src_coord.x];
|
||||
}
|
||||
};
|
||||
|
||||
/** Shared storage that can store intermediate results in an SRGB encoded uint. */
|
||||
struct SharedSRGB {
|
||||
/**
|
||||
* When generating 2 levels, the results the first level are cached here; this is the input tile
|
||||
* needed to generate the 8x8 tile of the second level.
|
||||
*/
|
||||
[[shared]] uint intermediate_level[MAX_SHARED_SAMPLES][MAX_SHARED_SAMPLES];
|
||||
|
||||
void store_sample(int2 dst_coord, float4 color)
|
||||
{
|
||||
float4 srgba;
|
||||
srgba.r = linearrgb_to_srgb(color.r);
|
||||
srgba.g = linearrgb_to_srgb(color.g);
|
||||
srgba.b = linearrgb_to_srgb(color.b);
|
||||
srgba.a = color.a;
|
||||
uint srgb_packed = packUnorm4x8(srgba);
|
||||
intermediate_level[dst_coord.y][dst_coord.x] = srgb_packed;
|
||||
}
|
||||
|
||||
float4 load_sample(int2 src_coord)
|
||||
{
|
||||
uint srgb_packed = intermediate_level[src_coord.y][src_coord.x];
|
||||
float4 srgba = unpackUnorm4x8(srgb_packed);
|
||||
float4 linear_color;
|
||||
linear_color.r = srgb_to_linearrgb(srgba.r);
|
||||
linear_color.g = srgb_to_linearrgb(srgba.g);
|
||||
linear_color.b = srgb_to_linearrgb(srgba.b);
|
||||
linear_color.a = srgba.a;
|
||||
return linear_color;
|
||||
}
|
||||
};
|
||||
|
||||
int2 kernel_size_from_input_size(int2 input_size)
|
||||
{
|
||||
return int2(input_size.x == 1 ? 1 : (2 | (input_size.x & 1)),
|
||||
input_size.y == 1 ? 1 : (2 | (input_size.y & 1)));
|
||||
}
|
||||
|
||||
/**
|
||||
* \brief Templated struct for bindings and performing the mipmap generation.
|
||||
*
|
||||
* The mipmap generation is based on the general algorithm of
|
||||
* https://github.com/nvpro-samples/vk_compute_mipmaps/tree/main/nvpro_pyramid
|
||||
* It can generate 2 mipmap levels per dispatch.
|
||||
*
|
||||
* \param format is the texture format of the mipmap images.
|
||||
*
|
||||
* \param SharedStorage is the storage class to store intermediate levels. Depending on the texture
|
||||
* format an optimal storage class can be selected.
|
||||
*
|
||||
* \param InnerType the type to use for computation. Depending on the number of samples that a
|
||||
* texture format has a more memory efficient type can be used.
|
||||
*/
|
||||
template<enum TextureWriteFormat format, typename SharedStorage, typename InnerType>
|
||||
struct Resources {
|
||||
[[push_constant]] const int num_levels;
|
||||
[[image(0, read, format)]] image2D mip_in;
|
||||
[[image(1, write, format)]] image2D mip_out1;
|
||||
[[image(2, write, format)]] image2D mip_out2;
|
||||
[[resource_table]] srt_t<SharedStorage> shared_storage;
|
||||
|
||||
/** Store sample result into an output mip image. */
|
||||
void store_sample(int2 dst_coord, int dst_level, InnerType color)
|
||||
{
|
||||
float4 color_out;
|
||||
convert<float4, InnerType>(color_out, color);
|
||||
|
||||
if (dst_level == 1) {
|
||||
imageStore(mip_out1, dst_coord, color_out);
|
||||
}
|
||||
else if (dst_level == 2) {
|
||||
imageStore(mip_out2, dst_coord, color_out);
|
||||
}
|
||||
}
|
||||
|
||||
void store_shared_sample(int2 dst_coord, InnerType color)
|
||||
{
|
||||
SharedStorage &storage = shared_storage;
|
||||
storage.store_sample(dst_coord, color);
|
||||
}
|
||||
|
||||
InnerType load_sample(int2 src_coord, bool load_from_shared)
|
||||
{
|
||||
InnerType color;
|
||||
if (load_from_shared) {
|
||||
SharedStorage &storage = shared_storage;
|
||||
color = storage.load_sample(src_coord);
|
||||
}
|
||||
else {
|
||||
float4 loaded_color = imageLoad(mip_in, src_coord);
|
||||
convert<InnerType, float4>(color, loaded_color);
|
||||
}
|
||||
return color;
|
||||
}
|
||||
|
||||
int2 level_size(int level)
|
||||
{
|
||||
int2 mip_in_size = imageSize(mip_in);
|
||||
int2 mip_size = max((mip_in_size >> level), int2(1));
|
||||
return mip_size;
|
||||
}
|
||||
|
||||
InnerType pyramid_reduce_3(
|
||||
float a0, InnerType v0, float a1, InnerType v1, float a2, InnerType v2)
|
||||
{
|
||||
return a0 * v0 + a1 * v1 + a2 * v2;
|
||||
}
|
||||
|
||||
InnerType pyramid_reduce_2(InnerType v0, InnerType v1)
|
||||
{
|
||||
return 0.5 * (v0 + v1);
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle loading and reducing a rectangle of size kernel_size
|
||||
* with the given upper-left coordinate src_coord. Samples read from
|
||||
* mip level src_level if !loadFromShared_, sharedLevel_ otherwise.
|
||||
*
|
||||
* kernel_size must range from 1x1 to 3x3.
|
||||
*
|
||||
* Once computed, the sample is written to the given coordinate of the
|
||||
* specified destination mip level, and returned. The destination
|
||||
* image size is needed to compute the kernel weights.
|
||||
*/
|
||||
template<bool load_from_shared>
|
||||
InnerType reduce_store_sample(int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level)
|
||||
{
|
||||
float num_dst_pixels = dst_image_size.y;
|
||||
float rcp = 1.0f / (2 * num_dst_pixels + 1);
|
||||
float w0 = rcp * (num_dst_pixels - dst_coord.y);
|
||||
float w1 = rcp * num_dst_pixels;
|
||||
float w2 = 1.0f - w0 - w1;
|
||||
|
||||
InnerType v0, v1, v2, h0, h1, h2, out_pixel;
|
||||
|
||||
/* Reduce vertically up to 3 times (depending on kernel horizontal size) */
|
||||
switch (kernel_size.x) {
|
||||
case 3:
|
||||
switch (kernel_size.y) {
|
||||
case 3:
|
||||
v2 = load_sample(src_coord + int2(2, 2), load_from_shared);
|
||||
ATTR_FALLTHROUGH;
|
||||
case 2:
|
||||
v1 = load_sample(src_coord + int2(2, 1), load_from_shared);
|
||||
ATTR_FALLTHROUGH;
|
||||
case 1:
|
||||
v0 = load_sample(src_coord + int2(2, 0), load_from_shared);
|
||||
break;
|
||||
}
|
||||
switch (kernel_size.y) {
|
||||
case 3:
|
||||
h2 = pyramid_reduce_3(w0, v0, w1, v1, w2, v2);
|
||||
break;
|
||||
case 2:
|
||||
h2 = pyramid_reduce_2(v0, v1);
|
||||
break;
|
||||
case 1:
|
||||
h2 = v0;
|
||||
break;
|
||||
}
|
||||
ATTR_FALLTHROUGH;
|
||||
case 2:
|
||||
switch (kernel_size.y) {
|
||||
case 3:
|
||||
v2 = load_sample(src_coord + int2(1, 2), load_from_shared);
|
||||
ATTR_FALLTHROUGH;
|
||||
case 2:
|
||||
v1 = load_sample(src_coord + int2(1, 1), load_from_shared);
|
||||
ATTR_FALLTHROUGH;
|
||||
case 1:
|
||||
v0 = load_sample(src_coord + int2(1, 0), load_from_shared);
|
||||
break;
|
||||
}
|
||||
switch (kernel_size.y) {
|
||||
case 3:
|
||||
h1 = pyramid_reduce_3(w0, v0, w1, v1, w2, v2);
|
||||
break;
|
||||
case 2:
|
||||
h1 = pyramid_reduce_2(v0, v1);
|
||||
break;
|
||||
case 1:
|
||||
h1 = v0;
|
||||
break;
|
||||
}
|
||||
ATTR_FALLTHROUGH;
|
||||
case 1:
|
||||
switch (kernel_size.y) {
|
||||
case 3:
|
||||
v2 = load_sample(src_coord + int2(0, 2), load_from_shared);
|
||||
ATTR_FALLTHROUGH;
|
||||
case 2:
|
||||
v1 = load_sample(src_coord + int2(0, 1), load_from_shared);
|
||||
ATTR_FALLTHROUGH;
|
||||
case 1:
|
||||
v0 = load_sample(src_coord + int2(0, 0), load_from_shared);
|
||||
break;
|
||||
}
|
||||
switch (kernel_size.y) {
|
||||
case 3:
|
||||
h0 = pyramid_reduce_3(w0, v0, w1, v1, w2, v2);
|
||||
break;
|
||||
case 2:
|
||||
h0 = pyramid_reduce_2(v0, v1);
|
||||
break;
|
||||
case 1:
|
||||
h0 = v0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
/* Reduce up to 3 samples horizontally. */
|
||||
switch (kernel_size.x) {
|
||||
case 3:
|
||||
num_dst_pixels = dst_image_size.x;
|
||||
rcp = 1.0f / (2 * num_dst_pixels + 1);
|
||||
w0 = rcp * (num_dst_pixels - dst_coord.x);
|
||||
w1 = rcp * num_dst_pixels;
|
||||
w2 = 1.0f - w0 - w1;
|
||||
out_pixel = pyramid_reduce_3(w0, h0, w1, h1, w2, h2);
|
||||
break;
|
||||
case 2:
|
||||
out_pixel = pyramid_reduce_2(h0, h1);
|
||||
break;
|
||||
case 1:
|
||||
out_pixel = h0;
|
||||
}
|
||||
|
||||
/* Write out sample. */
|
||||
store_sample(dst_coord, dst_level, out_pixel);
|
||||
return out_pixel;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute and write out (to the 1st mip level generated) the samples
|
||||
* at coordinates
|
||||
* init_dst_coord,
|
||||
* init_dst_coord + step, ...
|
||||
* init_dst_coord + (iterations-1) * step
|
||||
* and cache them at in the sharedLevel_ tile at coordinates
|
||||
* init_shared_coord,
|
||||
* init_shared_coord + step, ...
|
||||
* init_shared_coord + (iterations-1) * step
|
||||
* If use_bounds_check is true, skip coordinates that are out of bounds.
|
||||
*/
|
||||
void intermediate_level_loop(int2 init_dst_coord,
|
||||
int2 init_shared_coord,
|
||||
int2 step,
|
||||
int iterations,
|
||||
bool use_bounds_check)
|
||||
{
|
||||
int2 dst_coord = init_dst_coord;
|
||||
int2 shared_coord = init_shared_coord;
|
||||
int src_level = INPUT_LEVEL;
|
||||
int dst_level = src_level + 1;
|
||||
int2 src_image_size = level_size(src_level);
|
||||
int2 dst_image_size = level_size(dst_level);
|
||||
int2 kernel_size = kernel_size_from_input_size(src_image_size);
|
||||
|
||||
for (int i_ = 0; i_ < iterations; ++i_) {
|
||||
int2 src_coord = dst_coord * 2;
|
||||
|
||||
if (use_bounds_check) {
|
||||
if (uint(dst_coord.x) >= uint(dst_image_size.x)) {
|
||||
continue;
|
||||
}
|
||||
if (uint(dst_coord.y) >= uint(dst_image_size.y)) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
InnerType result = reduce_store_sample<false>(
|
||||
src_coord, src_level, kernel_size, dst_image_size, dst_coord, dst_level);
|
||||
|
||||
/* `reduce_store_sample` handles writing to the actual output; manually
|
||||
* cache into shared memory here. */
|
||||
store_shared_sample(shared_coord, result);
|
||||
dst_coord += step;
|
||||
shared_coord += step;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Function for the workgroup that handles filling the intermediate level
|
||||
* (caching it in shared memory as well).
|
||||
*
|
||||
* We need somewhere from 16x16 to 17x17 samples, depending
|
||||
* on what the kernel size for the 2nd mip level generation will be.
|
||||
*
|
||||
* dst_tile_coord : upper left coordinate of the tile to generate.
|
||||
* use_bounds_check : whether to skip samples that are out-of-bounds.
|
||||
*/
|
||||
void fill_intermediate_tile(uint local_index, int2 dst_tile_coord, bool use_bounds_check)
|
||||
{
|
||||
int2 init_thread_offset;
|
||||
int2 step;
|
||||
int iterations;
|
||||
|
||||
int2 dst_image_size = level_size(INPUT_LEVEL + 1);
|
||||
int2 future_kernel_size = kernel_size_from_input_size(dst_image_size);
|
||||
|
||||
if (future_kernel_size.x == 3) {
|
||||
if (future_kernel_size.y == 3) {
|
||||
/* Fill in 2 17x7 steps and 1 17x3 step (9 idle threads) */
|
||||
init_thread_offset = int2(local_index % 17u, local_index / 17u);
|
||||
step = int2(0, 7);
|
||||
iterations = local_index >= 7 * 17 ? 0 : local_index < 3 * 17 ? 3 : 2;
|
||||
}
|
||||
else {
|
||||
/* Future 3x[2,1] kernel
|
||||
* Fill in 2 8x16 steps and 1 1x16 step */
|
||||
init_thread_offset = int2(local_index / 16u, local_index % 16u);
|
||||
step = int2(8, 0);
|
||||
iterations = local_index < 1 * 16 ? 3 : 2;
|
||||
}
|
||||
}
|
||||
else {
|
||||
if (future_kernel_size.y == 3) {
|
||||
/* Fill in 2 16x8 steps and 1 16x1 step */
|
||||
init_thread_offset = int2(local_index % 16u, local_index / 16u);
|
||||
step = int2(0, 8);
|
||||
iterations = local_index < 1 * 16 ? 3 : 2;
|
||||
}
|
||||
else {
|
||||
/* Fill in 2 16x8 steps */
|
||||
init_thread_offset = int2(local_index % 16u, local_index / 16u);
|
||||
step = int2(0, 8);
|
||||
iterations = 2;
|
||||
}
|
||||
}
|
||||
|
||||
intermediate_level_loop(dst_tile_coord + init_thread_offset,
|
||||
init_thread_offset,
|
||||
step,
|
||||
iterations,
|
||||
use_bounds_check);
|
||||
}
|
||||
|
||||
/**
|
||||
* Function for the workgroup that handles filling the last level tile
|
||||
* (2nd level after the original input level), using as input the
|
||||
* tile in shared memory.
|
||||
*
|
||||
* dst_tile_coord : upper left coordinate of the tile to generate.
|
||||
* use_bounds_check : whether to skip samples that are out-of-bounds.
|
||||
*/
|
||||
void fill_last_tile(uint local_index, int2 dst_tile_coord, bool use_bounds_check)
|
||||
{
|
||||
|
||||
if (local_index < 8 * 8) {
|
||||
int2 thread_offset = int2(local_index % 8u, local_index / 8u);
|
||||
int src_level = INPUT_LEVEL + 1;
|
||||
int dst_level = INPUT_LEVEL + 2;
|
||||
int2 src_image_size = level_size(src_level);
|
||||
int2 dst_image_size = level_size(dst_level);
|
||||
|
||||
int2 src_shared_coord = thread_offset * 2;
|
||||
int2 kernel_size = kernel_size_from_input_size(src_image_size);
|
||||
int2 dst_coord = thread_offset + dst_tile_coord;
|
||||
|
||||
bool within_bounds = true;
|
||||
if (use_bounds_check) {
|
||||
within_bounds = (uint(dst_coord.x) < uint(dst_image_size.x)) &&
|
||||
(uint(dst_coord.y) < uint(dst_image_size.y));
|
||||
}
|
||||
if (within_bounds) {
|
||||
reduce_store_sample<true>(
|
||||
src_shared_coord, 0, kernel_size, dst_image_size, dst_coord, dst_level);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template<typename SRT>
|
||||
[[local_size(LOCAL_SIZE_X)]] [[compute]]
|
||||
void update_mipmaps([[global_invocation_id]] const uint3 global_id,
|
||||
[[work_group_id]] const uint3 group_id,
|
||||
[[local_invocation_id]] const uint3 local_index,
|
||||
[[resource_table]] SRT &srt)
|
||||
{
|
||||
if (srt.num_levels == 1u) {
|
||||
int2 kernel_size = kernel_size_from_input_size(srt.level_size(INPUT_LEVEL));
|
||||
int2 dst_image_size = srt.level_size(INPUT_LEVEL + 1);
|
||||
int2 dst_coord = int2(int(global_id.x) % dst_image_size.x,
|
||||
int(global_id.x) / dst_image_size.x);
|
||||
int2 src_coord = dst_coord * 2;
|
||||
|
||||
if (dst_coord.y < dst_image_size.y) {
|
||||
srt.template reduce_store_sample<false>(
|
||||
src_coord, INPUT_LEVEL, kernel_size, dst_image_size, dst_coord, INPUT_LEVEL + 1);
|
||||
}
|
||||
}
|
||||
else {
|
||||
/* Handling two levels.
|
||||
* Assign a 8x8 tile of mip level inputLevel_ + 2 to this workgroup. */
|
||||
int level2 = INPUT_LEVEL + 2;
|
||||
int2 level2_size = srt.level_size(level2);
|
||||
int2 tile_count;
|
||||
tile_count.x = int(uint(level2_size.x + 7) / 8u);
|
||||
tile_count.y = int(uint(level2_size.y + 7) / 8u);
|
||||
int2 tile_index = int2(group_id.x % uint(tile_count.x), group_id.x / uint(tile_count.x));
|
||||
|
||||
/* Determine if bounds checking is needed; this is only the case
|
||||
* for tiles at the right or bottom fringe that might be cut off
|
||||
* by the image border. Note that later, I use if statements rather
|
||||
* than passing use_bounds_check directly to convince the compiler
|
||||
* to inline everything. */
|
||||
bool use_bounds_check = tile_index.x >= tile_count.x - 1 || tile_index.y >= tile_count.y - 1;
|
||||
|
||||
if (use_bounds_check) {
|
||||
/* Compute the tile in level inputLevel_ + 1 that's needed to
|
||||
* compute the above 8x8 tile. */
|
||||
srt.fill_intermediate_tile(local_index.x, tile_index * 2 * int2(8, 8), true);
|
||||
barrier();
|
||||
|
||||
/* Compute the inputLevel_ + 2 tile of size 8x8, loading
|
||||
* inputs from shared memory. */
|
||||
srt.fill_last_tile(local_index.x, tile_index * int2(8, 8), true);
|
||||
}
|
||||
else {
|
||||
/* Same but without bounds checking. */
|
||||
srt.fill_intermediate_tile(local_index.x, tile_index * 2 * int2(8, 8), false);
|
||||
barrier();
|
||||
srt.fill_last_tile(local_index.x, tile_index * int2(8, 8), false);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template struct Shared<float>;
|
||||
template struct Shared<float4>;
|
||||
|
||||
template struct Resources<UNORM_8_8_8_8, SharedSRGB, float4>;
|
||||
template struct Resources<SFLOAT_16, Shared<float>, float>;
|
||||
template struct Resources<SFLOAT_16_16_16_16, Shared<float4>, float4>;
|
||||
|
||||
template float4 Resources<UNORM_8_8_8_8, SharedSRGB, float4>::reduce_store_sample<true>(
|
||||
int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level);
|
||||
template float4 Resources<UNORM_8_8_8_8, SharedSRGB, float4>::reduce_store_sample<false>(
|
||||
int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level);
|
||||
template float Resources<SFLOAT_16, Shared<float>, float>::reduce_store_sample<true>(
|
||||
int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level);
|
||||
template float Resources<SFLOAT_16, Shared<float>, float>::reduce_store_sample<false>(
|
||||
int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level);
|
||||
template float4 Resources<SFLOAT_16_16_16_16, Shared<float4>, float4>::reduce_store_sample<true>(
|
||||
int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level);
|
||||
template float4 Resources<SFLOAT_16_16_16_16, Shared<float4>, float4>::reduce_store_sample<false>(
|
||||
int2 src_coord,
|
||||
int src_level,
|
||||
int2 kernel_size,
|
||||
int2 dst_image_size,
|
||||
int2 dst_coord,
|
||||
int dst_level);
|
||||
|
||||
template void update_mipmaps<Resources<UNORM_8_8_8_8, SharedSRGB, float4>>(
|
||||
const uint3 global_id,
|
||||
const uint3 group_id,
|
||||
const uint3 local_index,
|
||||
Resources<UNORM_8_8_8_8, SharedSRGB, float4> &srt);
|
||||
template void update_mipmaps<Resources<SFLOAT_16, Shared<float>, float>>(
|
||||
const uint3 global_id,
|
||||
const uint3 group_id,
|
||||
const uint3 local_index,
|
||||
Resources<SFLOAT_16, Shared<float>, float> &srt);
|
||||
template void update_mipmaps<Resources<SFLOAT_16_16_16_16, Shared<float4>, float4>>(
|
||||
const uint3 global_id,
|
||||
const uint3 group_id,
|
||||
const uint3 local_index,
|
||||
Resources<SFLOAT_16_16_16_16, Shared<float4>, float4> &srt);
|
||||
|
||||
} // namespace builtin::mipmaps
|
||||
|
||||
PipelineCompute gpu_shader_2D_update_mipmaps_unorm_8_8_8_8(
|
||||
builtin::mipmaps::update_mipmaps<
|
||||
builtin::mipmaps::Resources<UNORM_8_8_8_8, builtin::mipmaps::SharedSRGB, float4>>);
|
||||
PipelineCompute gpu_shader_2D_update_mipmaps_sfloat_16(
|
||||
builtin::mipmaps::update_mipmaps<
|
||||
builtin::mipmaps::Resources<SFLOAT_16, builtin::mipmaps::Shared<float>, float>>);
|
||||
PipelineCompute gpu_shader_2D_update_mipmaps_sfloat_16_16_16_16(
|
||||
builtin::mipmaps::update_mipmaps<builtin::mipmaps::Resources<SFLOAT_16_16_16_16,
|
||||
builtin::mipmaps::Shared<float4>,
|
||||
float4>>);
|
||||
|
|
@ -275,3 +275,75 @@ struct PipelineCompute {
|
|||
};
|
||||
|
||||
#include "GPU_shader_shared_utils.hh"
|
||||
|
||||
/* -------------------------------------------------------------------- */
|
||||
/** \name Enums
|
||||
*
|
||||
* Enums should be defined in the root namespace when used directly in the pipeline, as they will
|
||||
* not be fully qualified when generating the template name substitution. Defining in the root
|
||||
* works around this limitation
|
||||
*
|
||||
* \{ */
|
||||
|
||||
/**
|
||||
* TextureWriteFormat.
|
||||
*
|
||||
* We can not use GPU_TEXTURE_WRITE_FORMAT_EXPAND as other parts are included that will intervene
|
||||
* with the compatibility defines.
|
||||
*/
|
||||
enum TextureWriteFormat : uint32_t {
|
||||
SNORM_8,
|
||||
SNORM_8_8,
|
||||
SNORM_8_8_8_8,
|
||||
|
||||
SNORM_16,
|
||||
SNORM_16_16,
|
||||
SNORM_16_16_16_16,
|
||||
|
||||
UNORM_8,
|
||||
UNORM_8_8,
|
||||
UNORM_8_8_8_8,
|
||||
|
||||
UNORM_16,
|
||||
UNORM_16_16,
|
||||
UNORM_16_16_16_16,
|
||||
|
||||
SINT_8,
|
||||
SINT_8_8,
|
||||
SINT_8_8_8_8,
|
||||
|
||||
SINT_16,
|
||||
SINT_16_16,
|
||||
SINT_16_16_16_16,
|
||||
|
||||
SINT_32,
|
||||
SINT_32_32,
|
||||
SINT_32_32_32_32,
|
||||
|
||||
UINT_8,
|
||||
UINT_8_8,
|
||||
UINT_8_8_8_8,
|
||||
|
||||
UINT_16,
|
||||
UINT_16_16,
|
||||
UINT_16_16_16_16,
|
||||
|
||||
UINT_32,
|
||||
UINT_32_32,
|
||||
UINT_32_32_32_32,
|
||||
|
||||
SFLOAT_16,
|
||||
SFLOAT_16_16,
|
||||
SFLOAT_16_16_16_16,
|
||||
|
||||
SFLOAT_32,
|
||||
SFLOAT_32_32,
|
||||
SFLOAT_32_32_32_32,
|
||||
|
||||
UNORM_10_10_10_2,
|
||||
UINT_10_10_10_2,
|
||||
|
||||
UFLOAT_11_11_10,
|
||||
};
|
||||
|
||||
/** \} */
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@
|
|||
|
||||
using namespace metal;
|
||||
|
||||
#define packUnorm4x8 pack_float_to_unorm4x8
|
||||
#define unpackUnorm4x8 unpack_unorm4x8_to_float
|
||||
#define unpackSnorm4x8 unpack_snorm4x8_to_float
|
||||
#define unpackUnorm2x16 unpack_unorm2x16_to_float
|
||||
|
|
|
|||
|
|
@ -64,6 +64,9 @@ BLOCKLIST_VULKAN = [
|
|||
"image.blend",
|
||||
]
|
||||
|
||||
BLOCKLIST_OPENGL = [
|
||||
]
|
||||
|
||||
BLOCKLIST_INTEL = [
|
||||
]
|
||||
|
||||
|
|
@ -237,6 +240,8 @@ def main():
|
|||
blocklist += BLOCKLIST_METAL
|
||||
elif args.gpu_backend == "vulkan":
|
||||
blocklist += BLOCKLIST_VULKAN
|
||||
elif args.gpu_backend == "opengl":
|
||||
blocklist += BLOCKLIST_OPENGL
|
||||
|
||||
if os.getenv("BLENDER_TEST_IGNORE_VENDOR_BLOCKLIST") is None:
|
||||
gpu_vendor = render_report.get_gpu_device_vendor(args.blender)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue