Fix #162523: GPU: Add mipmap workaround for Intel Arc Windows OpenGL

The compute-shader based function `update_mipmaps` can generate broken
mipmaps for Intel Arc Windows OpenGL. A workaround for falling back to
`generate_mipmap` is added to mitigate the issue. Furthermore, a render
test is added that catches the issue.

Pull Request: https://projects.blender.org/blender/blender/pulls/162524
This commit is contained in:
Christoph Neuhauser 2026-08-13 16:03:06 +02:00
parent f6c6ee2405
commit 9dfdae1c36
7 changed files with 89 additions and 24 deletions

View file

@ -8,6 +8,7 @@
#include "BLI_index_range.hh"
#include "GPU_capabilities.hh"
#include "GPU_compute.hh"
#include "GPU_debug.hh"
#include "GPU_shader.hh"
@ -74,6 +75,42 @@ static Shader *get_update_mipmap_shader(TextureFormat texture_format, bool is_la
return nullptr;
}
constexpr int max_levels_per_dispatch = 2;
/**
* Compute the number of work groups required to dispatch a single pass of the mipmap update
* shader.
*
* \param mip_start: The first mipmap level that is processed by the dispatch.
*/
uint mipmap_dispatch_group_len(Texture &texture, int mip_start)
{
const int num_mipmaps = texture.mip_count();
if (mip_start >= num_mipmaps - 1) {
return 0;
}
int num_levels = min_ii(num_mipmaps - mip_start - 1, max_levels_per_dispatch);
int3 mip_size(1, 1, 1);
texture.mip_size_get(mip_start + num_levels, mip_size);
if (num_levels == 1u) {
/* Each thread writes one sample. */
constexpr uint32_t warps = 4;
const uint32_t samples = mip_size.x * mip_size.y;
const uint32_t threads = warps * 32U;
return divide_ceil_u(samples, threads);
}
else {
/* Each workgroup handles a tile. */
constexpr uint32_t TileWidth = 8;
constexpr uint32_t TileHeight = 8;
const uint32_t horizontalTiles = divide_ceil_u(mip_size.x, TileWidth);
const uint32_t verticalTiles = divide_ceil_u(mip_size.y, TileHeight);
return horizontalTiles * verticalTiles;
}
}
static void update_mipmaps(Texture &texture, Shader &shader, int layer)
{
const int num_mipmaps = texture.mip_count();
@ -84,8 +121,6 @@ static void update_mipmaps(Texture &texture, Shader &shader, int layer)
__func__, &texture, view_format, mipmap, 1, layer, 1, false, false));
}
constexpr int max_levels_per_dispatch = 2;
for (int mip_start = 0; mip_start < num_mipmaps - 1; mip_start += max_levels_per_dispatch) {
GPU_memory_barrier(GPU_BARRIER_SHADER_IMAGE_ACCESS);
GPU_texture_image_bind(views[mip_start], 0);
@ -95,26 +130,8 @@ static void update_mipmaps(Texture &texture, Shader &shader, int layer)
int num_levels = min_ii(views.size() - mip_start - 1, max_levels_per_dispatch);
GPU_shader_uniform_1i(&shader, "num_levels", num_levels);
int3 mip_size(1, 1, 1);
texture.mip_size_get(mip_start + num_levels, mip_size);
if (num_levels == 1u) {
/* Each thread writes one sample. */
constexpr uint32_t warps = 4;
const uint32_t samples = mip_size.x * mip_size.y;
const uint32_t threads = warps * 32U;
int group_len = divide_ceil_u(samples, threads);
GPU_compute_dispatch(&shader, group_len, 1, 1);
}
else {
/* Each workgroup handles a tile. */
constexpr uint32_t TileWidth = 8;
constexpr uint32_t TileHeight = 8;
const uint32_t horizontalTiles = divide_ceil_u(mip_size.x, TileWidth);
const uint32_t verticalTiles = divide_ceil_u(mip_size.y, TileHeight);
int group_len = horizontalTiles * verticalTiles;
GPU_compute_dispatch(&shader, group_len, 1, 1);
}
uint group_len = mipmap_dispatch_group_len(texture, mip_start);
GPU_compute_dispatch(&shader, group_len, 1, 1);
}
for (Texture *view : views) {
@ -211,6 +228,16 @@ void GPU_texture_update_mipmap_chain(Texture *tex)
use_compute_shaders = false;
}
/* The number of work groups is bounded by `GPU_max_work_group_count()`. Only the dispatch of
* the largest mipmap levels has to be checked, as each subsequent dispatch covers halved
* dimensions and the required work group count only decreases. */
if (use_compute_shaders && mipmap_dispatch_group_len(*tex, 0) > GPU_max_work_group_count(0)) {
use_compute_shaders = false;
CLOG_INFO(&LOG,
"Texture size exceeds `maxComputeWorkGroupCount[0]`. Fallback to backend "
"implementation, this could lead to different results between platforms.");
}
if (use_compute_shaders) {
const TextureFormat texture_format = tex->format_get();
const bool is_layered = tex->type_get() & GPU_TEXTURE_ARRAY;

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1908e97e865bdf1cff2d88a73c09172e1aabbb702fe31cfe7b2fcd17ad407499
size 98916

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:70217c9568f286654082b94d702f32e12b7f7f11fb2009a4d3492c073d344f2a
size 91776

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3663e5fe45ea25ab2953efb3890484754df61e50073ed93a5e82ad4f4e622f5c
size 184002

View file

@ -915,6 +915,7 @@ if((WITH_CYCLES OR WITH_GPU_RENDER_TESTS) AND TEST_SRC_DIR_EXISTS)
image_colorspace
image_data_types
image_mapping
image_mipmap
image_texture_limit
integrator
light
@ -998,7 +999,7 @@ if((WITH_CYCLES OR WITH_GPU_RENDER_TESTS) AND TEST_SRC_DIR_EXISTS)
"Supported devices are: ${_cycles_all_test_devices}")
endif()
string(TOLOWER "${_cycles_device}" _cycles_device_lower)
set(_cycles_render_tests bake;bake_raytrace;${render_tests};osl;image_mipmap;updates)
set(_cycles_render_tests bake;bake_raytrace;${render_tests};osl;updates)
set(_cycles_osl_test_type none)
if(WITH_CYCLES_OSL)

View file

@ -67,6 +67,17 @@ BLOCKLIST = [
"light_path_is_camera_ray.blend",
# Exhibit non-deterministic (to be fixed).
"background_scene.blend",
# Currently, the only image_mipmap test enabled for EEVEE is image_mipmap_large_tex.blend.
"image_cache_evict.blend",
"image_mipmap_area_light.blend",
"image_mipmap_filter_glossy.blend",
"image_mipmap_incomplete_derivs.blend",
"image_mipmap_light_tree.blend",
"image_mipmap_transparent_shadow.blend",
"image_mipmap_volume.blend",
"image_mipmap_working_space.blend",
"image_mipmap_world.blend",
"image_mipmap_world_sun.blend",
### Cycles only tests go here ###
]
@ -384,6 +395,10 @@ def main():
elif test_dir_name in {"texture"}:
report.set_fail_percent(0.14)
report.set_fail_threshold(6.0 / 255.0)
elif test_dir_name.startswith('image_mipmap'):
# Reference images on the CI worker seem to differ slightly from images on
# an NVIDIA RTX 4060 Ti with driver 610.74
report.set_fail_threshold(6.0 / 255.0)
elif test_dir_name.startswith('camera'):
# camera_stereo_panoramic have some platform specific small differences
report.set_fail_percent(0.14)
@ -474,6 +489,13 @@ def main():
# Some shadow difference, to be investigated
report.set_fail_percent(0.09)
report.set_fail_threshold(6.0 / 255.0)
elif test_dir_name.startswith('image_mipmap'):
# Texture interpolation can look slightly different.
if gpu_vendor == "AMD":
report.set_fail_percent(0.39)
else:
report.set_fail_percent(0.08)
report.set_fail_threshold(8.0 / 255.0)
ok = report.run(args.testdir, args.blender, get_arguments, batch=args.batch)
sys.exit(not ok)

View file

@ -24,6 +24,12 @@ except ImportError:
# this script is run during preparation steps.
pass
BLOCKLIST = [
# Currently, image_mipmap tests are not enabled for storm-usd
"image_cache_evict.blend",
"image_mipmap_.*.blend",
]
# Unsupported or broken scenarios for the Storm render engine
BLOCKLIST_HYDRA = [
# Corrupted output around borders
@ -235,7 +241,7 @@ def main():
parser = create_argparse()
args = parser.parse_args()
blocklist = []
blocklist = BLOCKLIST
if args.gpu_backend == "metal":
blocklist += BLOCKLIST_METAL
elif args.gpu_backend == "vulkan":