Fix: Cycles: Ensure integrator working memory is not host mapped

The integrator state was already device only, but some other memory is
also frequently accessed and should be on the GPU for best performance.

This is a follow up for a pre-existing issue found reviewing #163437 and
#163930.

Pull Request: https://projects.blender.org/blender/blender/pulls/164306
This commit is contained in:
Brecht Van Lommel 2026-09-25 10:49:13 +02:00 • committed by Brecht Van Lommel
parent 603236005a
commit 067ed4498f
5 changed files with 48 additions and 20 deletions

View file

@ -766,6 +766,9 @@ GPUDevice::Mem *GPUDevice::generic_alloc(device_memory &mem, const size_t pitch_
const size_t headroom = (is_texture) ? device_image_headroom : device_working_headroom;
const bool no_host_fallback = (mem.type == MEM_DEVICE_ONLY) ||
(mem.flags & MEM_FLAG_NO_HOST_FALLBACK);
/* Move textures to host memory if needed. */
if (!mem.move_to_host && !is_image && can_map_host) {
move_textures_to_host(size, headroom, is_texture);
@ -776,7 +779,7 @@ GPUDevice::Mem *GPUDevice::generic_alloc(device_memory &mem, const size_t pitch_
get_device_memory_info(total, free);
/* Allocate in device memory. */
if ((!mem.move_to_host && (size + headroom) < free) || (mem.type == MEM_DEVICE_ONLY)) {
if ((!mem.move_to_host && (size + headroom) < free) || no_host_fallback) {
mem_alloc_result = alloc_device(device_pointer, size);
if (mem_alloc_result) {
status = " in device memory";
@ -787,7 +790,7 @@ GPUDevice::Mem *GPUDevice::generic_alloc(device_memory &mem, const size_t pitch_
void *shared_pointer = nullptr;
if (!mem_alloc_result && can_map_host && mem.type != MEM_DEVICE_ONLY) {
if (!mem_alloc_result && can_map_host && !no_host_fallback) {
if (mem.shared_pointer) {
/* Another device already allocated host memory. */
mem_alloc_result = true;
@ -809,7 +812,7 @@ GPUDevice::Mem *GPUDevice::generic_alloc(device_memory &mem, const size_t pitch_
}
if (!mem_alloc_result) {
if (mem.type == MEM_DEVICE_ONLY) {
if (no_host_fallback) {
status = " failed, out of device memory";
set_error("System is out of GPU memory");
}

View file

@ -48,7 +48,10 @@ static const char *name_from_type(ImageDataType type)
/* Device Memory */
device_memory::device_memory(Device *device, const char *name, MemoryType type)
device_memory::device_memory(Device *device,
const char *name,
MemoryType type,
const uint32_t flags)
: data_type(device_type_traits<uchar>::data_type),
data_elements(device_type_traits<uchar>::num_elements),
data_size(0),
@ -61,6 +64,7 @@ device_memory::device_memory(Device *device, const char *name, MemoryType type)
host_pointer(nullptr),
shared_pointer(nullptr),
shared_counter(0),
flags(flags),
name_(name),
original_device_ptr(0),
original_device_size(0),

View file

@ -33,6 +33,13 @@ enum MemoryType {
MEM_IMAGE_TEXTURE,
};
enum MemoryFlag {
/* Never map to host memory when the device runs out of memory, where using
* GPU memory is essential for performance. Scene data will then be moved to
* the host instead. */
MEM_FLAG_NO_HOST_FALLBACK = (1 << 0),
};
/* Supported Data Types */
enum DataType {
@ -269,6 +276,9 @@ class device_memory {
int shared_counter;
bool move_to_host = false;
/* MemoryFlag. */
uint32_t flags;
virtual ~device_memory();
void swap_device(Device *new_device, const size_t new_device_size, device_ptr new_device_ptr);
@ -298,7 +308,7 @@ class device_memory {
friend class OneapiDevice;
/* Only create through subclasses. */
device_memory(Device *device, const char *name, MemoryType type);
device_memory(Device *device, const char *name, MemoryType type, uint32_t flags = 0);
/* Host allocation on the device. All host_pointer memory should be
* allocated with these functions, for devices that support using
@ -393,8 +403,8 @@ template<typename T> class device_only_memory : public device_memory {
template<typename T> class device_vector : public device_memory {
public:
device_vector(Device *device, const char *name, MemoryType type)
: device_memory(device, name, type)
device_vector(Device *device, const char *name, MemoryType type, const uint32_t flags = 0)
: device_memory(device, name, type, flags)
{
data_type = device_type_traits<T>::data_type;
data_elements = device_type_traits<T>::num_elements;

View file

@ -57,8 +57,10 @@ class PathTrace {
* The progress is reported to the currently configure progress object (via `set_progress`). */
void load_kernels();
/* Allocate working memory. This runs before allocating scene memory so that we can estimate
* more accurately which scene device memory may need to allocated on the host. */
/* Allocate working memory. This runs after scene memory allocation, so that the number of
* path states can take the updated scene into account. Scene memory can be moved back to
* the host as part of this allocation, as work memory is not host mapped and allocating
* it may request more space to be freed on the GPU. */
void alloc_work_memory();
/* Check whether now it is a good time to reset rendering.

View file

@ -88,19 +88,28 @@ PathTraceWorkGPU::PathTraceWorkGPU(Device *device,
: PathTraceWork(device, film, device_scene, cancel_requested_flag),
queue_(device->gpu_queue_create()),
integrator_state_soa_kernel_features_(0),
integrator_queue_counter_(device, "integrator_queue_counter", MEM_READ_WRITE),
integrator_shader_sort_counter_(device, "integrator_shader_sort_counter", MEM_READ_WRITE),
integrator_shader_raytrace_sort_counter_(
device, "integrator_shader_raytrace_sort_counter", MEM_READ_WRITE),
/* Use MEM_FLAG_NO_HOST_FALLBACK since having this on the GPU is more important
* than scene memory for performance. */
integrator_queue_counter_(
device, "integrator_queue_counter", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
integrator_shader_sort_counter_(
device, "integrator_shader_sort_counter", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
integrator_shader_raytrace_sort_counter_(device,
"integrator_shader_raytrace_sort_counter",
MEM_READ_WRITE,
MEM_FLAG_NO_HOST_FALLBACK),
integrator_shader_sort_prefix_sum_(
device, "integrator_shader_sort_prefix_sum", MEM_READ_WRITE),
integrator_shader_sort_partition_key_offsets_(
device, "integrator_shader_sort_partition_key_offsets", MEM_READ_WRITE),
integrator_next_main_path_index_(device, "integrator_next_main_path_index", MEM_READ_WRITE),
device, "integrator_shader_sort_prefix_sum", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
integrator_shader_sort_partition_key_offsets_(device,
"integrator_shader_sort_partition_key_offsets",
MEM_READ_WRITE,
MEM_FLAG_NO_HOST_FALLBACK),
integrator_next_main_path_index_(
device, "integrator_next_main_path_index", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
integrator_next_shadow_path_index_(
device, "integrator_next_shadow_path_index", MEM_READ_WRITE),
queued_paths_(device, "queued_paths", MEM_READ_WRITE),
num_queued_paths_(device, "num_queued_paths", MEM_READ_WRITE),
device, "integrator_next_shadow_path_index", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
queued_paths_(device, "queued_paths", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
num_queued_paths_(device, "num_queued_paths", MEM_READ_WRITE, MEM_FLAG_NO_HOST_FALLBACK),
work_tiles_(device, "work_tiles", MEM_READ_WRITE),
display_rgba_half_(device, "display buffer half", MEM_READ_WRITE),
max_num_paths_(0),