blender/intern/cycles/device/cpu/device_impl.h
Sergey Sharybin a70b34065d Refactor: Cycles: Make kernel_features 64bit
This change adds 32 more bit to store kernel features.

While for a short term it might be possible to make a space for one or
two extra bits, it seems going 64bit is inevitable.

Expanding the field to 64bit might introduce some slowdown due to less
optimal cache, but so is consolidation of existing flags could also
lead to performance drop in certain configurations.

The main tricky part of the change is Metal where function constants
are used to store kernel_features, and 64bit constants are only
available on macOS 12. There is a runtime check for it. On older macOS
versions the flags are stored as a pair of 32bit values. It is slower,
but there are unlikely to be many Cycles users on macOS 11.

Ref #159470

Pull Request: https://projects.blender.org/blender/blender/pulls/162737
2026-08-18 10:41:27 +02:00

101 lines
3 KiB
C++

/* SPDX-FileCopyrightText: 2011-2022 Blender Foundation
*
* SPDX-License-Identifier: Apache-2.0 */
#pragma once
/* So ImathMath is included before our kernel_cpu_compat. */
#ifdef WITH_OSL
# include <cstdint> /* Needed before `sdlexec.h` for `int32_t` with GCC 15.1. */
/* So no context pollution happens from indirectly included windows.h */
# ifdef _WIN32
# include "util/windows.h"
# endif
# include <OSL/oslexec.h>
#endif
#ifdef WITH_EMBREE
# include <embree4/rtcore.h>
#endif
#include "device/cpu/kernel.h"
#include "device/device.h"
#include "device/memory.h"
// clang-format off
#include "kernel/device/cpu/kernel.h"
#include "kernel/globals.h"
#include "kernel/osl/globals.h"
// clang-format on
#include "util/guiding.h" // IWYU pragma: keep
#include "util/list.h"
#include "util/unique_ptr.h"
CCL_NAMESPACE_BEGIN
class CPUDevice : public Device {
public:
KernelGlobalsCPU kernel_globals;
vector<ThreadKernelGlobalsCPU> kernel_thread_globals_;
unique_ptr<device_vector<KernelImageInfo>> image_info;
list<unique_ptr<device_vector<KernelImageInfo>>> old_image_infos;
#ifdef WITH_OSL
OSLGlobals osl_globals;
#endif
#ifdef WITH_EMBREE
# if RTC_VERSION >= 40400
RTCTraversable embree_traversable = nullptr;
# else
RTCScene embree_traversable = nullptr;
# endif
RTCDevice embree_device;
#endif
#if defined(WITH_PATH_GUIDING)
mutable unique_ptr<openpgl::cpp::Device> guiding_device;
#endif
CPUDevice(const DeviceInfo &info_, Stats &stats_, Profiler &profiler_, bool headless_);
~CPUDevice() override;
BVHLayoutMask get_bvh_layout_mask(uint64_t kernel_features) const override;
void mem_alloc(device_memory &mem) override;
void mem_copy_to(device_memory &mem) override;
void mem_move_to_host(device_memory &mem) override;
void mem_copy_from(
device_memory &mem, const size_t y, size_t w, const size_t h, size_t elem) override;
void mem_zero(device_memory &mem) override;
void mem_free(device_memory &mem) override;
void mem_or_from_device(device_memory &mem) override;
device_ptr mem_alloc_sub_ptr(device_memory &mem, const size_t offset, size_t /*size*/) override;
void const_copy_to(const char *name, void *host, const size_t size) override;
void global_alloc(device_memory &mem);
void global_free(device_memory &mem);
void image_alloc(device_image &mem);
void image_free(device_image &mem);
bool has_unified_memory_any() const override;
bool has_unified_image_memory_all() const override;
void build_bvh(BVH *bvh, Progress &progress, bool refit) override;
void *get_guiding_device() const override;
vector<ThreadKernelGlobalsCPU> *acquire_cpu_kernel_thread_globals() override;
void release_cpu_kernel_thread_globals() override;
OSLGlobals *get_cpu_osl_memory() override;
void set_image_cache_func(KernelImageLoadRequestedCPU image_load_requested_cpu,
KernelImageLoadRequestedGPU image_load_requested_gpu) override;
protected:
bool load_kernels(uint64_t kernel_features) override;
};
CCL_NAMESPACE_END