Cycles: Reduce GPU integrator state with smaller data types

Some values don't need full float precision, and with millions of states
reducing memory usage is important, while the math to pack/unpack these
is quite cheap. On CPU full precision is used since there are few states
and the extra conversion cost only hurts.

This gives a 2% reduction in state size with just
KERNEL_FEATURE_PATH_TRACING, and 9% reduction when enabling more
features like LIGHT_PASSES + DENOISING or SUBSURFACE + VOLUME.

Benchmarks do not show any significant impact on GPU render time either
way.

Pull Request: https://projects.blender.org/blender/blender/pulls/161876
This commit is contained in:
Brecht Van Lommel 2026-09-20 00:13:07 +02:00 • committed by Brecht Van Lommel
parent 4aeb35f183
commit 33b629d65b
14 changed files with 307 additions and 71 deletions

View file

@ -203,6 +203,12 @@ template<> struct device_type_traits<uint16_t> {
static_assert(sizeof(uint16_t) == num_elements * datatype_size(data_type));
};
template<> struct device_type_traits<packed_half3> {
static const DataType data_type = TYPE_HALF;
static const size_t num_elements = 3;
static_assert(sizeof(packed_half3) == num_elements * datatype_size(data_type));
};
template<> struct device_type_traits<half4> {
static const DataType data_type = TYPE_HALF;
static const size_t num_elements = 4;

View file

@ -29,20 +29,26 @@ static size_t estimate_single_state_size(const uint64_t kernel_features)
#ifdef __INTEGRATOR_GPU_PACKED_STATE__
# define KERNEL_STRUCT_MEMBER(parent_struct, type, name, feature) \
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? sizeof(type) : 0;
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? \
sizeof(gpu_state_storage<type>::gpu_type) : \
0;
# define KERNEL_STRUCT_MEMBER_PACKED(parent_struct, type, name, feature)
# define KERNEL_STRUCT_BEGIN_PACKED(parent_struct, feature) \
KERNEL_STRUCT_BEGIN(parent_struct) \
KERNEL_STRUCT_MEMBER(parent_struct, packed_##parent_struct, packed, feature)
#else
# define KERNEL_STRUCT_MEMBER(parent_struct, type, name, feature) \
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? sizeof(type) : 0;
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? \
sizeof(gpu_state_storage<type>::gpu_type) : \
0;
# define KERNEL_STRUCT_MEMBER_PACKED KERNEL_STRUCT_MEMBER
# define KERNEL_STRUCT_BEGIN_PACKED(parent_struct, feature) KERNEL_STRUCT_BEGIN(parent_struct)
#endif
#define KERNEL_STRUCT_ARRAY_MEMBER(parent_struct, type, name, feature) \
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? sizeof(type) : 0;
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? \
sizeof(gpu_state_storage<type>::gpu_type) : \
0;
#define KERNEL_STRUCT_END(name) \
(void)array_index; \
break; \
@ -148,7 +154,8 @@ void PathTraceWorkGPU::alloc_integrator_soa()
{ \
string name_str = string_printf("%sintegrator_state_" #parent_struct "_" #name, \
shadow ? "shadow_" : ""); \
auto array = make_unique<device_only_memory<type>>(device_, name_str.c_str()); \
auto array = make_unique<device_only_memory<gpu_state_storage<type>::gpu_type>>( \
device_, name_str.c_str()); \
array->alloc_to_device(max_num_paths_); \
memcpy(&integrator_state_gpu_.parent_struct.name, \
&array->device_pointer, \
@ -177,7 +184,8 @@ void PathTraceWorkGPU::alloc_integrator_soa()
{ \
string name_str = string_printf( \
"%sintegrator_state_" #name "_%d", shadow ? "shadow_" : "", array_index); \
auto array = make_unique<device_only_memory<type>>(device_, name_str.c_str()); \
auto array = make_unique<device_only_memory<gpu_state_storage<type>::gpu_type>>( \
device_, name_str.c_str()); \
array->alloc_to_device(max_num_paths_); \
memcpy(&integrator_state_gpu_.parent_struct[array_index].name, \
&array->device_pointer, \

View file

@ -314,6 +314,7 @@ set(SRC_UTIL_HEADERS
../util/types_float3.h
../util/types_float4.h
../util/types_float8.h
../util/types_gpu_compressed.h
../util/types_image.h
../util/types_int2.h
../util/types_int3.h

View file

@ -213,7 +213,9 @@ ccl_device_forceinline void film_write_denoising_features_surface(KernelGlobals
if (!follow_reflections) {
deferred_albedo = transparent_albedo;
}
INTEGRATOR_STATE_WRITE(state, path, denoising_feature_throughput) *= deferred_albedo;
const Spectrum throughput = INTEGRATOR_STATE(state, path, denoising_feature_throughput);
INTEGRATOR_STATE_WRITE(state, path, denoising_feature_throughput) = throughput *
deferred_albedo;
}
else {
INTEGRATOR_STATE_WRITE(state, path, flag) &= ~PATH_RAY_DENOISING_FEATURES;

View file

@ -661,8 +661,8 @@ ccl_device_forceinline bool integrate_surface_terminate(IntegratorState state,
{
const float continuation_probability = (path_flag & PATH_RAY_TERMINATE_ON_NEXT_SURFACE) ?
0.0f :
INTEGRATOR_STATE(
state, path, continuation_probability);
float(INTEGRATOR_STATE(
state, path, continuation_probability));
if (continuation_probability == 0.0f) {
return true;
}

View file

@ -2742,8 +2742,8 @@ volume_integrate_event(KernelGlobals kg,
const uint32_t path_flag = INTEGRATOR_STATE(state, path, flag);
const float continuation_probability = (path_flag & PATH_RAY_TERMINATE_IN_NEXT_VOLUME) ?
0.0f :
INTEGRATOR_STATE(
state, path, continuation_probability);
float(INTEGRATOR_STATE(
state, path, continuation_probability));
if (continuation_probability == 0.0f) {
return VOLUME_PATH_MISSED;
}

View file

@ -41,8 +41,14 @@ KERNEL_STRUCT_MEMBER(shadow_path,
unshadowed_throughput,
KERNEL_FEATURE_AO_ADDITIVE)
/* Ratio of throughput to distinguish diffuse / glossy / transmission render passes. */
KERNEL_STRUCT_MEMBER(shadow_path, PackedSpectrum, pass_diffuse_weight, KERNEL_FEATURE_LIGHT_PASSES)
KERNEL_STRUCT_MEMBER(shadow_path, PackedSpectrum, pass_glossy_weight, KERNEL_FEATURE_LIGHT_PASSES)
KERNEL_STRUCT_MEMBER(shadow_path,
SpectrumCompressedOnGPU,
pass_diffuse_weight,
KERNEL_FEATURE_LIGHT_PASSES)
KERNEL_STRUCT_MEMBER(shadow_path,
SpectrumCompressedOnGPU,
pass_glossy_weight,
KERNEL_FEATURE_LIGHT_PASSES)
/* Packed number of intersections found by ray-tracing, and on GPU also the resume hit index
* and skip_volume flag for cache miss handling.
* Note that this is the total number of intersections for the shadow ray.
@ -52,7 +58,10 @@ KERNEL_STRUCT_MEMBER(shadow_path, uint16_t, packed_num_hits, KERNEL_FEATURE_PATH
/* Light group. */
KERNEL_STRUCT_MEMBER(shadow_path, uint8_t, lightgroup, KERNEL_FEATURE_PATH_TRACING)
/* Path guiding. */
KERNEL_STRUCT_MEMBER(shadow_path, PackedSpectrum, unlit_throughput, KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_MEMBER(shadow_path,
SpectrumCompressedOnGPU,
unlit_throughput,
KERNEL_FEATURE_PATH_GUIDING)
#if defined(__PATH_GUIDING__)
KERNEL_STRUCT_MEMBER(shadow_path,
openpgl::cpp::PathSegment *,
@ -80,11 +89,11 @@ KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, packed_float3, P, KERNEL_FEATURE_PATH_TR
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, packed_float3, D, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, tmin, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, tmax, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, time, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, dP, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, dD, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, int, self_light_object, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, int, self_light_prim, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, FloatCompressedOnGPU, time, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, FloatCompressedOnGPU, dD, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_END(shadow_ray)
/*********************** Shadow Intersection result **************************/

View file

@ -57,17 +57,29 @@ KERNEL_STRUCT_MEMBER(path, packed_float3, mis_origin_n, KERNEL_FEATURE_PATH_TRAC
/* Filter glossy. */
KERNEL_STRUCT_MEMBER(path, float, min_ray_pdf, KERNEL_FEATURE_PATH_TRACING)
/* Continuation probability for path termination. */
KERNEL_STRUCT_MEMBER(path, float, continuation_probability, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER(path,
FloatCompressedOnGPU,
continuation_probability,
KERNEL_FEATURE_PATH_TRACING)
/* Throughput. */
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, throughput, KERNEL_FEATURE_PATH_TRACING)
/* Factor to multiple with throughput to get remove any guiding PDFS.
* Such throughput without guiding PDFS is used for Russian roulette termination. */
KERNEL_STRUCT_MEMBER(path, float, unguided_throughput, KERNEL_FEATURE_PATH_GUIDING)
/* Ratio of throughput to distinguish diffuse / glossy / transmission render passes. */
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, pass_diffuse_weight, KERNEL_FEATURE_LIGHT_PASSES)
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, pass_glossy_weight, KERNEL_FEATURE_LIGHT_PASSES)
KERNEL_STRUCT_MEMBER(path,
SpectrumCompressedOnGPU,
pass_diffuse_weight,
KERNEL_FEATURE_LIGHT_PASSES)
KERNEL_STRUCT_MEMBER(path,
SpectrumCompressedOnGPU,
pass_glossy_weight,
KERNEL_FEATURE_LIGHT_PASSES)
/* Denoising. */
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, denoising_feature_throughput, KERNEL_FEATURE_DENOISING)
KERNEL_STRUCT_MEMBER(path,
SpectrumCompressedOnGPU,
denoising_feature_throughput,
KERNEL_FEATURE_DENOISING)
/* Shader sorting. */
/* TODO: compress as uint16? or leave out entirely and recompute key in sorting code? */
KERNEL_STRUCT_MEMBER(path, uint32_t, shader_sort_key, KERNEL_FEATURE_PATH_TRACING)
@ -77,12 +89,12 @@ KERNEL_STRUCT_END(path)
KERNEL_STRUCT_BEGIN_PACKED(ray, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, packed_float3, P, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, float, dP, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, packed_float3, D, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, float, dD, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, float, tmin, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, float, tmax, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, float, time, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, float, dP, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, FloatCompressedOnGPU, dD, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER_PACKED(ray, FloatCompressedOnGPU, time, KERNEL_FEATURE_PATH_TRACING)
KERNEL_STRUCT_MEMBER(ray, float, previous_dt, KERNEL_FEATURE_LIGHT_TREE)
KERNEL_STRUCT_END(ray)
@ -101,10 +113,13 @@ KERNEL_STRUCT_END(isect)
/*************** Subsurface closure state for subsurface kernel ***************/
KERNEL_STRUCT_BEGIN_PACKED(subsurface, KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_MEMBER_PACKED(subsurface, PackedSpectrum, albedo, KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_MEMBER_PACKED(subsurface, PackedSpectrum, radius, KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_MEMBER_PACKED(subsurface, float, anisotropy, KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_MEMBER_PACKED(subsurface, packed_float3, N, KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_MEMBER_PACKED(subsurface, SpectrumCompressedOnGPU, albedo, KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_MEMBER_PACKED(subsurface,
FloatCompressedOnGPU,
anisotropy,
KERNEL_FEATURE_SUBSURFACE)
KERNEL_STRUCT_END(subsurface)
/********************************** Volume Stack ******************************/
@ -132,24 +147,40 @@ KERNEL_STRUCT_MEMBER(guiding, uint64_t, path_segment, KERNEL_FEATURE_PATH_GUIDIN
KERNEL_STRUCT_MEMBER(guiding, bool, use_surface_guiding, KERNEL_FEATURE_PATH_GUIDING)
/* Random number used for additional guiding decisions (e.g., cache query, selection to use guiding
* or BSDF sampling) */
KERNEL_STRUCT_MEMBER(guiding, float, sample_surface_guiding_rand, KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_MEMBER(guiding,
FloatCompressedOnGPU,
sample_surface_guiding_rand,
KERNEL_FEATURE_PATH_GUIDING)
/* The probability to use surface guiding (i.e., diffuse sampling prob * guiding prob). */
KERNEL_STRUCT_MEMBER(guiding, float, surface_guiding_sampling_prob, KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_MEMBER(guiding,
FloatCompressedOnGPU,
surface_guiding_sampling_prob,
KERNEL_FEATURE_PATH_GUIDING)
/* Probability of sampling a BSSRDF closure instead of a BSDF closure. */
KERNEL_STRUCT_MEMBER(guiding, float, bssrdf_sampling_prob, KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_MEMBER(guiding,
FloatCompressedOnGPU,
bssrdf_sampling_prob,
KERNEL_FEATURE_PATH_GUIDING)
/* If volume guiding is enabled */
KERNEL_STRUCT_MEMBER(guiding, bool, use_volume_guiding, KERNEL_FEATURE_PATH_GUIDING)
/* Random number used for additional guiding decisions (e.g., cache query, selection to use guiding
* or BSDF sampling) */
KERNEL_STRUCT_MEMBER(guiding, float, sample_volume_guiding_rand, KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_MEMBER(guiding,
FloatCompressedOnGPU,
sample_volume_guiding_rand,
KERNEL_FEATURE_PATH_GUIDING)
/* The probability to use surface guiding (i.e., diffuse sampling prob * guiding prob). */
KERNEL_STRUCT_MEMBER(guiding, float, volume_guiding_sampling_prob, KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_MEMBER(guiding,
FloatCompressedOnGPU,
volume_guiding_sampling_prob,
KERNEL_FEATURE_PATH_GUIDING)
KERNEL_STRUCT_END(guiding)
/******************************* Shadow linking *******************************/
KERNEL_STRUCT_BEGIN(shadow_link)
KERNEL_STRUCT_MEMBER(shadow_link, float, dedicated_light_weight, KERNEL_FEATURE_SHADOW_LINKING)
/* Number of dedicated light hits along the shadow ray. */
KERNEL_STRUCT_MEMBER(shadow_link, uint16_t, dedicated_light_weight, KERNEL_FEATURE_SHADOW_LINKING)
/* Copy of primitive and object from the last main path intersection. */
KERNEL_STRUCT_MEMBER(shadow_link, int, last_isect_prim, KERNEL_FEATURE_SHADOW_LINKING)
KERNEL_STRUCT_MEMBER(shadow_link, int, last_isect_object, KERNEL_FEATURE_SHADOW_LINKING)

View file

@ -26,29 +26,15 @@ ccl_device_forceinline void integrator_state_write_ray(IntegratorState state,
const ccl_private Ray *ccl_restrict ray)
{
#if defined(__INTEGRATOR_GPU_PACKED_STATE__) && defined(__KERNEL_GPU__)
static_assert(sizeof(ray->P) == sizeof(float4), "Bad assumption about float3 padding");
/* dP and dP are packed based on the assumption that float3 is padded to 16 bytes.
* This assumption hold trues on Metal, but not CUDA.
*/
((ccl_private float4 &)ray->P).w = ray->dP;
((ccl_private float4 &)ray->D).w = ray->dD;
INTEGRATOR_STATE_WRITE(state, ray, packed) = (ccl_private packed_ray &)*ray;
/* Ensure that we can correctly cast between Ray and the generated packed_ray struct. */
static_assert(offsetof(packed_ray, P) == offsetof(Ray, P),
"Generated packed_ray struct is misaligned with Ray struct");
static_assert(offsetof(packed_ray, D) == offsetof(Ray, D),
"Generated packed_ray struct is misaligned with Ray struct");
static_assert(offsetof(packed_ray, tmin) == offsetof(Ray, tmin),
"Generated packed_ray struct is misaligned with Ray struct");
static_assert(offsetof(packed_ray, tmax) == offsetof(Ray, tmax),
"Generated packed_ray struct is misaligned with Ray struct");
static_assert(offsetof(packed_ray, time) == offsetof(Ray, time),
"Generated packed_ray struct is misaligned with Ray struct");
static_assert(offsetof(packed_ray, dP) == 12 + offsetof(Ray, P),
"Generated packed_ray struct is misaligned with Ray struct");
static_assert(offsetof(packed_ray, dD) == 12 + offsetof(Ray, D),
"Generated packed_ray struct is misaligned with Ray struct");
packed_ray packed;
packed.P = ray->P;
packed.dP = ray->dP;
packed.D = ray->D;
packed.dD = ray->dD;
packed.tmin = ray->tmin;
packed.tmax = ray->tmax;
packed.time = ray->time;
INTEGRATOR_STATE_WRITE(state, ray, packed) = packed;
#else
INTEGRATOR_STATE_WRITE(state, ray, P) = ray->P;
INTEGRATOR_STATE_WRITE(state, ray, D) = ray->D;
@ -64,9 +50,14 @@ ccl_device_forceinline void integrator_state_read_ray(ConstIntegratorState state
ccl_private Ray *ccl_restrict ray)
{
#if defined(__INTEGRATOR_GPU_PACKED_STATE__) && defined(__KERNEL_GPU__)
*((ccl_private packed_ray *)ray) = INTEGRATOR_STATE(state, ray, packed);
ray->dP = ((ccl_private float4 &)ray->P).w;
ray->dD = ((ccl_private float4 &)ray->D).w;
const packed_ray packed = INTEGRATOR_STATE(state, ray, packed);
ray->P = packed.P;
ray->dP = packed.dP;
ray->D = packed.D;
ray->dD = packed.dD;
ray->tmin = packed.tmin;
ray->tmax = packed.tmax;
ray->time = packed.time;
#else
ray->P = INTEGRATOR_STATE(state, ray, P);
ray->D = INTEGRATOR_STATE(state, ray, D);
@ -358,7 +349,7 @@ ccl_device_forceinline void integrator_state_write_mnee(IntegratorState state,
const Spectrum mnee_throughput,
const float3 mnee_wo)
{
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 2);
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 3);
# ifdef __KERNEL_GPU__
INTEGRATOR_STATE_WRITE(state, path, mnee_shadow_state) = (int)shadow_state;
@ -376,7 +367,7 @@ ccl_device_forceinline void integrator_state_write_mnee(IntegratorState state,
INTEGRATOR_STATE_ARRAY_WRITE(shadow_state, shadow_isect, 1, v) = mnee_wo.z;
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, tmin) = ls->t;
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, tmax) = ls->pdf;
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, time) = ls->eval_fac;
INTEGRATOR_STATE_ARRAY_WRITE(shadow_state, shadow_isect, 2, t) = ls->eval_fac;
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, self_light_object) = ls->object;
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, self_light_prim) = ls->prim;
INTEGRATOR_STATE_ARRAY_WRITE(shadow_state, shadow_isect, 0, object) = ls->shader;
@ -401,7 +392,7 @@ ccl_device_forceinline void integrator_state_read_mnee(ConstIntegratorState stat
ccl_private LightSample *ls,
ccl_private int *mnee_vertex_count)
{
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 2);
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 3);
ConstIntegratorShadowState shadow_state = integrator_state_get_mnee_shadow_state(state);
@ -413,7 +404,7 @@ ccl_device_forceinline void integrator_state_read_mnee(ConstIntegratorState stat
INTEGRATOR_STATE_ARRAY(shadow_state, shadow_isect, 1, v));
ls->t = INTEGRATOR_STATE(shadow_state, shadow_ray, tmin);
ls->pdf = INTEGRATOR_STATE(shadow_state, shadow_ray, tmax);
ls->eval_fac = INTEGRATOR_STATE(shadow_state, shadow_ray, time);
ls->eval_fac = INTEGRATOR_STATE_ARRAY(shadow_state, shadow_isect, 2, t);
ls->object = INTEGRATOR_STATE(shadow_state, shadow_ray, self_light_object);
ls->prim = INTEGRATOR_STATE(shadow_state, shadow_ray, self_light_prim);
ls->shader = INTEGRATOR_STATE_ARRAY(shadow_state, shadow_isect, 0, object);

View file

@ -67,7 +67,7 @@ TEST(TEST_CATEGORY_NAME, float_to_half)
}
}
TEST(TEST_CATEGORY_NAME, float3_to_half3)
TEST(TEST_CATEGORY_NAME, float3_to_packed_half3)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
@ -80,8 +80,8 @@ TEST(TEST_CATEGORY_NAME, float3_to_half3)
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
const half3 h = float3_to_half3(in);
const float3 out = half3_to_float3(h);
const packed_half3 h = float3_to_packed_half3(in);
const float3 out = packed_half3_to_float3(h);
EXPECT_EQ(out.x, in.x);
EXPECT_EQ(out.y, in.y);
@ -174,7 +174,7 @@ TEST(TEST_CATEGORY_NAME, half_to_float_flush_to_zero)
}
}
TEST(TEST_CATEGORY_NAME, fallback_float3_to_half3)
TEST(TEST_CATEGORY_NAME, fallback_float3_to_packed_half3)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
@ -187,8 +187,8 @@ TEST(TEST_CATEGORY_NAME, fallback_float3_to_half3)
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
const half3 h = fallback_float3_to_half3(in);
const float3 out = fallback_half3_to_float3(h);
const packed_half3 h = fallback_float3_to_packed_half3(in);
const float3 out = fallback_packed_half3_to_float3(h);
EXPECT_EQ(out.x, in.x);
EXPECT_EQ(out.y, in.y);

View file

@ -128,6 +128,7 @@ set(SRC_HEADERS
types_float3.h
types_float4.h
types_float8.h
types_gpu_compressed.h
types_image.h
types_int2.h
types_int3.h

View file

@ -46,7 +46,7 @@ class half {
#endif
#if !defined(__KERNEL_METAL__)
struct half3 {
struct packed_half3 {
half x, y, z;
};
@ -55,6 +55,10 @@ struct half4 {
};
#endif
static_assert(sizeof(half) == 2);
static_assert(sizeof(packed_half3) == 6);
static_assert(sizeof(half4) == 8);
#if !defined(__KERNEL_GPU__)
/* Optimized fallback implementations with fast path for normal and denormal numbers, assuming
* no Infs or NaNs. Based on public domain functions from.
@ -110,7 +114,7 @@ ccl_device_inline float4 fallback_half4_to_float4(const half4 h)
return cast(f | s);
}
ccl_device_inline float3 fallback_half3_to_float3(const half3 h)
ccl_device_inline float3 fallback_packed_half3_to_float3(const packed_half3 h)
{
return make_float3(fallback_half4_to_float4({h.x, h.y, h.z, 0}));
}
@ -184,7 +188,7 @@ ccl_device_inline half4 fallback_float4_to_half4(const float4 f)
half(uint16_t(res.x)), half(uint16_t(res.y)), half(uint16_t(res.z)), half(uint16_t(res.w))};
}
ccl_device_inline half3 fallback_float3_to_half3(const float3 f)
ccl_device_inline packed_half3 fallback_float3_to_packed_half3(const float3 f)
{
const half4 h = fallback_float4_to_half4(make_float4(f));
return {h.x, h.y, h.z};
@ -272,7 +276,7 @@ ccl_device_inline float4 half4_to_float4(const half4 h)
#endif
}
ccl_device_inline half3 float3_to_half3(const float3 f)
ccl_device_inline packed_half3 float3_to_packed_half3(const float3 f)
{
#if defined(__KERNEL_GPU__)
return {float_to_half(f.x), float_to_half(f.y), float_to_half(f.z)};
@ -291,7 +295,7 @@ ccl_device_inline half3 float3_to_half3(const float3 f)
#endif
}
ccl_device_inline float3 half3_to_float3(const half3 h)
ccl_device_inline float3 packed_half3_to_float3(const packed_half3 h)
{
#if defined(__KERNEL_GPU__)
return make_float3(half_to_float(h.x), half_to_float(h.y), half_to_float(h.z));

View file

@ -38,3 +38,5 @@
#include "util/types_float3x3.h" // IWYU pragma: export
#include "util/types_spherical_harmonics.h" // IWYU pragma: export
#include "util/types_gpu_compressed.h" // IWYU pragma: export

View file

@ -0,0 +1,181 @@
/* SPDX-FileCopyrightText: 2011-2026 Blender Foundation
*
* SPDX-License-Identifier: Apache-2.0 */
#pragma once
#include "util/half.h"
#include "util/types_spectrum.h"
CCL_NAMESPACE_BEGIN
/* Compact storage of data types on GPU, stored in full precision on CPU.
* Under the assumption that reducing state size is important on GPU and
* math is relatively cheap. */
/* A float stored in half precision. */
struct FloatCompressedOnGPU {
using value_type = float;
using compressed_type = half;
#ifdef __KERNEL_GPU__
half v;
#else
float v;
#endif
FloatCompressedOnGPU() = default;
ccl_device_inline_method FloatCompressedOnGPU(const float a)
{
#ifdef __KERNEL_GPU__
v = float_to_half(a);
#else
v = a;
#endif
}
ccl_device_inline_method operator float() const ccl_global
{
#ifdef __KERNEL_GPU__
return half_to_float(v);
#else
return v;
#endif
}
ccl_device_inline_method ccl_global FloatCompressedOnGPU &operator=(const float a) ccl_global
{
#ifdef __KERNEL_GPU__
v = float_to_half(a);
#else
v = a;
#endif
return *this;
}
ccl_device_inline_method ccl_global FloatCompressedOnGPU &operator=(FloatCompressedOnGPU a)
ccl_global
{
v = a.v;
return *this;
}
#ifdef __KERNEL_METAL__
ccl_device_inline_method operator float() const ccl_private
{
# ifdef __KERNEL_GPU__
return half_to_float(v);
# else
return v;
# endif
}
ccl_device_inline_method ccl_private FloatCompressedOnGPU &operator=(const float a) ccl_private
{
# ifdef __KERNEL_GPU__
v = float_to_half(a);
# else
v = a;
# endif
return *this;
}
ccl_device_inline_method ccl_private FloatCompressedOnGPU &operator=(FloatCompressedOnGPU a)
ccl_private
{
v = a.v;
return *this;
}
#endif
};
/* Spectrum stored in half precision. */
struct SpectrumCompressedOnGPU {
using value_type = Spectrum;
using compressed_type = packed_half3;
#ifdef __KERNEL_GPU__
packed_half3 v;
#else
Spectrum v;
#endif
SpectrumCompressedOnGPU() = default;
ccl_device_inline_method SpectrumCompressedOnGPU(const Spectrum a)
{
#ifdef __KERNEL_GPU__
v = float3_to_packed_half3(a);
#else
v = a;
#endif
}
ccl_device_inline_method operator Spectrum() const ccl_global
{
#ifdef __KERNEL_GPU__
return packed_half3_to_float3(v);
#else
return v;
#endif
}
ccl_device_inline_method ccl_global SpectrumCompressedOnGPU &operator=(const Spectrum a)
ccl_global
{
#ifdef __KERNEL_GPU__
v = float3_to_packed_half3(a);
#else
v = a;
#endif
return *this;
}
ccl_device_inline_method ccl_global SpectrumCompressedOnGPU &operator=(SpectrumCompressedOnGPU a)
ccl_global
{
v = a.v;
return *this;
}
#ifdef __KERNEL_METAL__
ccl_device_inline_method operator Spectrum() const ccl_private
{
# ifdef __KERNEL_GPU__
return packed_half3_to_float3(v);
# else
return v;
# endif
}
ccl_device_inline_method ccl_private SpectrumCompressedOnGPU &operator=(const Spectrum a)
ccl_private
{
# ifdef __KERNEL_GPU__
v = float3_to_packed_half3(a);
# else
v = a;
# endif
return *this;
}
ccl_device_inline_method ccl_private SpectrumCompressedOnGPU &operator=(
SpectrumCompressedOnGPU a) ccl_private
{
v = a.v;
return *this;
}
#endif
};
/* Templates for determining GPU storage type and size from the host. */
template<typename...> using gpu_void_t = void;
template<typename T, typename = void> struct gpu_state_storage {
using gpu_type = T;
};
template<typename T> struct gpu_state_storage<T, gpu_void_t<typename T::compressed_type>> {
using gpu_type = typename T::compressed_type;
};
CCL_NAMESPACE_END