mirror of
https://github.com/blender/blender
synced 2026-09-29 04:37:17 +03:00
Cycles: Reduce GPU integrator state with smaller data types
Some values don't need full float precision, and with millions of states reducing memory usage is important, while the math to pack/unpack these is quite cheap. On CPU full precision is used since there are few states and the extra conversion cost only hurts. This gives a 2% reduction in state size with just KERNEL_FEATURE_PATH_TRACING, and 9% reduction when enabling more features like LIGHT_PASSES + DENOISING or SUBSURFACE + VOLUME. Benchmarks do not show any significant impact on GPU render time either way. Pull Request: https://projects.blender.org/blender/blender/pulls/161876
This commit is contained in:
parent
4aeb35f183
commit
33b629d65b
14 changed files with 307 additions and 71 deletions
|
|
@ -203,6 +203,12 @@ template<> struct device_type_traits<uint16_t> {
|
|||
static_assert(sizeof(uint16_t) == num_elements * datatype_size(data_type));
|
||||
};
|
||||
|
||||
template<> struct device_type_traits<packed_half3> {
|
||||
static const DataType data_type = TYPE_HALF;
|
||||
static const size_t num_elements = 3;
|
||||
static_assert(sizeof(packed_half3) == num_elements * datatype_size(data_type));
|
||||
};
|
||||
|
||||
template<> struct device_type_traits<half4> {
|
||||
static const DataType data_type = TYPE_HALF;
|
||||
static const size_t num_elements = 4;
|
||||
|
|
|
|||
|
|
@ -29,20 +29,26 @@ static size_t estimate_single_state_size(const uint64_t kernel_features)
|
|||
|
||||
#ifdef __INTEGRATOR_GPU_PACKED_STATE__
|
||||
# define KERNEL_STRUCT_MEMBER(parent_struct, type, name, feature) \
|
||||
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? sizeof(type) : 0;
|
||||
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? \
|
||||
sizeof(gpu_state_storage<type>::gpu_type) : \
|
||||
0;
|
||||
# define KERNEL_STRUCT_MEMBER_PACKED(parent_struct, type, name, feature)
|
||||
# define KERNEL_STRUCT_BEGIN_PACKED(parent_struct, feature) \
|
||||
KERNEL_STRUCT_BEGIN(parent_struct) \
|
||||
KERNEL_STRUCT_MEMBER(parent_struct, packed_##parent_struct, packed, feature)
|
||||
#else
|
||||
# define KERNEL_STRUCT_MEMBER(parent_struct, type, name, feature) \
|
||||
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? sizeof(type) : 0;
|
||||
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? \
|
||||
sizeof(gpu_state_storage<type>::gpu_type) : \
|
||||
0;
|
||||
# define KERNEL_STRUCT_MEMBER_PACKED KERNEL_STRUCT_MEMBER
|
||||
# define KERNEL_STRUCT_BEGIN_PACKED(parent_struct, feature) KERNEL_STRUCT_BEGIN(parent_struct)
|
||||
#endif
|
||||
|
||||
#define KERNEL_STRUCT_ARRAY_MEMBER(parent_struct, type, name, feature) \
|
||||
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? sizeof(type) : 0;
|
||||
state_size += (KernelFeatureRequest(feature).test(kernel_features)) ? \
|
||||
sizeof(gpu_state_storage<type>::gpu_type) : \
|
||||
0;
|
||||
#define KERNEL_STRUCT_END(name) \
|
||||
(void)array_index; \
|
||||
break; \
|
||||
|
|
@ -148,7 +154,8 @@ void PathTraceWorkGPU::alloc_integrator_soa()
|
|||
{ \
|
||||
string name_str = string_printf("%sintegrator_state_" #parent_struct "_" #name, \
|
||||
shadow ? "shadow_" : ""); \
|
||||
auto array = make_unique<device_only_memory<type>>(device_, name_str.c_str()); \
|
||||
auto array = make_unique<device_only_memory<gpu_state_storage<type>::gpu_type>>( \
|
||||
device_, name_str.c_str()); \
|
||||
array->alloc_to_device(max_num_paths_); \
|
||||
memcpy(&integrator_state_gpu_.parent_struct.name, \
|
||||
&array->device_pointer, \
|
||||
|
|
@ -177,7 +184,8 @@ void PathTraceWorkGPU::alloc_integrator_soa()
|
|||
{ \
|
||||
string name_str = string_printf( \
|
||||
"%sintegrator_state_" #name "_%d", shadow ? "shadow_" : "", array_index); \
|
||||
auto array = make_unique<device_only_memory<type>>(device_, name_str.c_str()); \
|
||||
auto array = make_unique<device_only_memory<gpu_state_storage<type>::gpu_type>>( \
|
||||
device_, name_str.c_str()); \
|
||||
array->alloc_to_device(max_num_paths_); \
|
||||
memcpy(&integrator_state_gpu_.parent_struct[array_index].name, \
|
||||
&array->device_pointer, \
|
||||
|
|
|
|||
|
|
@ -314,6 +314,7 @@ set(SRC_UTIL_HEADERS
|
|||
../util/types_float3.h
|
||||
../util/types_float4.h
|
||||
../util/types_float8.h
|
||||
../util/types_gpu_compressed.h
|
||||
../util/types_image.h
|
||||
../util/types_int2.h
|
||||
../util/types_int3.h
|
||||
|
|
|
|||
|
|
@ -213,7 +213,9 @@ ccl_device_forceinline void film_write_denoising_features_surface(KernelGlobals
|
|||
if (!follow_reflections) {
|
||||
deferred_albedo = transparent_albedo;
|
||||
}
|
||||
INTEGRATOR_STATE_WRITE(state, path, denoising_feature_throughput) *= deferred_albedo;
|
||||
const Spectrum throughput = INTEGRATOR_STATE(state, path, denoising_feature_throughput);
|
||||
INTEGRATOR_STATE_WRITE(state, path, denoising_feature_throughput) = throughput *
|
||||
deferred_albedo;
|
||||
}
|
||||
else {
|
||||
INTEGRATOR_STATE_WRITE(state, path, flag) &= ~PATH_RAY_DENOISING_FEATURES;
|
||||
|
|
|
|||
|
|
@ -661,8 +661,8 @@ ccl_device_forceinline bool integrate_surface_terminate(IntegratorState state,
|
|||
{
|
||||
const float continuation_probability = (path_flag & PATH_RAY_TERMINATE_ON_NEXT_SURFACE) ?
|
||||
0.0f :
|
||||
INTEGRATOR_STATE(
|
||||
state, path, continuation_probability);
|
||||
float(INTEGRATOR_STATE(
|
||||
state, path, continuation_probability));
|
||||
if (continuation_probability == 0.0f) {
|
||||
return true;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2742,8 +2742,8 @@ volume_integrate_event(KernelGlobals kg,
|
|||
const uint32_t path_flag = INTEGRATOR_STATE(state, path, flag);
|
||||
const float continuation_probability = (path_flag & PATH_RAY_TERMINATE_IN_NEXT_VOLUME) ?
|
||||
0.0f :
|
||||
INTEGRATOR_STATE(
|
||||
state, path, continuation_probability);
|
||||
float(INTEGRATOR_STATE(
|
||||
state, path, continuation_probability));
|
||||
if (continuation_probability == 0.0f) {
|
||||
return VOLUME_PATH_MISSED;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -41,8 +41,14 @@ KERNEL_STRUCT_MEMBER(shadow_path,
|
|||
unshadowed_throughput,
|
||||
KERNEL_FEATURE_AO_ADDITIVE)
|
||||
/* Ratio of throughput to distinguish diffuse / glossy / transmission render passes. */
|
||||
KERNEL_STRUCT_MEMBER(shadow_path, PackedSpectrum, pass_diffuse_weight, KERNEL_FEATURE_LIGHT_PASSES)
|
||||
KERNEL_STRUCT_MEMBER(shadow_path, PackedSpectrum, pass_glossy_weight, KERNEL_FEATURE_LIGHT_PASSES)
|
||||
KERNEL_STRUCT_MEMBER(shadow_path,
|
||||
SpectrumCompressedOnGPU,
|
||||
pass_diffuse_weight,
|
||||
KERNEL_FEATURE_LIGHT_PASSES)
|
||||
KERNEL_STRUCT_MEMBER(shadow_path,
|
||||
SpectrumCompressedOnGPU,
|
||||
pass_glossy_weight,
|
||||
KERNEL_FEATURE_LIGHT_PASSES)
|
||||
/* Packed number of intersections found by ray-tracing, and on GPU also the resume hit index
|
||||
* and skip_volume flag for cache miss handling.
|
||||
* Note that this is the total number of intersections for the shadow ray.
|
||||
|
|
@ -52,7 +58,10 @@ KERNEL_STRUCT_MEMBER(shadow_path, uint16_t, packed_num_hits, KERNEL_FEATURE_PATH
|
|||
/* Light group. */
|
||||
KERNEL_STRUCT_MEMBER(shadow_path, uint8_t, lightgroup, KERNEL_FEATURE_PATH_TRACING)
|
||||
/* Path guiding. */
|
||||
KERNEL_STRUCT_MEMBER(shadow_path, PackedSpectrum, unlit_throughput, KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_MEMBER(shadow_path,
|
||||
SpectrumCompressedOnGPU,
|
||||
unlit_throughput,
|
||||
KERNEL_FEATURE_PATH_GUIDING)
|
||||
#if defined(__PATH_GUIDING__)
|
||||
KERNEL_STRUCT_MEMBER(shadow_path,
|
||||
openpgl::cpp::PathSegment *,
|
||||
|
|
@ -80,11 +89,11 @@ KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, packed_float3, P, KERNEL_FEATURE_PATH_TR
|
|||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, packed_float3, D, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, tmin, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, tmax, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, time, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, dP, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, float, dD, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, int, self_light_object, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, int, self_light_prim, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, FloatCompressedOnGPU, time, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(shadow_ray, FloatCompressedOnGPU, dD, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_END(shadow_ray)
|
||||
|
||||
/*********************** Shadow Intersection result **************************/
|
||||
|
|
|
|||
|
|
@ -57,17 +57,29 @@ KERNEL_STRUCT_MEMBER(path, packed_float3, mis_origin_n, KERNEL_FEATURE_PATH_TRAC
|
|||
/* Filter glossy. */
|
||||
KERNEL_STRUCT_MEMBER(path, float, min_ray_pdf, KERNEL_FEATURE_PATH_TRACING)
|
||||
/* Continuation probability for path termination. */
|
||||
KERNEL_STRUCT_MEMBER(path, float, continuation_probability, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER(path,
|
||||
FloatCompressedOnGPU,
|
||||
continuation_probability,
|
||||
KERNEL_FEATURE_PATH_TRACING)
|
||||
/* Throughput. */
|
||||
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, throughput, KERNEL_FEATURE_PATH_TRACING)
|
||||
/* Factor to multiple with throughput to get remove any guiding PDFS.
|
||||
* Such throughput without guiding PDFS is used for Russian roulette termination. */
|
||||
KERNEL_STRUCT_MEMBER(path, float, unguided_throughput, KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* Ratio of throughput to distinguish diffuse / glossy / transmission render passes. */
|
||||
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, pass_diffuse_weight, KERNEL_FEATURE_LIGHT_PASSES)
|
||||
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, pass_glossy_weight, KERNEL_FEATURE_LIGHT_PASSES)
|
||||
KERNEL_STRUCT_MEMBER(path,
|
||||
SpectrumCompressedOnGPU,
|
||||
pass_diffuse_weight,
|
||||
KERNEL_FEATURE_LIGHT_PASSES)
|
||||
KERNEL_STRUCT_MEMBER(path,
|
||||
SpectrumCompressedOnGPU,
|
||||
pass_glossy_weight,
|
||||
KERNEL_FEATURE_LIGHT_PASSES)
|
||||
/* Denoising. */
|
||||
KERNEL_STRUCT_MEMBER(path, PackedSpectrum, denoising_feature_throughput, KERNEL_FEATURE_DENOISING)
|
||||
KERNEL_STRUCT_MEMBER(path,
|
||||
SpectrumCompressedOnGPU,
|
||||
denoising_feature_throughput,
|
||||
KERNEL_FEATURE_DENOISING)
|
||||
/* Shader sorting. */
|
||||
/* TODO: compress as uint16? or leave out entirely and recompute key in sorting code? */
|
||||
KERNEL_STRUCT_MEMBER(path, uint32_t, shader_sort_key, KERNEL_FEATURE_PATH_TRACING)
|
||||
|
|
@ -77,12 +89,12 @@ KERNEL_STRUCT_END(path)
|
|||
|
||||
KERNEL_STRUCT_BEGIN_PACKED(ray, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, packed_float3, P, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, float, dP, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, packed_float3, D, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, float, dD, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, float, tmin, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, float, tmax, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, float, time, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, float, dP, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, FloatCompressedOnGPU, dD, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(ray, FloatCompressedOnGPU, time, KERNEL_FEATURE_PATH_TRACING)
|
||||
KERNEL_STRUCT_MEMBER(ray, float, previous_dt, KERNEL_FEATURE_LIGHT_TREE)
|
||||
KERNEL_STRUCT_END(ray)
|
||||
|
||||
|
|
@ -101,10 +113,13 @@ KERNEL_STRUCT_END(isect)
|
|||
/*************** Subsurface closure state for subsurface kernel ***************/
|
||||
|
||||
KERNEL_STRUCT_BEGIN_PACKED(subsurface, KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(subsurface, PackedSpectrum, albedo, KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(subsurface, PackedSpectrum, radius, KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(subsurface, float, anisotropy, KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(subsurface, packed_float3, N, KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(subsurface, SpectrumCompressedOnGPU, albedo, KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_MEMBER_PACKED(subsurface,
|
||||
FloatCompressedOnGPU,
|
||||
anisotropy,
|
||||
KERNEL_FEATURE_SUBSURFACE)
|
||||
KERNEL_STRUCT_END(subsurface)
|
||||
|
||||
/********************************** Volume Stack ******************************/
|
||||
|
|
@ -132,24 +147,40 @@ KERNEL_STRUCT_MEMBER(guiding, uint64_t, path_segment, KERNEL_FEATURE_PATH_GUIDIN
|
|||
KERNEL_STRUCT_MEMBER(guiding, bool, use_surface_guiding, KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* Random number used for additional guiding decisions (e.g., cache query, selection to use guiding
|
||||
* or BSDF sampling) */
|
||||
KERNEL_STRUCT_MEMBER(guiding, float, sample_surface_guiding_rand, KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_MEMBER(guiding,
|
||||
FloatCompressedOnGPU,
|
||||
sample_surface_guiding_rand,
|
||||
KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* The probability to use surface guiding (i.e., diffuse sampling prob * guiding prob). */
|
||||
KERNEL_STRUCT_MEMBER(guiding, float, surface_guiding_sampling_prob, KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_MEMBER(guiding,
|
||||
FloatCompressedOnGPU,
|
||||
surface_guiding_sampling_prob,
|
||||
KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* Probability of sampling a BSSRDF closure instead of a BSDF closure. */
|
||||
KERNEL_STRUCT_MEMBER(guiding, float, bssrdf_sampling_prob, KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_MEMBER(guiding,
|
||||
FloatCompressedOnGPU,
|
||||
bssrdf_sampling_prob,
|
||||
KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* If volume guiding is enabled */
|
||||
KERNEL_STRUCT_MEMBER(guiding, bool, use_volume_guiding, KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* Random number used for additional guiding decisions (e.g., cache query, selection to use guiding
|
||||
* or BSDF sampling) */
|
||||
KERNEL_STRUCT_MEMBER(guiding, float, sample_volume_guiding_rand, KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_MEMBER(guiding,
|
||||
FloatCompressedOnGPU,
|
||||
sample_volume_guiding_rand,
|
||||
KERNEL_FEATURE_PATH_GUIDING)
|
||||
/* The probability to use surface guiding (i.e., diffuse sampling prob * guiding prob). */
|
||||
KERNEL_STRUCT_MEMBER(guiding, float, volume_guiding_sampling_prob, KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_MEMBER(guiding,
|
||||
FloatCompressedOnGPU,
|
||||
volume_guiding_sampling_prob,
|
||||
KERNEL_FEATURE_PATH_GUIDING)
|
||||
KERNEL_STRUCT_END(guiding)
|
||||
|
||||
/******************************* Shadow linking *******************************/
|
||||
|
||||
KERNEL_STRUCT_BEGIN(shadow_link)
|
||||
KERNEL_STRUCT_MEMBER(shadow_link, float, dedicated_light_weight, KERNEL_FEATURE_SHADOW_LINKING)
|
||||
/* Number of dedicated light hits along the shadow ray. */
|
||||
KERNEL_STRUCT_MEMBER(shadow_link, uint16_t, dedicated_light_weight, KERNEL_FEATURE_SHADOW_LINKING)
|
||||
/* Copy of primitive and object from the last main path intersection. */
|
||||
KERNEL_STRUCT_MEMBER(shadow_link, int, last_isect_prim, KERNEL_FEATURE_SHADOW_LINKING)
|
||||
KERNEL_STRUCT_MEMBER(shadow_link, int, last_isect_object, KERNEL_FEATURE_SHADOW_LINKING)
|
||||
|
|
|
|||
|
|
@ -26,29 +26,15 @@ ccl_device_forceinline void integrator_state_write_ray(IntegratorState state,
|
|||
const ccl_private Ray *ccl_restrict ray)
|
||||
{
|
||||
#if defined(__INTEGRATOR_GPU_PACKED_STATE__) && defined(__KERNEL_GPU__)
|
||||
static_assert(sizeof(ray->P) == sizeof(float4), "Bad assumption about float3 padding");
|
||||
/* dP and dP are packed based on the assumption that float3 is padded to 16 bytes.
|
||||
* This assumption hold trues on Metal, but not CUDA.
|
||||
*/
|
||||
((ccl_private float4 &)ray->P).w = ray->dP;
|
||||
((ccl_private float4 &)ray->D).w = ray->dD;
|
||||
INTEGRATOR_STATE_WRITE(state, ray, packed) = (ccl_private packed_ray &)*ray;
|
||||
|
||||
/* Ensure that we can correctly cast between Ray and the generated packed_ray struct. */
|
||||
static_assert(offsetof(packed_ray, P) == offsetof(Ray, P),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
static_assert(offsetof(packed_ray, D) == offsetof(Ray, D),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
static_assert(offsetof(packed_ray, tmin) == offsetof(Ray, tmin),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
static_assert(offsetof(packed_ray, tmax) == offsetof(Ray, tmax),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
static_assert(offsetof(packed_ray, time) == offsetof(Ray, time),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
static_assert(offsetof(packed_ray, dP) == 12 + offsetof(Ray, P),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
static_assert(offsetof(packed_ray, dD) == 12 + offsetof(Ray, D),
|
||||
"Generated packed_ray struct is misaligned with Ray struct");
|
||||
packed_ray packed;
|
||||
packed.P = ray->P;
|
||||
packed.dP = ray->dP;
|
||||
packed.D = ray->D;
|
||||
packed.dD = ray->dD;
|
||||
packed.tmin = ray->tmin;
|
||||
packed.tmax = ray->tmax;
|
||||
packed.time = ray->time;
|
||||
INTEGRATOR_STATE_WRITE(state, ray, packed) = packed;
|
||||
#else
|
||||
INTEGRATOR_STATE_WRITE(state, ray, P) = ray->P;
|
||||
INTEGRATOR_STATE_WRITE(state, ray, D) = ray->D;
|
||||
|
|
@ -64,9 +50,14 @@ ccl_device_forceinline void integrator_state_read_ray(ConstIntegratorState state
|
|||
ccl_private Ray *ccl_restrict ray)
|
||||
{
|
||||
#if defined(__INTEGRATOR_GPU_PACKED_STATE__) && defined(__KERNEL_GPU__)
|
||||
*((ccl_private packed_ray *)ray) = INTEGRATOR_STATE(state, ray, packed);
|
||||
ray->dP = ((ccl_private float4 &)ray->P).w;
|
||||
ray->dD = ((ccl_private float4 &)ray->D).w;
|
||||
const packed_ray packed = INTEGRATOR_STATE(state, ray, packed);
|
||||
ray->P = packed.P;
|
||||
ray->dP = packed.dP;
|
||||
ray->D = packed.D;
|
||||
ray->dD = packed.dD;
|
||||
ray->tmin = packed.tmin;
|
||||
ray->tmax = packed.tmax;
|
||||
ray->time = packed.time;
|
||||
#else
|
||||
ray->P = INTEGRATOR_STATE(state, ray, P);
|
||||
ray->D = INTEGRATOR_STATE(state, ray, D);
|
||||
|
|
@ -358,7 +349,7 @@ ccl_device_forceinline void integrator_state_write_mnee(IntegratorState state,
|
|||
const Spectrum mnee_throughput,
|
||||
const float3 mnee_wo)
|
||||
{
|
||||
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 2);
|
||||
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 3);
|
||||
|
||||
# ifdef __KERNEL_GPU__
|
||||
INTEGRATOR_STATE_WRITE(state, path, mnee_shadow_state) = (int)shadow_state;
|
||||
|
|
@ -376,7 +367,7 @@ ccl_device_forceinline void integrator_state_write_mnee(IntegratorState state,
|
|||
INTEGRATOR_STATE_ARRAY_WRITE(shadow_state, shadow_isect, 1, v) = mnee_wo.z;
|
||||
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, tmin) = ls->t;
|
||||
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, tmax) = ls->pdf;
|
||||
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, time) = ls->eval_fac;
|
||||
INTEGRATOR_STATE_ARRAY_WRITE(shadow_state, shadow_isect, 2, t) = ls->eval_fac;
|
||||
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, self_light_object) = ls->object;
|
||||
INTEGRATOR_STATE_WRITE(shadow_state, shadow_ray, self_light_prim) = ls->prim;
|
||||
INTEGRATOR_STATE_ARRAY_WRITE(shadow_state, shadow_isect, 0, object) = ls->shader;
|
||||
|
|
@ -401,7 +392,7 @@ ccl_device_forceinline void integrator_state_read_mnee(ConstIntegratorState stat
|
|||
ccl_private LightSample *ls,
|
||||
ccl_private int *mnee_vertex_count)
|
||||
{
|
||||
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 2);
|
||||
static_assert(INTEGRATOR_SHADOW_ISECT_SIZE >= 3);
|
||||
|
||||
ConstIntegratorShadowState shadow_state = integrator_state_get_mnee_shadow_state(state);
|
||||
|
||||
|
|
@ -413,7 +404,7 @@ ccl_device_forceinline void integrator_state_read_mnee(ConstIntegratorState stat
|
|||
INTEGRATOR_STATE_ARRAY(shadow_state, shadow_isect, 1, v));
|
||||
ls->t = INTEGRATOR_STATE(shadow_state, shadow_ray, tmin);
|
||||
ls->pdf = INTEGRATOR_STATE(shadow_state, shadow_ray, tmax);
|
||||
ls->eval_fac = INTEGRATOR_STATE(shadow_state, shadow_ray, time);
|
||||
ls->eval_fac = INTEGRATOR_STATE_ARRAY(shadow_state, shadow_isect, 2, t);
|
||||
ls->object = INTEGRATOR_STATE(shadow_state, shadow_ray, self_light_object);
|
||||
ls->prim = INTEGRATOR_STATE(shadow_state, shadow_ray, self_light_prim);
|
||||
ls->shader = INTEGRATOR_STATE_ARRAY(shadow_state, shadow_isect, 0, object);
|
||||
|
|
|
|||
|
|
@ -67,7 +67,7 @@ TEST(TEST_CATEGORY_NAME, float_to_half)
|
|||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, float3_to_half3)
|
||||
TEST(TEST_CATEGORY_NAME, float3_to_packed_half3)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
|
|
@ -80,8 +80,8 @@ TEST(TEST_CATEGORY_NAME, float3_to_half3)
|
|||
|
||||
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
|
||||
|
||||
const half3 h = float3_to_half3(in);
|
||||
const float3 out = half3_to_float3(h);
|
||||
const packed_half3 h = float3_to_packed_half3(in);
|
||||
const float3 out = packed_half3_to_float3(h);
|
||||
|
||||
EXPECT_EQ(out.x, in.x);
|
||||
EXPECT_EQ(out.y, in.y);
|
||||
|
|
@ -174,7 +174,7 @@ TEST(TEST_CATEGORY_NAME, half_to_float_flush_to_zero)
|
|||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, fallback_float3_to_half3)
|
||||
TEST(TEST_CATEGORY_NAME, fallback_float3_to_packed_half3)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
|
|
@ -187,8 +187,8 @@ TEST(TEST_CATEGORY_NAME, fallback_float3_to_half3)
|
|||
|
||||
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
|
||||
|
||||
const half3 h = fallback_float3_to_half3(in);
|
||||
const float3 out = fallback_half3_to_float3(h);
|
||||
const packed_half3 h = fallback_float3_to_packed_half3(in);
|
||||
const float3 out = fallback_packed_half3_to_float3(h);
|
||||
|
||||
EXPECT_EQ(out.x, in.x);
|
||||
EXPECT_EQ(out.y, in.y);
|
||||
|
|
|
|||
|
|
@ -128,6 +128,7 @@ set(SRC_HEADERS
|
|||
types_float3.h
|
||||
types_float4.h
|
||||
types_float8.h
|
||||
types_gpu_compressed.h
|
||||
types_image.h
|
||||
types_int2.h
|
||||
types_int3.h
|
||||
|
|
|
|||
|
|
@ -46,7 +46,7 @@ class half {
|
|||
#endif
|
||||
|
||||
#if !defined(__KERNEL_METAL__)
|
||||
struct half3 {
|
||||
struct packed_half3 {
|
||||
half x, y, z;
|
||||
};
|
||||
|
||||
|
|
@ -55,6 +55,10 @@ struct half4 {
|
|||
};
|
||||
#endif
|
||||
|
||||
static_assert(sizeof(half) == 2);
|
||||
static_assert(sizeof(packed_half3) == 6);
|
||||
static_assert(sizeof(half4) == 8);
|
||||
|
||||
#if !defined(__KERNEL_GPU__)
|
||||
/* Optimized fallback implementations with fast path for normal and denormal numbers, assuming
|
||||
* no Infs or NaNs. Based on public domain functions from.
|
||||
|
|
@ -110,7 +114,7 @@ ccl_device_inline float4 fallback_half4_to_float4(const half4 h)
|
|||
return cast(f | s);
|
||||
}
|
||||
|
||||
ccl_device_inline float3 fallback_half3_to_float3(const half3 h)
|
||||
ccl_device_inline float3 fallback_packed_half3_to_float3(const packed_half3 h)
|
||||
{
|
||||
return make_float3(fallback_half4_to_float4({h.x, h.y, h.z, 0}));
|
||||
}
|
||||
|
|
@ -184,7 +188,7 @@ ccl_device_inline half4 fallback_float4_to_half4(const float4 f)
|
|||
half(uint16_t(res.x)), half(uint16_t(res.y)), half(uint16_t(res.z)), half(uint16_t(res.w))};
|
||||
}
|
||||
|
||||
ccl_device_inline half3 fallback_float3_to_half3(const float3 f)
|
||||
ccl_device_inline packed_half3 fallback_float3_to_packed_half3(const float3 f)
|
||||
{
|
||||
const half4 h = fallback_float4_to_half4(make_float4(f));
|
||||
return {h.x, h.y, h.z};
|
||||
|
|
@ -272,7 +276,7 @@ ccl_device_inline float4 half4_to_float4(const half4 h)
|
|||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline half3 float3_to_half3(const float3 f)
|
||||
ccl_device_inline packed_half3 float3_to_packed_half3(const float3 f)
|
||||
{
|
||||
#if defined(__KERNEL_GPU__)
|
||||
return {float_to_half(f.x), float_to_half(f.y), float_to_half(f.z)};
|
||||
|
|
@ -291,7 +295,7 @@ ccl_device_inline half3 float3_to_half3(const float3 f)
|
|||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline float3 half3_to_float3(const half3 h)
|
||||
ccl_device_inline float3 packed_half3_to_float3(const packed_half3 h)
|
||||
{
|
||||
#if defined(__KERNEL_GPU__)
|
||||
return make_float3(half_to_float(h.x), half_to_float(h.y), half_to_float(h.z));
|
||||
|
|
|
|||
|
|
@ -38,3 +38,5 @@
|
|||
#include "util/types_float3x3.h" // IWYU pragma: export
|
||||
|
||||
#include "util/types_spherical_harmonics.h" // IWYU pragma: export
|
||||
|
||||
#include "util/types_gpu_compressed.h" // IWYU pragma: export
|
||||
|
|
|
|||
181
intern/cycles/util/types_gpu_compressed.h
Normal file
181
intern/cycles/util/types_gpu_compressed.h
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
/* SPDX-FileCopyrightText: 2011-2026 Blender Foundation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "util/half.h"
|
||||
#include "util/types_spectrum.h"
|
||||
|
||||
CCL_NAMESPACE_BEGIN
|
||||
|
||||
/* Compact storage of data types on GPU, stored in full precision on CPU.
|
||||
* Under the assumption that reducing state size is important on GPU and
|
||||
* math is relatively cheap. */
|
||||
|
||||
/* A float stored in half precision. */
|
||||
struct FloatCompressedOnGPU {
|
||||
using value_type = float;
|
||||
using compressed_type = half;
|
||||
|
||||
#ifdef __KERNEL_GPU__
|
||||
half v;
|
||||
#else
|
||||
float v;
|
||||
#endif
|
||||
|
||||
FloatCompressedOnGPU() = default;
|
||||
|
||||
ccl_device_inline_method FloatCompressedOnGPU(const float a)
|
||||
{
|
||||
#ifdef __KERNEL_GPU__
|
||||
v = float_to_half(a);
|
||||
#else
|
||||
v = a;
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline_method operator float() const ccl_global
|
||||
{
|
||||
#ifdef __KERNEL_GPU__
|
||||
return half_to_float(v);
|
||||
#else
|
||||
return v;
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_global FloatCompressedOnGPU &operator=(const float a) ccl_global
|
||||
{
|
||||
#ifdef __KERNEL_GPU__
|
||||
v = float_to_half(a);
|
||||
#else
|
||||
v = a;
|
||||
#endif
|
||||
return *this;
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_global FloatCompressedOnGPU &operator=(FloatCompressedOnGPU a)
|
||||
ccl_global
|
||||
{
|
||||
v = a.v;
|
||||
return *this;
|
||||
}
|
||||
|
||||
#ifdef __KERNEL_METAL__
|
||||
ccl_device_inline_method operator float() const ccl_private
|
||||
{
|
||||
# ifdef __KERNEL_GPU__
|
||||
return half_to_float(v);
|
||||
# else
|
||||
return v;
|
||||
# endif
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_private FloatCompressedOnGPU &operator=(const float a) ccl_private
|
||||
{
|
||||
# ifdef __KERNEL_GPU__
|
||||
v = float_to_half(a);
|
||||
# else
|
||||
v = a;
|
||||
# endif
|
||||
return *this;
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_private FloatCompressedOnGPU &operator=(FloatCompressedOnGPU a)
|
||||
ccl_private
|
||||
{
|
||||
v = a.v;
|
||||
return *this;
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
/* Spectrum stored in half precision. */
|
||||
struct SpectrumCompressedOnGPU {
|
||||
using value_type = Spectrum;
|
||||
using compressed_type = packed_half3;
|
||||
|
||||
#ifdef __KERNEL_GPU__
|
||||
packed_half3 v;
|
||||
#else
|
||||
Spectrum v;
|
||||
#endif
|
||||
|
||||
SpectrumCompressedOnGPU() = default;
|
||||
|
||||
ccl_device_inline_method SpectrumCompressedOnGPU(const Spectrum a)
|
||||
{
|
||||
#ifdef __KERNEL_GPU__
|
||||
v = float3_to_packed_half3(a);
|
||||
#else
|
||||
v = a;
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline_method operator Spectrum() const ccl_global
|
||||
{
|
||||
#ifdef __KERNEL_GPU__
|
||||
return packed_half3_to_float3(v);
|
||||
#else
|
||||
return v;
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_global SpectrumCompressedOnGPU &operator=(const Spectrum a)
|
||||
ccl_global
|
||||
{
|
||||
#ifdef __KERNEL_GPU__
|
||||
v = float3_to_packed_half3(a);
|
||||
#else
|
||||
v = a;
|
||||
#endif
|
||||
return *this;
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_global SpectrumCompressedOnGPU &operator=(SpectrumCompressedOnGPU a)
|
||||
ccl_global
|
||||
{
|
||||
v = a.v;
|
||||
return *this;
|
||||
}
|
||||
|
||||
#ifdef __KERNEL_METAL__
|
||||
ccl_device_inline_method operator Spectrum() const ccl_private
|
||||
{
|
||||
# ifdef __KERNEL_GPU__
|
||||
return packed_half3_to_float3(v);
|
||||
# else
|
||||
return v;
|
||||
# endif
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_private SpectrumCompressedOnGPU &operator=(const Spectrum a)
|
||||
ccl_private
|
||||
{
|
||||
# ifdef __KERNEL_GPU__
|
||||
v = float3_to_packed_half3(a);
|
||||
# else
|
||||
v = a;
|
||||
# endif
|
||||
return *this;
|
||||
}
|
||||
|
||||
ccl_device_inline_method ccl_private SpectrumCompressedOnGPU &operator=(
|
||||
SpectrumCompressedOnGPU a) ccl_private
|
||||
{
|
||||
v = a.v;
|
||||
return *this;
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
/* Templates for determining GPU storage type and size from the host. */
|
||||
template<typename...> using gpu_void_t = void;
|
||||
template<typename T, typename = void> struct gpu_state_storage {
|
||||
using gpu_type = T;
|
||||
};
|
||||
template<typename T> struct gpu_state_storage<T, gpu_void_t<typename T::compressed_type>> {
|
||||
using gpu_type = typename T::compressed_type;
|
||||
};
|
||||
|
||||
CCL_NAMESPACE_END
|
||||
Loading…
Add table
Add a link
Reference in a new issue