Cycles: Half float denormals and optimizations

* Switch to more accurate half and float conversion supporting denormals
  to match native instructions. This makes CPU and GPU match more
  closely in some tests and avoids clipping some low values.
* Inf and NaN are not supported still, as we already filter these out
  and there is no reason to have the overhead.
* Change avx2 kernel to require f16c. For all physical CPUs avx2 implies
  f16c, and it's only for emulation and virtual machines that this would
  not be the case. So it's fine to fall back to the sse4.1 kernel then.
* Assume half instructions are available with ARM NEON. There is no
  defined minimum architecture, but Blender assumes the same and ARMv8.2-A
  is relatively old.
* For everything else there are SIMD optimized fallbacks.
* New unit tests were added, coverting both native instructions and
  fallback implementations, and half/half3/half4.

Fix #152763: Half float image low values are clipped

Pull Request: https://projects.blender.org/blender/blender/pulls/154042
This commit is contained in:
Brecht Van Lommel 2026-02-09 13:01:08 +01:00 • committed by Brecht Van Lommel
parent 646f919cc9
commit 110de4f488
13 changed files with 510 additions and 141 deletions

View file

@ -34,6 +34,7 @@ endif()
if(WITH_CYCLES_NATIVE_ONLY)
set(CXX_HAS_SSE42 FALSE)
set(CXX_HAS_AVX2 FALSE)
set(CXX_HAS_F16C FALSE)
add_definitions(
-DWITH_KERNEL_NATIVE
)
@ -71,16 +72,15 @@ elseif(SUPPORTS_NEON_BUILD AND SSE2NEON_FOUND)
# Disable the CXX_HAS_* flags so we don't add any SSE or AVX compiler flags.
set(CXX_HAS_SSE42 FALSE)
set(CXX_HAS_AVX2 FALSE)
set(CXX_HAS_F16C FALSE)
elseif(WIN32 AND MSVC AND NOT CMAKE_CXX_COMPILER_ID MATCHES "Clang")
set(CXX_HAS_SSE42 TRUE)
set(CXX_HAS_AVX2 TRUE)
set(CXX_HAS_F16C TRUE)
# /arch:AVX for VC2012 and above
if(NOT MSVC_VERSION LESS 1700)
set(CYCLES_AVX2_FLAGS "/arch:AVX /arch:AVX2")
elseif(NOT CMAKE_CL_64)
set(CYCLES_AVX2_FLAGS "/arch:SSE2")
endif()
# No separate f16c flag for MSVC.
set(CYCLES_AVX2_FLAGS "/arch:AVX2")
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS}")
# there is no /arch:SSE3, but intrinsics are available anyway
if(CMAKE_CL_64)
@ -91,34 +91,40 @@ elseif(WIN32 AND MSVC AND NOT CMAKE_CXX_COMPILER_ID MATCHES "Clang")
elseif((CMAKE_C_COMPILER_ID STREQUAL "GNU") OR (CMAKE_CXX_COMPILER_ID MATCHES "Clang"))
check_cxx_compiler_flag(-msse4.2 CXX_HAS_SSE42)
check_cxx_compiler_flag(-mavx2 CXX_HAS_AVX2)
check_cxx_compiler_flag(-mf16c CXX_HAS_F16C)
if(CXX_HAS_SSE42)
set(CYCLES_SSE42_FLAGS "-msse -msse2 -msse3 -mssse3 -msse4.1 -msse4.2")
if(CXX_HAS_AVX2)
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
set(CYCLES_AVX2_FLAGS "${CYCLES_SSE42_FLAGS} -mavx -mavx2 -mfma -mlzcnt -mbmi -mbmi2 -mf16c")
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS} -mf16c")
endif()
endif()
elseif(WIN32 AND CMAKE_CXX_COMPILER_ID STREQUAL "Intel")
check_cxx_compiler_flag(/QxSSE4.2 CXX_HAS_SSE42)
check_cxx_compiler_flag(/QxCORE-AVX2 CXX_HAS_AVX2)
set(CXX_HAS_F16C CXX_HAS_AVX) # Same flag enables AVX, FMA and F16C
if(CXX_HAS_SSE42)
set(CYCLES_SSE42_FLAGS "/QxSSE4.2")
if(CXX_HAS_AVX2)
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
set(CYCLES_AVX2_FLAGS "/QxCORE-AVX2")
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS}")
endif()
endif()
elseif(CMAKE_CXX_COMPILER_ID STREQUAL "Intel")
check_cxx_compiler_flag(-xsse4.2 CXX_HAS_SSE42)
check_cxx_compiler_flag(-xcore-avx2 CXX_HAS_AVX2)
set(CXX_HAS_F16C CXX_HAS_AVX) # Same flag enables AVX, FMA and F16C
if(CXX_HAS_SSE42)
set(CYCLES_SSE42_FLAGS "-xsse4.2")
if(CXX_HAS_AVX2)
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
set(CYCLES_AVX2_FLAGS "-xcore-avx2")
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS}")
endif()
endif()
endif()

View file

@ -40,8 +40,8 @@ if(DEFINED CYCLES_KERNEL_FLAGS)
set_source_files_properties(kernel.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_KERNEL_FLAGS}")
endif()
if(CXX_HAS_AVX2)
set_source_files_properties(kernel_avx2.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_FLAGS}")
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
set_source_files_properties(kernel_avx2.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_F16C_FLAGS}")
endif()
# Warnings to avoid using doubles in the kernel.

View file

@ -5,6 +5,7 @@
#pragma once
#include "util/half.h"
#include "util/math_int2.h"
#include "util/types.h"
#include "util/vector.h"

View file

@ -34,6 +34,7 @@ set(SRC
render_graph_finalize_test.cpp
util_aligned_malloc_test.cpp
util_boundbox_test.cpp
util_half_test.cpp
util_ies_test.cpp
util_math_test.cpp
util_math_fast_test.cpp
@ -54,8 +55,10 @@ if(NOT APPLE)
if(CXX_HAS_AVX2)
list(APPEND SRC
util_float8_avx2_test.cpp
util_half_avx2_test.cpp
)
set_source_files_properties(util_float8_avx2_test.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_FLAGS}")
set_source_files_properties(util_float8_avx2_test.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_F16C_FLAGS}")
set_source_files_properties(util_half_avx2_test.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_F16C_FLAGS}")
endif()
endif()

View file

@ -12,7 +12,6 @@ CCL_NAMESPACE_BEGIN
static bool validate_cpu_capabilities()
{
#if defined(__KERNEL_AVX2__)
return system_cpu_support_avx2();
#elif defined(__KERNEL_AVX__)

View file

@ -0,0 +1,14 @@
/* SPDX-FileCopyrightText: 2026 Blender Foundation
*
* SPDX-License-Identifier: Apache-2.0 */
#define __KERNEL_SSE__
#define __KERNEL_AVX__
#define __KERNEL_AVX2__
#define TEST_CATEGORY_NAME util_half_avx2
#if (defined(i386) || defined(_M_IX86) || defined(__x86_64__) || defined(_M_X64)) && \
defined(__AVX2__)
# include "util_half_test.h"
#endif

View file

@ -0,0 +1,6 @@
/* SPDX-FileCopyrightText: 2026 Blender Foundation
*
* SPDX-License-Identifier: Apache-2.0 */
#define TEST_CATEGORY_NAME util_half
#include "util_half_test.h"

View file

@ -0,0 +1,193 @@
/* SPDX-FileCopyrightText: 2026 Blender Foundation
*
* SPDX-License-Identifier: Apache-2.0 */
#include <gtest/gtest.h>
#include "util/half.h"
#include "util/system.h"
#include <array>
CCL_NAMESPACE_BEGIN
static bool validate_cpu_capabilities()
{
#if defined(__KERNEL_AVX2__)
return system_cpu_support_avx2();
#else
return true;
#endif
}
#define EXPECT_HALF_EQ(a, b) \
do { \
const uint16_t a_ = (a); \
const uint16_t b_ = (b); \
if ((a_ & 0x7fff) == 0 && (b_ & 0x7fff) == 0) { \
/* Ignore sign of zero, -0.0 getting converted to 0.0 is fine. */ \
} \
else { \
EXPECT_EQ(a_, b_); \
} \
} while (0)
struct HalfTestVal {
float f;
uint16_t h;
};
/* We don't support NaN and inf, so not tested here. */
static const std::array<HalfTestVal, 13> test_values = {{{0.0f, 0x0000},
{-0.0f, 0x8000},
{1.0f, 0x3c00},
{2.0f, 0x4000},
{10.0f, 0x4900},
{100.0f, 0x5640},
{0.0999755859375f, 0x2e66},
{-0.0999755859375f, 0xae66},
{-0.5f, 0xb800},
{-1.0f, 0xbc00},
/* Smallest positive normal (2^-14) */
{6.103515625e-5f, 0x0400},
/* Largest denormal (2^-14 - 2^-24) */
{6.0975551605e-5f, 0x03ff},
/* Smallest positive denormal (2^-24) */
{5.9604644775e-8f, 0x0001}}};
TEST(TEST_CATEGORY_NAME, float_to_half)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
return;
}
for (const HalfTestVal &val : test_values) {
EXPECT_HALF_EQ(float_to_half(val.f), val.h);
EXPECT_EQ(half_to_float(half(val.h)), val.f);
}
}
TEST(TEST_CATEGORY_NAME, float3_to_half3)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
return;
}
for (size_t i = 0; i < test_values.size(); i++) {
const size_t i0 = i;
const size_t i1 = (i + 1) % test_values.size();
const size_t i2 = (i + 2) % test_values.size();
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
const half3 h = float3_to_half3(in);
const float3 out = half3_to_float3(h);
EXPECT_EQ(out.x, in.x);
EXPECT_EQ(out.y, in.y);
EXPECT_EQ(out.z, in.z);
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
}
}
TEST(TEST_CATEGORY_NAME, float4_to_half4)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
return;
}
for (size_t i = 0; i < test_values.size(); i += 4) {
const size_t i0 = i;
const size_t i1 = (i + 1) % test_values.size();
const size_t i2 = (i + 2) % test_values.size();
const size_t i3 = (i + 3) % test_values.size();
const float4 in = make_float4(
test_values[i0].f, test_values[i1].f, test_values[i2].f, test_values[i3].f);
const half4 h = float4_to_half4(in);
const float4 out = half4_to_float4(h);
EXPECT_EQ(out.x, in.x);
EXPECT_EQ(out.y, in.y);
EXPECT_EQ(out.z, in.z);
EXPECT_EQ(out.w, in.w);
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
EXPECT_HALF_EQ(uint16_t(h.w), test_values[i3].h);
}
}
TEST(TEST_CATEGORY_NAME, fallback_float_to_half)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
return;
}
for (const HalfTestVal &val : test_values) {
EXPECT_HALF_EQ(fallback_float_to_half(val.f), val.h);
EXPECT_EQ(fallback_half_to_float(half(val.h)), val.f);
}
}
TEST(TEST_CATEGORY_NAME, fallback_float3_to_half3)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
return;
}
for (size_t i = 0; i < test_values.size(); i++) {
const size_t i0 = i;
const size_t i1 = (i + 1) % test_values.size();
const size_t i2 = (i + 2) % test_values.size();
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
const half3 h = fallback_float3_to_half3(in);
const float3 out = fallback_half3_to_float3(h);
EXPECT_EQ(out.x, in.x);
EXPECT_EQ(out.y, in.y);
EXPECT_EQ(out.z, in.z);
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
}
}
TEST(TEST_CATEGORY_NAME, fallback_float4_to_half4)
{
if (!validate_cpu_capabilities()) {
GTEST_SKIP();
return;
}
for (size_t i = 0; i < test_values.size(); i += 4) {
const size_t i0 = i;
const size_t i1 = (i + 1) % test_values.size();
const size_t i2 = (i + 2) % test_values.size();
const size_t i3 = (i + 3) % test_values.size();
const float4 in = make_float4(
test_values[i0].f, test_values[i1].f, test_values[i2].f, test_values[i3].f);
const half4 h = fallback_float4_to_half4(in);
const float4 out = fallback_half4_to_float4(h);
EXPECT_EQ(out.x, in.x);
EXPECT_EQ(out.y, in.y);
EXPECT_EQ(out.z, in.z);
EXPECT_EQ(out.w, in.w);
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
EXPECT_HALF_EQ(uint16_t(h.w), test_values[i3].h);
}
}
CCL_NAMESPACE_END

View file

@ -1,184 +1,316 @@
/* SPDX-FileCopyrightText: 2011-2022 Blender Foundation
/* SPDX-FileCopyrightText: 2011-2026 Blender Foundation
*
* SPDX-License-Identifier: Apache-2.0 */
#pragma once
#include "util/math.h"
#include "util/types.h"
#include "util/defines.h"
#include "util/math_base.h"
#include "util/math_float4.h"
#include "util/math_int4.h"
#include "util/types_base.h"
#include "util/types_float4.h"
#include "util/types_int4.h"
#include "util/types_uint4.h"
#if !defined(__KERNEL_GPU__) && defined(__KERNEL_SSE2__)
# include "util/simd.h" // IWYU pragma: keep
# include "util/optimization.h" // IWYU pragma: keep
# include "util/simd.h" // IWYU pragma: keep
#endif
CCL_NAMESPACE_BEGIN
/* Half Floats */
#if defined(__KERNEL_METAL__)
ccl_device_inline float half_to_float(half h_in)
{
float f;
union {
half h;
uint16_t s;
} val;
val.h = h_in;
*((ccl_private int *)&f) = ((val.s & 0x8000) << 16) | (((val.s & 0x7c00) + 0x1C000) << 13) |
((val.s & 0x03FF) << 13);
return f;
}
#else
/* CUDA has its own half data type, no need to define then */
# if !defined(__KERNEL_CUDA__) && !defined(__KERNEL_HIP__) && !defined(__KERNEL_ONEAPI__)
/* Implementing this as a class rather than a typedef so that the compiler can tell it apart from
* unsigned shorts. */
#if !defined(__KERNEL_GPU__)
/* GPUs have native support for this type.
* Implementing this as a class rather than a typedef so that the compiler can tell it apart from
* uint16_ts. */
class half {
public:
half() = default;
half(const unsigned short &i) : v(i) {}
operator unsigned short() const
half(const uint16_t &i) : v(i) {}
operator uint16_t() const
{
return v;
}
half &operator=(const unsigned short &i)
half &operator=(const uint16_t &i)
{
v = i;
return *this;
}
private:
unsigned short v;
uint16_t v;
};
#endif
#if !defined(__KERNEL_METAL__)
struct half3 {
half x, y, z;
};
# endif
struct half4 {
half x, y, z, w;
};
#endif
/* Conversion to/from half float for image textures
#if !defined(__KERNEL_GPU__)
/* Optimized fallback implementations with fast path for normal and denormal numbers, assuming
* no Infs or NaNs. Based on public domain functions from.
*
* Simplified float to half for fast sampling on processor without a native
* instruction, and eliminating any NaN and inf values. */
* https://fgiesen.wordpress.com/2012/03/28/half-to-float-done-quic/
* https://gist.github.com/rygorous/2144712
* https://gist.github.com/rygorous/2156668
* https://gist.github.com/rygorous/4d9e9e88cab13c703773dc767a23575f
*/
ccl_device_inline float fallback_half_to_float(const half h)
{
const uint32_t bits = uint16_t(h);
const uint32_t s = (bits & 0x8000) << 16;
const uint32_t em = (bits & 0x7fff) << 13;
const float f = __int_as_float(em) * __int_as_float(0x77800000 /* 2^112 */);
return __int_as_float(__float_as_uint(f) | s);
}
ccl_device_inline half float_to_half_image(const float f)
ccl_device_inline float4 fallback_half4_to_float4(const half4 h)
{
const int4 i = make_int4(uint16_t(h.x), uint16_t(h.y), uint16_t(h.z), uint16_t(h.w));
const int4 s = (i & 0x8000) << 16;
const int4 em = (i & 0x7fff) << 13;
const float4 f = cast(em) * __int_as_float(0x77800000 /* 2^112 */);
return cast(cast(f) | s);
}
ccl_device_inline float3 fallback_half3_to_float3(const half3 h)
{
return make_float3(fallback_half4_to_float4({h.x, h.y, h.z, 0}));
}
ccl_device_inline half fallback_float_to_half(const float f)
{
const int c_f16max = (127 + 16) << 23;
const int c_infty_as_fp16 = 0x7c00;
const int c_min_normal = (127 - 14) << 23;
const int c_denorm_magic = ((127 - 15) + (23 - 10) + 1) << 23;
const int c_normal_bias = 0xfff - ((127 - 15) << 23);
const uint f_i = __float_as_uint(f);
const uint sign_i = f_i & 0x80000000u;
const int abs_i = int(f_i ^ sign_i);
uint16_t res;
if (abs_i >= c_f16max) {
/* Overflows to infinity. */
res = uint16_t(c_infty_as_fp16);
}
else if (abs_i < c_min_normal) {
/* Denormal. */
float denorm_f = __uint_as_float(uint(abs_i));
denorm_f += __int_as_float(c_denorm_magic);
const int denorm_i = int(__float_as_uint(denorm_f)) - c_denorm_magic;
res = uint16_t(denorm_i);
}
else {
/* Normal. */
const int mant_odd = int(uint(abs_i) >> 13) & 1;
res = uint16_t((abs_i + c_normal_bias + mant_odd) >> 13);
}
return half(res | uint16_t(sign_i >> 16));
}
ccl_device_inline half4 fallback_float4_to_half4(const float4 f)
{
const int4 c_f16max = make_int4((127 + 16) << 23);
const int4 c_infty_as_fp16 = make_int4(0x7c00);
const int4 c_min_normal = make_int4((127 - 14) << 23);
const int4 c_denorm_magic = make_int4(((127 - 15) + (23 - 10) + 1) << 23);
const int4 c_normal_bias = make_int4(0xfff - ((127 - 15) << 23));
const float4 abs_f = fabs(f);
const int4 abs_i = __float4_as_int4(abs_f);
const int4 b_isregular = c_f16max > abs_i;
const int4 b_isdenorm = c_min_normal > abs_i;
/* Denormal. */
const float4 denorm_f = abs_f + __int4_as_float4(c_denorm_magic);
const int4 denorm_i = __float4_as_int4(denorm_f) - c_denorm_magic;
/* Normal. */
const int4 mant_odd = (abs_i << (31 - 13)) >> 31;
const int4 normal = srl(abs_i + c_normal_bias - mant_odd, 13);
/* Combined normal and denormal. */
const int4 nonspecial = select(b_isdenorm, denorm_i, normal);
/* Combine overflow to infinity. */
const int4 combined = select(b_isregular, nonspecial, c_infty_as_fp16);
const int4 sign_i = __float4_as_int4(f ^ abs_f);
const int4 res = combined | (sign_i >> 16);
return {
half(uint16_t(res.x)), half(uint16_t(res.y)), half(uint16_t(res.z)), half(uint16_t(res.w))};
}
ccl_device_inline half3 fallback_float3_to_half3(const float3 f)
{
const half4 h = fallback_float4_to_half4(make_float4(f));
return {h.x, h.y, h.z};
}
#endif
ccl_device_inline float half_to_float(half h)
{
#if defined(__KERNEL_METAL__) || defined(__KERNEL_ONEAPI__)
return half(min(f, 65504.0f));
return float(h);
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
return __float2half(min(f, 65504.0f));
return __half2float(h);
/* We assume half instructions are always supported when there is ARM Neon,
* which implies ARMv8.2-A+. There is no official Blender minimum, but is
* already assumed elsewhere in Blender and not that recent. */
#elif defined(__ARM_NEON) || defined(_M_ARM64)
uint16x4_t v = vdup_n_u16(uint16_t(h));
return vgetq_lane_f32(vcvt_f32_f16(vreinterpret_f16_u16(v)), 0);
#elif defined(__F16C__)
return _cvtsh_ss(uint16_t(h));
#else
const uint u = __float_as_uint(f);
/* Sign bit, shifted to its position. */
uint sign_bit = u & 0x80000000;
sign_bit >>= 16;
/* Exponent. */
const uint exponent_bits = u & 0x7f800000;
/* Non-sign bits. */
uint value_bits = u & 0x7fffffff;
value_bits >>= 13; /* Align mantissa on MSB. */
value_bits -= 0x1c000; /* Adjust bias. */
/* Flush-to-zero. */
value_bits = (exponent_bits < 0x38800000) ? 0 : value_bits;
/* Clamp-to-max. */
value_bits = (exponent_bits > 0x47000000) ? 0x7bff : value_bits;
/* Denormals-as-zero. */
value_bits = (exponent_bits == 0 ? 0 : value_bits);
/* Re-insert sign bit and return. */
return (value_bits | sign_bit);
/* The fallback is fast so don't bother with native instructions. */
return fallback_half_to_float(h);
#endif
}
ccl_device_inline half float_to_half(const float f)
{
#if defined(__KERNEL_METAL__) || defined(__KERNEL_ONEAPI__)
return half(f);
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
return __float2half(f);
#elif defined(__ARM_NEON) || defined(_M_ARM64)
return half(vget_lane_u16(vreinterpret_u16_f16(vcvt_f16_f32(vdupq_n_f32(f))), 0));
#elif defined(__F16C__)
return half((uint16_t)_cvtss_sh(f, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
#else
return fallback_float_to_half(f);
#endif
}
ccl_device_inline half4 float4_to_half4(const float4 f)
{
#if defined(__KERNEL_METAL__)
return {half(f.x), half(f.y), half(f.z), half(f.w)};
#elif defined(__KERNEL_ONEAPI__)
return {half(f.x), half(f.y), half(f.z), half(f.w)};
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
return {__float2half(f.x), __float2half(f.y), __float2half(f.z), __float2half(f.w)};
#elif defined(__ARM_NEON) || defined(_M_ARM64)
half4 r;
vst1_u16(reinterpret_cast<uint16_t *>(&r),
vreinterpret_u16_f16(vcvt_f16_f32(float32x4_t{f.x, f.y, f.z, f.w})));
return r;
#elif defined(__F16C__)
half4 r;
const __m128i h = _mm_cvtps_ph(_mm_loadu_ps(&f.x),
_MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC);
_mm_storel_epi64(reinterpret_cast<__m128i *>(&r), h);
return r;
#else
return fallback_float4_to_half4(f);
#endif
}
ccl_device_inline float4 half4_to_float4(const half4 h)
{
#if defined(__KERNEL_METAL__)
return {float(h.x), float(h.y), float(h.z), float(h.w)};
#elif defined(__KERNEL_ONEAPI__)
return {float(h.x), float(h.y), float(h.z), float(h.w)};
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
return {__half2float(h.x), __half2float(h.y), __half2float(h.z), __half2float(h.w)};
#elif defined(__ARM_NEON) || defined(_M_ARM64)
float4 r;
vst1q_f32(&r.x,
vcvt_f32_f16(vreinterpret_f16_u16(vld1_u16(reinterpret_cast<const uint16_t *>(&h)))));
return r;
#elif defined(__F16C__)
float4 r;
_mm_storeu_ps(&r.x, _mm_cvtph_ps(_mm_loadl_epi64(reinterpret_cast<const __m128i *>(&h))));
return r;
#else
return fallback_half4_to_float4(h);
#endif
}
ccl_device_inline half3 float3_to_half3(const float3 f)
{
#if defined(__KERNEL_GPU__)
return {float_to_half(f.x), float_to_half(f.y), float_to_half(f.z)};
#elif defined(__ARM_NEON) || defined(_M_ARM64)
const uint16x4_t h = vreinterpret_u16_f16(vcvt_f16_f32(float32x4_t{f.x, f.y, f.z, 0.0f}));
return {half(vget_lane_u16(h, 0)), half(vget_lane_u16(h, 1)), half(vget_lane_u16(h, 2))};
#elif defined(__F16C__)
const __m128i h = _mm_cvtps_ph(_mm_set_ps(0.0f, f.z, f.y, f.x),
_MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC);
return {half(uint16_t(_mm_extract_epi16(h, 0))),
half(uint16_t(_mm_extract_epi16(h, 1))),
half(uint16_t(_mm_extract_epi16(h, 2)))};
#else
half4 h = fallback_float4_to_half4(make_float4(f));
return {h.x, h.y, h.z};
#endif
}
ccl_device_inline float3 half3_to_float3(const half3 h)
{
#if defined(__KERNEL_GPU__)
return make_float3(half_to_float(h.x), half_to_float(h.y), half_to_float(h.z));
#elif defined(__ARM_NEON) || defined(_M_ARM64)
const float32x4_t f = vcvt_f32_f16(
vreinterpret_f16_u16(uint16x4_t{uint16_t(h.x), uint16_t(h.y), uint16_t(h.z), 0}));
return make_float3(vgetq_lane_f32(f, 0), vgetq_lane_f32(f, 1), vgetq_lane_f32(f, 2));
#elif defined(__F16C__)
const __m128i v = _mm_set_epi16(0, 0, 0, 0, 0, uint16_t(h.z), uint16_t(h.y), uint16_t(h.x));
const __m128 f = _mm_cvtph_ps(v);
return make_float3(float4(f));
#else
return make_float3(fallback_half4_to_float4({h.x, h.y, h.z, 0}));
#endif
}
/* For image textutes */
ccl_device_inline half float_to_half_image(const float f)
{
return float_to_half(clamp(f, -65504.0f, 65504.0f));
}
ccl_device_inline float half_to_float_image(half h)
{
#if defined(__KERNEL_METAL__)
return half_to_float(h);
#elif defined(__KERNEL_ONEAPI__)
return float(h);
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
return __half2float(h);
#else
const int x = ((h & 0x8000) << 16) | (((h & 0x7c00) + 0x1C000) << 13) | ((h & 0x03FF) << 13);
return __int_as_float(x);
#endif
}
ccl_device_inline float4 half4_to_float4_image(const half4 h)
{
/* Unable to use because it gives different results half_to_float_image, can we
* modify float_to_half_image so the conversion results are identical? */
#if 0 /* defined(__KERNEL_AVX2__) */
/* CPU: AVX. */
__m128i x = _mm_castpd_si128(_mm_load_sd((const double *)&h));
return float4(_mm_cvtph_ps(x));
#endif
const float4 f = make_float4(half_to_float_image(h.x),
half_to_float_image(h.y),
half_to_float_image(h.z),
half_to_float_image(h.w));
return f;
return half4_to_float4(h);
}
/* Conversion to half float texture for display.
*
* Simplified float to half for fast display texture conversion on processors
* without a native instruction. Assumes no negative, no NaN, no inf, and sets
* denormal to 0. */
/* For render display */
ccl_device_inline half float_to_half_display(const float f)
{
#if defined(__KERNEL_METAL__) || defined(__KERNEL_ONEAPI__)
return half(min(f, 65504.0f));
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
return __float2half(min(f, 65504.0f));
#else
const int x = __float_as_int((f > 0.0f) ? ((f < 65504.0f) ? f : 65504.0f) : 0.0f);
const int absolute = x & 0x7FFFFFFF;
const int Z = absolute + 0xC8000000;
const int result = (absolute < 0x38800000) ? 0 : Z;
const int rshift = (result >> 13);
return (rshift & 0x7FFF);
#endif
return float_to_half(clamp(f, 0.0f, 65504.0f));
}
ccl_device_inline half4 float4_to_half4_display(const float4 f)
{
#ifdef __KERNEL_SSE__
/* CPU: SSE and AVX. */
const float4 x = min(max(f, make_float4(0.0f)), make_float4(65504.0f));
# ifdef __KERNEL_AVX2__
int4 rpack = int4(_mm_cvtps_ph(x, 0));
# else
const int4 absolute = cast(x) & make_int4(0x7FFFFFFF);
const int4 Z = absolute + make_int4(0xC8000000);
const int4 result = andnot(absolute < make_int4(0x38800000), Z);
int4 rshift = (result >> 13) & make_int4(0x7FFF);
int4 rpack = int4(_mm_packs_epi32(rshift, rshift));
# endif
half4 h;
_mm_storel_pi((__m64 *)&h, _mm_castsi128_ps(rpack));
return h;
#else
/* GPU and scalar fallback. */
const half4 h = {float_to_half_display(f.x),
float_to_half_display(f.y),
float_to_half_display(f.z),
float_to_half_display(f.w)};
return h;
#endif
return float4_to_half4(clamp(f, make_float4(0.0f), make_float4(65504.0f)));
}
#ifndef __KERNEL_GPU__
ccl_device_inline float half_is_finite(const half h)
{
const int exponent = (h >> 10) & 0x001f;
const int exponent = (uint16_t(h) >> 10) & 0x001f;
return exponent < 31;
}
#endif

View file

@ -76,6 +76,11 @@ ccl_device_inline int4 operator<(const int4 a, const int b)
return a < make_int4(b);
}
ccl_device_inline int4 operator>(const int4 a, const int4 b)
{
return b < a;
}
ccl_device_inline int4 operator==(const int4 a, const int4 b)
{
# ifdef __KERNEL_SSE__
@ -197,12 +202,14 @@ ccl_device_inline int4 &operator>>=(int4 &a, const int32_t b)
return a = a >> b;
}
# ifdef __KERNEL_SSE__
ccl_device_forceinline int4 srl(const int4 a, const int32_t b)
ccl_device_inline int4 srl(const int4 a, const int i)
{
return int4(_mm_srli_epi32(a.m128, b));
}
# ifdef __KERNEL_SSE__
return int4(_mm_srli_epi32(a.m128, i));
# else
return make_int4(uint32_t(a.x) >> i, uint32_t(a.y) >> i, uint32_t(a.z) >> i, uint32_t(a.w) >> i);
# endif
}
ccl_device_inline int4 min(const int4 a, const int4 b)
{

View file

@ -138,6 +138,7 @@ int system_cpu_bits()
struct CPUCapabilities {
bool sse42;
bool avx2;
bool f16c;
};
static CPUCapabilities &system_cpu_capabilities()
@ -184,6 +185,8 @@ static CPUCapabilities &system_cpu_capabilities()
const bool avx = (xcr_feature_mask & 0x6) == 0x6;
const bool f16c = (result[2] & ((int)1 << 29)) != 0;
caps.f16c = avx && f16c;
__cpuid(result, 0x00000007);
bool bmi1 = (result[1] & ((int)1 << 3)) != 0;
bool bmi2 = (result[1] & ((int)1 << 8)) != 0;
@ -208,9 +211,15 @@ bool system_cpu_support_sse42()
bool system_cpu_support_avx2()
{
/* F16C is considered part of AVX2 for our purpose, as all physical CPUs with
* AVX2 support also support it so there is no point having a separate kernel.
*
* Some cases where it might be missing is Rosetta or virtual machines, so we
* check for it just to be safe. */
CPUCapabilities &caps = system_cpu_capabilities();
return caps.avx2;
return caps.avx2 && caps.f16c;
}
#else
bool system_cpu_support_sse42()

View file

@ -1,3 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:be955bdb956bf643b5f2a947bdc15c74a256669f52802b9a5acbba6bc0f867d6
size 35635
oid sha256:5dd8955b116a94be88ce81be906af659e2fee3b6daafa1c1e0c2445deff1ae8a
size 37092

View file

@ -122,7 +122,6 @@ if platform.system() == "Darwin":
BLOCKLIST_GPU = [
# Uninvestigated differences with GPU.
'image_log.blend',
'glass_mix_40964.blend',
'filter_glossy_refraction_45609.blend',
'bevel_mblur.blend',