mirror of
https://github.com/blender/blender
synced 2026-09-29 04:37:17 +03:00
Cycles: Half float denormals and optimizations
* Switch to more accurate half and float conversion supporting denormals to match native instructions. This makes CPU and GPU match more closely in some tests and avoids clipping some low values. * Inf and NaN are not supported still, as we already filter these out and there is no reason to have the overhead. * Change avx2 kernel to require f16c. For all physical CPUs avx2 implies f16c, and it's only for emulation and virtual machines that this would not be the case. So it's fine to fall back to the sse4.1 kernel then. * Assume half instructions are available with ARM NEON. There is no defined minimum architecture, but Blender assumes the same and ARMv8.2-A is relatively old. * For everything else there are SIMD optimized fallbacks. * New unit tests were added, coverting both native instructions and fallback implementations, and half/half3/half4. Fix #152763: Half float image low values are clipped Pull Request: https://projects.blender.org/blender/blender/pulls/154042
This commit is contained in:
parent
646f919cc9
commit
110de4f488
13 changed files with 510 additions and 141 deletions
|
|
@ -34,6 +34,7 @@ endif()
|
|||
if(WITH_CYCLES_NATIVE_ONLY)
|
||||
set(CXX_HAS_SSE42 FALSE)
|
||||
set(CXX_HAS_AVX2 FALSE)
|
||||
set(CXX_HAS_F16C FALSE)
|
||||
add_definitions(
|
||||
-DWITH_KERNEL_NATIVE
|
||||
)
|
||||
|
|
@ -71,16 +72,15 @@ elseif(SUPPORTS_NEON_BUILD AND SSE2NEON_FOUND)
|
|||
# Disable the CXX_HAS_* flags so we don't add any SSE or AVX compiler flags.
|
||||
set(CXX_HAS_SSE42 FALSE)
|
||||
set(CXX_HAS_AVX2 FALSE)
|
||||
set(CXX_HAS_F16C FALSE)
|
||||
elseif(WIN32 AND MSVC AND NOT CMAKE_CXX_COMPILER_ID MATCHES "Clang")
|
||||
set(CXX_HAS_SSE42 TRUE)
|
||||
set(CXX_HAS_AVX2 TRUE)
|
||||
set(CXX_HAS_F16C TRUE)
|
||||
|
||||
# /arch:AVX for VC2012 and above
|
||||
if(NOT MSVC_VERSION LESS 1700)
|
||||
set(CYCLES_AVX2_FLAGS "/arch:AVX /arch:AVX2")
|
||||
elseif(NOT CMAKE_CL_64)
|
||||
set(CYCLES_AVX2_FLAGS "/arch:SSE2")
|
||||
endif()
|
||||
# No separate f16c flag for MSVC.
|
||||
set(CYCLES_AVX2_FLAGS "/arch:AVX2")
|
||||
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS}")
|
||||
|
||||
# there is no /arch:SSE3, but intrinsics are available anyway
|
||||
if(CMAKE_CL_64)
|
||||
|
|
@ -91,34 +91,40 @@ elseif(WIN32 AND MSVC AND NOT CMAKE_CXX_COMPILER_ID MATCHES "Clang")
|
|||
elseif((CMAKE_C_COMPILER_ID STREQUAL "GNU") OR (CMAKE_CXX_COMPILER_ID MATCHES "Clang"))
|
||||
check_cxx_compiler_flag(-msse4.2 CXX_HAS_SSE42)
|
||||
check_cxx_compiler_flag(-mavx2 CXX_HAS_AVX2)
|
||||
check_cxx_compiler_flag(-mf16c CXX_HAS_F16C)
|
||||
|
||||
if(CXX_HAS_SSE42)
|
||||
set(CYCLES_SSE42_FLAGS "-msse -msse2 -msse3 -mssse3 -msse4.1 -msse4.2")
|
||||
if(CXX_HAS_AVX2)
|
||||
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
|
||||
set(CYCLES_AVX2_FLAGS "${CYCLES_SSE42_FLAGS} -mavx -mavx2 -mfma -mlzcnt -mbmi -mbmi2 -mf16c")
|
||||
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS} -mf16c")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
elseif(WIN32 AND CMAKE_CXX_COMPILER_ID STREQUAL "Intel")
|
||||
check_cxx_compiler_flag(/QxSSE4.2 CXX_HAS_SSE42)
|
||||
check_cxx_compiler_flag(/QxCORE-AVX2 CXX_HAS_AVX2)
|
||||
set(CXX_HAS_F16C CXX_HAS_AVX) # Same flag enables AVX, FMA and F16C
|
||||
|
||||
if(CXX_HAS_SSE42)
|
||||
set(CYCLES_SSE42_FLAGS "/QxSSE4.2")
|
||||
|
||||
if(CXX_HAS_AVX2)
|
||||
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
|
||||
set(CYCLES_AVX2_FLAGS "/QxCORE-AVX2")
|
||||
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS}")
|
||||
endif()
|
||||
endif()
|
||||
elseif(CMAKE_CXX_COMPILER_ID STREQUAL "Intel")
|
||||
check_cxx_compiler_flag(-xsse4.2 CXX_HAS_SSE42)
|
||||
check_cxx_compiler_flag(-xcore-avx2 CXX_HAS_AVX2)
|
||||
set(CXX_HAS_F16C CXX_HAS_AVX) # Same flag enables AVX, FMA and F16C
|
||||
|
||||
if(CXX_HAS_SSE42)
|
||||
set(CYCLES_SSE42_FLAGS "-xsse4.2")
|
||||
|
||||
if(CXX_HAS_AVX2)
|
||||
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
|
||||
set(CYCLES_AVX2_FLAGS "-xcore-avx2")
|
||||
set(CYCLES_AVX2_F16C_FLAGS "${CYCLES_AVX2_FLAGS}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
|
|
|||
|
|
@ -40,8 +40,8 @@ if(DEFINED CYCLES_KERNEL_FLAGS)
|
|||
set_source_files_properties(kernel.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_KERNEL_FLAGS}")
|
||||
endif()
|
||||
|
||||
if(CXX_HAS_AVX2)
|
||||
set_source_files_properties(kernel_avx2.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_FLAGS}")
|
||||
if(CXX_HAS_AVX2 AND CXX_HAS_F16C)
|
||||
set_source_files_properties(kernel_avx2.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_F16C_FLAGS}")
|
||||
endif()
|
||||
|
||||
# Warnings to avoid using doubles in the kernel.
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@
|
|||
#pragma once
|
||||
|
||||
#include "util/half.h"
|
||||
#include "util/math_int2.h"
|
||||
#include "util/types.h"
|
||||
#include "util/vector.h"
|
||||
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ set(SRC
|
|||
render_graph_finalize_test.cpp
|
||||
util_aligned_malloc_test.cpp
|
||||
util_boundbox_test.cpp
|
||||
util_half_test.cpp
|
||||
util_ies_test.cpp
|
||||
util_math_test.cpp
|
||||
util_math_fast_test.cpp
|
||||
|
|
@ -54,8 +55,10 @@ if(NOT APPLE)
|
|||
if(CXX_HAS_AVX2)
|
||||
list(APPEND SRC
|
||||
util_float8_avx2_test.cpp
|
||||
util_half_avx2_test.cpp
|
||||
)
|
||||
set_source_files_properties(util_float8_avx2_test.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_FLAGS}")
|
||||
set_source_files_properties(util_float8_avx2_test.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_F16C_FLAGS}")
|
||||
set_source_files_properties(util_half_avx2_test.cpp PROPERTIES COMPILE_FLAGS "${CYCLES_AVX2_F16C_FLAGS}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
|
|
|
|||
|
|
@ -12,7 +12,6 @@ CCL_NAMESPACE_BEGIN
|
|||
|
||||
static bool validate_cpu_capabilities()
|
||||
{
|
||||
|
||||
#if defined(__KERNEL_AVX2__)
|
||||
return system_cpu_support_avx2();
|
||||
#elif defined(__KERNEL_AVX__)
|
||||
|
|
|
|||
14
intern/cycles/test/util_half_avx2_test.cpp
Normal file
14
intern/cycles/test/util_half_avx2_test.cpp
Normal file
|
|
@ -0,0 +1,14 @@
|
|||
/* SPDX-FileCopyrightText: 2026 Blender Foundation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#define __KERNEL_SSE__
|
||||
#define __KERNEL_AVX__
|
||||
#define __KERNEL_AVX2__
|
||||
|
||||
#define TEST_CATEGORY_NAME util_half_avx2
|
||||
|
||||
#if (defined(i386) || defined(_M_IX86) || defined(__x86_64__) || defined(_M_X64)) && \
|
||||
defined(__AVX2__)
|
||||
# include "util_half_test.h"
|
||||
#endif
|
||||
6
intern/cycles/test/util_half_test.cpp
Normal file
6
intern/cycles/test/util_half_test.cpp
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
/* SPDX-FileCopyrightText: 2026 Blender Foundation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#define TEST_CATEGORY_NAME util_half
|
||||
#include "util_half_test.h"
|
||||
193
intern/cycles/test/util_half_test.h
Normal file
193
intern/cycles/test/util_half_test.h
Normal file
|
|
@ -0,0 +1,193 @@
|
|||
/* SPDX-FileCopyrightText: 2026 Blender Foundation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
|
||||
#include "util/half.h"
|
||||
#include "util/system.h"
|
||||
#include <array>
|
||||
|
||||
CCL_NAMESPACE_BEGIN
|
||||
|
||||
static bool validate_cpu_capabilities()
|
||||
{
|
||||
#if defined(__KERNEL_AVX2__)
|
||||
return system_cpu_support_avx2();
|
||||
#else
|
||||
return true;
|
||||
#endif
|
||||
}
|
||||
|
||||
#define EXPECT_HALF_EQ(a, b) \
|
||||
do { \
|
||||
const uint16_t a_ = (a); \
|
||||
const uint16_t b_ = (b); \
|
||||
if ((a_ & 0x7fff) == 0 && (b_ & 0x7fff) == 0) { \
|
||||
/* Ignore sign of zero, -0.0 getting converted to 0.0 is fine. */ \
|
||||
} \
|
||||
else { \
|
||||
EXPECT_EQ(a_, b_); \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
struct HalfTestVal {
|
||||
float f;
|
||||
uint16_t h;
|
||||
};
|
||||
|
||||
/* We don't support NaN and inf, so not tested here. */
|
||||
static const std::array<HalfTestVal, 13> test_values = {{{0.0f, 0x0000},
|
||||
{-0.0f, 0x8000},
|
||||
{1.0f, 0x3c00},
|
||||
{2.0f, 0x4000},
|
||||
{10.0f, 0x4900},
|
||||
{100.0f, 0x5640},
|
||||
{0.0999755859375f, 0x2e66},
|
||||
{-0.0999755859375f, 0xae66},
|
||||
{-0.5f, 0xb800},
|
||||
{-1.0f, 0xbc00},
|
||||
/* Smallest positive normal (2^-14) */
|
||||
{6.103515625e-5f, 0x0400},
|
||||
/* Largest denormal (2^-14 - 2^-24) */
|
||||
{6.0975551605e-5f, 0x03ff},
|
||||
/* Smallest positive denormal (2^-24) */
|
||||
{5.9604644775e-8f, 0x0001}}};
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, float_to_half)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
return;
|
||||
}
|
||||
for (const HalfTestVal &val : test_values) {
|
||||
EXPECT_HALF_EQ(float_to_half(val.f), val.h);
|
||||
EXPECT_EQ(half_to_float(half(val.h)), val.f);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, float3_to_half3)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
return;
|
||||
}
|
||||
for (size_t i = 0; i < test_values.size(); i++) {
|
||||
const size_t i0 = i;
|
||||
const size_t i1 = (i + 1) % test_values.size();
|
||||
const size_t i2 = (i + 2) % test_values.size();
|
||||
|
||||
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
|
||||
|
||||
const half3 h = float3_to_half3(in);
|
||||
const float3 out = half3_to_float3(h);
|
||||
|
||||
EXPECT_EQ(out.x, in.x);
|
||||
EXPECT_EQ(out.y, in.y);
|
||||
EXPECT_EQ(out.z, in.z);
|
||||
|
||||
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, float4_to_half4)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
return;
|
||||
}
|
||||
for (size_t i = 0; i < test_values.size(); i += 4) {
|
||||
const size_t i0 = i;
|
||||
const size_t i1 = (i + 1) % test_values.size();
|
||||
const size_t i2 = (i + 2) % test_values.size();
|
||||
const size_t i3 = (i + 3) % test_values.size();
|
||||
|
||||
const float4 in = make_float4(
|
||||
test_values[i0].f, test_values[i1].f, test_values[i2].f, test_values[i3].f);
|
||||
|
||||
const half4 h = float4_to_half4(in);
|
||||
const float4 out = half4_to_float4(h);
|
||||
|
||||
EXPECT_EQ(out.x, in.x);
|
||||
EXPECT_EQ(out.y, in.y);
|
||||
EXPECT_EQ(out.z, in.z);
|
||||
EXPECT_EQ(out.w, in.w);
|
||||
|
||||
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.w), test_values[i3].h);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, fallback_float_to_half)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
return;
|
||||
}
|
||||
for (const HalfTestVal &val : test_values) {
|
||||
EXPECT_HALF_EQ(fallback_float_to_half(val.f), val.h);
|
||||
EXPECT_EQ(fallback_half_to_float(half(val.h)), val.f);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, fallback_float3_to_half3)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
return;
|
||||
}
|
||||
for (size_t i = 0; i < test_values.size(); i++) {
|
||||
const size_t i0 = i;
|
||||
const size_t i1 = (i + 1) % test_values.size();
|
||||
const size_t i2 = (i + 2) % test_values.size();
|
||||
|
||||
const float3 in = make_float3(test_values[i0].f, test_values[i1].f, test_values[i2].f);
|
||||
|
||||
const half3 h = fallback_float3_to_half3(in);
|
||||
const float3 out = fallback_half3_to_float3(h);
|
||||
|
||||
EXPECT_EQ(out.x, in.x);
|
||||
EXPECT_EQ(out.y, in.y);
|
||||
EXPECT_EQ(out.z, in.z);
|
||||
|
||||
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
|
||||
}
|
||||
}
|
||||
|
||||
TEST(TEST_CATEGORY_NAME, fallback_float4_to_half4)
|
||||
{
|
||||
if (!validate_cpu_capabilities()) {
|
||||
GTEST_SKIP();
|
||||
return;
|
||||
}
|
||||
for (size_t i = 0; i < test_values.size(); i += 4) {
|
||||
const size_t i0 = i;
|
||||
const size_t i1 = (i + 1) % test_values.size();
|
||||
const size_t i2 = (i + 2) % test_values.size();
|
||||
const size_t i3 = (i + 3) % test_values.size();
|
||||
|
||||
const float4 in = make_float4(
|
||||
test_values[i0].f, test_values[i1].f, test_values[i2].f, test_values[i3].f);
|
||||
|
||||
const half4 h = fallback_float4_to_half4(in);
|
||||
const float4 out = fallback_half4_to_float4(h);
|
||||
|
||||
EXPECT_EQ(out.x, in.x);
|
||||
EXPECT_EQ(out.y, in.y);
|
||||
EXPECT_EQ(out.z, in.z);
|
||||
EXPECT_EQ(out.w, in.w);
|
||||
|
||||
EXPECT_HALF_EQ(uint16_t(h.x), test_values[i0].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.y), test_values[i1].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.z), test_values[i2].h);
|
||||
EXPECT_HALF_EQ(uint16_t(h.w), test_values[i3].h);
|
||||
}
|
||||
}
|
||||
|
||||
CCL_NAMESPACE_END
|
||||
|
|
@ -1,184 +1,316 @@
|
|||
/* SPDX-FileCopyrightText: 2011-2022 Blender Foundation
|
||||
/* SPDX-FileCopyrightText: 2011-2026 Blender Foundation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "util/math.h"
|
||||
#include "util/types.h"
|
||||
#include "util/defines.h"
|
||||
#include "util/math_base.h"
|
||||
#include "util/math_float4.h"
|
||||
#include "util/math_int4.h"
|
||||
#include "util/types_base.h"
|
||||
#include "util/types_float4.h"
|
||||
#include "util/types_int4.h"
|
||||
#include "util/types_uint4.h"
|
||||
|
||||
#if !defined(__KERNEL_GPU__) && defined(__KERNEL_SSE2__)
|
||||
# include "util/simd.h" // IWYU pragma: keep
|
||||
# include "util/optimization.h" // IWYU pragma: keep
|
||||
# include "util/simd.h" // IWYU pragma: keep
|
||||
#endif
|
||||
|
||||
CCL_NAMESPACE_BEGIN
|
||||
|
||||
/* Half Floats */
|
||||
|
||||
#if defined(__KERNEL_METAL__)
|
||||
|
||||
ccl_device_inline float half_to_float(half h_in)
|
||||
{
|
||||
float f;
|
||||
union {
|
||||
half h;
|
||||
uint16_t s;
|
||||
} val;
|
||||
val.h = h_in;
|
||||
|
||||
*((ccl_private int *)&f) = ((val.s & 0x8000) << 16) | (((val.s & 0x7c00) + 0x1C000) << 13) |
|
||||
((val.s & 0x03FF) << 13);
|
||||
|
||||
return f;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
/* CUDA has its own half data type, no need to define then */
|
||||
# if !defined(__KERNEL_CUDA__) && !defined(__KERNEL_HIP__) && !defined(__KERNEL_ONEAPI__)
|
||||
/* Implementing this as a class rather than a typedef so that the compiler can tell it apart from
|
||||
* unsigned shorts. */
|
||||
#if !defined(__KERNEL_GPU__)
|
||||
/* GPUs have native support for this type.
|
||||
* Implementing this as a class rather than a typedef so that the compiler can tell it apart from
|
||||
* uint16_ts. */
|
||||
class half {
|
||||
public:
|
||||
half() = default;
|
||||
half(const unsigned short &i) : v(i) {}
|
||||
operator unsigned short() const
|
||||
half(const uint16_t &i) : v(i) {}
|
||||
operator uint16_t() const
|
||||
{
|
||||
return v;
|
||||
}
|
||||
half &operator=(const unsigned short &i)
|
||||
half &operator=(const uint16_t &i)
|
||||
{
|
||||
v = i;
|
||||
return *this;
|
||||
}
|
||||
|
||||
private:
|
||||
unsigned short v;
|
||||
uint16_t v;
|
||||
};
|
||||
#endif
|
||||
|
||||
#if !defined(__KERNEL_METAL__)
|
||||
struct half3 {
|
||||
half x, y, z;
|
||||
};
|
||||
# endif
|
||||
|
||||
struct half4 {
|
||||
half x, y, z, w;
|
||||
};
|
||||
#endif
|
||||
|
||||
/* Conversion to/from half float for image textures
|
||||
#if !defined(__KERNEL_GPU__)
|
||||
/* Optimized fallback implementations with fast path for normal and denormal numbers, assuming
|
||||
* no Infs or NaNs. Based on public domain functions from.
|
||||
*
|
||||
* Simplified float to half for fast sampling on processor without a native
|
||||
* instruction, and eliminating any NaN and inf values. */
|
||||
* https://fgiesen.wordpress.com/2012/03/28/half-to-float-done-quic/
|
||||
* https://gist.github.com/rygorous/2144712
|
||||
* https://gist.github.com/rygorous/2156668
|
||||
* https://gist.github.com/rygorous/4d9e9e88cab13c703773dc767a23575f
|
||||
*/
|
||||
ccl_device_inline float fallback_half_to_float(const half h)
|
||||
{
|
||||
const uint32_t bits = uint16_t(h);
|
||||
const uint32_t s = (bits & 0x8000) << 16;
|
||||
const uint32_t em = (bits & 0x7fff) << 13;
|
||||
const float f = __int_as_float(em) * __int_as_float(0x77800000 /* 2^112 */);
|
||||
return __int_as_float(__float_as_uint(f) | s);
|
||||
}
|
||||
|
||||
ccl_device_inline half float_to_half_image(const float f)
|
||||
ccl_device_inline float4 fallback_half4_to_float4(const half4 h)
|
||||
{
|
||||
const int4 i = make_int4(uint16_t(h.x), uint16_t(h.y), uint16_t(h.z), uint16_t(h.w));
|
||||
const int4 s = (i & 0x8000) << 16;
|
||||
const int4 em = (i & 0x7fff) << 13;
|
||||
const float4 f = cast(em) * __int_as_float(0x77800000 /* 2^112 */);
|
||||
return cast(cast(f) | s);
|
||||
}
|
||||
|
||||
ccl_device_inline float3 fallback_half3_to_float3(const half3 h)
|
||||
{
|
||||
return make_float3(fallback_half4_to_float4({h.x, h.y, h.z, 0}));
|
||||
}
|
||||
|
||||
ccl_device_inline half fallback_float_to_half(const float f)
|
||||
{
|
||||
const int c_f16max = (127 + 16) << 23;
|
||||
const int c_infty_as_fp16 = 0x7c00;
|
||||
const int c_min_normal = (127 - 14) << 23;
|
||||
const int c_denorm_magic = ((127 - 15) + (23 - 10) + 1) << 23;
|
||||
const int c_normal_bias = 0xfff - ((127 - 15) << 23);
|
||||
|
||||
const uint f_i = __float_as_uint(f);
|
||||
const uint sign_i = f_i & 0x80000000u;
|
||||
const int abs_i = int(f_i ^ sign_i);
|
||||
|
||||
uint16_t res;
|
||||
|
||||
if (abs_i >= c_f16max) {
|
||||
/* Overflows to infinity. */
|
||||
res = uint16_t(c_infty_as_fp16);
|
||||
}
|
||||
else if (abs_i < c_min_normal) {
|
||||
/* Denormal. */
|
||||
float denorm_f = __uint_as_float(uint(abs_i));
|
||||
denorm_f += __int_as_float(c_denorm_magic);
|
||||
const int denorm_i = int(__float_as_uint(denorm_f)) - c_denorm_magic;
|
||||
res = uint16_t(denorm_i);
|
||||
}
|
||||
else {
|
||||
/* Normal. */
|
||||
const int mant_odd = int(uint(abs_i) >> 13) & 1;
|
||||
res = uint16_t((abs_i + c_normal_bias + mant_odd) >> 13);
|
||||
}
|
||||
|
||||
return half(res | uint16_t(sign_i >> 16));
|
||||
}
|
||||
|
||||
ccl_device_inline half4 fallback_float4_to_half4(const float4 f)
|
||||
{
|
||||
const int4 c_f16max = make_int4((127 + 16) << 23);
|
||||
const int4 c_infty_as_fp16 = make_int4(0x7c00);
|
||||
const int4 c_min_normal = make_int4((127 - 14) << 23);
|
||||
const int4 c_denorm_magic = make_int4(((127 - 15) + (23 - 10) + 1) << 23);
|
||||
const int4 c_normal_bias = make_int4(0xfff - ((127 - 15) << 23));
|
||||
|
||||
const float4 abs_f = fabs(f);
|
||||
const int4 abs_i = __float4_as_int4(abs_f);
|
||||
|
||||
const int4 b_isregular = c_f16max > abs_i;
|
||||
const int4 b_isdenorm = c_min_normal > abs_i;
|
||||
|
||||
/* Denormal. */
|
||||
const float4 denorm_f = abs_f + __int4_as_float4(c_denorm_magic);
|
||||
const int4 denorm_i = __float4_as_int4(denorm_f) - c_denorm_magic;
|
||||
|
||||
/* Normal. */
|
||||
const int4 mant_odd = (abs_i << (31 - 13)) >> 31;
|
||||
const int4 normal = srl(abs_i + c_normal_bias - mant_odd, 13);
|
||||
|
||||
/* Combined normal and denormal. */
|
||||
const int4 nonspecial = select(b_isdenorm, denorm_i, normal);
|
||||
|
||||
/* Combine overflow to infinity. */
|
||||
const int4 combined = select(b_isregular, nonspecial, c_infty_as_fp16);
|
||||
|
||||
const int4 sign_i = __float4_as_int4(f ^ abs_f);
|
||||
const int4 res = combined | (sign_i >> 16);
|
||||
|
||||
return {
|
||||
half(uint16_t(res.x)), half(uint16_t(res.y)), half(uint16_t(res.z)), half(uint16_t(res.w))};
|
||||
}
|
||||
|
||||
ccl_device_inline half3 fallback_float3_to_half3(const float3 f)
|
||||
{
|
||||
const half4 h = fallback_float4_to_half4(make_float4(f));
|
||||
return {h.x, h.y, h.z};
|
||||
}
|
||||
#endif
|
||||
|
||||
ccl_device_inline float half_to_float(half h)
|
||||
{
|
||||
#if defined(__KERNEL_METAL__) || defined(__KERNEL_ONEAPI__)
|
||||
return half(min(f, 65504.0f));
|
||||
return float(h);
|
||||
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
|
||||
return __float2half(min(f, 65504.0f));
|
||||
return __half2float(h);
|
||||
/* We assume half instructions are always supported when there is ARM Neon,
|
||||
* which implies ARMv8.2-A+. There is no official Blender minimum, but is
|
||||
* already assumed elsewhere in Blender and not that recent. */
|
||||
#elif defined(__ARM_NEON) || defined(_M_ARM64)
|
||||
uint16x4_t v = vdup_n_u16(uint16_t(h));
|
||||
return vgetq_lane_f32(vcvt_f32_f16(vreinterpret_f16_u16(v)), 0);
|
||||
#elif defined(__F16C__)
|
||||
return _cvtsh_ss(uint16_t(h));
|
||||
#else
|
||||
const uint u = __float_as_uint(f);
|
||||
/* Sign bit, shifted to its position. */
|
||||
uint sign_bit = u & 0x80000000;
|
||||
sign_bit >>= 16;
|
||||
/* Exponent. */
|
||||
const uint exponent_bits = u & 0x7f800000;
|
||||
/* Non-sign bits. */
|
||||
uint value_bits = u & 0x7fffffff;
|
||||
value_bits >>= 13; /* Align mantissa on MSB. */
|
||||
value_bits -= 0x1c000; /* Adjust bias. */
|
||||
/* Flush-to-zero. */
|
||||
value_bits = (exponent_bits < 0x38800000) ? 0 : value_bits;
|
||||
/* Clamp-to-max. */
|
||||
value_bits = (exponent_bits > 0x47000000) ? 0x7bff : value_bits;
|
||||
/* Denormals-as-zero. */
|
||||
value_bits = (exponent_bits == 0 ? 0 : value_bits);
|
||||
/* Re-insert sign bit and return. */
|
||||
return (value_bits | sign_bit);
|
||||
/* The fallback is fast so don't bother with native instructions. */
|
||||
return fallback_half_to_float(h);
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline half float_to_half(const float f)
|
||||
{
|
||||
#if defined(__KERNEL_METAL__) || defined(__KERNEL_ONEAPI__)
|
||||
return half(f);
|
||||
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
|
||||
return __float2half(f);
|
||||
#elif defined(__ARM_NEON) || defined(_M_ARM64)
|
||||
return half(vget_lane_u16(vreinterpret_u16_f16(vcvt_f16_f32(vdupq_n_f32(f))), 0));
|
||||
#elif defined(__F16C__)
|
||||
return half((uint16_t)_cvtss_sh(f, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
|
||||
#else
|
||||
return fallback_float_to_half(f);
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline half4 float4_to_half4(const float4 f)
|
||||
{
|
||||
#if defined(__KERNEL_METAL__)
|
||||
return {half(f.x), half(f.y), half(f.z), half(f.w)};
|
||||
#elif defined(__KERNEL_ONEAPI__)
|
||||
return {half(f.x), half(f.y), half(f.z), half(f.w)};
|
||||
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
|
||||
return {__float2half(f.x), __float2half(f.y), __float2half(f.z), __float2half(f.w)};
|
||||
#elif defined(__ARM_NEON) || defined(_M_ARM64)
|
||||
half4 r;
|
||||
vst1_u16(reinterpret_cast<uint16_t *>(&r),
|
||||
vreinterpret_u16_f16(vcvt_f16_f32(float32x4_t{f.x, f.y, f.z, f.w})));
|
||||
return r;
|
||||
#elif defined(__F16C__)
|
||||
half4 r;
|
||||
const __m128i h = _mm_cvtps_ph(_mm_loadu_ps(&f.x),
|
||||
_MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC);
|
||||
_mm_storel_epi64(reinterpret_cast<__m128i *>(&r), h);
|
||||
return r;
|
||||
#else
|
||||
return fallback_float4_to_half4(f);
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline float4 half4_to_float4(const half4 h)
|
||||
{
|
||||
#if defined(__KERNEL_METAL__)
|
||||
return {float(h.x), float(h.y), float(h.z), float(h.w)};
|
||||
#elif defined(__KERNEL_ONEAPI__)
|
||||
return {float(h.x), float(h.y), float(h.z), float(h.w)};
|
||||
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
|
||||
return {__half2float(h.x), __half2float(h.y), __half2float(h.z), __half2float(h.w)};
|
||||
#elif defined(__ARM_NEON) || defined(_M_ARM64)
|
||||
float4 r;
|
||||
vst1q_f32(&r.x,
|
||||
vcvt_f32_f16(vreinterpret_f16_u16(vld1_u16(reinterpret_cast<const uint16_t *>(&h)))));
|
||||
return r;
|
||||
#elif defined(__F16C__)
|
||||
float4 r;
|
||||
_mm_storeu_ps(&r.x, _mm_cvtph_ps(_mm_loadl_epi64(reinterpret_cast<const __m128i *>(&h))));
|
||||
return r;
|
||||
#else
|
||||
return fallback_half4_to_float4(h);
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline half3 float3_to_half3(const float3 f)
|
||||
{
|
||||
#if defined(__KERNEL_GPU__)
|
||||
return {float_to_half(f.x), float_to_half(f.y), float_to_half(f.z)};
|
||||
#elif defined(__ARM_NEON) || defined(_M_ARM64)
|
||||
const uint16x4_t h = vreinterpret_u16_f16(vcvt_f16_f32(float32x4_t{f.x, f.y, f.z, 0.0f}));
|
||||
return {half(vget_lane_u16(h, 0)), half(vget_lane_u16(h, 1)), half(vget_lane_u16(h, 2))};
|
||||
#elif defined(__F16C__)
|
||||
const __m128i h = _mm_cvtps_ph(_mm_set_ps(0.0f, f.z, f.y, f.x),
|
||||
_MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC);
|
||||
return {half(uint16_t(_mm_extract_epi16(h, 0))),
|
||||
half(uint16_t(_mm_extract_epi16(h, 1))),
|
||||
half(uint16_t(_mm_extract_epi16(h, 2)))};
|
||||
#else
|
||||
half4 h = fallback_float4_to_half4(make_float4(f));
|
||||
return {h.x, h.y, h.z};
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline float3 half3_to_float3(const half3 h)
|
||||
{
|
||||
#if defined(__KERNEL_GPU__)
|
||||
return make_float3(half_to_float(h.x), half_to_float(h.y), half_to_float(h.z));
|
||||
#elif defined(__ARM_NEON) || defined(_M_ARM64)
|
||||
const float32x4_t f = vcvt_f32_f16(
|
||||
vreinterpret_f16_u16(uint16x4_t{uint16_t(h.x), uint16_t(h.y), uint16_t(h.z), 0}));
|
||||
return make_float3(vgetq_lane_f32(f, 0), vgetq_lane_f32(f, 1), vgetq_lane_f32(f, 2));
|
||||
#elif defined(__F16C__)
|
||||
const __m128i v = _mm_set_epi16(0, 0, 0, 0, 0, uint16_t(h.z), uint16_t(h.y), uint16_t(h.x));
|
||||
const __m128 f = _mm_cvtph_ps(v);
|
||||
return make_float3(float4(f));
|
||||
#else
|
||||
return make_float3(fallback_half4_to_float4({h.x, h.y, h.z, 0}));
|
||||
#endif
|
||||
}
|
||||
|
||||
/* For image textutes */
|
||||
ccl_device_inline half float_to_half_image(const float f)
|
||||
{
|
||||
return float_to_half(clamp(f, -65504.0f, 65504.0f));
|
||||
}
|
||||
|
||||
ccl_device_inline float half_to_float_image(half h)
|
||||
{
|
||||
#if defined(__KERNEL_METAL__)
|
||||
return half_to_float(h);
|
||||
#elif defined(__KERNEL_ONEAPI__)
|
||||
return float(h);
|
||||
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
|
||||
return __half2float(h);
|
||||
#else
|
||||
const int x = ((h & 0x8000) << 16) | (((h & 0x7c00) + 0x1C000) << 13) | ((h & 0x03FF) << 13);
|
||||
return __int_as_float(x);
|
||||
#endif
|
||||
}
|
||||
|
||||
ccl_device_inline float4 half4_to_float4_image(const half4 h)
|
||||
{
|
||||
/* Unable to use because it gives different results half_to_float_image, can we
|
||||
* modify float_to_half_image so the conversion results are identical? */
|
||||
#if 0 /* defined(__KERNEL_AVX2__) */
|
||||
/* CPU: AVX. */
|
||||
__m128i x = _mm_castpd_si128(_mm_load_sd((const double *)&h));
|
||||
return float4(_mm_cvtph_ps(x));
|
||||
#endif
|
||||
|
||||
const float4 f = make_float4(half_to_float_image(h.x),
|
||||
half_to_float_image(h.y),
|
||||
half_to_float_image(h.z),
|
||||
half_to_float_image(h.w));
|
||||
return f;
|
||||
return half4_to_float4(h);
|
||||
}
|
||||
|
||||
/* Conversion to half float texture for display.
|
||||
*
|
||||
* Simplified float to half for fast display texture conversion on processors
|
||||
* without a native instruction. Assumes no negative, no NaN, no inf, and sets
|
||||
* denormal to 0. */
|
||||
|
||||
/* For render display */
|
||||
ccl_device_inline half float_to_half_display(const float f)
|
||||
{
|
||||
#if defined(__KERNEL_METAL__) || defined(__KERNEL_ONEAPI__)
|
||||
return half(min(f, 65504.0f));
|
||||
#elif defined(__KERNEL_CUDA__) || defined(__KERNEL_HIP__)
|
||||
return __float2half(min(f, 65504.0f));
|
||||
#else
|
||||
const int x = __float_as_int((f > 0.0f) ? ((f < 65504.0f) ? f : 65504.0f) : 0.0f);
|
||||
const int absolute = x & 0x7FFFFFFF;
|
||||
const int Z = absolute + 0xC8000000;
|
||||
const int result = (absolute < 0x38800000) ? 0 : Z;
|
||||
const int rshift = (result >> 13);
|
||||
return (rshift & 0x7FFF);
|
||||
#endif
|
||||
return float_to_half(clamp(f, 0.0f, 65504.0f));
|
||||
}
|
||||
|
||||
ccl_device_inline half4 float4_to_half4_display(const float4 f)
|
||||
{
|
||||
#ifdef __KERNEL_SSE__
|
||||
/* CPU: SSE and AVX. */
|
||||
const float4 x = min(max(f, make_float4(0.0f)), make_float4(65504.0f));
|
||||
# ifdef __KERNEL_AVX2__
|
||||
int4 rpack = int4(_mm_cvtps_ph(x, 0));
|
||||
# else
|
||||
const int4 absolute = cast(x) & make_int4(0x7FFFFFFF);
|
||||
const int4 Z = absolute + make_int4(0xC8000000);
|
||||
const int4 result = andnot(absolute < make_int4(0x38800000), Z);
|
||||
int4 rshift = (result >> 13) & make_int4(0x7FFF);
|
||||
int4 rpack = int4(_mm_packs_epi32(rshift, rshift));
|
||||
# endif
|
||||
half4 h;
|
||||
_mm_storel_pi((__m64 *)&h, _mm_castsi128_ps(rpack));
|
||||
return h;
|
||||
#else
|
||||
/* GPU and scalar fallback. */
|
||||
const half4 h = {float_to_half_display(f.x),
|
||||
float_to_half_display(f.y),
|
||||
float_to_half_display(f.z),
|
||||
float_to_half_display(f.w)};
|
||||
return h;
|
||||
#endif
|
||||
return float4_to_half4(clamp(f, make_float4(0.0f), make_float4(65504.0f)));
|
||||
}
|
||||
|
||||
#ifndef __KERNEL_GPU__
|
||||
ccl_device_inline float half_is_finite(const half h)
|
||||
{
|
||||
const int exponent = (h >> 10) & 0x001f;
|
||||
const int exponent = (uint16_t(h) >> 10) & 0x001f;
|
||||
return exponent < 31;
|
||||
}
|
||||
#endif
|
||||
|
|
|
|||
|
|
@ -76,6 +76,11 @@ ccl_device_inline int4 operator<(const int4 a, const int b)
|
|||
return a < make_int4(b);
|
||||
}
|
||||
|
||||
ccl_device_inline int4 operator>(const int4 a, const int4 b)
|
||||
{
|
||||
return b < a;
|
||||
}
|
||||
|
||||
ccl_device_inline int4 operator==(const int4 a, const int4 b)
|
||||
{
|
||||
# ifdef __KERNEL_SSE__
|
||||
|
|
@ -197,12 +202,14 @@ ccl_device_inline int4 &operator>>=(int4 &a, const int32_t b)
|
|||
return a = a >> b;
|
||||
}
|
||||
|
||||
# ifdef __KERNEL_SSE__
|
||||
ccl_device_forceinline int4 srl(const int4 a, const int32_t b)
|
||||
ccl_device_inline int4 srl(const int4 a, const int i)
|
||||
{
|
||||
return int4(_mm_srli_epi32(a.m128, b));
|
||||
}
|
||||
# ifdef __KERNEL_SSE__
|
||||
return int4(_mm_srli_epi32(a.m128, i));
|
||||
# else
|
||||
return make_int4(uint32_t(a.x) >> i, uint32_t(a.y) >> i, uint32_t(a.z) >> i, uint32_t(a.w) >> i);
|
||||
# endif
|
||||
}
|
||||
|
||||
ccl_device_inline int4 min(const int4 a, const int4 b)
|
||||
{
|
||||
|
|
|
|||
|
|
@ -138,6 +138,7 @@ int system_cpu_bits()
|
|||
struct CPUCapabilities {
|
||||
bool sse42;
|
||||
bool avx2;
|
||||
bool f16c;
|
||||
};
|
||||
|
||||
static CPUCapabilities &system_cpu_capabilities()
|
||||
|
|
@ -184,6 +185,8 @@ static CPUCapabilities &system_cpu_capabilities()
|
|||
const bool avx = (xcr_feature_mask & 0x6) == 0x6;
|
||||
const bool f16c = (result[2] & ((int)1 << 29)) != 0;
|
||||
|
||||
caps.f16c = avx && f16c;
|
||||
|
||||
__cpuid(result, 0x00000007);
|
||||
bool bmi1 = (result[1] & ((int)1 << 3)) != 0;
|
||||
bool bmi2 = (result[1] & ((int)1 << 8)) != 0;
|
||||
|
|
@ -208,9 +211,15 @@ bool system_cpu_support_sse42()
|
|||
|
||||
bool system_cpu_support_avx2()
|
||||
{
|
||||
/* F16C is considered part of AVX2 for our purpose, as all physical CPUs with
|
||||
* AVX2 support also support it so there is no point having a separate kernel.
|
||||
*
|
||||
* Some cases where it might be missing is Rosetta or virtual machines, so we
|
||||
* check for it just to be safe. */
|
||||
CPUCapabilities &caps = system_cpu_capabilities();
|
||||
return caps.avx2;
|
||||
return caps.avx2 && caps.f16c;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
bool system_cpu_support_sse42()
|
||||
|
|
|
|||
|
|
@ -1,3 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:be955bdb956bf643b5f2a947bdc15c74a256669f52802b9a5acbba6bc0f867d6
|
||||
size 35635
|
||||
oid sha256:5dd8955b116a94be88ce81be906af659e2fee3b6daafa1c1e0c2445deff1ae8a
|
||||
size 37092
|
||||
|
|
|
|||
|
|
@ -122,7 +122,6 @@ if platform.system() == "Darwin":
|
|||
|
||||
BLOCKLIST_GPU = [
|
||||
# Uninvestigated differences with GPU.
|
||||
'image_log.blend',
|
||||
'glass_mix_40964.blend',
|
||||
'filter_glossy_refraction_45609.blend',
|
||||
'bevel_mblur.blend',
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue