mirror of
https://github.com/blender/blender
synced 2026-09-29 04:37:17 +03:00
This change adds 32 more bit to store kernel features. While for a short term it might be possible to make a space for one or two extra bits, it seems going 64bit is inevitable. Expanding the field to 64bit might introduce some slowdown due to less optimal cache, but so is consolidation of existing flags could also lead to performance drop in certain configurations. The main tricky part of the change is Metal where function constants are used to store kernel_features, and 64bit constants are only available on macOS 12. There is a runtime check for it. On older macOS versions the flags are stored as a pair of 32bit values. It is slower, but there are unlikely to be many Cycles users on macOS 11. Ref #159470 Pull Request: https://projects.blender.org/blender/blender/pulls/162737
138 lines
3.4 KiB
C++
138 lines
3.4 KiB
C++
/* SPDX-FileCopyrightText: 2011-2022 Blender Foundation
|
|
*
|
|
* SPDX-License-Identifier: Apache-2.0 */
|
|
|
|
#pragma once
|
|
|
|
#include "kernel/globals.h"
|
|
|
|
#include "kernel/integrator/path_state.h"
|
|
|
|
#include "kernel/bvh/bvh.h"
|
|
|
|
#include "kernel/sample/mapping.h"
|
|
|
|
#include "kernel/svm/node_types.h"
|
|
#include "kernel/svm/util.h"
|
|
|
|
CCL_NAMESPACE_BEGIN
|
|
|
|
#ifdef __SHADER_RAYTRACE__
|
|
|
|
# ifdef __KERNEL_OPTIX__
|
|
extern "C" __device__ float __direct_callable__svm_node_ao(
|
|
# else
|
|
ccl_device float svm_ao(
|
|
# endif
|
|
KernelGlobals kg,
|
|
ConstIntegratorState state,
|
|
ccl_private ShaderData *sd,
|
|
float3 N,
|
|
float max_dist,
|
|
const int num_samples,
|
|
const int flags)
|
|
{
|
|
if (flags & NODE_AO_GLOBAL_RADIUS) {
|
|
max_dist = kernel_data.integrator.ao_bounces_distance;
|
|
}
|
|
|
|
/* Early out if no sampling needed. */
|
|
if (max_dist <= 0.0f || num_samples < 1 || sd->object == OBJECT_NONE) {
|
|
return 1.0f;
|
|
}
|
|
|
|
/* Can't ray-trace from shaders like displacement, before BVH exists. */
|
|
if (kernel_data.bvh.bvh_layout == BVH_LAYOUT_NONE) {
|
|
return 1.0f;
|
|
}
|
|
|
|
if (flags & NODE_AO_INSIDE) {
|
|
N = -N;
|
|
}
|
|
|
|
float3 T;
|
|
float3 B;
|
|
make_orthonormals(N, &T, &B);
|
|
|
|
/* TODO: support ray-tracing in shadow shader evaluation? */
|
|
RNGState rng_state;
|
|
path_state_rng_load(state, &rng_state);
|
|
|
|
int unoccluded = 0;
|
|
for (int sample = 0; sample < num_samples; sample++) {
|
|
const float2 rand_disk = path_branched_rng_2D(
|
|
kg, &rng_state, sample, num_samples, PRNG_SURFACE_AO);
|
|
|
|
const float2 d = sample_uniform_disk(rand_disk);
|
|
const float3 D = make_float3(d.x, d.y, safe_sqrtf(1.0f - dot(d, d)));
|
|
|
|
/* Create ray. */
|
|
Ray ray;
|
|
ray.P = sd->P;
|
|
ray.D = to_global(D, T, B, N);
|
|
ray.tmin = 0.0f;
|
|
ray.tmax = max_dist;
|
|
ray.time = sd->time;
|
|
ray.self.object = sd->object;
|
|
ray.self.prim = sd->prim;
|
|
ray.self.light_object = OBJECT_NONE;
|
|
ray.self.light_prim = PRIM_NONE;
|
|
ray.dP = differential_zero_compact();
|
|
ray.dD = differential_zero_compact();
|
|
|
|
if (flags & NODE_AO_ONLY_LOCAL) {
|
|
if (!scene_intersect_local(kg, &ray, nullptr, sd->object, nullptr, 0)) {
|
|
unoccluded++;
|
|
}
|
|
}
|
|
else {
|
|
if (!scene_intersect_shadow(kg, &ray, PATH_RAY_VISIBILITY_SHADOW_OPAQUE)) {
|
|
unoccluded++;
|
|
}
|
|
}
|
|
}
|
|
|
|
return ((float)unoccluded) / num_samples;
|
|
}
|
|
|
|
template<uint64_t node_feature_mask, typename ConstIntegratorGenericState>
|
|
# if defined(__KERNEL_OPTIX__)
|
|
ccl_device_inline
|
|
# else
|
|
ccl_device_noinline
|
|
# endif
|
|
void
|
|
svm_node_ao(KernelGlobals kg,
|
|
ConstIntegratorGenericState state,
|
|
ccl_private ShaderData *sd,
|
|
ccl_private float *ccl_restrict stack,
|
|
const ccl_global SVMNodeAmbientOcclusion &ccl_restrict node)
|
|
{
|
|
float ao = 1.0f;
|
|
|
|
IF_KERNEL_NODES_FEATURE(RAYTRACE)
|
|
{
|
|
float dist = stack_load(stack, node.dist);
|
|
float3 normal = stack_load_float3_default(stack, node.normal_offset, sd->N);
|
|
normal = safe_normalize(normal);
|
|
|
|
# ifdef __KERNEL_OPTIX__
|
|
ao = optixDirectCall<float>(0, kg, state, sd, normal, dist, node.samples, node.flags);
|
|
# else
|
|
ao = svm_ao(kg, state, sd, normal, dist, node.samples, node.flags);
|
|
# endif
|
|
}
|
|
|
|
if (stack_valid(node.out_ao_offset)) {
|
|
stack_store_float(stack, node.out_ao_offset, ao);
|
|
}
|
|
|
|
if (stack_valid(node.out_color_offset)) {
|
|
const float3 color = stack_load(stack, node.color);
|
|
stack_store_float3(stack, node.out_color_offset, ao * color);
|
|
}
|
|
}
|
|
|
|
#endif /* __SHADER_RAYTRACE__ */
|
|
|
|
CCL_NAMESPACE_END
|