mirror of
https://github.com/blender/blender
synced 2026-09-29 04:37:17 +03:00
Refactor: Cycles: Disable force inline in MNEE kernels
This appears to be causing problems for HIP. It's not clear that there is a need to force inline this rather than letting the compiler decide. The force inlining had been here since the initial MNEE implementation. Pull Request: https://projects.blender.org/blender/blender/pulls/153836
This commit is contained in:
parent
8f90dd3475
commit
6d6303f85c
1 changed files with 51 additions and 51 deletions
|
|
@ -117,14 +117,14 @@ ccl_device_inline float mat22_inverse(const float4 m, ccl_private float4 &m_inve
|
|||
}
|
||||
|
||||
/* Manifold vertex setup from ray and intersection data */
|
||||
ccl_device_forceinline void mnee_setup_manifold_vertex(KernelGlobals kg,
|
||||
ccl_private ManifoldVertex *vtx,
|
||||
ccl_private ShaderClosure *bsdf,
|
||||
const float eta,
|
||||
const float2 n_offset,
|
||||
const ccl_private Ray *ray,
|
||||
const ccl_private Intersection *isect,
|
||||
ccl_private ShaderData *sd_vtx)
|
||||
ccl_device_inline void mnee_setup_manifold_vertex(KernelGlobals kg,
|
||||
ccl_private ManifoldVertex *vtx,
|
||||
ccl_private ShaderClosure *bsdf,
|
||||
const float eta,
|
||||
const float2 n_offset,
|
||||
const ccl_private Ray *ray,
|
||||
const ccl_private Intersection *isect,
|
||||
ccl_private ShaderData *sd_vtx)
|
||||
{
|
||||
sd_vtx->object = (isect->object == OBJECT_NONE) ? kernel_data_fetch(prim_object, isect->prim) :
|
||||
isect->object;
|
||||
|
|
@ -251,7 +251,7 @@ ccl_device_forceinline void mnee_setup_manifold_vertex(KernelGlobals kg,
|
|||
* inlined). */
|
||||
__attribute__((noinline))
|
||||
#else
|
||||
ccl_device_forceinline
|
||||
ccl_device_inline
|
||||
#endif
|
||||
bool mnee_compute_constraint_derivatives(
|
||||
const int vertex_count,
|
||||
|
|
@ -373,9 +373,9 @@ bool mnee_compute_constraint_derivatives(
|
|||
* to use for specular manifold walk
|
||||
* (See for example http://faculty.washington.edu/finlayso/ebook/algebraic/advanced/LUtri.htm
|
||||
* for block tridiagonal matrix based linear system solve) */
|
||||
ccl_device_forceinline bool mnee_solve_matrix_h_to_x(const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices,
|
||||
ccl_private float2 *dx)
|
||||
ccl_device_inline bool mnee_solve_matrix_h_to_x(const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices,
|
||||
ccl_private float2 *dx)
|
||||
{
|
||||
float4 Li[MNEE_MAX_CAUSTIC_CASTERS];
|
||||
float2 C[MNEE_MAX_CAUSTIC_CASTERS];
|
||||
|
|
@ -408,13 +408,13 @@ ccl_device_forceinline bool mnee_solve_matrix_h_to_x(const int vertex_count,
|
|||
}
|
||||
|
||||
/* Newton solver to walk on specular manifold. */
|
||||
ccl_device_forceinline bool mnee_newton_solver(KernelGlobals kg,
|
||||
const ccl_private ShaderData *sd,
|
||||
ccl_private ShaderData *sd_vtx,
|
||||
const ccl_private LightSample *ls,
|
||||
const bool light_fixed_direction,
|
||||
const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices)
|
||||
ccl_device_inline bool mnee_newton_solver(KernelGlobals kg,
|
||||
const ccl_private ShaderData *sd,
|
||||
ccl_private ShaderData *sd_vtx,
|
||||
const ccl_private LightSample *ls,
|
||||
const bool light_fixed_direction,
|
||||
const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices)
|
||||
{
|
||||
float2 dx[MNEE_MAX_CAUSTIC_CASTERS];
|
||||
ManifoldVertex tentative[MNEE_MAX_CAUSTIC_CASTERS];
|
||||
|
|
@ -581,11 +581,11 @@ ccl_device_forceinline bool mnee_newton_solver(KernelGlobals kg,
|
|||
}
|
||||
|
||||
/* Sample bsdf in half-vector measure. */
|
||||
ccl_device_forceinline float2 mnee_sample_bsdf_dh(ClosureType type,
|
||||
const float alpha_x,
|
||||
const float alpha_y,
|
||||
const float sample_u,
|
||||
const float sample_v)
|
||||
ccl_device_inline float2 mnee_sample_bsdf_dh(ClosureType type,
|
||||
const float alpha_x,
|
||||
const float alpha_y,
|
||||
const float sample_u,
|
||||
const float sample_v)
|
||||
{
|
||||
float alpha2;
|
||||
float cos_phi;
|
||||
|
|
@ -626,10 +626,10 @@ ccl_device_forceinline float2 mnee_sample_bsdf_dh(ClosureType type,
|
|||
* We assume here that the pdf (in half-vector measure) is the same as
|
||||
* the one calculation when sampling the microfacet normals from the
|
||||
* specular chain above: this allows us to simplify the bsdf weight */
|
||||
ccl_device_forceinline Spectrum mnee_eval_bsdf_contribution(KernelGlobals kg,
|
||||
ccl_private ShaderClosure *closure,
|
||||
const float3 wi,
|
||||
const float3 wo)
|
||||
ccl_device_inline Spectrum mnee_eval_bsdf_contribution(KernelGlobals kg,
|
||||
ccl_private ShaderClosure *closure,
|
||||
const float3 wi,
|
||||
const float3 wo)
|
||||
{
|
||||
ccl_private MicrofacetBsdf *bsdf = (ccl_private MicrofacetBsdf *)closure;
|
||||
|
||||
|
|
@ -668,13 +668,13 @@ ccl_device_forceinline Spectrum mnee_eval_bsdf_contribution(KernelGlobals kg,
|
|||
}
|
||||
|
||||
/* Compute transfer matrix determinant |T1| = |dx1/dxn| (and |dh/dx| in the process) */
|
||||
ccl_device_forceinline bool mnee_compute_transfer_matrix(const ccl_private ShaderData *sd,
|
||||
const ccl_private LightSample *ls,
|
||||
const bool light_fixed_direction,
|
||||
const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices,
|
||||
ccl_private float *dx1_dxlight,
|
||||
ccl_private float *dh_dx)
|
||||
ccl_device_inline bool mnee_compute_transfer_matrix(const ccl_private ShaderData *sd,
|
||||
const ccl_private LightSample *ls,
|
||||
const bool light_fixed_direction,
|
||||
const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices,
|
||||
ccl_private float *dx1_dxlight,
|
||||
ccl_private float *dh_dx)
|
||||
{
|
||||
/* Simplified block tridiagonal LU factorization. */
|
||||
float4 Li;
|
||||
|
|
@ -796,15 +796,15 @@ ccl_device_forceinline bool mnee_compute_transfer_matrix(const ccl_private Shade
|
|||
}
|
||||
|
||||
/* Calculate the path contribution. */
|
||||
ccl_device_forceinline bool mnee_path_contribution(KernelGlobals kg,
|
||||
IntegratorState state,
|
||||
ccl_private ShaderData *sd,
|
||||
ccl_private ShaderData *sd_mnee,
|
||||
ccl_private LightSample *ls,
|
||||
const bool light_fixed_direction,
|
||||
const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices,
|
||||
ccl_private BsdfEval *throughput)
|
||||
ccl_device_inline bool mnee_path_contribution(KernelGlobals kg,
|
||||
IntegratorState state,
|
||||
ccl_private ShaderData *sd,
|
||||
ccl_private ShaderData *sd_mnee,
|
||||
ccl_private LightSample *ls,
|
||||
const bool light_fixed_direction,
|
||||
const int vertex_count,
|
||||
ccl_private ManifoldVertex *vertices,
|
||||
ccl_private BsdfEval *throughput)
|
||||
{
|
||||
float wo_len;
|
||||
float3 wo = normalize_len(vertices[0].p - sd->P, &wo_len);
|
||||
|
|
@ -935,13 +935,13 @@ ccl_device_forceinline bool mnee_path_contribution(KernelGlobals kg,
|
|||
}
|
||||
|
||||
/* Manifold next event estimation path sampling. */
|
||||
ccl_device_forceinline int kernel_path_mnee_sample(KernelGlobals kg,
|
||||
IntegratorState state,
|
||||
ccl_private ShaderData *sd,
|
||||
ccl_private ShaderData *sd_mnee,
|
||||
const ccl_private RNGState *rng_state,
|
||||
ccl_private LightSample *ls,
|
||||
ccl_private BsdfEval *throughput)
|
||||
ccl_device_inline int kernel_path_mnee_sample(KernelGlobals kg,
|
||||
IntegratorState state,
|
||||
ccl_private ShaderData *sd,
|
||||
ccl_private ShaderData *sd_mnee,
|
||||
const ccl_private RNGState *rng_state,
|
||||
ccl_private LightSample *ls,
|
||||
ccl_private BsdfEval *throughput)
|
||||
{
|
||||
/*
|
||||
* 1. send seed ray from shading point to light sample position (or along sampled light
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue