Cycles: Improve compilation time of tricubic interpolation

by using a loop instead of unrolling all 64 evaluations.
Seems that two nested loops has the best performance.

Compilation time measured on Metal M2 Ultra:
|                              Kernel| Before|  After|
|                                  --|     --|     --|
|             integrator_shade_volume|148.13s|114.71s|
|integrator_shade_volume_ray_marching| 44.30s| 14.27s|
|             integrator_shade_shadow| 87.83s| 58.82s|
|          shader_eval_volume_density| 32.69s|  6.63s|

Also added test file because we were not testing deterministic tricubic
interpolation before

Ref: #150119
This commit is contained in:
Weizhen Huang 2025-11-21 11:46:04 +01:00 • committed by Thomas Dinges
parent ffb0fc48e9
commit c0ad7c16dc
9 changed files with 48 additions and 38 deletions

View file

@ -24,6 +24,16 @@ namespace {
#endif
#ifdef WITH_NANOVDB
/* Cubic interpolation weights. */
ccl_device_inline void fill_cubic_weights(float3 w[4], float3 t)
{
w[0] = (((-1.0f / 6.0f) * t + 0.5f) * t - 0.5f) * t + (1.0f / 6.0f);
w[1] = ((0.5f * t - 1.0f) * t) * t + (2.0f / 3.0f);
w[2] = ((-0.5f * t + 0.5f) * t + 0.5f) * t + (1.0f / 6.0f);
w[3] = (1.0f / 6.0f) * t * t * t;
}
/* -------------------------------------------------------------------- */
/** Return the sample position for stochastical one-tap sampling.
* From "Stochastic Texture Filtering": https://arxiv.org/abs/2305.05810
@ -33,11 +43,8 @@ ccl_device_inline float3 interp_tricubic_stochastic(const float3 P, ccl_private
const float3 p = floor(P);
const float3 t = P - p;
/* Cubic interpolation weights. */
const float3 w[4] = {(((-1.0f / 6.0f) * t + 0.5f) * t - 0.5f) * t + (1.0f / 6.0f),
((0.5f * t - 1.0f) * t) * t + (2.0f / 3.0f),
((-0.5f * t + 0.5f) * t + 0.5f) * t + (1.0f / 6.0f),
(1.0f / 6.0f) * t * t * t};
float3 w[4];
fill_cubic_weights(w, t);
/* For reservoir sampling, always accept the first in the stream. */
float3 total_weight = w[0];
@ -113,43 +120,23 @@ ccl_device OutT kernel_tex_image_interp_tricubic_nanovdb(ccl_private Acc &acc, c
{
const float3 floor_P = floor(P);
const float3 t = P - floor_P;
const int3 index = make_int3(floor_P);
const int3 index = make_int3(floor_P) - make_int3(1);
const int xc[4] = {index.x - 1, index.x, index.x + 1, index.x + 2};
const int yc[4] = {index.y - 1, index.y, index.y + 1, index.y + 2};
const int zc[4] = {index.z - 1, index.z, index.z + 1, index.z + 2};
float u[4], v[4], w[4];
float3 w[4];
fill_cubic_weights(w, t);
/* Some helper macros to keep code size reasonable.
* Lets the compiler inline all the matrix multiplications.
*/
# define SET_CUBIC_SPLINE_WEIGHTS(u, t) \
{ \
u[0] = (((-1.0f / 6.0f) * t + 0.5f) * t - 0.5f) * t + (1.0f / 6.0f); \
u[1] = ((0.5f * t - 1.0f) * t) * t + (2.0f / 3.0f); \
u[2] = ((-0.5f * t + 0.5f) * t + 0.5f) * t + (1.0f / 6.0f); \
u[3] = (1.0f / 6.0f) * t * t * t; \
} \
(void)0
OutT result = make_zero<OutT>();
# define DATA(x, y, z) (OutT(acc.getValue(make_int3(xc[x], yc[y], zc[z]))))
# define COL_TERM(col, row) \
(v[col] * (u[0] * DATA(0, col, row) + u[1] * DATA(1, col, row) + u[2] * DATA(2, col, row) + \
u[3] * DATA(3, col, row)))
# define ROW_TERM(row) \
(w[row] * (COL_TERM(0, row) + COL_TERM(1, row) + COL_TERM(2, row) + COL_TERM(3, row)))
for (int k = 0; k < 4; k++) {
for (int j = 0; j < 4; j++) {
result += w[k].z * (w[j].y * (w[0].x * (OutT(acc.getValue(index + make_int3(0, j, k)))) +
w[1].x * (OutT(acc.getValue(index + make_int3(1, j, k)))) +
w[2].x * (OutT(acc.getValue(index + make_int3(2, j, k)))) +
w[3].x * (OutT(acc.getValue(index + make_int3(3, j, k))))));
}
}
SET_CUBIC_SPLINE_WEIGHTS(u, t.x);
SET_CUBIC_SPLINE_WEIGHTS(v, t.y);
SET_CUBIC_SPLINE_WEIGHTS(w, t.z);
/* Actual interpolation. */
return ROW_TERM(0) + ROW_TERM(1) + ROW_TERM(2) + ROW_TERM(3);
# undef COL_TERM
# undef ROW_TERM
# undef DATA
# undef SET_CUBIC_SPLINE_WEIGHTS
return result;
}
template<typename OutT, typename T>

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5abf94ecd91c53568f63170688f94139b81b06fec5d349392a7c0e7757098071
size 31535

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:65221312af4665920c637c608d288a0f93cdb44211d62eecf71459971ded80ee
size 22322

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:dba967b5f4ef79269c7450e1a0cd0a9cf9087427df7cc9e8f8442df49a8dbdb5
size 615318

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:96f8f74d958e4095af3c78da06d44aebb90fcec6ed012ad91ea3982fc7cef2e0
size 25035

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:29ed24bc5f4b2d3bcae82c872a5ded353f3b07e774e17811bd3ae51f799cc12b
size 25203

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2b206a469368ec322bf451498a65199729cf43ebc98cf827aa278bfe160902e6
size 1241250

View file

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3db6cd9ea280965b5d7859760e06334d9834b6830008cdf66615a793490ebc84
size 13040

View file

@ -57,6 +57,8 @@ BLOCKLIST_OSL_ALL = BLOCKLIST_OSL_LIMITED + [
# Tests that need investigating into why they're failing:
# Noise differences due to Principled BSDF mixing/layering used in some of these scenes
'render_passes_.*.blend',
# OSL can not specify parameters when reading attribute, which we need for stochastic sampling
'volume_tricubic_interpolation.blend',
]
BLOCKLIST_OPTIX = [