mirror of
https://github.com/blender/blender
synced 2026-09-29 04:37:17 +03:00
Cycles: Improve compilation time of tricubic interpolation
by using a loop instead of unrolling all 64 evaluations. Seems that two nested loops has the best performance. Compilation time measured on Metal M2 Ultra: | Kernel| Before| After| | --| --| --| | integrator_shade_volume|148.13s|114.71s| |integrator_shade_volume_ray_marching| 44.30s| 14.27s| | integrator_shade_shadow| 87.83s| 58.82s| | shader_eval_volume_density| 32.69s| 6.63s| Also added test file because we were not testing deterministic tricubic interpolation before Ref: #150119
This commit is contained in:
parent
ffb0fc48e9
commit
c0ad7c16dc
9 changed files with 48 additions and 38 deletions
|
|
@ -24,6 +24,16 @@ namespace {
|
|||
#endif
|
||||
|
||||
#ifdef WITH_NANOVDB
|
||||
|
||||
/* Cubic interpolation weights. */
|
||||
ccl_device_inline void fill_cubic_weights(float3 w[4], float3 t)
|
||||
{
|
||||
w[0] = (((-1.0f / 6.0f) * t + 0.5f) * t - 0.5f) * t + (1.0f / 6.0f);
|
||||
w[1] = ((0.5f * t - 1.0f) * t) * t + (2.0f / 3.0f);
|
||||
w[2] = ((-0.5f * t + 0.5f) * t + 0.5f) * t + (1.0f / 6.0f);
|
||||
w[3] = (1.0f / 6.0f) * t * t * t;
|
||||
}
|
||||
|
||||
/* -------------------------------------------------------------------- */
|
||||
/** Return the sample position for stochastical one-tap sampling.
|
||||
* From "Stochastic Texture Filtering": https://arxiv.org/abs/2305.05810
|
||||
|
|
@ -33,11 +43,8 @@ ccl_device_inline float3 interp_tricubic_stochastic(const float3 P, ccl_private
|
|||
const float3 p = floor(P);
|
||||
const float3 t = P - p;
|
||||
|
||||
/* Cubic interpolation weights. */
|
||||
const float3 w[4] = {(((-1.0f / 6.0f) * t + 0.5f) * t - 0.5f) * t + (1.0f / 6.0f),
|
||||
((0.5f * t - 1.0f) * t) * t + (2.0f / 3.0f),
|
||||
((-0.5f * t + 0.5f) * t + 0.5f) * t + (1.0f / 6.0f),
|
||||
(1.0f / 6.0f) * t * t * t};
|
||||
float3 w[4];
|
||||
fill_cubic_weights(w, t);
|
||||
|
||||
/* For reservoir sampling, always accept the first in the stream. */
|
||||
float3 total_weight = w[0];
|
||||
|
|
@ -113,43 +120,23 @@ ccl_device OutT kernel_tex_image_interp_tricubic_nanovdb(ccl_private Acc &acc, c
|
|||
{
|
||||
const float3 floor_P = floor(P);
|
||||
const float3 t = P - floor_P;
|
||||
const int3 index = make_int3(floor_P);
|
||||
const int3 index = make_int3(floor_P) - make_int3(1);
|
||||
|
||||
const int xc[4] = {index.x - 1, index.x, index.x + 1, index.x + 2};
|
||||
const int yc[4] = {index.y - 1, index.y, index.y + 1, index.y + 2};
|
||||
const int zc[4] = {index.z - 1, index.z, index.z + 1, index.z + 2};
|
||||
float u[4], v[4], w[4];
|
||||
float3 w[4];
|
||||
fill_cubic_weights(w, t);
|
||||
|
||||
/* Some helper macros to keep code size reasonable.
|
||||
* Lets the compiler inline all the matrix multiplications.
|
||||
*/
|
||||
# define SET_CUBIC_SPLINE_WEIGHTS(u, t) \
|
||||
{ \
|
||||
u[0] = (((-1.0f / 6.0f) * t + 0.5f) * t - 0.5f) * t + (1.0f / 6.0f); \
|
||||
u[1] = ((0.5f * t - 1.0f) * t) * t + (2.0f / 3.0f); \
|
||||
u[2] = ((-0.5f * t + 0.5f) * t + 0.5f) * t + (1.0f / 6.0f); \
|
||||
u[3] = (1.0f / 6.0f) * t * t * t; \
|
||||
} \
|
||||
(void)0
|
||||
OutT result = make_zero<OutT>();
|
||||
|
||||
# define DATA(x, y, z) (OutT(acc.getValue(make_int3(xc[x], yc[y], zc[z]))))
|
||||
# define COL_TERM(col, row) \
|
||||
(v[col] * (u[0] * DATA(0, col, row) + u[1] * DATA(1, col, row) + u[2] * DATA(2, col, row) + \
|
||||
u[3] * DATA(3, col, row)))
|
||||
# define ROW_TERM(row) \
|
||||
(w[row] * (COL_TERM(0, row) + COL_TERM(1, row) + COL_TERM(2, row) + COL_TERM(3, row)))
|
||||
for (int k = 0; k < 4; k++) {
|
||||
for (int j = 0; j < 4; j++) {
|
||||
result += w[k].z * (w[j].y * (w[0].x * (OutT(acc.getValue(index + make_int3(0, j, k)))) +
|
||||
w[1].x * (OutT(acc.getValue(index + make_int3(1, j, k)))) +
|
||||
w[2].x * (OutT(acc.getValue(index + make_int3(2, j, k)))) +
|
||||
w[3].x * (OutT(acc.getValue(index + make_int3(3, j, k))))));
|
||||
}
|
||||
}
|
||||
|
||||
SET_CUBIC_SPLINE_WEIGHTS(u, t.x);
|
||||
SET_CUBIC_SPLINE_WEIGHTS(v, t.y);
|
||||
SET_CUBIC_SPLINE_WEIGHTS(w, t.z);
|
||||
|
||||
/* Actual interpolation. */
|
||||
return ROW_TERM(0) + ROW_TERM(1) + ROW_TERM(2) + ROW_TERM(3);
|
||||
|
||||
# undef COL_TERM
|
||||
# undef ROW_TERM
|
||||
# undef DATA
|
||||
# undef SET_CUBIC_SPLINE_WEIGHTS
|
||||
return result;
|
||||
}
|
||||
|
||||
template<typename OutT, typename T>
|
||||
|
|
|
|||
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5abf94ecd91c53568f63170688f94139b81b06fec5d349392a7c0e7757098071
|
||||
size 31535
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:65221312af4665920c637c608d288a0f93cdb44211d62eecf71459971ded80ee
|
||||
size 22322
|
||||
3
tests/files/render/openvdb/intelCloudLib_sparse.4.S.vdb
Normal file
3
tests/files/render/openvdb/intelCloudLib_sparse.4.S.vdb
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:dba967b5f4ef79269c7450e1a0cd0a9cf9087427df7cc9e8f8442df49a8dbdb5
|
||||
size 615318
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:96f8f74d958e4095af3c78da06d44aebb90fcec6ed012ad91ea3982fc7cef2e0
|
||||
size 25035
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:29ed24bc5f4b2d3bcae82c872a5ded353f3b07e774e17811bd3ae51f799cc12b
|
||||
size 25203
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2b206a469368ec322bf451498a65199729cf43ebc98cf827aa278bfe160902e6
|
||||
size 1241250
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3db6cd9ea280965b5d7859760e06334d9834b6830008cdf66615a793490ebc84
|
||||
size 13040
|
||||
|
|
@ -57,6 +57,8 @@ BLOCKLIST_OSL_ALL = BLOCKLIST_OSL_LIMITED + [
|
|||
# Tests that need investigating into why they're failing:
|
||||
# Noise differences due to Principled BSDF mixing/layering used in some of these scenes
|
||||
'render_passes_.*.blend',
|
||||
# OSL can not specify parameters when reading attribute, which we need for stochastic sampling
|
||||
'volume_tricubic_interpolation.blend',
|
||||
]
|
||||
|
||||
BLOCKLIST_OPTIX = [
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue