Fix #162183: Cycles texture cache broken for textures larger than 2^31 px

Fix integer overflows that were causing this, in the kernel and in tx
generation. The overflow caused both wrong render results and slowness
due to too many tiles being loaded due to wrong derivatives.

Also fix an additional overflows in CPU image sampling without the cache.

Pull Request: https://projects.blender.org/blender/blender/pulls/162195
This commit is contained in:
Brecht Van Lommel 2026-08-05 21:35:48 +02:00 • committed by Brecht Van Lommel
parent f207c2ddf3
commit 41e2d2dadd
3 changed files with 12 additions and 12 deletions

View file

@ -93,7 +93,7 @@ template<typename TexT, typename OutT = float4> struct ImageInterpolator {
static ccl_always_inline OutT
read(const TexT *data, const int x, int y, const int width, const int /*height*/)
{
return read(data[y * width + x]);
return read(data[int64_t(y) * width + x]);
}
/* Read 2D Texture Data Clip
@ -104,7 +104,7 @@ template<typename TexT, typename OutT = float4> struct ImageInterpolator {
if (x < 0 || x >= width || y < 0 || y >= height) {
return zero();
}
return read(data[y * width + x]);
return read(data[int64_t(y) * width + x]);
}
static ccl_always_inline int wrap_periodic(int x, const int width)

View file

@ -81,10 +81,10 @@ kernel_image_tile_map(KernelGlobals kg,
ccl_private float2 &xy)
{
/* Find mipmap level. Use squared lengths to avoid two sqrt operations,
* compensating with 0.5 factor on the log2. */
const float dudxy_sq = len_squared(make_float2(uv.dx.x, uv.dy.x)) * float(tex.width * tex.width);
const float dvdxy_sq = len_squared(make_float2(uv.dx.y, uv.dy.y)) *
float(tex.height * tex.height);
* compensating with 0.5 factor on the log2. Square as float to avoid
* integer overflow. */
const float dudxy_sq = len_squared(make_float2(uv.dx.x, uv.dy.x)) * sqr(float(tex.width));
const float dvdxy_sq = len_squared(make_float2(uv.dx.y, uv.dy.y)) * sqr(float(tex.height));
/* Limit max anisotropy ratio, to avoid loading too high mip resolutions
* for stretched UV coordinates, which don't really benefit from it anyway. */

View file

@ -189,10 +189,10 @@ static bool resize_block_2pass(OIIO::ImageBuf &dst, const OIIO::ImageBuf &src, O
* any NDC -> pixel math, and just directly traverse pixels. */
const SRCTYPE *s = (const SRCTYPE *)src.localpixels();
SRCTYPE *d = (SRCTYPE *)dst.localpixels();
assert(s && d); /* Assume contig bufs */
d += roi.ybegin * dst.spec().width * nchannels; /* Top of dst OIIO::ROI */
const size_t ystride = src.spec().width * nchannels; /* Scanline offset */
s += 2 * roi.ybegin * ystride; /* Top of src OIIO::ROI */
assert(s && d); /* Assume contig bufs */
d += int64_t(roi.ybegin) * dst.spec().width * nchannels; /* Top of dst OIIO::ROI */
const size_t ystride = src.spec().width * nchannels; /* Scanline offset */
s += 2 * roi.ybegin * ystride; /* Top of src OIIO::ROI */
/* Run through destination rows, doing the two-pass bilerp filter. */
const size_t dw = roi.width(), dh = roi.height(); /* Loop invariants */
@ -578,7 +578,7 @@ static void clamp_half_tx(OIIO::ImageBuf &buf, const TypeDesc out_format)
assert(buf.spec().format == TypeFloat);
const int64_t num_values = buf.spec().width * buf.spec().height * buf.spec().nchannels;
const int64_t num_values = int64_t(buf.spec().width) * buf.spec().height * buf.spec().nchannels;
float *pixels = static_cast<float *>(buf.localpixels());
for (int64_t i = 0; i < num_values; i++) {
pixels[i] = clamp(pixels[i], -HALF_MAX, HALF_MAX);
@ -590,7 +590,7 @@ static void convert_srgb_tx(OIIO::ImageBuf &buf, const bool from_srgb)
assert(buf.spec().format == TypeFloat);
assert(buf.spec().nchannels == 1 || buf.spec().nchannels == 4);
const int64_t num_pixels = buf.spec().width * buf.spec().height;
const int64_t num_pixels = int64_t(buf.spec().width) * buf.spec().height;
if (buf.spec().nchannels == 1) {
float *pixels = static_cast<float *>(buf.localpixels());