Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions esmvalcore/preprocessor/_regrid.py
Original file line number Diff line number Diff line change
Expand Up @@ -986,6 +986,24 @@ def regrid(
# Rechunk and actually perform the regridding
cube = _rechunk(cube, target_grid_cube)
result = regridder(cube)
for ancillary_var in cube.ancillary_variables():
ancillary_dims = cube.ancillary_variable_dims(ancillary_var)
ancillary_slice = tuple(
slice(None) if i in ancillary_dims else 0 for i in range(cube.ndim)
)
ancillary_cube = cube[ancillary_slice].copy(ancillary_var.core_data())
ancillary_cube = _rechunk(ancillary_cube, target_grid_cube)
ancillary_result = regridder(ancillary_cube)
if result.has_lazy_data() and ancillary_result.has_lazy_data():
# Keep the chunks of the ancillary variable aligned with the
# regridded data variable.
ancillary_result.data = ancillary_result.lazy_data().rechunk(
result[ancillary_slice].lazy_data().chunks,
)

@valeriupredoi valeriupredoi Sep 16, 2026 •

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

thanks @bouweandela - I am a bit puzzled as to why all the rechunking - why not perform a single rechunk of the ancillary_result at the end, after regridding? What are you going to do if the cube data is not chunked (one single slab)?

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The initial rechunk is needed to avoid getting huge chunks out of regridding when regridding to a higher resolution. The final rechunk is to align the chunks inside the cube, this gives better performance when combining one of the ancillary variables with the main variable in some later preprocessing step (e.g. area statistics).

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

yes, that much I understood myself - but the question is why is the first rechunk needed since I thought the ancil data is chunked like the main var data? They have different chunking? That's insane if they do - they are both produced by the same model with the same data specs

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

side note: huge chunks will not be a problem for CMIP7 anymore since the data will pass through cmip7_repack, so we'll have to be a bit more careful then

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

BTW I saw your comment from the other PR #3204 - don't hold the merge of this hanging on my rechunking comments, we'll have to rethink that for CMIP7 light anyway - I am just being a bit anal 😁

@bouweandela bouweandela Sep 16, 2026 •

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I thought the ancil data is chunked like the main var data? They have different chunking?

There is nothing in iris that enforces that.

huge chunks will not be a problem for CMIP7 anymore

Regridding is done on a per chunk basis: one chunk is regridded to one other chunk. Even if the input chunks are reasonably sized, the output chunks may be too large when the resolution is increased by regridding. I believe this has been addressed in iris in SciTools/iris#6730, but I do not see the corresponding changes in iris-esmf-regrid, so it's probably best to keep our own regridding code. This applies to CMIP7 just the same.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

that's mental! Thanks, bud - not auto scaling chunks upon regridding is bad, not having chunks harmonized between main var and ancil var, equally. OK nevermind my comments above 😁

result.add_ancillary_variable(
ancillary_var.copy(ancillary_result.core_data()),
ancillary_dims,
)
# Iris only supports regridding of 1D coordinates and iris-esmf-regrid
# uses only the DimCoords if both grid_latitude/grid_longitude or
# projection_x_coordinate/projection_y_coordinate DimCoords and latitude
Expand Down
25 changes: 25 additions & 0 deletions tests/integration/preprocessor/_regrid/test_regrid.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,11 @@
"""Integration tests for :func:`esmvalcore.preprocessor.regrid`."""

import dask.array as da
import iris
import iris.coord_systems
import iris.coords
import iris.cube
import iris.fileformats.pp
import numpy as np
import pytest
from numpy import ma
Expand Down Expand Up @@ -364,6 +369,26 @@ def test_regrid__linear_with_mask(self, cache_weights):
expected[:, 1, 1] = np.array([1.5, 5.5, 9.5])
assert_array_equal(result.data, expected)

def test_regrid__linear_with_ancillary(self) -> None:
"""Test that ancillary coordinates are also regridded."""
cube = self.cube.copy()
cube.data = cube.lazy_data()
cube.add_ancillary_variable(
iris.coords.AncillaryVariable(
da.arange(2, 6).astype(np.float32).reshape(2, 2),
var_name="ancillary",
),
(1, 2),
)
result = regrid(cube, self.grid_for_linear, "linear")
ancillary_result = result.ancillary_variable("ancillary")
assert isinstance(ancillary_result, iris.coords.AncillaryVariable)
assert ancillary_result.has_lazy_data()
assert_array_equal(
ancillary_result.data,
np.array([3.5], dtype=np.float32).reshape(1, 1),
)

@pytest.mark.parametrize("cache_weights", [True, False])
def test_regrid__nearest(self, cache_weights):
data = np.empty((1, 1))
Expand Down