update normalize based on PR reviews

justincdavis · justincdavis · commit 778ad3240db5 · 2025-12-01T18:01:33.000-08:00
diff --git a/test/common_utils.py b/test/common_utils.py
@@ -20,13 +20,15 @@
 from torch.testing._comparison import BooleanPair, NonePair, not_close_error_metas, NumberPair, TensorLikePair
 from torchvision import io, tv_tensors
 from torchvision.transforms._functional_tensor import _max_value as get_max_value
-from torchvision.transforms.v2.functional import to_cvcuda_tensor, to_image, to_pil_image
+from torchvision.transforms.v2.functional import cvcuda_to_tensor, to_cvcuda_tensor, to_image, to_pil_image
+from torchvision.transforms.v2.functional._utils import _import_cvcuda, _is_cvcuda_available
 from torchvision.utils import _Image_fromarray
 
 
 IN_OSS_CI = any(os.getenv(var) == "true" for var in ["CIRCLECI", "GITHUB_ACTIONS"])
 IN_RE_WORKER = os.environ.get("INSIDE_RE_WORKER") is not None
 IN_FBCODE = os.environ.get("IN_FBCODE_TORCHVISION") == "1"
+CVCUDA_AVAILABLE = _is_cvcuda_available()
 CUDA_NOT_AVAILABLE_MSG = "CUDA device not available"
 MPS_NOT_AVAILABLE_MSG = "MPS device not available"
 OSS_CI_GPU_NO_CUDA_MSG = "We're in an OSS GPU machine, and this test doesn't need cuda."
@@ -275,6 +277,17 @@ def combinations_grid(**kwargs):
     return [dict(zip(kwargs.keys(), values)) for values in itertools.product(*kwargs.values())]
 
 
+def cvcuda_to_pil_compatible_tensor(tensor: "cvcuda.Tensor") -> torch.Tensor:
+    tensor = cvcuda_to_tensor(tensor)
+    if tensor.ndim != 4:
+        raise ValueError(f"CV-CUDA Tensor should be 4 dimensional. Got {tensor.ndim} dimensions.")
+    if tensor.shape[0] != 1:
+        raise ValueError(
+            f"CV-CUDA Tensor should have batch dimension 1 for comparison with PIL.Image.Image. Got {tensor.shape[0]}."
+        )
+    return tensor.squeeze(0).cpu()
+
+
 class ImagePair(TensorLikePair):
     def __init__(
         self,
@@ -287,6 +300,11 @@ def __init__(
         if all(isinstance(input, PIL.Image.Image) for input in [actual, expected]):
             actual, expected = (to_image(input) for input in [actual, expected])
 
+        # handle check for CV-CUDA Tensors
+        if CVCUDA_AVAILABLE and isinstance(actual, _import_cvcuda().Tensor):
+            # Use the PIL compatible tensor, so we can always compare with PIL.Image.Image
+            actual = cvcuda_to_pil_compatible_tensor(actual)
+
         super().__init__(actual, expected, **other_parameters)
         self.mae = mae
 
@@ -401,7 +419,6 @@ def make_image_pil(*args, **kwargs):
 
 
 def make_image_cvcuda(*args, batch_dims=(1,), **kwargs):
-    # explicitly default batch_dims to (1,) since to_cvcuda_tensor requires a batch dimension (ndims == 4)
     return to_cvcuda_tensor(make_image(*args, batch_dims=batch_dims, **kwargs))
 
 
diff --git a/test/test_transforms_v2.py b/test/test_transforms_v2.py
@@ -21,9 +21,11 @@
 import torchvision.transforms.v2 as transforms
 
 from common_utils import (
+    assert_close,
     assert_equal,
     cache,
     cpu_and_cuda,
+    cvcuda_to_pil_compatible_tensor,
     freeze_rng_state,
     ignore_jit_no_profile_information_warning,
     make_bounding_boxes,
@@ -41,7 +43,6 @@
 )
 
 from torch import nn
-from torch.testing import assert_close
 from torch.utils._pytree import tree_flatten, tree_map
 from torch.utils.data import DataLoader, default_collate
 from torchvision import tv_tensors
@@ -5500,17 +5501,17 @@ def test_kernel_image(self, mean, std, device):
 
     @pytest.mark.parametrize("device", cpu_and_cuda())
     def test_kernel_image_inplace(self, device):
-        input = make_image_tensor(dtype=torch.float32, device=device)
-        input_version = input._version
+        inpt = make_image_tensor(dtype=torch.float32, device=device)
+        input_version = inpt._version
 
-        output_out_of_place = F.normalize_image(input, mean=self.MEAN, std=self.STD)
-        assert output_out_of_place.data_ptr() != input.data_ptr()
-        assert output_out_of_place is not input
+        output_out_of_place = F.normalize_image(inpt, mean=self.MEAN, std=self.STD)
+        assert output_out_of_place.data_ptr() != inpt.data_ptr()
+        assert output_out_of_place is not inpt
 
-        output_inplace = F.normalize_image(input, mean=self.MEAN, std=self.STD, inplace=True)
-        assert output_inplace.data_ptr() == input.data_ptr()
+        output_inplace = F.normalize_image(inpt, mean=self.MEAN, std=self.STD, inplace=True)
+        assert output_inplace.data_ptr() == inpt.data_ptr()
         assert output_inplace._version > input_version
-        assert output_inplace is input
+        assert output_inplace is inpt
 
         assert_equal(output_inplace, output_out_of_place)
 
@@ -5560,9 +5561,9 @@ def test_functional_error(self):
             with pytest.raises(ValueError, match="std evaluated to zero, leading to division by zero"):
                 F.normalize_image(make_image(dtype=torch.float32), mean=self.MEAN, std=std)
 
-    def _sample_input_adapter(self, transform, input, device):
+    def _sample_input_adapter(self, transform, inpt, device):
         adapted_input = {}
-        for key, value in input.items():
+        for key, value in inpt.items():
             if isinstance(value, PIL.Image.Image):
                 # normalize doesn't support PIL images
                 continue
@@ -5616,15 +5617,12 @@ def test_correctness_image(self, mean, std, dtype, make_input, fn):
         actual = fn(image, mean=mean, std=std)
 
         if make_input == make_image_cvcuda:
-            image = F.cvcuda_to_tensor(image).to(device="cpu")
-            image = image.squeeze(0)
-            actual = F.cvcuda_to_tensor(actual).to(device="cpu")
-            actual = actual.squeeze(0)
+            image = cvcuda_to_pil_compatible_tensor(image)
 
         expected = self._reference_normalize_image(image, mean=mean, std=std)
 
         if make_input == make_image_cvcuda:
-            torch.testing.assert_close(actual, expected, rtol=0, atol=1e-6)
+            assert_close(actual, expected, rtol=0, atol=1e-6)
         else:
             assert_equal(actual, expected)
 
diff --git a/torchvision/transforms/v2/_misc.py b/torchvision/transforms/v2/_misc.py
@@ -17,6 +17,7 @@
     get_bounding_boxes,
     get_keypoints,
     has_any,
+    is_cvcuda_tensor,
     is_pure_tensor,
 )
 
@@ -160,6 +161,8 @@ class Normalize(Transform):
 
     _v1_transform_cls = _transforms.Normalize
 
+    _transformed_types = Transform._transformed_types + (is_cvcuda_tensor,)
+
     def __init__(self, mean: Sequence[float], std: Sequence[float], inplace: bool = False):
         super().__init__()
         self.mean = list(mean)