From ea4c1d770170791a9eb1f70399d08e54aebc39d1 Mon Sep 17 00:00:00 2001 From: N0AHZACH Date: Tue, 15 Sep 2026 22:16:25 +0530 Subject: [PATCH 1/2] Reject mismatched out= dtype in histc instead of casting silently _histc_out_xpu computed into a fresh tensor of the input dtype and then did result.copy_(ret). copy_ converts, so a float64 histogram written into an int64 out= was silently truncated. CPU rejects the same call in histogramdd_prepare_out, which the XPU out variant does not go through. Add the CPU-matching TORCH_CHECK before resize_output so a rejected call leaves the caller tensor untouched. Remove test_out_histc from the skip list. De-scope the tuple device_type xfail in align_db_decorators so the now-passing test does not report an unexpected success. Fixes #5237. Test Plan: ``` python -m py_compile test/regressions/test_histc.py test/xpu/xpu_test_utils.py test/xpu/skip_list_common.py git diff --check ``` --- src/ATen/native/xpu/SummaryOps.cpp | 7 ++++++ test/regressions/test_histc.py | 35 ++++++++++++++++++++++++++++++ test/xpu/skip_list_common.py | 1 - test/xpu/xpu_test_utils.py | 17 +++++++++++++++ 4 files changed, 59 insertions(+), 1 deletion(-) create mode 100644 test/regressions/test_histc.py diff --git a/src/ATen/native/xpu/SummaryOps.cpp b/src/ATen/native/xpu/SummaryOps.cpp index 0f5cbe43506..98f0d8c42ad 100644 --- a/src/ATen/native/xpu/SummaryOps.cpp +++ b/src/ATen/native/xpu/SummaryOps.cpp @@ -55,6 +55,13 @@ Tensor& _histc_out_xpu( const Scalar& min, const Scalar& max, Tensor& result) { + TORCH_CHECK( + self.dtype() == result.dtype(), + "torch.histogram: input tensor and hist tensor should", + " have the same dtype, but got input ", + self.dtype(), + " and hist ", + result.dtype()); auto ret = _histc_xpu(self, bins, min, max); at::native::resize_output(result, ret.sizes()); result.copy_(ret); diff --git a/test/regressions/test_histc.py b/test/regressions/test_histc.py new file mode 100644 index 00000000000..65646bbbda4 --- /dev/null +++ b/test/regressions/test_histc.py @@ -0,0 +1,35 @@ +# Copyright 2020-2026 Intel Corporation +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 + +# Owner(s): ["module: intel"] + +import torch +from torch.testing._internal.common_utils import run_tests, TestCase + + +class TestHistc(TestCase): + def test_histc_out_rejects_mismatched_dtype(self): + # Regression for intel/torch-xpu-ops#5237: _histc_out_xpu finished + # with result.copy_(ret), so a float histogram written into an + # integral out= was silently truncated. CPU rejects this call. + x = torch.linspace(1, 8, 8, device="xpu", dtype=torch.float32) + out = torch.empty(4, device="xpu", dtype=torch.int64) + + with self.assertRaisesRegex(RuntimeError, "should have the same dtype"): + torch.histc(x, bins=4, min=0, max=8, out=out) + + def test_histc_out_matching_dtype_still_works(self): + x = torch.linspace(1, 8, 8, device="xpu", dtype=torch.float32) + out = torch.empty(0, device="xpu", dtype=torch.float32) + + torch.histc(x, bins=4, min=0, max=8, out=out) + self.assertEqual(out, torch.histc(x, bins=4, min=0, max=8)) + + +if __name__ == "__main__": + run_tests() diff --git a/test/xpu/skip_list_common.py b/test/xpu/skip_list_common.py index 055205ec2b8..44f32569372 100644 --- a/test/xpu/skip_list_common.py +++ b/test/xpu/skip_list_common.py @@ -180,7 +180,6 @@ # Exception: The supported dtypes for linalg.multi_dot on device type xpu are incorrect! "test_dtypes_linalg_multi_dot_xpu", # For CUDA it's skipped explicitly in common_methods_invocations.py in upstream. We can skip it here - "test_out_histc_xpu_float32", "test_out_mean_xpu_float32", # FakeTensor mismatch in outputs_alias_inputs for aten.view.default # Known upstream issue: https://github.com/pytorch/pytorch/issues/159150 diff --git a/test/xpu/xpu_test_utils.py b/test/xpu/xpu_test_utils.py index ae3b325c1a1..bdb6af08a66 100644 --- a/test/xpu/xpu_test_utils.py +++ b/test/xpu/xpu_test_utils.py @@ -1070,6 +1070,23 @@ def gen_xpu_wrappers(op_name, wrappers): else: wrapper.device_type = "xpu" replaced = True + elif ( + isinstance(wrapper.device_type, (list, tuple)) + and "xpu" in wrapper.device_type + and unittest.expectedFailure in wrapper.decorators + and (op_name, wrapper.test_name) in _cuda_xfail_xpu_pass + ): + # Upstream may scope one xfail to several devices at + # once (device_type=("cuda", "xpu")). Drop XPU from + # the scope so a test that now passes on XPU does + # not report an unexpected success. + replaced = True + new_wrapper = copy.copy(wrapper) + new_wrapper.device_type = tuple( + d for d in wrapper.device_type if d != "xpu" + ) + wrapper_xpu.append(new_wrapper) + continue elif ( wrapper.device_type is None and unittest.expectedFailure in wrapper.decorators From 49b65a7c6df9fc95eefe6bf139d780945597a6a6 Mon Sep 17 00:00:00 2001 From: N0AHZACH Date: Sat, 3 Oct 2026 00:11:56 +0530 Subject: [PATCH 2/2] Address review: drop dead histc xfail scaffolding and duplicate test The upstream ("cuda", "xpu") xfail for (histc, test_out) was removed in pytorch/pytorch#196100, so the tuple de-scope branch added to align_db_decorators never matches a wrapper, and ("histc", "test_out") in _cuda_xfail_xpu_pass no longer refers to any DecorateInfo. Remove both, per review. Delete test/regressions/test_histc.py: upstream test_histc_out_dtype in test/test_reductions.py covers the same mismatched out= dtype regression and torch-xpu-ops runs it through test_reductions_xpu.py. The fix itself is unchanged: the CPU-matching TORCH_CHECK in _histc_out_xpu before resize_output, plus dropping test_out_histc_xpu_float32 from the op_ut skip list. Test Plan: python -m py_compile test/xpu/xpu_test_utils.py python -m flake8 test/xpu/xpu_test_utils.py clang-format --dry-run --Werror src/ATen/native/xpu/SummaryOps.cpp git diff --check --- test/regressions/test_histc.py | 35 ---------------------------------- test/xpu/xpu_test_utils.py | 18 ----------------- 2 files changed, 53 deletions(-) delete mode 100644 test/regressions/test_histc.py diff --git a/test/regressions/test_histc.py b/test/regressions/test_histc.py deleted file mode 100644 index 65646bbbda4..00000000000 --- a/test/regressions/test_histc.py +++ /dev/null @@ -1,35 +0,0 @@ -# Copyright 2020-2026 Intel Corporation -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 - -# Owner(s): ["module: intel"] - -import torch -from torch.testing._internal.common_utils import run_tests, TestCase - - -class TestHistc(TestCase): - def test_histc_out_rejects_mismatched_dtype(self): - # Regression for intel/torch-xpu-ops#5237: _histc_out_xpu finished - # with result.copy_(ret), so a float histogram written into an - # integral out= was silently truncated. CPU rejects this call. - x = torch.linspace(1, 8, 8, device="xpu", dtype=torch.float32) - out = torch.empty(4, device="xpu", dtype=torch.int64) - - with self.assertRaisesRegex(RuntimeError, "should have the same dtype"): - torch.histc(x, bins=4, min=0, max=8, out=out) - - def test_histc_out_matching_dtype_still_works(self): - x = torch.linspace(1, 8, 8, device="xpu", dtype=torch.float32) - out = torch.empty(0, device="xpu", dtype=torch.float32) - - torch.histc(x, bins=4, min=0, max=8, out=out) - self.assertEqual(out, torch.histc(x, bins=4, min=0, max=8)) - - -if __name__ == "__main__": - run_tests() diff --git a/test/xpu/xpu_test_utils.py b/test/xpu/xpu_test_utils.py index 5c827224b44..d66b817ad13 100644 --- a/test/xpu/xpu_test_utils.py +++ b/test/xpu/xpu_test_utils.py @@ -349,7 +349,6 @@ ("_batch_norm_with_update", "test_dispatch_symbolic_meta_outplace_all_strides"), ("_native_batch_norm_legit", "test_out"), ("native_batch_norm", "test_out"), - ("histc", "test_out"), ("_refs.mul", "test_python_ref"), ("_refs.mul", "test_python_ref_torch_fallback"), ("nn.AvgPool2d", "test_memory_format"), @@ -1065,23 +1064,6 @@ def gen_xpu_wrappers(op_name, wrappers): else: wrapper.device_type = "xpu" replaced = True - elif ( - isinstance(wrapper.device_type, (list, tuple)) - and "xpu" in wrapper.device_type - and unittest.expectedFailure in wrapper.decorators - and (op_name, wrapper.test_name) in _cuda_xfail_xpu_pass - ): - # Upstream may scope one xfail to several devices at - # once (device_type=("cuda", "xpu")). Drop XPU from - # the scope so a test that now passes on XPU does - # not report an unexpected success. - replaced = True - new_wrapper = copy.copy(wrapper) - new_wrapper.device_type = tuple( - d for d in wrapper.device_type if d != "xpu" - ) - wrapper_xpu.append(new_wrapper) - continue elif ( wrapper.device_type is None and unittest.expectedFailure in wrapper.decorators