diff --git a/CHANGELOG.md b/CHANGELOG.md index 0240e4d5d9eb..bfb8b6e81e8b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -115,6 +115,7 @@ This release is compatible with NumPy 2.5. * Fixed `dpnp.nanmedian` dropping kept dimensions of size 1, which produced a wrong result shape [#3081](https://github.com/IntelPython/dpnp/pull/3081) * Fixed incorrect results of `dpnp.tensor.vecdot` in some cases with strided outputs and of `dpnp.tensor` reductions, `dpnp.tensor.vecdot` and `dpnp.tensor.matmul` on large inputs with some data types [#3082](https://github.com/IntelPython/dpnp/pull/3082) * Fixed `dpnp.median` and `dpnp.nanmedian` raising a `ValueError` for a tuple `axis` when a kept dimension has size 0 [#3081](https://github.com/IntelPython/dpnp/pull/3081) +* Fixed a `RuntimeError` raised by `dpnp.tensor` kernels launched over more than `INT_MAX` work-items when built with DPC++ compiler 2026.2 or newer, by also passing `-fno-sycl-id-queries-fit-in-int` to the linker [#3089](https://github.com/IntelPython/dpnp/pull/3089) ### Security diff --git a/dpnp/CMakeLists.txt b/dpnp/CMakeLists.txt index e616e1964c56..5a93c42a6108 100644 --- a/dpnp/CMakeLists.txt +++ b/dpnp/CMakeLists.txt @@ -100,7 +100,10 @@ function(build_dpnp_tensor_ext _trgt _src _dest) if(BUILD_DPNP_TENSOR_SYCL) add_sycl_to_target(TARGET ${_trgt} SOURCES ${_generated_src}) target_compile_options(${_trgt} PRIVATE -fno-sycl-id-queries-fit-in-int) - target_link_options(${_trgt} PRIVATE -fsycl-device-code-split=per_kernel) + target_link_options( + ${_trgt} + PRIVATE -fsycl-device-code-split=per_kernel -fno-sycl-id-queries-fit-in-int + ) if(DPNP_TENSOR_OFFLOAD_COMPRESS) target_link_options(${_trgt} PRIVATE --offload-compress) endif() diff --git a/dpnp/backend/extensions/indexing/CMakeLists.txt b/dpnp/backend/extensions/indexing/CMakeLists.txt index 4242f053d1b5..e43d3d647529 100644 --- a/dpnp/backend/extensions/indexing/CMakeLists.txt +++ b/dpnp/backend/extensions/indexing/CMakeLists.txt @@ -96,7 +96,10 @@ else() endif() target_compile_options(${python_module_name} PUBLIC -fno-sycl-id-queries-fit-in-int) -target_link_options(${python_module_name} PUBLIC -fsycl-device-code-split=per_kernel) +target_link_options( + ${python_module_name} + PUBLIC -fsycl-device-code-split=per_kernel -fno-sycl-id-queries-fit-in-int +) if(DPNP_GENERATE_COVERAGE) target_link_options( diff --git a/dpnp/tensor/CMakeLists.txt b/dpnp/tensor/CMakeLists.txt index 5345b672397d..67fa250ef630 100644 --- a/dpnp/tensor/CMakeLists.txt +++ b/dpnp/tensor/CMakeLists.txt @@ -346,7 +346,7 @@ foreach(python_module_name ${_py_trgts}) ) target_link_options( ${python_module_name} - PRIVATE -fsycl-device-code-split=per_kernel + PRIVATE -fsycl-device-code-split=per_kernel -fno-sycl-id-queries-fit-in-int ) if(DPNP_TENSOR_OFFLOAD_COMPRESS) target_link_options(${python_module_name} PRIVATE --offload-compress) diff --git a/dpnp/tests/third_party/cupy/binary_tests/test_packing.py b/dpnp/tests/third_party/cupy/binary_tests/test_packing.py index 518e74d98868..6ffd64420d93 100644 --- a/dpnp/tests/third_party/cupy/binary_tests/test_packing.py +++ b/dpnp/tests/third_party/cupy/binary_tests/test_packing.py @@ -59,6 +59,24 @@ def test_pack_invalid_array(self): fa = cupy.array([10, 20, 30], dtype=float) pytest.raises(TypeError, cupy.packbits, fa) + @testing.slow + # @pytest.mark.thread_unsafe(reason="Allocation too large.") + @pytest.mark.parametrize( + "bitorder, expected", [("big", 128), ("little", 1)] + ) + def test_packbits_large_input(self, bitorder, expected): + a = packed = None + try: + a = cupy.zeros(2**31 + 1, dtype=cupy.uint8) + a[-1] = 1 + packed = cupy.packbits(a, bitorder=bitorder) + assert packed[-1] == expected + except MemoryError: + pytest.skip("out of memory in test.") + finally: + del packed, a + # cupy.get_default_memory_pool().free_all_blocks() + def test_unpackbits(self): self.check_unpackbits([]) self.check_unpackbits([0]) diff --git a/dpnp/tests/third_party/cupy/core_tests/test_ndarray_adv_indexing.py b/dpnp/tests/third_party/cupy/core_tests/test_ndarray_adv_indexing.py index f4917393cd2a..53ae7f0ced6e 100644 --- a/dpnp/tests/third_party/cupy/core_tests/test_ndarray_adv_indexing.py +++ b/dpnp/tests/third_party/cupy/core_tests/test_ndarray_adv_indexing.py @@ -967,7 +967,12 @@ def test_adv_setitem_transp(self, xp): class TestHugeArrays: - # These tests require a lot of memory + # These tests use arrays with large logical sizes. + def test_take_negative_index_at_int32_size_boundary(self): + # Only the broadcast view is huge, so this needs no real allocation. + arr = cupy.broadcast_to(cupy.array(42, dtype=cupy.uint8), (2**31,)) + assert arr.take(-1).item() == 42 + @testing.slow def test_advanced(self): try: @@ -985,28 +990,41 @@ def test_advanced(self): pytest.skip("out of memory in test.") @testing.slow + # @pytest.mark.thread_unsafe(reason="Allocation too large.") def test_take_array(self): + arr = indices = res = None try: - arr = cupy.ones((1, 2**32), dtype=cupy.int8) - arr[0, 2**30] = 0 # We should see each of these once - arr[0, -1] = 0 - res = arr.take(cupy.array([0, 0]), axis=0) - # sanity check, we mostly care about it not crashing. - assert res.sum() == 2 * (2**32 - 2) + arr = cupy.array([[[1, 2]], [[3, 4]]], dtype=cupy.uint8) + indices = cupy.broadcast_to( + cupy.array(0, dtype=cupy.int8), (2**30,) + ) + res = arr.take(indices, axis=1) + testing.assert_array_equal(res[0, 0], [1, 2]) + testing.assert_array_equal(res[0, -1], [1, 2]) + testing.assert_array_equal(res[1, 0], [3, 4]) + testing.assert_array_equal(res[1, -1], [3, 4]) except MemoryError: pytest.skip("out of memory in test.") + finally: + del res, indices, arr + # cupy.get_default_memory_pool().free_all_blocks() @testing.slow + # @pytest.mark.thread_unsafe(reason="Allocation too large.") def test_take_scalar(self): + arr = res = None try: - arr = cupy.ones((1, 2**32), dtype=cupy.int8) - arr[0, 2**30] = 0 # We should see each of these once - arr[0, -1] = 0 - res = arr.take(0, axis=0) - # sanity check, we mostly care about it not crashing. - assert res.sum() == 2**32 - 2 + arr = cupy.broadcast_to( + cupy.array([[1], [2]], dtype=cupy.uint8), (2, 2**31) + ) + res = arr.take(1, axis=0) + assert res[0] == 2 + assert res[-1] == 2 except MemoryError: pytest.skip("out of memory in test.") + finally: + del res, arr + # cupy.get_default_memory_pool().free_all_blocks() @testing.slow def test_choose(self): diff --git a/dpnp/tests/third_party/cupy/creation_tests/test_basic.py b/dpnp/tests/third_party/cupy/creation_tests/test_basic.py index 28f365026bff..4e42bc85e3e6 100644 --- a/dpnp/tests/third_party/cupy/creation_tests/test_basic.py +++ b/dpnp/tests/third_party/cupy/creation_tests/test_basic.py @@ -336,17 +336,17 @@ def test_full_like_subok(self): @pytest.mark.skip("_index_32_bits attribute is not supported by dpnp") @pytest.mark.slow - # thread_unsafe marker requires pytest-run-parallel, not used by dpnp # @pytest.mark.thread_unsafe(reason="large allocations") @pytest.mark.parametrize( "arr_factory,expected", [ (lambda: cupy.empty(2**31 - 1, dtype=cupy.int8), True), - (lambda: cupy.empty(2**31, dtype=cupy.int8), True), + # Array sizes should fit as well, so 2**31 is also rejected: + (lambda: cupy.empty(2**31, dtype=cupy.int8), False), (lambda: cupy.empty(2**31 + 1, dtype=cupy.int8)[::2], False), - (lambda: cupy.empty(2**31 // 8, dtype=cupy.complex64), True), + (lambda: cupy.empty(2**31 // 8, dtype=cupy.complex64), False), (lambda: cupy.empty(2**31 // 8 + 1, dtype=cupy.complex64), False), - # Regression test for gh-9750: + # Regression test for gh-9750 (first view spans 2**31 - 4 bytes) (lambda: cupy.empty(2**31 // 8, dtype=cupy.complex64).real, True), ( lambda: cupy.empty(2**31 // 8 + 1, dtype=cupy.complex64).real, diff --git a/dpnp/tests/third_party/cupy/creation_tests/test_matrix.py b/dpnp/tests/third_party/cupy/creation_tests/test_matrix.py index 799ed954c70e..15441dc9ebbf 100644 --- a/dpnp/tests/third_party/cupy/creation_tests/test_matrix.py +++ b/dpnp/tests/third_party/cupy/creation_tests/test_matrix.py @@ -126,6 +126,39 @@ def test_tri_posi(self, xp, dtype): return xp.tri(*self.shape, k=1, dtype=dtype) +class TestTriLargeDimensions(unittest.TestCase): + + @testing.slow + # @pytest.mark.thread_unsafe(reason="Allocation too large.") + def test_large_number_of_rows(self): + out = None + try: + # k must be 0: a nonzero k lets a truncated 32-bit `row` wrap + # back onto the expected answer instead of failing. + out = cupy.tri(2**31 + 1, 1, k=0, dtype=cupy.uint8) + assert out[0, 0] == 1 + assert out[-1, 0] == 1 + except MemoryError: + pytest.skip("out of memory in test.") + finally: + del out + # cupy.get_default_memory_pool().free_all_blocks() + + @testing.slow + # @pytest.mark.thread_unsafe(reason="Allocation too large.") + def test_large_number_of_columns(self): + out = None + try: + out = cupy.tri(1, 2**31, k=2**31 - 1, dtype=cupy.uint8) + assert out[0, 0] == 1 + assert out[0, -1] == 1 + except MemoryError: + pytest.skip("out of memory in test.") + finally: + del out + # cupy.get_default_memory_pool().free_all_blocks() + + @testing.parameterize( {"shape": (2,)}, {"shape": (3, 3)}, diff --git a/dpnp/tests/third_party/cupy/random_tests/test_generator.py b/dpnp/tests/third_party/cupy/random_tests/test_generator.py index 1cda3a3dc7cf..af33a006b52e 100644 --- a/dpnp/tests/third_party/cupy/random_tests/test_generator.py +++ b/dpnp/tests/third_party/cupy/random_tests/test_generator.py @@ -21,6 +21,15 @@ pytest.skip("random.generator() is not supported yet", allow_module_level=True) +def test_get_indices_preserves_uint64_cumsum_values(): + sentinel = numpy.iinfo(numpy.uint64).max + csum = cupy.array([2**32 + 1, 2**32 + 1], dtype=cupy.uint64) + indices = cupy.full(2, sentinel, dtype=cupy.uint64) + _generator.RandomState._kernel_get_indices(csum, indices, size=csum.size) + assert indices[0] == 0 + assert indices[1] == sentinel + + def numpy_cupy_equal_continuous_distribution(significance_level, name="xp"): """Decorator that tests the distributions of NumPy samples and CuPy ones. diff --git a/dpnp/tests/third_party/cupy/random_tests/test_sample.py b/dpnp/tests/third_party/cupy/random_tests/test_sample.py index b71ca1a43e7b..645d6a252ad5 100644 --- a/dpnp/tests/third_party/cupy/random_tests/test_sample.py +++ b/dpnp/tests/third_party/cupy/random_tests/test_sample.py @@ -7,6 +7,8 @@ import dpnp as cupy from dpnp import random from dpnp.tests.third_party.cupy import testing + +# from cupy.random import _sample from dpnp.tests.third_party.cupy.testing import _condition, _hypothesis @@ -293,6 +295,21 @@ def test_randn_invalid_argument(self): @pytest.mark.skip("random.multinomial() is not fully supported") class TestMultinomial(unittest.TestCase): + def test_kernel_accepts_large_n(self): + xs = cupy.array([0], dtype=cupy.int64) + ys = cupy.zeros(1, dtype="l") + _sample._multinominal_kernel(xs, 1, 2**31, ys) + assert ys[0] == 1 + + def test_kernel_uses_large_p_for_output_offset(self): + xs = cupy.array([0, 0], dtype=cupy.int64) + ys = cupy.zeros(2, dtype="l") + ys_view = cupy.lib.stride_tricks.as_strided( + ys, shape=(2, 2**31), strides=(ys.itemsize, 0) + ) + _sample._multinominal_kernel(xs, 2**31, 1, ys_view) + testing.assert_array_equal(ys, cupy.ones(2, dtype="l")) + @_condition.repeat(3, 10) @testing.for_float_dtypes() @testing.numpy_cupy_allclose(rtol=0.05) diff --git a/dpnp/tests/third_party/cupy/statistics_tests/test_histogram.py b/dpnp/tests/third_party/cupy/statistics_tests/test_histogram.py index edb713830e42..cfaeeecf55a3 100644 --- a/dpnp/tests/third_party/cupy/statistics_tests/test_histogram.py +++ b/dpnp/tests/third_party/cupy/statistics_tests/test_histogram.py @@ -8,6 +8,8 @@ from dpnp.tests.helper import has_support_aspect64, numpy_version from dpnp.tests.third_party.cupy import testing +# from cupy._statistics import histogram as histogram_module + # Note that numpy.bincount does not support uint64 on 64-bit environment # as it casts an input array to intp (planned to support since 2.2.4). # And it does not support uint32, int64 and uint64 on 32-bit environment. @@ -46,6 +48,43 @@ def for_all_dtypes_combination_bincount(names): class TestHistogram(unittest.TestCase): + # Number of bins that makes the searched index exceed 2**31. + _n_bins = 2**31 + 1 + + def _large_bins_and_counters(self, dtype): + # Equal bins send the binary search to the last one, `n_bins - 2`, + # without needing the gigabytes a monotonic `bins` would take. `y` + # only has to be indexable that far, and cycling it over three + # counters keeps it free while still recording which bin was picked. + bins = cupy.broadcast_to( + cupy.array([0], dtype=cupy.float32), (self._n_bins,) + ) + counters = cupy.zeros(3, dtype=dtype) + y = cupy.lib.stride_tricks.as_strided( + counters, + shape=(self._n_bins // counters.size + 1, counters.size), + strides=(0, counters.itemsize), + ) + return bins, counters, y + + @pytest.mark.skip("_histogram_kernel is not supported") + def test_kernel_accepts_large_number_of_bins(self): + x = cupy.zeros(1, dtype=cupy.float32) + bins, counters, y = self._large_bins_and_counters(cupy.int64) + histogram_module._histogram_kernel(x, bins, bins.size, y) + # (2**31 - 1) % 3 == 1 + testing.assert_array_equal(counters, [0, 1, 0]) + + @pytest.mark.skip("_weighted_histogram_kernel is not supported") + def test_weighted_kernel_accepts_large_number_of_bins(self): + x = cupy.zeros(1, dtype=cupy.float32) + weights = cupy.full(1, 2, dtype=cupy.float32) + bins, counters, y = self._large_bins_and_counters(cupy.float32) + histogram_module._weighted_histogram_kernel( + x, bins, bins.size, weights, y + ) + testing.assert_array_equal(counters, [0, 2, 0]) + @testing.for_all_dtypes(no_bool=True, no_complex=True) @testing.numpy_cupy_allclose(atol=1e-6, type_check=has_support_aspect64()) def test_histogram(self, xp, dtype): diff --git a/dpnp/tests/third_party/cupy/statistics_tests/test_order.py b/dpnp/tests/third_party/cupy/statistics_tests/test_order.py index 97d464fa63f2..2b3d22f9e41c 100644 --- a/dpnp/tests/third_party/cupy/statistics_tests/test_order.py +++ b/dpnp/tests/third_party/cupy/statistics_tests/test_order.py @@ -11,6 +11,8 @@ # from cupy import cuda from dpnp.tests.third_party.cupy import testing +# from cupy._statistics import order as order_module + _all_methods = ( "inverted_cdf", # 'averaged_inverted_cdf', # TODO(takagi) Not implemented @@ -51,6 +53,17 @@ def for_all_methods(name="method"): return pytest.mark.parametrize(name, _all_methods) +@pytest.mark.skip("_get_percentile_weightnening_kernel() is not supported") +def test_percentile_kernel_accepts_large_dimensions(): + indices = cupy.array([0], dtype=cupy.float64) + a = cupy.broadcast_to(cupy.array([1], dtype=cupy.float64), (2**31,)) + out = cupy.empty(1, dtype=cupy.float64) + order_module._get_percentile_weightnening_kernel()( + indices, a, 0, a.size, out + ) + assert out[0] == 1 + + @pytest.mark.skip("dpnp.quantile() is not implemented yet") @testing.with_requires("numpy>=1.22.0rc1") class TestQuantile: diff --git a/dpnp/tests/third_party/cupyx/scipy_tests/linalg_tests/test_decomp_lu.py b/dpnp/tests/third_party/cupyx/scipy_tests/linalg_tests/test_decomp_lu.py index 440521419652..251a5c7a450d 100644 --- a/dpnp/tests/third_party/cupyx/scipy_tests/linalg_tests/test_decomp_lu.py +++ b/dpnp/tests/third_party/cupyx/scipy_tests/linalg_tests/test_decomp_lu.py @@ -9,6 +9,8 @@ import dpnp as cupy from dpnp.tests.third_party.cupy import testing +# from cupyx.scipy.linalg import _decomp_lu + if cupy.tests.helper.is_scipy_available(): import scipy.linalg @@ -20,6 +22,35 @@ ) +@pytest.mark.skip("_kernel_cupy_split_lu is not supported") +def test_split_lu_kernel_accepts_large_dimensions(): + large = 2**31 + lu = cupy.lib.stride_tricks.as_strided( + cupy.array([2], dtype=cupy.float32), shape=(large, 1), strides=(0, 0) + ) + lower = cupy.empty(1, dtype=cupy.float32) + upper = cupy.empty(1, dtype=cupy.float32) + _decomp_lu._kernel_cupy_split_lu( + lu, large, 1, 1, lower._c_contiguous, lower, upper, size=1 + ) + assert lower[0] == 1 + assert upper[0] == 2 + + +@pytest.mark.skip("_kernel_cupy_laswp is not supported") +def test_laswp_kernel_accepts_large_dimensions(): + large = 2**31 + pivots = cupy.array([0], dtype=cupy.int64) + a_base = cupy.array([3], dtype=cupy.float32) + a = cupy.lib.stride_tricks.as_strided( + a_base, shape=(large, 1), strides=(0, 0) + ) + _decomp_lu._kernel_cupy_laswp( + large, 1, 0, 0, pivots, 1, a._c_contiguous, a, size=1 + ) + assert a_base[0] == 3 + + @testing.parameterize( *testing.product( {