Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,7 @@ This release is compatible with NumPy 2.5.
* Fixed `dpnp.nanmedian` dropping kept dimensions of size 1, which produced a wrong result shape [#3081](https://github.com/IntelPython/dpnp/pull/3081)
* Fixed incorrect results of `dpnp.tensor.vecdot` in some cases with strided outputs and of `dpnp.tensor` reductions, `dpnp.tensor.vecdot` and `dpnp.tensor.matmul` on large inputs with some data types [#3082](https://github.com/IntelPython/dpnp/pull/3082)
* Fixed `dpnp.median` and `dpnp.nanmedian` raising a `ValueError` for a tuple `axis` when a kept dimension has size 0 [#3081](https://github.com/IntelPython/dpnp/pull/3081)
* Fixed a `RuntimeError` raised by `dpnp.tensor` kernels launched over more than `INT_MAX` work-items when built with DPC++ compiler 2026.2 or newer, by also passing `-fno-sycl-id-queries-fit-in-int` to the linker [#3089](https://github.com/IntelPython/dpnp/pull/3089)

### Security

Expand Down
5 changes: 4 additions & 1 deletion dpnp/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -100,7 +100,10 @@ function(build_dpnp_tensor_ext _trgt _src _dest)
if(BUILD_DPNP_TENSOR_SYCL)
add_sycl_to_target(TARGET ${_trgt} SOURCES ${_generated_src})
target_compile_options(${_trgt} PRIVATE -fno-sycl-id-queries-fit-in-int)
target_link_options(${_trgt} PRIVATE -fsycl-device-code-split=per_kernel)
target_link_options(
${_trgt}
PRIVATE -fsycl-device-code-split=per_kernel -fno-sycl-id-queries-fit-in-int
)
if(DPNP_TENSOR_OFFLOAD_COMPRESS)
target_link_options(${_trgt} PRIVATE --offload-compress)
endif()
Expand Down
5 changes: 4 additions & 1 deletion dpnp/backend/extensions/indexing/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -96,7 +96,10 @@ else()
endif()

target_compile_options(${python_module_name} PUBLIC -fno-sycl-id-queries-fit-in-int)
target_link_options(${python_module_name} PUBLIC -fsycl-device-code-split=per_kernel)
target_link_options(
${python_module_name}
PUBLIC -fsycl-device-code-split=per_kernel -fno-sycl-id-queries-fit-in-int
)

if(DPNP_GENERATE_COVERAGE)
target_link_options(
Expand Down
2 changes: 1 addition & 1 deletion dpnp/tensor/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -346,7 +346,7 @@ foreach(python_module_name ${_py_trgts})
)
target_link_options(
${python_module_name}
PRIVATE -fsycl-device-code-split=per_kernel
PRIVATE -fsycl-device-code-split=per_kernel -fno-sycl-id-queries-fit-in-int
)
if(DPNP_TENSOR_OFFLOAD_COMPRESS)
target_link_options(${python_module_name} PRIVATE --offload-compress)
Expand Down
18 changes: 18 additions & 0 deletions dpnp/tests/third_party/cupy/binary_tests/test_packing.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,6 +59,24 @@ def test_pack_invalid_array(self):
fa = cupy.array([10, 20, 30], dtype=float)
pytest.raises(TypeError, cupy.packbits, fa)

@testing.slow
# @pytest.mark.thread_unsafe(reason="Allocation too large.")
@pytest.mark.parametrize(
"bitorder, expected", [("big", 128), ("little", 1)]
)
def test_packbits_large_input(self, bitorder, expected):
a = packed = None
try:
a = cupy.zeros(2**31 + 1, dtype=cupy.uint8)
a[-1] = 1
packed = cupy.packbits(a, bitorder=bitorder)
assert packed[-1] == expected
except MemoryError:
pytest.skip("out of memory in test.")
finally:
del packed, a
# cupy.get_default_memory_pool().free_all_blocks()

def test_unpackbits(self):
self.check_unpackbits([])
self.check_unpackbits([0])
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -967,7 +967,12 @@ def test_adv_setitem_transp(self, xp):


class TestHugeArrays:
# These tests require a lot of memory
# These tests use arrays with large logical sizes.
def test_take_negative_index_at_int32_size_boundary(self):
# Only the broadcast view is huge, so this needs no real allocation.
arr = cupy.broadcast_to(cupy.array(42, dtype=cupy.uint8), (2**31,))
assert arr.take(-1).item() == 42

@testing.slow
def test_advanced(self):
try:
Expand All @@ -985,28 +990,41 @@ def test_advanced(self):
pytest.skip("out of memory in test.")

@testing.slow
# @pytest.mark.thread_unsafe(reason="Allocation too large.")
def test_take_array(self):
arr = indices = res = None
try:
arr = cupy.ones((1, 2**32), dtype=cupy.int8)
arr[0, 2**30] = 0 # We should see each of these once
arr[0, -1] = 0
res = arr.take(cupy.array([0, 0]), axis=0)
# sanity check, we mostly care about it not crashing.
assert res.sum() == 2 * (2**32 - 2)
arr = cupy.array([[[1, 2]], [[3, 4]]], dtype=cupy.uint8)
indices = cupy.broadcast_to(
cupy.array(0, dtype=cupy.int8), (2**30,)
)
res = arr.take(indices, axis=1)
testing.assert_array_equal(res[0, 0], [1, 2])
testing.assert_array_equal(res[0, -1], [1, 2])
testing.assert_array_equal(res[1, 0], [3, 4])
testing.assert_array_equal(res[1, -1], [3, 4])
except MemoryError:
pytest.skip("out of memory in test.")
finally:
del res, indices, arr
# cupy.get_default_memory_pool().free_all_blocks()

@testing.slow
# @pytest.mark.thread_unsafe(reason="Allocation too large.")
def test_take_scalar(self):
arr = res = None
try:
arr = cupy.ones((1, 2**32), dtype=cupy.int8)
arr[0, 2**30] = 0 # We should see each of these once
arr[0, -1] = 0
res = arr.take(0, axis=0)
# sanity check, we mostly care about it not crashing.
assert res.sum() == 2**32 - 2
arr = cupy.broadcast_to(
cupy.array([[1], [2]], dtype=cupy.uint8), (2, 2**31)
)
res = arr.take(1, axis=0)
assert res[0] == 2
assert res[-1] == 2
except MemoryError:
pytest.skip("out of memory in test.")
finally:
del res, arr
# cupy.get_default_memory_pool().free_all_blocks()

@testing.slow
def test_choose(self):
Expand Down
8 changes: 4 additions & 4 deletions dpnp/tests/third_party/cupy/creation_tests/test_basic.py
Original file line number Diff line number Diff line change
Expand Up @@ -336,17 +336,17 @@ def test_full_like_subok(self):

@pytest.mark.skip("_index_32_bits attribute is not supported by dpnp")
@pytest.mark.slow
# thread_unsafe marker requires pytest-run-parallel, not used by dpnp
# @pytest.mark.thread_unsafe(reason="large allocations")
@pytest.mark.parametrize(
"arr_factory,expected",
[
(lambda: cupy.empty(2**31 - 1, dtype=cupy.int8), True),
(lambda: cupy.empty(2**31, dtype=cupy.int8), True),
# Array sizes should fit as well, so 2**31 is also rejected:
(lambda: cupy.empty(2**31, dtype=cupy.int8), False),
(lambda: cupy.empty(2**31 + 1, dtype=cupy.int8)[::2], False),
(lambda: cupy.empty(2**31 // 8, dtype=cupy.complex64), True),
(lambda: cupy.empty(2**31 // 8, dtype=cupy.complex64), False),
(lambda: cupy.empty(2**31 // 8 + 1, dtype=cupy.complex64), False),
# Regression test for gh-9750:
# Regression test for gh-9750 (first view spans 2**31 - 4 bytes)
(lambda: cupy.empty(2**31 // 8, dtype=cupy.complex64).real, True),
(
lambda: cupy.empty(2**31 // 8 + 1, dtype=cupy.complex64).real,
Expand Down
33 changes: 33 additions & 0 deletions dpnp/tests/third_party/cupy/creation_tests/test_matrix.py
Original file line number Diff line number Diff line change
Expand Up @@ -126,6 +126,39 @@ def test_tri_posi(self, xp, dtype):
return xp.tri(*self.shape, k=1, dtype=dtype)


class TestTriLargeDimensions(unittest.TestCase):

@testing.slow
# @pytest.mark.thread_unsafe(reason="Allocation too large.")
def test_large_number_of_rows(self):
out = None
try:
# k must be 0: a nonzero k lets a truncated 32-bit `row` wrap
# back onto the expected answer instead of failing.
out = cupy.tri(2**31 + 1, 1, k=0, dtype=cupy.uint8)
assert out[0, 0] == 1
assert out[-1, 0] == 1
except MemoryError:
pytest.skip("out of memory in test.")
finally:
del out
# cupy.get_default_memory_pool().free_all_blocks()

@testing.slow
# @pytest.mark.thread_unsafe(reason="Allocation too large.")
def test_large_number_of_columns(self):
out = None
try:
out = cupy.tri(1, 2**31, k=2**31 - 1, dtype=cupy.uint8)
assert out[0, 0] == 1
assert out[0, -1] == 1
except MemoryError:
pytest.skip("out of memory in test.")
finally:
del out
# cupy.get_default_memory_pool().free_all_blocks()


@testing.parameterize(
{"shape": (2,)},
{"shape": (3, 3)},
Expand Down
9 changes: 9 additions & 0 deletions dpnp/tests/third_party/cupy/random_tests/test_generator.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,15 @@
pytest.skip("random.generator() is not supported yet", allow_module_level=True)


def test_get_indices_preserves_uint64_cumsum_values():
sentinel = numpy.iinfo(numpy.uint64).max
csum = cupy.array([2**32 + 1, 2**32 + 1], dtype=cupy.uint64)
indices = cupy.full(2, sentinel, dtype=cupy.uint64)
_generator.RandomState._kernel_get_indices(csum, indices, size=csum.size)
assert indices[0] == 0
assert indices[1] == sentinel


def numpy_cupy_equal_continuous_distribution(significance_level, name="xp"):
"""Decorator that tests the distributions of NumPy samples and CuPy ones.

Expand Down
17 changes: 17 additions & 0 deletions dpnp/tests/third_party/cupy/random_tests/test_sample.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@
import dpnp as cupy
from dpnp import random
from dpnp.tests.third_party.cupy import testing

# from cupy.random import _sample
from dpnp.tests.third_party.cupy.testing import _condition, _hypothesis


Expand Down Expand Up @@ -293,6 +295,21 @@ def test_randn_invalid_argument(self):
@pytest.mark.skip("random.multinomial() is not fully supported")
class TestMultinomial(unittest.TestCase):

def test_kernel_accepts_large_n(self):
xs = cupy.array([0], dtype=cupy.int64)
ys = cupy.zeros(1, dtype="l")
_sample._multinominal_kernel(xs, 1, 2**31, ys)
assert ys[0] == 1

def test_kernel_uses_large_p_for_output_offset(self):
xs = cupy.array([0, 0], dtype=cupy.int64)
ys = cupy.zeros(2, dtype="l")
ys_view = cupy.lib.stride_tricks.as_strided(
ys, shape=(2, 2**31), strides=(ys.itemsize, 0)
)
_sample._multinominal_kernel(xs, 2**31, 1, ys_view)
testing.assert_array_equal(ys, cupy.ones(2, dtype="l"))

@_condition.repeat(3, 10)
@testing.for_float_dtypes()
@testing.numpy_cupy_allclose(rtol=0.05)
Expand Down
39 changes: 39 additions & 0 deletions dpnp/tests/third_party/cupy/statistics_tests/test_histogram.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,8 @@
from dpnp.tests.helper import has_support_aspect64, numpy_version
from dpnp.tests.third_party.cupy import testing

# from cupy._statistics import histogram as histogram_module

# Note that numpy.bincount does not support uint64 on 64-bit environment
# as it casts an input array to intp (planned to support since 2.2.4).
# And it does not support uint32, int64 and uint64 on 32-bit environment.
Expand Down Expand Up @@ -46,6 +48,43 @@ def for_all_dtypes_combination_bincount(names):

class TestHistogram(unittest.TestCase):

# Number of bins that makes the searched index exceed 2**31.
_n_bins = 2**31 + 1

def _large_bins_and_counters(self, dtype):
# Equal bins send the binary search to the last one, `n_bins - 2`,
# without needing the gigabytes a monotonic `bins` would take. `y`
# only has to be indexable that far, and cycling it over three
# counters keeps it free while still recording which bin was picked.
bins = cupy.broadcast_to(
cupy.array([0], dtype=cupy.float32), (self._n_bins,)
)
counters = cupy.zeros(3, dtype=dtype)
y = cupy.lib.stride_tricks.as_strided(
counters,
shape=(self._n_bins // counters.size + 1, counters.size),
strides=(0, counters.itemsize),
)
return bins, counters, y

@pytest.mark.skip("_histogram_kernel is not supported")
def test_kernel_accepts_large_number_of_bins(self):
x = cupy.zeros(1, dtype=cupy.float32)
bins, counters, y = self._large_bins_and_counters(cupy.int64)
histogram_module._histogram_kernel(x, bins, bins.size, y)
# (2**31 - 1) % 3 == 1
testing.assert_array_equal(counters, [0, 1, 0])

@pytest.mark.skip("_weighted_histogram_kernel is not supported")
def test_weighted_kernel_accepts_large_number_of_bins(self):
x = cupy.zeros(1, dtype=cupy.float32)
weights = cupy.full(1, 2, dtype=cupy.float32)
bins, counters, y = self._large_bins_and_counters(cupy.float32)
histogram_module._weighted_histogram_kernel(
x, bins, bins.size, weights, y
)
testing.assert_array_equal(counters, [0, 2, 0])

@testing.for_all_dtypes(no_bool=True, no_complex=True)
@testing.numpy_cupy_allclose(atol=1e-6, type_check=has_support_aspect64())
def test_histogram(self, xp, dtype):
Expand Down
13 changes: 13 additions & 0 deletions dpnp/tests/third_party/cupy/statistics_tests/test_order.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,8 @@
# from cupy import cuda
from dpnp.tests.third_party.cupy import testing

# from cupy._statistics import order as order_module

_all_methods = (
"inverted_cdf",
# 'averaged_inverted_cdf', # TODO(takagi) Not implemented
Expand Down Expand Up @@ -51,6 +53,17 @@ def for_all_methods(name="method"):
return pytest.mark.parametrize(name, _all_methods)


@pytest.mark.skip("_get_percentile_weightnening_kernel() is not supported")
def test_percentile_kernel_accepts_large_dimensions():
indices = cupy.array([0], dtype=cupy.float64)
a = cupy.broadcast_to(cupy.array([1], dtype=cupy.float64), (2**31,))
out = cupy.empty(1, dtype=cupy.float64)
order_module._get_percentile_weightnening_kernel()(
indices, a, 0, a.size, out
)
assert out[0] == 1


@pytest.mark.skip("dpnp.quantile() is not implemented yet")
@testing.with_requires("numpy>=1.22.0rc1")
class TestQuantile:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,8 @@
import dpnp as cupy
from dpnp.tests.third_party.cupy import testing

# from cupyx.scipy.linalg import _decomp_lu

if cupy.tests.helper.is_scipy_available():
import scipy.linalg

Expand All @@ -20,6 +22,35 @@
)


@pytest.mark.skip("_kernel_cupy_split_lu is not supported")
def test_split_lu_kernel_accepts_large_dimensions():
large = 2**31
lu = cupy.lib.stride_tricks.as_strided(
cupy.array([2], dtype=cupy.float32), shape=(large, 1), strides=(0, 0)
)
lower = cupy.empty(1, dtype=cupy.float32)
upper = cupy.empty(1, dtype=cupy.float32)
_decomp_lu._kernel_cupy_split_lu(
lu, large, 1, 1, lower._c_contiguous, lower, upper, size=1
)
assert lower[0] == 1
assert upper[0] == 2


@pytest.mark.skip("_kernel_cupy_laswp is not supported")
def test_laswp_kernel_accepts_large_dimensions():
large = 2**31
pivots = cupy.array([0], dtype=cupy.int64)
a_base = cupy.array([3], dtype=cupy.float32)
a = cupy.lib.stride_tricks.as_strided(
a_base, shape=(large, 1), strides=(0, 0)
)
_decomp_lu._kernel_cupy_laswp(
large, 1, 0, 0, pivots, 1, a._c_contiguous, a, size=1
)
assert a_base[0] == 3


@testing.parameterize(
*testing.product(
{
Expand Down
Loading