diff --git a/Project.toml b/Project.toml index 2fa3ee30a..9003240c6 100644 --- a/Project.toml +++ b/Project.toml @@ -9,7 +9,6 @@ projects = ["test", "docs", "perf"] [deps] AMDGPU_LLVM_Backend_jll = "cc5c0156-bd05-5a77-8a68-bb0aafb29019" AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" -AcceleratedKernels = "6a4ca0a5-0e36-4168-a932-d9be78d558f1" Adapt = "79e6a3ab-5dfb-504d-930d-738a2a938a0e" Atomix = "a9b6321e-bd34-4604-b9c9-b65b8de01458" BFloat16s = "ab4f0b2a-ad5b-11e8-123f-65d77653426b" @@ -51,7 +50,6 @@ AMDGPUSpecialFunctionsExt = "SpecialFunctions" [compat] AMDGPU_LLVM_Backend_jll = "23" AbstractFFTs = "1.0" -AcceleratedKernels = "0.3.1, 0.4" Adapt = "4" Atomix = "1.5" BFloat16s = "0.6.0" @@ -59,7 +57,7 @@ CEnum = "0.4, 0.5" ChainRulesCore = "1" EnzymeCore = "0.8" ExprTools = "0.1" -GPUArrays = "11.5.14" +GPUArrays = "12" GPUCompiler = "2.13" GPUToolbox = "3.3.1" KernelAbstractions = "0.9.2" diff --git a/src/AMDGPU.jl b/src/AMDGPU.jl index c713d4385..799d958eb 100644 --- a/src/AMDGPU.jl +++ b/src/AMDGPU.jl @@ -11,7 +11,6 @@ using LLVM using Preferences using Printf -import AcceleratedKernels as AK import UnsafeAtomics import Atomix import Atomix: @atomic, @atomicswap, @atomicreplace @@ -123,12 +122,6 @@ include("conversions.jl") include("broadcast.jl") include("exception_handler.jl") -include("kernels/mapreduce.jl") -include("kernels/indexing.jl") -include("kernels/accumulate.jl") -include("kernels/sorting.jl") -include("kernels/reverse.jl") - include("blas/rocBLAS.jl") include("solver/rocSOLVER.jl") include("sparse/rocSPARSE.jl") diff --git a/src/array.jl b/src/array.jl index f57a52ccd..40e266e20 100644 --- a/src/array.jl +++ b/src/array.jl @@ -189,7 +189,34 @@ Base.similar(::ROCArray{<:Any, <:Any, B}, ::Type{T}, dims::Base.Dims{N}) where { Base.elsize(::Type{<:ROCArray{T}}) where {T} = aligned_sizeof(T) Base.size(x::ROCArray) = x.dims Base.sizeof(x::ROCArray) = Base.elsize(x) * length(x) -Base.dataids(A::ROCArray) = (UInt(pointer(A)),) + +## alias detection + +# The device address of the memory `x` lives in. Not using `pointer(x)`, which takes +# ownership of that memory for the current stream. +function base_address(x::ROCArray) + mem = x.buf[].mem + return UInt(mem isa Mem.HostBuffer ? mem.dev_ptr : mem.ptr) +end + +# Identify the underlying memory, not just where this array starts in it: derived arrays +# (contiguous views, reshapes, reinterprets) have an offset into their parent's memory, and +# Base compares `dataids` whenever one side is wrapped (e.g., a `SubArray`). The start address +# is included too, to match memory that was `unsafe_wrap`ped from it. Empty arrays alias +# nothing, which Julia 1.10's `mightalias` does not check (and they all have address 0). +function Base.dataids(x::ROCArray) + isempty(x) && return () + base = base_address(x) + return (base, base + x.offset) +end + +Base.unaliascopy(x::ROCArray) = copy(x) + +function Base.mightalias(A::ROCArray, B::ROCArray) + startA = base_address(A) + A.offset + startB = base_address(B) + B.offset + return startA <= startB < startA + sizeof(A) || startB <= startA < startB + sizeof(B) +end ## interop with Julia arrays diff --git a/src/kernels/accumulate.jl b/src/kernels/accumulate.jl deleted file mode 100644 index 30b12df8a..000000000 --- a/src/kernels/accumulate.jl +++ /dev/null @@ -1,8 +0,0 @@ -Base.accumulate!(op, B::AnyROCArray, A::AnyROCArray; init=zero(eltype(A)), kwargs...) = - AK.accumulate!(op, B, A, ROCBackend(); init, kwargs...) - -Base.accumulate(op, A::AnyROCArray; init=zero(eltype(A)), kwargs...) = - AK.accumulate(op, A, ROCBackend(); init, kwargs...) - -Base.cumsum(src::AnyROCArray; kwargs...) = AK.cumsum(src, ROCBackend(); kwargs...) -Base.cumprod(src::AnyROCArray; kwargs...) = AK.cumprod(src, ROCBackend(); kwargs...) diff --git a/src/kernels/indexing.jl b/src/kernels/indexing.jl deleted file mode 100644 index 6b8eca9b7..000000000 --- a/src/kernels/indexing.jl +++ /dev/null @@ -1,36 +0,0 @@ -Base.to_index(::ROCArray, I::AbstractArray{Bool}) = findall(I) - -if VERSION >= v"1.11.0-DEV.1157" - Base.to_indices(x::ROCArray, I::Tuple{AbstractArray{Bool}}) = - (Base.to_index(x, I[1]),) -end - -function Base.findall(bools::AnyROCArray{Bool}) - I = keytype(bools) - indices = cumsum(reshape(bools, prod(size(bools)))) - - n = isempty(indices) ? 0 : @allowscalar indices[end] - ys = ROCArray{I}(undef, n) - - if n > 0 - function _ker!(ys, bools, indices) - i = workitemIdx().x + (workgroupIdx().x - Int32(1)) * workgroupDim().x - - @inbounds if i ≤ length(bools) && bools[i] - ii = CartesianIndices(bools)[i] - b = indices[i] # new position - ys[b] = ii - end - return - end - - kernel = @roc launch=false _ker!(ys, bools, indices) - config = launch_configuration(kernel) - groupsize = min(length(indices), config.groupsize) - gridsize = cld(length(indices), groupsize) - kernel(ys, bools, indices; groupsize, gridsize) - end - unsafe_free!(indices) - - return ys -end diff --git a/src/kernels/mapreduce.jl b/src/kernels/mapreduce.jl deleted file mode 100644 index cae6b9e41..000000000 --- a/src/kernels/mapreduce.jl +++ /dev/null @@ -1,169 +0,0 @@ -# TODO -# - serial version for lower latency -# - group-stride loop to delay need for second kernel launch - -# Reduce a value across a group, using local memory for communication -@inline function reduce_group(op, val::T, neutral) where T - items = workgroupDim().x - item = workitemIdx().x - - # Shared mem for a complete reduction. - shared = @ROCDynamicLocalArray(T, items, false) - @inbounds shared[item] = val - - # Perform a reduction. - d = 1 - while d < items - sync_workgroup() - index = 2 * d * (item - 1) + 1 - @inbounds if index ≤ items - other_val = (index + d) ≤ items ? shared[index + d] : neutral - shared[index] = op(shared[index], other_val) - end - d *= 2 - end - - # Load the final value on the first item. - if item == 1 - val = @inbounds shared[item] - end - - return val -end - -Base.@propagate_inbounds _map_getindex(args::Tuple, I) = ((args[1][I]), _map_getindex(Base.tail(args), I)...) -Base.@propagate_inbounds _map_getindex(args::Tuple{Any}, I) = ((args[1][I]),) -Base.@propagate_inbounds _map_getindex(args::Tuple{}, I) = () - -# Reduce an array across the grid. All elements to be processed can be addressed by the -# product of the two iterators `Rreduce` and `Rother`, where the latter iterator will have -# singleton entries for the dimensions that should be reduced (and vice versa). -function partial_mapreduce_device(f, op, neutral, Rreduce, Rother, R, As...) - # decompose the 1D hardware indices into separate ones for reduction (across items - # and possibly groups if it doesn't fit) and other elements (remaining groups) - localIdx_reduce = workitemIdx().x - localDim_reduce = workgroupDim().x - - groupIdx_reduce, groupIdx_other = fldmod1(workgroupIdx().x, length(Rother)) - groupDim_reduce = gridGroupDim().x ÷ length(Rother) - - # group-based indexing into the values outside of the reduction dimension - # (that means we can safely synchronize items within this group) - iother = groupIdx_other - @inbounds if iother ≤ length(Rother) - Iother = Rother[iother] - - # load the neutral value - Iout = CartesianIndex(Tuple(Iother)..., groupIdx_reduce) - neutral = neutral ≡ nothing ? R[Iout] : neutral - - val = op(neutral, neutral) - - # reduce serially across chunks of input vector that don't fit in a group - ireduce = localIdx_reduce + (groupIdx_reduce - 1) * localDim_reduce - while ireduce ≤ length(Rreduce) - Ireduce = Rreduce[ireduce] - J = max(Iother, Ireduce) - val = op(val, f(_map_getindex(As, J)...)) - ireduce += localDim_reduce * groupDim_reduce - end - - val = reduce_group(op, val, neutral) - - # write back to memory - if localIdx_reduce == 1 - R[Iout] = val - end - end - - return -end - -function GPUArrays.mapreducedim!( - f::F, op::OP, R::AnyROCArray{T}, - A::Union{AbstractArray,Broadcast.Broadcasted}; init=nothing, -) where {F, OP, T} - Base.check_reducedims(R, A) - length(A) == 0 && return R # isempty(::Broadcasted) iterates - - R_old = R - # add singleton dimensions to the output container, if needed - if ndims(R) < ndims(A) - dims = Base.fill_to_length(size(R), 1, Val(ndims(A))) - R = reshape(R, dims) - end - - # iteration domain, split in two: one part covers the dimensions that should - # be reduced, and the other covers the rest. combining both covers all values. - Rall = CartesianIndices(axes(A)) - Rother = CartesianIndices(axes(R)) - Rreduce = CartesianIndices(ifelse.(axes(A) .== axes(R), Ref(Base.OneTo(1)), axes(A))) - # NOTE: we hard-code `OneTo` (`first.(axes(A))` would work too) or we get a - # CartesianIndices object with UnitRanges that behave badly on the GPU. - @assert length(Rall) == length(Rother) * length(Rreduce) - - # allocate an additional, empty dimension to write the reduced value to. - # this does not affect the actual location in memory of the final values, - # but allows us to write a generalized kernel supporting partial reductions. - R′ = reshape(R, (size(R)..., 1)) - - # how many items do we want? - # - # items in a group work together to reduce values across the reduction dimensions; - # we want as many as possible to improve algorithm efficiency and execution occupancy. - wanted_items = nextpow(2, length(Rreduce)) - function compute_items(max_items) - if wanted_items > max_items - prevpow(2, max_items) - else - wanted_items - end - end - - # how many items can we launch? - # - # we might not be able to launch all those items to reduce each slice in one go. - # that's why each items also loops across their inputs, processing multiple values - # so that we can span the entire reduction dimension using a single item group. - max_block_size = 256 - compute_shmem(items) = items * aligned_sizeof(T) - max_shmem = max_block_size |> compute_items |> compute_shmem - kernel = @roc launch=false partial_mapreduce_device( - f, op, init, Rreduce, Rother, R′, A) - kernel_config = launch_configuration(kernel; shmem=max_shmem, max_block_size) - reduce_items = compute_items(kernel_config.groupsize) - reduce_shmem = compute_shmem(reduce_items) - - # how many groups should we launch? - # - # even though we can always reduce each slice in a single item group, that may not be - # optimal as it might not saturate the GPU. we already launch some groups to process - # independent dimensions in parallel; pad that number to ensure full occupancy. - other_groups = length(Rother) - reduce_groups = cld(length(Rreduce), reduce_items) - - # determine the launch configuration - blocks = reduce_items - grid = reduce_groups * other_groups - - # perform the actual reduction - if reduce_groups == 1 - # we can cover the dimensions to reduce using a single group - @roc gridsize=grid groupsize=blocks shmem=reduce_shmem partial_mapreduce_device( - f, op, init, Rreduce, Rother, R′, A) - else - # we need multiple steps to cover all values to reduce - partial = similar(R, (size(R)..., reduce_groups)) - if init === nothing - # without an explicit initializer we need to copy from the output container - partial .= R - end - @roc gridsize=grid groupsize=blocks shmem=reduce_shmem partial_mapreduce_device( - f, op, init, Rreduce, Rother, partial, A) - - GPUArrays.mapreducedim!(identity, op, R′, partial; init) - AMDGPU.unsafe_free!(partial) - end - - return R_old -end diff --git a/src/kernels/reverse.jl b/src/kernels/reverse.jl deleted file mode 100644 index bc933258a..000000000 --- a/src/kernels/reverse.jl +++ /dev/null @@ -1,124 +0,0 @@ -# Adapted from CUDA.jl. - -# 1D case. - -Base.reverse(x::AnyROCVector) = @inbounds reverse(x, 1, length(x)) - -Base.@propagate_inbounds function Base.reverse( - x::AnyROCVector, start::Integer, stop::Integer = length(x), -) - y = similar(x) - - start > 1 && copyto!(y, 1, x, 1, start - 1) - stop < length(x) && copyto!(y, stop + 1, x, stop + 1) - _reverse!(@view(y[start:stop]), @view(x[start:stop])) - - return y -end - -Base.@propagate_inbounds function Base.reverse!( - x::AnyROCVector, start::Integer, stop::Integer = length(x), -) - _reverse!(@view(x[start:stop])) - return x -end - -Base.reverse!(x::AnyROCVector) = @inbounds reverse!(x, 1, length(x)) - -# N-D case. - -function Base.reverse(x::AnyROCArray; dims=:) - isa(dims, Colon) && (dims = 1:ndims(x);) - - applicable(iterate, dims) || throw(ArgumentError( - "Dimension `$dims` is not an iterable.")) - all(1 .≤ dims .≤ ndims(x)) || throw(ArgumentError( - "Dimension `$dims` is not 1 ≤ $dims ≤ $(ndims(x))")) - - all(size(x)[[dims...]] .== 1) && return copy(x) # Dims to reverse are 1. - - y = similar(x) - _reverse!(y, x; dims) - return y -end - -function Base.reverse!(x::AnyROCArray; dims=:) - isa(dims, Colon) && (dims = 1:ndims(x);) - - applicable(iterate, dims) || throw(ArgumentError( - "Dimension `$dims` is not an iterable.")) - all(1 .≤ dims .≤ ndims(x)) || throw(ArgumentError( - "Dimension `$dims` is not 1 ≤ $dims ≤ $(ndims(x))")) - - _reverse!(x; dims) - return x -end - -# Out-of-place kernel, swapping a single element per thread. - -function _reverse!( - y::AnyROCArray{T, N}, x::AnyROCArray{T, N}; dims=1:ndims(x), -) where {T, N} - size(x) == size(y) || throw(DimensionMismatch( - "Input and output arrays must equal in size, instead: " * - "`$(size(x))` vs `$(size(y))`.")) - - rev_dims = ntuple(d -> (d ∈ dims) && (size(x, d) > 1), N) - - ref = size(x) .+ 1 - lin_ids = LinearIndices(x) - nd_ids = CartesianIndices(x) - - function _kernel!(y::AbstractArray{T, N}, x::AbstractArray{T, N}) where {T, N} - i = workitemIdx().x + (workgroupIdx().x - 0x1) * workgroupDim().x - - if i ≤ length(x) - idx = Tuple(nd_ids[i]) - idx = ifelse.(rev_dims, ref .- idx, idx) - idx_out = lin_ids[idx...] - y[idx_out] = x[i] - end - return - end - - groupsize = 256 - gridsize = cld(length(x), groupsize) - iszero(gridsize) && return - @roc groupsize=groupsize gridsize=gridsize _kernel!(y, x) -end - -# In-place kernel, swapping elements on half the number of threads. - -function _reverse!(x::AnyROCArray{T, N}; dims=1:ndims(x)) where {T, N} - rev_dims = ntuple(d -> (d ∈ dims) && (size(x, d) > 1), N) - half_dim = findlast(rev_dims) - half_dim ≡ nothing && return # No reverse needed in this case. - - reduced_sz = ntuple(d -> (d == half_dim) ? cld(size(x, d), 2) : size(x, d), N) - reduced_len = prod(reduced_sz) - - ref = size(x) .+ 1 - lin_ids = LinearIndices(x) - nd_ids = CartesianIndices(reduced_sz) - - function _kernel!(x::AbstractArray{T, N}) where {T, N} - i = workitemIdx().x + (workgroupIdx().x - 0x1) * workgroupDim().x - - if i ≤ reduced_len - idx = Tuple(nd_ids[i]) - idx_in = lin_ids[idx...] - idx = ifelse.(rev_dims, ref .- idx, idx) - idx_out = lin_ids[idx...] - - if idx_in < idx_out - x[idx_in], x[idx_out] = x[idx_out], x[idx_in] - end - end - return - end - - groupsize = 256 - gridsize = cld(prod(reduced_sz), groupsize) - iszero(gridsize) && return - @roc groupsize=groupsize gridsize=gridsize _kernel!(x) -end diff --git a/src/kernels/sorting.jl b/src/kernels/sorting.jl deleted file mode 100644 index e65b06d94..000000000 --- a/src/kernels/sorting.jl +++ /dev/null @@ -1,92 +0,0 @@ -Base.sort!(x::AnyROCArray; dims::Union{Nothing, Integer}=nothing, kwargs...) = - isnothing(dims) ? (AK.sort!(x; kwargs...); x) : _sort_dims!(x, dims; kwargs...) - -# Base's out-of-place `sort(x; dims)` sorts chunks with the CPU algorithm, which trips -# scalar indexing - route it through `sort!` instead. -Base.sort(x::AnyROCArray; kwargs...) = sort!(copy(x); kwargs...) - -Base.sortperm!(ix::AnyROCArray, x::AnyROCArray; dims::Union{Nothing, Integer}=nothing, kwargs...) = - isnothing(dims) ? - (AK.sortperm!(ix, x; kwargs...); ix) : - _sortperm_dims!(ix, x, dims; kwargs...) - -Base.sortperm(x::AnyROCArray; dims::Union{Nothing, Integer}=nothing, kwargs...) = - isnothing(dims) ? - sortperm!(ROCArray(1:length(x)), x; kwargs...) : - sortperm!(reshape(ROCArray(1:length(x)), size(x)), x; dims, kwargs...) - -# Sorting along `dims`, a stopgap until AcceleratedKernels support for it reaches AMDGPU. -# Each element is tagged with the index of its slice and the array is sorted once, ordered -# lexicographically by `(slice, element)`; slices then come out grouped and internally sorted, -# and are scattered back. - -# Tag element `(i, j - 1)` of the `(sd * n, rest)` view of `x` with its slice index, plus its -# linear index for `sortperm`. Broadcast alongside `x` itself, so the ranges stay lazy. -_slice_kv(::Type{K}, sd, i, j, v) where K = (K(mod1(i, sd) + sd * j), v) -_slice_kvi(::Type{K}, ::Type{I}, sd, n, i, j, v) where {K, I} = - (K(mod1(i, sd) + sd * j), v, I(i + sd * n * j)) - -_slice_value(kv) = kv[2] -_slice_index(kv) = kv[3] - -# `x` seen as `(sd, n, rest)`: `n` elements along `dims`, strided by `sd`, over `rest` groups. -function _slice_layout(x::AnyROCArray, dims::Integer) - 1 ≤ dims ≤ ndims(x) || throw(ArgumentError("dimension out of range")) - n = size(x, dims) - sd = prod(size(x)[1:dims - 1]) - return n, sd, length(x) ÷ (n * sd) -end - -# Smallest key type that can enumerate the slices of `x`. -_slice_keytype(x::AnyROCArray) = length(x) ≤ typemax(Int32) ? Int32 : Int64 - -# `(slice, element)` ordering, ties broken by the user-supplied ordering. -_slice_lt(ord) = (a, b) -> a[1] == b[1] ? Base.Order.lt(ord, a[2], b[2]) : a[1] < b[1] - -# Scatter the sorted - and therefore slice-grouped - `kv` back over `dst`. -function _slice_scatter!(f::F, dst, kv, n, sd, rest) where F - if sd == 1 - reshape(dst, n, rest) .= f.(reshape(kv, n, rest)) - else - reshape(dst, sd, n, rest) .= f.(PermutedDimsArray(reshape(kv, n, sd, rest), (2, 1, 3))) - end - return dst -end - -function _sort_dims!( - x::AnyROCArray, dims::Integer; - lt=isless, by=identity, rev::Union{Nothing, Bool}=nothing, - order::Base.Order.Ordering=Base.Order.Forward, kwargs..., -) - isempty(x) && return x - n, sd, rest = _slice_layout(x, dims) - K = _slice_keytype(x) - - kv = similar(x, Tuple{K, eltype(x)}, length(x)) - reshape(kv, sd * n, rest) .= _slice_kv.( - K, sd, 1:sd * n, (0:rest - 1)', reshape(x, sd * n, rest)) - - AK.sort!(kv; lt=_slice_lt(Base.Order.ord(lt, by, rev, order)), kwargs...) - - return _slice_scatter!(_slice_value, x, kv, n, sd, rest) -end - -function _sortperm_dims!( - ix::AnyROCArray, x::AnyROCArray, dims::Integer; - lt=isless, by=identity, rev::Union{Nothing, Bool}=nothing, - order::Base.Order.Ordering=Base.Order.Forward, kwargs..., -) - axes(ix) == axes(x) || throw(ArgumentError( - "index array has axes $(axes(ix)), but sorted array has axes $(axes(x))")) - isempty(x) && return ix - n, sd, rest = _slice_layout(x, dims) - K, I = _slice_keytype(x), eltype(ix) - - kv = similar(x, Tuple{K, eltype(x), I}, length(x)) - reshape(kv, sd * n, rest) .= _slice_kvi.( - K, I, sd, n, 1:sd * n, (0:rest - 1)', reshape(x, sd * n, rest)) - - AK.sort!(kv; lt=_slice_lt(Base.Order.ord(lt, by, rev, order)), kwargs...) - - return _slice_scatter!(_slice_index, ix, kv, n, sd, rest) -end diff --git a/src/random.jl b/src/random.jl index 69c9005f1..8a5e46df6 100644 --- a/src/random.jl +++ b/src/random.jl @@ -4,15 +4,13 @@ using Random const GPUARRAY_RNG_KEY = :AMDGPU_GPUARRAY_RNG -function GPUArrays.default_rng(::Type{<:ROCArray}) +function gpuarrays_rng() rngs = get!(() -> Dict{HIPDevice,GPUArrays.RNG}(), task_local_storage(), GPUARRAY_RNG_KEY) return get!(rngs, device()) do GPUArrays.RNG{ROCArray}() end end - -gpuarrays_rng() = GPUArrays.default_rng(ROCArray) rocrand_rng() = rocRAND.handle() # the interface is split in two levels: diff --git a/test/core/rocarray_base.jl b/test/core/rocarray_base.jl index 18e2f0f37..106f2d80d 100644 --- a/test/core/rocarray_base.jl +++ b/test/core/rocarray_base.jl @@ -61,6 +61,55 @@ end @test Array(r) == reinterpret(Int64, @view Array(a)[2:7]) end +@testset "aliasing" begin + x = ROCArray([1, 2]) + y = view(x, 2:2) + @test Base.mightalias(x, x) + @test Base.mightalias(x, y) + z = view(x, 1:1) + @test Base.mightalias(x, z) + @test !Base.mightalias(y, z) + + a = copy(y)::typeof(x) + @test !Base.mightalias(x, a) + a .= 3 + @test Array(y) == [2] + + b = Base.unaliascopy(y)::typeof(y) + @test !Base.mightalias(x, b) + b .= 3 + @test Array(y) == [2] + + # contiguous views are ROCArrays with an offset into the parent's memory, + # which should still alias wrapped arrays (like SubArrays) of that memory + x = ROCArray(1:16) + @test Base.mightalias(view(x, 2:16), view(x, 15:-1:1)) + @test Base.mightalias(view(x, 1:2:15), view(x, 2:9)) + @test Base.mightalias(view(x, 2:16), view(reinterpret(Int32, x), 1:2:31)) + @test !Base.mightalias(view(x, 2:16), view(ROCArray(1:16), 15:-1:1)) + + # memory wrapped from a view's pointer + y = view(x, 2:16) + z = unsafe_wrap(ROCArray, pointer(y), size(y)) + @test Base.mightalias(y, view(z, 15:-1:1)) + + # so in-place broadcasts between them should make a copy first + n = 2^20 + x = ROCArray{Float32}(1:n) + view(x, 2:n) .= view(x, n-1:-1:1) + @test Array(x) == [1; n-1:-1:1] + + # empty arrays alias nothing, also on Julia 1.10 (where all have address 0) + @test !Base.mightalias(ROCArray(Int[]), ROCArray(Float32[])) + @test !Base.mightalias(view(ROCArray(zeros(2, 0)), 1:1, :), ROCArray(Int[])) + @test isempty(cumsum(ROCArray(Int[]))) + + # disjoint parts of one array may be each other's source and destination + x = ROCArray(collect(1:10)) + @test Array(sum!(view(x, 1:1), view(x, 2:10))) == [54] + @test Array(cumsum!(view(x, 1:5), view(x, 6:10))) == cumsum(6:10) +end + @testset "resize!" begin a_h = Array(range(1, 10)) a_d = a_h |> roc @@ -378,10 +427,4 @@ end @test Array(x) == [true, true] end -@testset "mapreducedim! returning same type" begin - R = transpose(AMDGPU.zeros(Float32, 2, 3)) - A = ROCArray(rand(Float32, 3, 2, 10)) - @test @inferred(GPUArrays.mapreducedim!(identity, +, R, A)) === R -end - end