From c77460e860b35f21277d8f1253e37304bd8b8993 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 7 Oct 2026 07:07:06 +0200 Subject: [PATCH 1/3] Keep the GPUArrays RNG in gpuarrays_rng instead of GPUArrays.default_rng GPUArrays 12 removes default_rng, and nothing outside AMDGPU called AMDGPU's method for it. --- src/random.jl | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/src/random.jl b/src/random.jl index 69c9005f1..8a5e46df6 100644 --- a/src/random.jl +++ b/src/random.jl @@ -4,15 +4,13 @@ using Random const GPUARRAY_RNG_KEY = :AMDGPU_GPUARRAY_RNG -function GPUArrays.default_rng(::Type{<:ROCArray}) +function gpuarrays_rng() rngs = get!(() -> Dict{HIPDevice,GPUArrays.RNG}(), task_local_storage(), GPUARRAY_RNG_KEY) return get!(rngs, device()) do GPUArrays.RNG{ROCArray}() end end - -gpuarrays_rng() = GPUArrays.default_rng(ROCArray) rocrand_rng() = rocRAND.handle() # the interface is split in two levels: From efb39dfab4e9a143bbfea60d9b50b80503e4a520 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 7 Oct 2026 07:07:06 +0200 Subject: [PATCH 2/3] Use GPUArrays 12's reductions, sorting, scans, findall and reverse GPUArrays 12 implements these once for every GPU array on top of AcceleratedKernels and removes the mapreducedim! hook, so delete AMDGPU's own reduction kernel and its sort!, sortperm!, accumulate, findall, logical indexing and reverse methods. The vector sort! and sortperm! methods were ambiguous with GPUArrays', and the scans called AcceleratedKernels 0.4's API. AMDGPU no longer depends on AcceleratedKernels directly; its AMDGPU extension still loads through GPUArrays. --- Project.toml | 4 +- src/AMDGPU.jl | 7 -- src/kernels/accumulate.jl | 8 -- src/kernels/indexing.jl | 36 -------- src/kernels/mapreduce.jl | 169 ------------------------------------- src/kernels/reverse.jl | 124 --------------------------- src/kernels/sorting.jl | 92 -------------------- test/core/rocarray_base.jl | 6 -- 8 files changed, 1 insertion(+), 445 deletions(-) delete mode 100644 src/kernels/accumulate.jl delete mode 100644 src/kernels/indexing.jl delete mode 100644 src/kernels/mapreduce.jl delete mode 100644 src/kernels/reverse.jl delete mode 100644 src/kernels/sorting.jl diff --git a/Project.toml b/Project.toml index 6c0a799f0..931826ad8 100644 --- a/Project.toml +++ b/Project.toml @@ -9,7 +9,6 @@ projects = ["test", "docs", "perf"] [deps] AMDGPU_LLVM_Backend_jll = "cc5c0156-bd05-5a77-8a68-bb0aafb29019" AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" -AcceleratedKernels = "6a4ca0a5-0e36-4168-a932-d9be78d558f1" Adapt = "79e6a3ab-5dfb-504d-930d-738a2a938a0e" Atomix = "a9b6321e-bd34-4604-b9c9-b65b8de01458" BFloat16s = "ab4f0b2a-ad5b-11e8-123f-65d77653426b" @@ -51,7 +50,6 @@ AMDGPUSpecialFunctionsExt = "SpecialFunctions" [compat] AMDGPU_LLVM_Backend_jll = "23" AbstractFFTs = "1.0" -AcceleratedKernels = "0.3.1, 0.4" Adapt = "4" Atomix = "1.5" BFloat16s = "0.6.0" @@ -59,7 +57,7 @@ CEnum = "0.4, 0.5" ChainRulesCore = "1" EnzymeCore = "0.8" ExprTools = "0.1" -GPUArrays = "11.5.14" +GPUArrays = "12" GPUCompiler = "2.13" GPUToolbox = "3.3.1" KernelAbstractions = "0.9.2" diff --git a/src/AMDGPU.jl b/src/AMDGPU.jl index c713d4385..799d958eb 100644 --- a/src/AMDGPU.jl +++ b/src/AMDGPU.jl @@ -11,7 +11,6 @@ using LLVM using Preferences using Printf -import AcceleratedKernels as AK import UnsafeAtomics import Atomix import Atomix: @atomic, @atomicswap, @atomicreplace @@ -123,12 +122,6 @@ include("conversions.jl") include("broadcast.jl") include("exception_handler.jl") -include("kernels/mapreduce.jl") -include("kernels/indexing.jl") -include("kernels/accumulate.jl") -include("kernels/sorting.jl") -include("kernels/reverse.jl") - include("blas/rocBLAS.jl") include("solver/rocSOLVER.jl") include("sparse/rocSPARSE.jl") diff --git a/src/kernels/accumulate.jl b/src/kernels/accumulate.jl deleted file mode 100644 index 30b12df8a..000000000 --- a/src/kernels/accumulate.jl +++ /dev/null @@ -1,8 +0,0 @@ -Base.accumulate!(op, B::AnyROCArray, A::AnyROCArray; init=zero(eltype(A)), kwargs...) = - AK.accumulate!(op, B, A, ROCBackend(); init, kwargs...) - -Base.accumulate(op, A::AnyROCArray; init=zero(eltype(A)), kwargs...) = - AK.accumulate(op, A, ROCBackend(); init, kwargs...) - -Base.cumsum(src::AnyROCArray; kwargs...) = AK.cumsum(src, ROCBackend(); kwargs...) -Base.cumprod(src::AnyROCArray; kwargs...) = AK.cumprod(src, ROCBackend(); kwargs...) diff --git a/src/kernels/indexing.jl b/src/kernels/indexing.jl deleted file mode 100644 index 6b8eca9b7..000000000 --- a/src/kernels/indexing.jl +++ /dev/null @@ -1,36 +0,0 @@ -Base.to_index(::ROCArray, I::AbstractArray{Bool}) = findall(I) - -if VERSION >= v"1.11.0-DEV.1157" - Base.to_indices(x::ROCArray, I::Tuple{AbstractArray{Bool}}) = - (Base.to_index(x, I[1]),) -end - -function Base.findall(bools::AnyROCArray{Bool}) - I = keytype(bools) - indices = cumsum(reshape(bools, prod(size(bools)))) - - n = isempty(indices) ? 0 : @allowscalar indices[end] - ys = ROCArray{I}(undef, n) - - if n > 0 - function _ker!(ys, bools, indices) - i = workitemIdx().x + (workgroupIdx().x - Int32(1)) * workgroupDim().x - - @inbounds if i ≤ length(bools) && bools[i] - ii = CartesianIndices(bools)[i] - b = indices[i] # new position - ys[b] = ii - end - return - end - - kernel = @roc launch=false _ker!(ys, bools, indices) - config = launch_configuration(kernel) - groupsize = min(length(indices), config.groupsize) - gridsize = cld(length(indices), groupsize) - kernel(ys, bools, indices; groupsize, gridsize) - end - unsafe_free!(indices) - - return ys -end diff --git a/src/kernels/mapreduce.jl b/src/kernels/mapreduce.jl deleted file mode 100644 index cae6b9e41..000000000 --- a/src/kernels/mapreduce.jl +++ /dev/null @@ -1,169 +0,0 @@ -# TODO -# - serial version for lower latency -# - group-stride loop to delay need for second kernel launch - -# Reduce a value across a group, using local memory for communication -@inline function reduce_group(op, val::T, neutral) where T - items = workgroupDim().x - item = workitemIdx().x - - # Shared mem for a complete reduction. - shared = @ROCDynamicLocalArray(T, items, false) - @inbounds shared[item] = val - - # Perform a reduction. - d = 1 - while d < items - sync_workgroup() - index = 2 * d * (item - 1) + 1 - @inbounds if index ≤ items - other_val = (index + d) ≤ items ? shared[index + d] : neutral - shared[index] = op(shared[index], other_val) - end - d *= 2 - end - - # Load the final value on the first item. - if item == 1 - val = @inbounds shared[item] - end - - return val -end - -Base.@propagate_inbounds _map_getindex(args::Tuple, I) = ((args[1][I]), _map_getindex(Base.tail(args), I)...) -Base.@propagate_inbounds _map_getindex(args::Tuple{Any}, I) = ((args[1][I]),) -Base.@propagate_inbounds _map_getindex(args::Tuple{}, I) = () - -# Reduce an array across the grid. All elements to be processed can be addressed by the -# product of the two iterators `Rreduce` and `Rother`, where the latter iterator will have -# singleton entries for the dimensions that should be reduced (and vice versa). -function partial_mapreduce_device(f, op, neutral, Rreduce, Rother, R, As...) - # decompose the 1D hardware indices into separate ones for reduction (across items - # and possibly groups if it doesn't fit) and other elements (remaining groups) - localIdx_reduce = workitemIdx().x - localDim_reduce = workgroupDim().x - - groupIdx_reduce, groupIdx_other = fldmod1(workgroupIdx().x, length(Rother)) - groupDim_reduce = gridGroupDim().x ÷ length(Rother) - - # group-based indexing into the values outside of the reduction dimension - # (that means we can safely synchronize items within this group) - iother = groupIdx_other - @inbounds if iother ≤ length(Rother) - Iother = Rother[iother] - - # load the neutral value - Iout = CartesianIndex(Tuple(Iother)..., groupIdx_reduce) - neutral = neutral ≡ nothing ? R[Iout] : neutral - - val = op(neutral, neutral) - - # reduce serially across chunks of input vector that don't fit in a group - ireduce = localIdx_reduce + (groupIdx_reduce - 1) * localDim_reduce - while ireduce ≤ length(Rreduce) - Ireduce = Rreduce[ireduce] - J = max(Iother, Ireduce) - val = op(val, f(_map_getindex(As, J)...)) - ireduce += localDim_reduce * groupDim_reduce - end - - val = reduce_group(op, val, neutral) - - # write back to memory - if localIdx_reduce == 1 - R[Iout] = val - end - end - - return -end - -function GPUArrays.mapreducedim!( - f::F, op::OP, R::AnyROCArray{T}, - A::Union{AbstractArray,Broadcast.Broadcasted}; init=nothing, -) where {F, OP, T} - Base.check_reducedims(R, A) - length(A) == 0 && return R # isempty(::Broadcasted) iterates - - R_old = R - # add singleton dimensions to the output container, if needed - if ndims(R) < ndims(A) - dims = Base.fill_to_length(size(R), 1, Val(ndims(A))) - R = reshape(R, dims) - end - - # iteration domain, split in two: one part covers the dimensions that should - # be reduced, and the other covers the rest. combining both covers all values. - Rall = CartesianIndices(axes(A)) - Rother = CartesianIndices(axes(R)) - Rreduce = CartesianIndices(ifelse.(axes(A) .== axes(R), Ref(Base.OneTo(1)), axes(A))) - # NOTE: we hard-code `OneTo` (`first.(axes(A))` would work too) or we get a - # CartesianIndices object with UnitRanges that behave badly on the GPU. - @assert length(Rall) == length(Rother) * length(Rreduce) - - # allocate an additional, empty dimension to write the reduced value to. - # this does not affect the actual location in memory of the final values, - # but allows us to write a generalized kernel supporting partial reductions. - R′ = reshape(R, (size(R)..., 1)) - - # how many items do we want? - # - # items in a group work together to reduce values across the reduction dimensions; - # we want as many as possible to improve algorithm efficiency and execution occupancy. - wanted_items = nextpow(2, length(Rreduce)) - function compute_items(max_items) - if wanted_items > max_items - prevpow(2, max_items) - else - wanted_items - end - end - - # how many items can we launch? - # - # we might not be able to launch all those items to reduce each slice in one go. - # that's why each items also loops across their inputs, processing multiple values - # so that we can span the entire reduction dimension using a single item group. - max_block_size = 256 - compute_shmem(items) = items * aligned_sizeof(T) - max_shmem = max_block_size |> compute_items |> compute_shmem - kernel = @roc launch=false partial_mapreduce_device( - f, op, init, Rreduce, Rother, R′, A) - kernel_config = launch_configuration(kernel; shmem=max_shmem, max_block_size) - reduce_items = compute_items(kernel_config.groupsize) - reduce_shmem = compute_shmem(reduce_items) - - # how many groups should we launch? - # - # even though we can always reduce each slice in a single item group, that may not be - # optimal as it might not saturate the GPU. we already launch some groups to process - # independent dimensions in parallel; pad that number to ensure full occupancy. - other_groups = length(Rother) - reduce_groups = cld(length(Rreduce), reduce_items) - - # determine the launch configuration - blocks = reduce_items - grid = reduce_groups * other_groups - - # perform the actual reduction - if reduce_groups == 1 - # we can cover the dimensions to reduce using a single group - @roc gridsize=grid groupsize=blocks shmem=reduce_shmem partial_mapreduce_device( - f, op, init, Rreduce, Rother, R′, A) - else - # we need multiple steps to cover all values to reduce - partial = similar(R, (size(R)..., reduce_groups)) - if init === nothing - # without an explicit initializer we need to copy from the output container - partial .= R - end - @roc gridsize=grid groupsize=blocks shmem=reduce_shmem partial_mapreduce_device( - f, op, init, Rreduce, Rother, partial, A) - - GPUArrays.mapreducedim!(identity, op, R′, partial; init) - AMDGPU.unsafe_free!(partial) - end - - return R_old -end diff --git a/src/kernels/reverse.jl b/src/kernels/reverse.jl deleted file mode 100644 index bc933258a..000000000 --- a/src/kernels/reverse.jl +++ /dev/null @@ -1,124 +0,0 @@ -# Adapted from CUDA.jl. - -# 1D case. - -Base.reverse(x::AnyROCVector) = @inbounds reverse(x, 1, length(x)) - -Base.@propagate_inbounds function Base.reverse( - x::AnyROCVector, start::Integer, stop::Integer = length(x), -) - y = similar(x) - - start > 1 && copyto!(y, 1, x, 1, start - 1) - stop < length(x) && copyto!(y, stop + 1, x, stop + 1) - _reverse!(@view(y[start:stop]), @view(x[start:stop])) - - return y -end - -Base.@propagate_inbounds function Base.reverse!( - x::AnyROCVector, start::Integer, stop::Integer = length(x), -) - _reverse!(@view(x[start:stop])) - return x -end - -Base.reverse!(x::AnyROCVector) = @inbounds reverse!(x, 1, length(x)) - -# N-D case. - -function Base.reverse(x::AnyROCArray; dims=:) - isa(dims, Colon) && (dims = 1:ndims(x);) - - applicable(iterate, dims) || throw(ArgumentError( - "Dimension `$dims` is not an iterable.")) - all(1 .≤ dims .≤ ndims(x)) || throw(ArgumentError( - "Dimension `$dims` is not 1 ≤ $dims ≤ $(ndims(x))")) - - all(size(x)[[dims...]] .== 1) && return copy(x) # Dims to reverse are 1. - - y = similar(x) - _reverse!(y, x; dims) - return y -end - -function Base.reverse!(x::AnyROCArray; dims=:) - isa(dims, Colon) && (dims = 1:ndims(x);) - - applicable(iterate, dims) || throw(ArgumentError( - "Dimension `$dims` is not an iterable.")) - all(1 .≤ dims .≤ ndims(x)) || throw(ArgumentError( - "Dimension `$dims` is not 1 ≤ $dims ≤ $(ndims(x))")) - - _reverse!(x; dims) - return x -end - -# Out-of-place kernel, swapping a single element per thread. - -function _reverse!( - y::AnyROCArray{T, N}, x::AnyROCArray{T, N}; dims=1:ndims(x), -) where {T, N} - size(x) == size(y) || throw(DimensionMismatch( - "Input and output arrays must equal in size, instead: " * - "`$(size(x))` vs `$(size(y))`.")) - - rev_dims = ntuple(d -> (d ∈ dims) && (size(x, d) > 1), N) - - ref = size(x) .+ 1 - lin_ids = LinearIndices(x) - nd_ids = CartesianIndices(x) - - function _kernel!(y::AbstractArray{T, N}, x::AbstractArray{T, N}) where {T, N} - i = workitemIdx().x + (workgroupIdx().x - 0x1) * workgroupDim().x - - if i ≤ length(x) - idx = Tuple(nd_ids[i]) - idx = ifelse.(rev_dims, ref .- idx, idx) - idx_out = lin_ids[idx...] - y[idx_out] = x[i] - end - return - end - - groupsize = 256 - gridsize = cld(length(x), groupsize) - iszero(gridsize) && return - @roc groupsize=groupsize gridsize=gridsize _kernel!(y, x) -end - -# In-place kernel, swapping elements on half the number of threads. - -function _reverse!(x::AnyROCArray{T, N}; dims=1:ndims(x)) where {T, N} - rev_dims = ntuple(d -> (d ∈ dims) && (size(x, d) > 1), N) - half_dim = findlast(rev_dims) - half_dim ≡ nothing && return # No reverse needed in this case. - - reduced_sz = ntuple(d -> (d == half_dim) ? cld(size(x, d), 2) : size(x, d), N) - reduced_len = prod(reduced_sz) - - ref = size(x) .+ 1 - lin_ids = LinearIndices(x) - nd_ids = CartesianIndices(reduced_sz) - - function _kernel!(x::AbstractArray{T, N}) where {T, N} - i = workitemIdx().x + (workgroupIdx().x - 0x1) * workgroupDim().x - - if i ≤ reduced_len - idx = Tuple(nd_ids[i]) - idx_in = lin_ids[idx...] - idx = ifelse.(rev_dims, ref .- idx, idx) - idx_out = lin_ids[idx...] - - if idx_in < idx_out - x[idx_in], x[idx_out] = x[idx_out], x[idx_in] - end - end - return - end - - groupsize = 256 - gridsize = cld(prod(reduced_sz), groupsize) - iszero(gridsize) && return - @roc groupsize=groupsize gridsize=gridsize _kernel!(x) -end diff --git a/src/kernels/sorting.jl b/src/kernels/sorting.jl deleted file mode 100644 index e65b06d94..000000000 --- a/src/kernels/sorting.jl +++ /dev/null @@ -1,92 +0,0 @@ -Base.sort!(x::AnyROCArray; dims::Union{Nothing, Integer}=nothing, kwargs...) = - isnothing(dims) ? (AK.sort!(x; kwargs...); x) : _sort_dims!(x, dims; kwargs...) - -# Base's out-of-place `sort(x; dims)` sorts chunks with the CPU algorithm, which trips -# scalar indexing - route it through `sort!` instead. -Base.sort(x::AnyROCArray; kwargs...) = sort!(copy(x); kwargs...) - -Base.sortperm!(ix::AnyROCArray, x::AnyROCArray; dims::Union{Nothing, Integer}=nothing, kwargs...) = - isnothing(dims) ? - (AK.sortperm!(ix, x; kwargs...); ix) : - _sortperm_dims!(ix, x, dims; kwargs...) - -Base.sortperm(x::AnyROCArray; dims::Union{Nothing, Integer}=nothing, kwargs...) = - isnothing(dims) ? - sortperm!(ROCArray(1:length(x)), x; kwargs...) : - sortperm!(reshape(ROCArray(1:length(x)), size(x)), x; dims, kwargs...) - -# Sorting along `dims`, a stopgap until AcceleratedKernels support for it reaches AMDGPU. -# Each element is tagged with the index of its slice and the array is sorted once, ordered -# lexicographically by `(slice, element)`; slices then come out grouped and internally sorted, -# and are scattered back. - -# Tag element `(i, j - 1)` of the `(sd * n, rest)` view of `x` with its slice index, plus its -# linear index for `sortperm`. Broadcast alongside `x` itself, so the ranges stay lazy. -_slice_kv(::Type{K}, sd, i, j, v) where K = (K(mod1(i, sd) + sd * j), v) -_slice_kvi(::Type{K}, ::Type{I}, sd, n, i, j, v) where {K, I} = - (K(mod1(i, sd) + sd * j), v, I(i + sd * n * j)) - -_slice_value(kv) = kv[2] -_slice_index(kv) = kv[3] - -# `x` seen as `(sd, n, rest)`: `n` elements along `dims`, strided by `sd`, over `rest` groups. -function _slice_layout(x::AnyROCArray, dims::Integer) - 1 ≤ dims ≤ ndims(x) || throw(ArgumentError("dimension out of range")) - n = size(x, dims) - sd = prod(size(x)[1:dims - 1]) - return n, sd, length(x) ÷ (n * sd) -end - -# Smallest key type that can enumerate the slices of `x`. -_slice_keytype(x::AnyROCArray) = length(x) ≤ typemax(Int32) ? Int32 : Int64 - -# `(slice, element)` ordering, ties broken by the user-supplied ordering. -_slice_lt(ord) = (a, b) -> a[1] == b[1] ? Base.Order.lt(ord, a[2], b[2]) : a[1] < b[1] - -# Scatter the sorted - and therefore slice-grouped - `kv` back over `dst`. -function _slice_scatter!(f::F, dst, kv, n, sd, rest) where F - if sd == 1 - reshape(dst, n, rest) .= f.(reshape(kv, n, rest)) - else - reshape(dst, sd, n, rest) .= f.(PermutedDimsArray(reshape(kv, n, sd, rest), (2, 1, 3))) - end - return dst -end - -function _sort_dims!( - x::AnyROCArray, dims::Integer; - lt=isless, by=identity, rev::Union{Nothing, Bool}=nothing, - order::Base.Order.Ordering=Base.Order.Forward, kwargs..., -) - isempty(x) && return x - n, sd, rest = _slice_layout(x, dims) - K = _slice_keytype(x) - - kv = similar(x, Tuple{K, eltype(x)}, length(x)) - reshape(kv, sd * n, rest) .= _slice_kv.( - K, sd, 1:sd * n, (0:rest - 1)', reshape(x, sd * n, rest)) - - AK.sort!(kv; lt=_slice_lt(Base.Order.ord(lt, by, rev, order)), kwargs...) - - return _slice_scatter!(_slice_value, x, kv, n, sd, rest) -end - -function _sortperm_dims!( - ix::AnyROCArray, x::AnyROCArray, dims::Integer; - lt=isless, by=identity, rev::Union{Nothing, Bool}=nothing, - order::Base.Order.Ordering=Base.Order.Forward, kwargs..., -) - axes(ix) == axes(x) || throw(ArgumentError( - "index array has axes $(axes(ix)), but sorted array has axes $(axes(x))")) - isempty(x) && return ix - n, sd, rest = _slice_layout(x, dims) - K, I = _slice_keytype(x), eltype(ix) - - kv = similar(x, Tuple{K, eltype(x), I}, length(x)) - reshape(kv, sd * n, rest) .= _slice_kvi.( - K, I, sd, n, 1:sd * n, (0:rest - 1)', reshape(x, sd * n, rest)) - - AK.sort!(kv; lt=_slice_lt(Base.Order.ord(lt, by, rev, order)), kwargs...) - - return _slice_scatter!(_slice_index, ix, kv, n, sd, rest) -end diff --git a/test/core/rocarray_base.jl b/test/core/rocarray_base.jl index 18e2f0f37..fdb3834e1 100644 --- a/test/core/rocarray_base.jl +++ b/test/core/rocarray_base.jl @@ -378,10 +378,4 @@ end @test Array(x) == [true, true] end -@testset "mapreducedim! returning same type" begin - R = transpose(AMDGPU.zeros(Float32, 2, 3)) - A = ROCArray(rand(Float32, 3, 2, 10)) - @test @inferred(GPUArrays.mapreducedim!(identity, +, R, A)) === R -end - end From 3c2458d9ef6b79c00c1be72b431faccece491b6f Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Thu, 8 Oct 2026 09:38:47 +0200 Subject: [PATCH 3/3] Detect aliasing through GPUArrays' memory location hook Base.dataids returned the array's own pointer, which missed overlaps between a contiguous view (a ROCArray at an offset) and a wrapped array of the same memory, made every empty ROCArray alias every other one (all have address 0; AcceleratedKernels' alias checks rejected that on Julia 1.10), and made disjoint views of one buffer alias. GPUArrays 12.1 defines `Base.dataids` and `Base.mightalias` for every GPU array from where its elements live, so implement that hook instead, using the buffer's device address rather than `pointer`, which takes stream ownership of the memory. --- Project.toml | 2 +- src/array.jl | 12 +++++++++- test/core/rocarray_base.jl | 49 ++++++++++++++++++++++++++++++++++++++ 3 files changed, 61 insertions(+), 2 deletions(-) diff --git a/Project.toml b/Project.toml index 931826ad8..41229fed4 100644 --- a/Project.toml +++ b/Project.toml @@ -57,7 +57,7 @@ CEnum = "0.4, 0.5" ChainRulesCore = "1" EnzymeCore = "0.8" ExprTools = "0.1" -GPUArrays = "12" +GPUArrays = "12.1" GPUCompiler = "2.13" GPUToolbox = "3.3.1" KernelAbstractions = "0.9.2" diff --git a/src/array.jl b/src/array.jl index f57a52ccd..680b457be 100644 --- a/src/array.jl +++ b/src/array.jl @@ -189,7 +189,17 @@ Base.similar(::ROCArray{<:Any, <:Any, B}, ::Type{T}, dims::Base.Dims{N}) where { Base.elsize(::Type{<:ROCArray{T}}) where {T} = aligned_sizeof(T) Base.size(x::ROCArray) = x.dims Base.sizeof(x::ROCArray) = Base.elsize(x) * length(x) -Base.dataids(A::ROCArray) = (UInt(pointer(A)),) + +## alias detection + +# GPUArrays implements `Base.dataids` and `Base.mightalias` from where an array lives. Not +# using `pointer(x)`, which takes ownership of the memory for the current stream. +function GPUArrays.memory_location(x::ROCArray) + mem = x.buf[].mem + return (UInt(mem isa Mem.HostBuffer ? mem.dev_ptr : mem.ptr), x.offset) +end + +Base.unaliascopy(x::ROCArray) = copy(x) ## interop with Julia arrays diff --git a/test/core/rocarray_base.jl b/test/core/rocarray_base.jl index fdb3834e1..106f2d80d 100644 --- a/test/core/rocarray_base.jl +++ b/test/core/rocarray_base.jl @@ -61,6 +61,55 @@ end @test Array(r) == reinterpret(Int64, @view Array(a)[2:7]) end +@testset "aliasing" begin + x = ROCArray([1, 2]) + y = view(x, 2:2) + @test Base.mightalias(x, x) + @test Base.mightalias(x, y) + z = view(x, 1:1) + @test Base.mightalias(x, z) + @test !Base.mightalias(y, z) + + a = copy(y)::typeof(x) + @test !Base.mightalias(x, a) + a .= 3 + @test Array(y) == [2] + + b = Base.unaliascopy(y)::typeof(y) + @test !Base.mightalias(x, b) + b .= 3 + @test Array(y) == [2] + + # contiguous views are ROCArrays with an offset into the parent's memory, + # which should still alias wrapped arrays (like SubArrays) of that memory + x = ROCArray(1:16) + @test Base.mightalias(view(x, 2:16), view(x, 15:-1:1)) + @test Base.mightalias(view(x, 1:2:15), view(x, 2:9)) + @test Base.mightalias(view(x, 2:16), view(reinterpret(Int32, x), 1:2:31)) + @test !Base.mightalias(view(x, 2:16), view(ROCArray(1:16), 15:-1:1)) + + # memory wrapped from a view's pointer + y = view(x, 2:16) + z = unsafe_wrap(ROCArray, pointer(y), size(y)) + @test Base.mightalias(y, view(z, 15:-1:1)) + + # so in-place broadcasts between them should make a copy first + n = 2^20 + x = ROCArray{Float32}(1:n) + view(x, 2:n) .= view(x, n-1:-1:1) + @test Array(x) == [1; n-1:-1:1] + + # empty arrays alias nothing, also on Julia 1.10 (where all have address 0) + @test !Base.mightalias(ROCArray(Int[]), ROCArray(Float32[])) + @test !Base.mightalias(view(ROCArray(zeros(2, 0)), 1:1, :), ROCArray(Int[])) + @test isempty(cumsum(ROCArray(Int[]))) + + # disjoint parts of one array may be each other's source and destination + x = ROCArray(collect(1:10)) + @test Array(sum!(view(x, 1:1), view(x, 2:10))) == [54] + @test Array(cumsum!(view(x, 1:5), view(x, 6:10))) == cumsum(6:10) +end + @testset "resize!" begin a_h = Array(range(1, 10)) a_d = a_h |> roc