diff --git a/src/array.jl b/src/array.jl index c0a183b6..accb2dac 100644 --- a/src/array.jl +++ b/src/array.jl @@ -370,14 +370,19 @@ end ## indexing -# Host-accessible arrays can be indexed from CPU, bypassing GPUArrays restrictions +# Host-accessible arrays can be indexed from CPU, bypassing GPUArrays restrictions. +# Wait for queued work first, e.g., the kernel computing the result of a reduction. This +# only synchronizes the current task's stream; work submitted by other tasks, or to an +# explicitly created queue, needs to be synchronized explicitly. function Base.getindex(x::oneArray{<:Any, <:Any, <:Union{oneL0.HostBuffer, oneL0.SharedBuffer}}, I::Int) @boundscheck checkbounds(x, I) + synchronize(global_stream(context(x), device())) return unsafe_load(pointer(x, I; type = oneL0.HostBuffer)) end function Base.setindex!(x::oneArray{<:Any, <:Any, <:Union{oneL0.HostBuffer, oneL0.SharedBuffer}}, v, I::Int) @boundscheck checkbounds(x, I) + synchronize(global_stream(context(x), device())) return unsafe_store!(pointer(x, I; type = oneL0.HostBuffer), v) end diff --git a/test/array.jl b/test/array.jl index 5d92d759..eb21d263 100644 --- a/test/array.jl +++ b/test/array.jl @@ -103,6 +103,14 @@ end @test b == [100, 200] end +@testset "reductions of host-accessible arrays" begin + for B in (oneL0.SharedBuffer, oneL0.HostBuffer) + a = oneArray{Float32, 1, B}(fill(1.0f0, 1024)) + @test sum(a) == 1024 + @test maximum(a) == 1 + end +end + # https://github.com/JuliaGPU/CUDA.jl/issues/2191 @testset "preserving buffer types" begin a = oneVector{Int,oneL0.SharedBuffer}([1])