diff --git a/test/core/rocarray_broadcast.jl b/test/core/rocarray_broadcast.jl index cbe859e38..362f12e17 100644 --- a/test/core/rocarray_broadcast.jl +++ b/test/core/rocarray_broadcast.jl @@ -45,28 +45,27 @@ end end # https://github.com/JuliaGPU/AMDGPU.jl/issues/1002 -if Base.datatype_alignment(Int128) == 16 - # note: miscompile only occurs with `julia --check-bounds=yes` - @testset "Int128 miscompilation" begin - function test_kernel(a::T, b) where T - c = a - for i=1:5 - c += T(b) - c = a * T(2) - end - return c +# note: miscompile only occurs with `julia --check-bounds=yes` +@testset "Int128 miscompilation" begin + function test_kernel(a::T, b) where T + c = a + for i=1:5 + c += T(b) + c = a * T(2) end - M = rand(Int128, 10, 10) - @test Array(test_kernel.(ROCArray(M), Int128(10))) == test_kernel.(M, Int128(10)) - AMDGPU.synchronize() + return c end + M = rand(Int128, 10, 10) + @test Array(test_kernel.(ROCArray(M), Int128(10))) == test_kernel.(M, Int128(10)) + AMDGPU.synchronize() +end - @testset "Int128 axpby! miscompilation" begin - a, b = rand(Int128), rand(Int128) - x, y = rand(Int128, 5), rand(Int128, 5) - gx, gy = ROCArray(x), ROCArray(y) - gy .= gx .* a .+ gy .* b - @test Array(gy) == x .* a .+ y .* b - AMDGPU.synchronize() - end +# https://github.com/JuliaGPU/AMDGPU.jl/issues/1002 +@testset "Int128 axpby! miscompilation" begin + a, b = rand(Int128), rand(Int128) + x, y = rand(Int128, 5), rand(Int128, 5) + gx, gy = ROCArray(x), ROCArray(y) + gy .= gx .* a .+ gy .* b + @test Array(gy) == x .* a .+ y .* b + AMDGPU.synchronize() end diff --git a/test/device/launch.jl b/test/device/launch.jl index e1eeabd91..e3725e705 100644 --- a/test/device/launch.jl +++ b/test/device/launch.jl @@ -60,6 +60,25 @@ end AMDGPU.synchronize() end +# https://github.com/JuliaGPU/AMDGPU.jl/issues/1002 +@testset "Int128 layout matches the host" begin + # a kernel argument with fields after an Int128 + function kernel(out, s) + out[1] = s[1] + out[2] = s[2] + out[3] = s[3] + return + end + s = (Int32(-7), typemax(Int128) - 3, Int64(42)) + out = ROCArray{Int128}(undef, 3) + @roc kernel(out, s) + @test Array(out) == [-7, typemax(Int128) - 3, 42] + + # device memory holding such values, as laid out by the host + xs = [(Int32(i), Int128(i) << 70, Int64(-i)) for i in 1:4] + @test Array(map(x -> x[1] + x[2] + x[3], ROCArray(xs))) == map(x -> x[1] + x[2] + x[3], xs) +end + @testset "Function/Argument Conversion" begin @testset "Closure as Argument" begin function kernel(closure) diff --git a/test/runtests.jl b/test/runtests.jl index 1888f60f0..6899ba7b0 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -80,9 +80,7 @@ init_code = quote include($gpuarrays_testsuite) testf(f, xs...; kwargs...) = TestSuite.compare(f, AMDGPU.ROCArray, xs...; kwargs...) - # Int128 kernel arguments need the host to align Int128 to 16 bytes like the device (Julia 1.12+), see #1002 - const eltypes = [Int16, Int32, Int64, - (Base.datatype_alignment(Int128) == 16 ? (Int128,) : ())..., + const eltypes = [Int16, Int32, Int64, Int128, Float16, Float32, Float64, ComplexF16, ComplexF32, ComplexF64, Complex{Int16}, Complex{Int32}, Complex{Int64}]