Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 3 additions & 2 deletions .buildkite/pipeline.yml
Original file line number Diff line number Diff line change
Expand Up @@ -106,12 +106,13 @@ steps:
(build.message =~ /\[only [^\]]*(tests|julia)/ ||
build.message !~ /\[only / && !build.pull_request.draft)
commands: |
git clone --depth 1 --branch main https://github.com/JuliaGPU/KernelAbstractions.jl ka
git clone --depth 1 --branch vc/ki-subgroup-ops https://github.com/JuliaGPU/KernelAbstractions.jl ka
git clone --depth 1 https://github.com/JuliaGPU/AcceleratedKernels.jl ak
git clone --depth 1 --branch vc/subgroup-votes https://github.com/JuliaGPU/OpenCL.jl ocl
julia --project=test -e '
using Pkg
if VERSION < v"1.11"
Pkg.develop([PackageSpec(; path) for path in (".", "ka", "ka/lib/KernelInterface", "ak")])
Pkg.develop([PackageSpec(; path) for path in (".", "ka", "ka/lib/KernelInterface", "ak", "ocl/lib/intrinsics")])
else
Pkg.instantiate()
end'
Expand Down
8 changes: 5 additions & 3 deletions Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -52,13 +52,14 @@ AMDGPUSpecialFunctionsExt = "SpecialFunctions"

[sources]
AcceleratedKernels = {url = "https://github.com/JuliaGPU/AcceleratedKernels.jl", rev = "main"}
KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main"}
KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main", subdir = "lib/KernelInterface"}
KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops"}
KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops", subdir = "lib/KernelInterface"}
SPIRVIntrinsics = {url = "https://github.com/JuliaGPU/OpenCL.jl", rev = "vc/subgroup-votes", subdir = "lib/intrinsics"}

[compat]
AMDGPU_LLVM_Backend_jll = "23"
AbstractFFTs = "1.0"
AcceleratedKernels = "0.3.1, 0.4"
AcceleratedKernels = "0.3.1, 0.4, 0.5"
Adapt = "4"
Atomix = "1"
BFloat16s = "0.6.0"
Expand Down Expand Up @@ -93,3 +94,4 @@ julia = "1.10"

[extras]
KernelAbstractions = "63c18a36-062a-441e-b654-da1e3ab1ce7c"
SPIRVIntrinsics = "71d1d633-e7e8-4a92-83a1-de8814b09ba8"
54 changes: 49 additions & 5 deletions src/ROCKernels.jl
Original file line number Diff line number Diff line change
Expand Up @@ -51,9 +51,10 @@ end
KI.supports_float64(::ROCBackend) = true
KI.supports_atomics(::ROCBackend) = true
KI.supports_subgroups(::ROCBackend) = true
# `shfl_down` decomposes other types into 32-bit shuffles
KI.supports_shuffle(::ROCBackend, ::Type{T}) where {T} =
T <: Union{Bool, Base.BitInteger, Base.IEEEFloat, Complex{<:Union{Base.BitInteger, Base.IEEEFloat}}}
# the types the shuffles below support, by decomposing them into 32-bit `ds_bpermute`s;
# KernelInterface shuffles other `isbits` types (e.g. `Complex`) field by field
const ShuffleTypes = Union{Bool, Base.BitInteger, Base.IEEEFloat}
KI.supports_shuffle(::ROCBackend, ::Type{T}) where {T <: ShuffleTypes} = true

function KI.priority!(::ROCBackend, priority::Symbol)
priority ∉ (:high, :normal, :low) && error(
Expand Down Expand Up @@ -221,6 +222,8 @@ end
return min(ws, workgroup_items() - (linear_workitem_id() ÷ ws) * ws) % T
end

# a constant: `wavefrontsize` is folded to the wavefront size the kernel is compiled for, see
# `fold_wavefrontsize!`
@device_override KI.get_max_sub_group_size(::Type{T}) where {T} = Device.wavefrontsize() % T

@device_override KI.get_num_sub_groups(::Type{T}) where {T} = cld(workgroup_items(), Device.wavefrontsize()) % T
Expand All @@ -247,10 +250,51 @@ end
Device.sync_wavefront()
end

@device_override function KI.shfl_down(val::T, offset::Integer) where T
@inline Device.shfl_down(val, offset % Cint)
## communication

# read `val` from the work-item in the 0-based hardware lane `lane`. `ds_bpermute` only uses
# the low bits of the address, so lanes out of range read some other lane instead of
# trapping. unlike `Device.shfl` etc., the lanes are the hardware lanes (see
# `hardware_lane`), not `activelane`.
@inline function bpermute_lane(val, lane::Cint)
return Device._shfl(x -> Device.bpermute(lane << 0x2, x), val)
end

@inline lane_id() = hardware_lane() % Cint

@device_override @inline KI.shfl(val::T, lane::Integer) where {T <: ShuffleTypes} =
bpermute_lane(val, (lane % Cint) - Cint(1))

# where the source lane is past the wavefront, `shfl_down` and `shfl_up` read from the
# work-item itself, like CUDA's shuffles (rather than from the lane `ds_bpermute` wraps around
# to). `wavefrontsize` is folded to a constant, see `fold_wavefrontsize!`.
@device_override @inline function KI.shfl_down(val::T, offset::Integer) where {T <: ShuffleTypes}
lane = lane_id()
ws = Device.wavefrontsize() % Cint
return bpermute_lane(val, ifelse(offset < ws - lane, lane + (offset % Cint), lane))
end

@device_override @inline function KI.shfl_up(val::T, offset::Integer) where {T <: ShuffleTypes}
lane = lane_id()
return bpermute_lane(val, ifelse(offset <= lane, lane - (offset % Cint), lane))
end

# `mask` is below the wavefront size, so the source lane is in the wavefront
@device_override @inline KI.shfl_xor(val::T, mask::Integer) where {T <: ShuffleTypes} =
bpermute_lane(val, lane_id() ⊻ (mask % Cint))

# the shuffles with a `width` use KernelInterface's fallbacks, a `ds_bpermute` from a lane
# computed with a few integer operations, like `Device.shfl` etc. (which use `activelane`).

# `ballot` only sets the bits of active lanes, i.e. of the work-items of the sub-group, and
# its result is uniform. `wavefrontsize`, which selects the 32- or 64-bit ballot, is folded
# to the wavefront size the kernel is compiled for (see `fold_wavefrontsize!`).
@device_override @inline KI.sub_group_ballot(pred::Bool) = Device.ballot(pred)

@device_override @inline KI.sub_group_any(pred::Bool) = Device.ballot(pred) != 0

@device_override @inline KI.sub_group_all(pred::Bool) = Device.ballot(!pred) == 0

# not supported, see the `ROCBackend` docstring
@device_override @inline KI._print(args...) = nothing

Expand Down
6 changes: 0 additions & 6 deletions src/device/gcn/math.jl
Original file line number Diff line number Diff line change
Expand Up @@ -60,12 +60,6 @@ for jltype in (Float64, Float32, Float16)
@eval @device_override Base.fma(x::$jltype, y::$jltype, z::$jltype) = ccall(
$("extern __ocml_fma_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype, $jltype), x, y, z)

@eval @device_override Base.min(x::$jltype, y::$jltype) = ccall(
$("extern __ocml_min_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype), x, y)

@eval @device_override Base.max(x::$jltype, y::$jltype) = ccall(
$("extern __ocml_max_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype), x, y)

@eval @device_override Base.copysign(x::$jltype, y::$jltype) = ccall(
$("extern __ocml_copysign_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype), x, y)

Expand Down
6 changes: 4 additions & 2 deletions test/Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@ ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc"
Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f"
PrettyTables = "08abe8d2-0d0c-5749-adfa-8a2ac140af0d"
Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c"
SPIRVIntrinsics = "71d1d633-e7e8-4a92-83a1-de8814b09ba8"
SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf"
SparseMatricesCSR = "a0a7dd2c-ebf4-11e9-1f05-cf50bc540ca1"
SpecialFunctions = "276daf66-3868-5448-9aa4-cd146d93841b"
Expand All @@ -28,5 +29,6 @@ Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40"

[sources]
AMDGPU = {path = ".."}
KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main"}
KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main", subdir = "lib/KernelInterface"}
KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops"}
KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops", subdir = "lib/KernelInterface"}
SPIRVIntrinsics = {url = "https://github.com/JuliaGPU/OpenCL.jl", rev = "vc/subgroup-votes", subdir = "lib/intrinsics"}
11 changes: 11 additions & 0 deletions test/device/math.jl
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,17 @@ using Base.FastMath
end
end

@testset "min/max" begin
# Julia's semantics: NaN propagates, and -0.0 < 0.0
a = [NaN, 1, -0.0, 0.0, 2, NaN, Inf, -Inf]
b = [1, NaN, 0.0, -0.0, 3, NaN, NaN, 1]
for T in (Float16, Float32, Float64)
x, y = T.(a), T.(b)
@test isequal(Array(max.(ROCArray(x), ROCArray(y))), max.(x, y))
@test isequal(Array(min.(ROCArray(x), ROCArray(y))), min.(x, y))
end
end

@testset "Fast min/max" begin
function ker!(x)
x[1] = @fastmath max(x[1], zero(eltype(x)))
Expand Down