From e4424db401cb35aadd4cfd5e0340edda6ed2586a Mon Sep 17 00:00:00 2001 From: Valentin Churavy Date: Sun, 4 Oct 2026 09:41:51 +0200 Subject: [PATCH 1/4] Implement KernelInterface's sub-group shuffles and votes Implement the shuffles `shfl`, `shfl_down`, `shfl_up` and `shfl_xor` and the votes `sub_group_any`, `sub_group_all` and `sub_group_ballot` of KernelInterface. The shuffles are `ds_bpermute`s from the hardware lane (`mbcnt`), like `get_sub_group_local_id`, rather than `activelane`. `ds_bpermute` only uses the low bits of the address, so a lane out of range reads an unspecified value instead of throwing, and the lane and offsets are truncated with `%` rather than converted with a check. They are only defined for the primitive types that `Device._shfl` decomposes into 32-bit shuffles (`Bool`, integers and IEEE floats), as is `supports_shuffle`, so that KernelInterface shuffles other `isbits` types, e.g. `Complex`, field by field. The votes are built on `ballot`, which only sets the bits of the active lanes, i.e. of the work-items of the sub-group. `get_max_sub_group_size` already is a constant of the generated code, as `fold_wavefrontsize!` folds `llvm.amdgcn.wavefrontsize` before optimization. Assisted-by: Claude Code (Opus 5.5) --- src/ROCKernels.jl | 42 +++++++++++++++++++++++++++++++++++++----- 1 file changed, 37 insertions(+), 5 deletions(-) diff --git a/src/ROCKernels.jl b/src/ROCKernels.jl index 9aefe1fbb..c7e9fbe82 100644 --- a/src/ROCKernels.jl +++ b/src/ROCKernels.jl @@ -51,9 +51,10 @@ end KI.supports_float64(::ROCBackend) = true KI.supports_atomics(::ROCBackend) = true KI.supports_subgroups(::ROCBackend) = true -# `shfl_down` decomposes other types into 32-bit shuffles -KI.supports_shuffle(::ROCBackend, ::Type{T}) where {T} = - T <: Union{Bool, Base.BitInteger, Base.IEEEFloat, Complex{<:Union{Base.BitInteger, Base.IEEEFloat}}} +# the types the shuffles below support, by decomposing them into 32-bit `ds_bpermute`s; +# KernelInterface shuffles other `isbits` types (e.g. `Complex`) field by field +const ShuffleTypes = Union{Bool, Base.BitInteger, Base.IEEEFloat} +KI.supports_shuffle(::ROCBackend, ::Type{T}) where {T <: ShuffleTypes} = true function KI.priority!(::ROCBackend, priority::Symbol) priority ∉ (:high, :normal, :low) && error( @@ -221,6 +222,8 @@ end return min(ws, workgroup_items() - (linear_workitem_id() ÷ ws) * ws) % T end +# a constant: `wavefrontsize` is folded to the wavefront size the kernel is compiled for, see +# `fold_wavefrontsize!` @device_override KI.get_max_sub_group_size(::Type{T}) where {T} = Device.wavefrontsize() % T @device_override KI.get_num_sub_groups(::Type{T}) where {T} = cld(workgroup_items(), Device.wavefrontsize()) % T @@ -247,10 +250,39 @@ end Device.sync_wavefront() end -@device_override function KI.shfl_down(val::T, offset::Integer) where T - @inline Device.shfl_down(val, offset % Cint) +## communication + +# read `val` from the work-item in the 0-based hardware lane `lane`. `ds_bpermute` only uses +# the low bits of the address, so lanes out of range read some other lane instead of +# trapping. unlike `Device.shfl` etc., the lanes are the hardware lanes (see +# `hardware_lane`), not `activelane`, and the offsets aren't clamped to the wavefront. +@inline function bpermute_lane(val, lane::Cint) + return Device._shfl(x -> Device.bpermute(lane << 0x2, x), val) end +@inline lane_id() = hardware_lane() % Cint + +@device_override @inline KI.shfl(val::T, lane::Integer) where {T <: ShuffleTypes} = + bpermute_lane(val, (lane % Cint) - Cint(1)) + +@device_override @inline KI.shfl_down(val::T, offset::Integer) where {T <: ShuffleTypes} = + bpermute_lane(val, lane_id() + (offset % Cint)) + +@device_override @inline KI.shfl_up(val::T, offset::Integer) where {T <: ShuffleTypes} = + bpermute_lane(val, lane_id() - (offset % Cint)) + +@device_override @inline KI.shfl_xor(val::T, mask::Integer) where {T <: ShuffleTypes} = + bpermute_lane(val, lane_id() ⊻ (mask % Cint)) + +# `ballot` only sets the bits of active lanes, i.e. of the work-items of the sub-group, and +# its result is uniform. `wavefrontsize`, which selects the 32- or 64-bit ballot, is folded +# to the wavefront size the kernel is compiled for (see `fold_wavefrontsize!`). +@device_override @inline KI.sub_group_ballot(pred::Bool) = Device.ballot(pred) + +@device_override @inline KI.sub_group_any(pred::Bool) = Device.ballot(pred) != 0 + +@device_override @inline KI.sub_group_all(pred::Bool) = Device.ballot(!pred) == 0 + # not supported, see the `ROCBackend` docstring @device_override @inline KI._print(args...) = nothing From ab2488702eb9697be3b1b3ba1b6a6ac94073207f Mon Sep 17 00:00:00 2001 From: Valentin Churavy Date: Sun, 4 Oct 2026 09:42:17 +0200 Subject: [PATCH 2/4] [TEMP] Test against KernelAbstractions' vc/ki-subgroup-ops branch Take KernelAbstractions and KernelInterface from the branch of JuliaGPU/KernelAbstractions.jl#831, which specifies the sub-group shuffles, votes and constant width, and the SPIRVIntrinsics it needs from OpenCL.jl's vc/subgroup-votes branch (JuliaGPU/OpenCL.jl#526), which KernelAbstractions gets from its [sources] that dependents don't pick up. Also allow AcceleratedKernels 0.5, which its main branch now is. Drop this commit once #831 is merged. Assisted-by: Claude Code (Opus 5.5) --- .buildkite/pipeline.yml | 5 +++-- Project.toml | 8 +++++--- test/Project.toml | 6 ++++-- 3 files changed, 12 insertions(+), 7 deletions(-) diff --git a/.buildkite/pipeline.yml b/.buildkite/pipeline.yml index a0c2da48a..436d03136 100644 --- a/.buildkite/pipeline.yml +++ b/.buildkite/pipeline.yml @@ -106,12 +106,13 @@ steps: (build.message =~ /\[only [^\]]*(tests|julia)/ || build.message !~ /\[only / && !build.pull_request.draft) commands: | - git clone --depth 1 --branch main https://github.com/JuliaGPU/KernelAbstractions.jl ka + git clone --depth 1 --branch vc/ki-subgroup-ops https://github.com/JuliaGPU/KernelAbstractions.jl ka git clone --depth 1 https://github.com/JuliaGPU/AcceleratedKernels.jl ak + git clone --depth 1 --branch vc/subgroup-votes https://github.com/JuliaGPU/OpenCL.jl ocl julia --project=test -e ' using Pkg if VERSION < v"1.11" - Pkg.develop([PackageSpec(; path) for path in (".", "ka", "ka/lib/KernelInterface", "ak")]) + Pkg.develop([PackageSpec(; path) for path in (".", "ka", "ka/lib/KernelInterface", "ak", "ocl/lib/intrinsics")]) else Pkg.instantiate() end' diff --git a/Project.toml b/Project.toml index 747de04da..eb9944042 100644 --- a/Project.toml +++ b/Project.toml @@ -52,13 +52,14 @@ AMDGPUSpecialFunctionsExt = "SpecialFunctions" [sources] AcceleratedKernels = {url = "https://github.com/JuliaGPU/AcceleratedKernels.jl", rev = "main"} -KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main"} -KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main", subdir = "lib/KernelInterface"} +KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops"} +KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops", subdir = "lib/KernelInterface"} +SPIRVIntrinsics = {url = "https://github.com/JuliaGPU/OpenCL.jl", rev = "vc/subgroup-votes", subdir = "lib/intrinsics"} [compat] AMDGPU_LLVM_Backend_jll = "23" AbstractFFTs = "1.0" -AcceleratedKernels = "0.3.1, 0.4" +AcceleratedKernels = "0.3.1, 0.4, 0.5" Adapt = "4" Atomix = "1" BFloat16s = "0.6.0" @@ -93,3 +94,4 @@ julia = "1.10" [extras] KernelAbstractions = "63c18a36-062a-441e-b654-da1e3ab1ce7c" +SPIRVIntrinsics = "71d1d633-e7e8-4a92-83a1-de8814b09ba8" diff --git a/test/Project.toml b/test/Project.toml index 16c56d2c0..5abf689fe 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -18,6 +18,7 @@ ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" PrettyTables = "08abe8d2-0d0c-5749-adfa-8a2ac140af0d" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" +SPIRVIntrinsics = "71d1d633-e7e8-4a92-83a1-de8814b09ba8" SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" SparseMatricesCSR = "a0a7dd2c-ebf4-11e9-1f05-cf50bc540ca1" SpecialFunctions = "276daf66-3868-5448-9aa4-cd146d93841b" @@ -28,5 +29,6 @@ Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" [sources] AMDGPU = {path = ".."} -KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main"} -KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "main", subdir = "lib/KernelInterface"} +KernelAbstractions = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops"} +KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "vc/ki-subgroup-ops", subdir = "lib/KernelInterface"} +SPIRVIntrinsics = {url = "https://github.com/JuliaGPU/OpenCL.jl", rev = "vc/subgroup-votes", subdir = "lib/intrinsics"} From 305ec4bc440ac2fd2ba1ba8d770bdc50d29f5ea5 Mon Sep 17 00:00:00 2001 From: Valentin Churavy Date: Sun, 4 Oct 2026 14:16:19 +0200 Subject: [PATCH 3/4] Return the own value from KI shfl_down/shfl_up past the wavefront KernelInterface now defines `shfl_down` and `shfl_up` to return the work-item's own value where the source lane is past the sub-group width (like CUDA's shuffles), rather than an unspecified value. `ds_bpermute` wraps the lane around, so select the work-item's own lane in that case, one compare and select with the constant wavefront size. The shuffles with a `width` use KernelInterface's fallbacks (a `ds_bpermute` from a computed lane), which is what `Device.shfl` etc. do too. Assisted-by: Claude Code (Opus 5.5) --- src/ROCKernels.jl | 22 +++++++++++++++++----- 1 file changed, 17 insertions(+), 5 deletions(-) diff --git a/src/ROCKernels.jl b/src/ROCKernels.jl index c7e9fbe82..069198252 100644 --- a/src/ROCKernels.jl +++ b/src/ROCKernels.jl @@ -255,7 +255,7 @@ end # read `val` from the work-item in the 0-based hardware lane `lane`. `ds_bpermute` only uses # the low bits of the address, so lanes out of range read some other lane instead of # trapping. unlike `Device.shfl` etc., the lanes are the hardware lanes (see -# `hardware_lane`), not `activelane`, and the offsets aren't clamped to the wavefront. +# `hardware_lane`), not `activelane`. @inline function bpermute_lane(val, lane::Cint) return Device._shfl(x -> Device.bpermute(lane << 0x2, x), val) end @@ -265,15 +265,27 @@ end @device_override @inline KI.shfl(val::T, lane::Integer) where {T <: ShuffleTypes} = bpermute_lane(val, (lane % Cint) - Cint(1)) -@device_override @inline KI.shfl_down(val::T, offset::Integer) where {T <: ShuffleTypes} = - bpermute_lane(val, lane_id() + (offset % Cint)) +# where the source lane is past the wavefront, `shfl_down` and `shfl_up` read from the +# work-item itself, like CUDA's shuffles (rather than from the lane `ds_bpermute` wraps around +# to). `wavefrontsize` is folded to a constant, see `fold_wavefrontsize!`. +@device_override @inline function KI.shfl_down(val::T, offset::Integer) where {T <: ShuffleTypes} + lane = lane_id() + ws = Device.wavefrontsize() % Cint + return bpermute_lane(val, ifelse(offset < ws - lane, lane + (offset % Cint), lane)) +end -@device_override @inline KI.shfl_up(val::T, offset::Integer) where {T <: ShuffleTypes} = - bpermute_lane(val, lane_id() - (offset % Cint)) +@device_override @inline function KI.shfl_up(val::T, offset::Integer) where {T <: ShuffleTypes} + lane = lane_id() + return bpermute_lane(val, ifelse(offset <= lane, lane - (offset % Cint), lane)) +end +# `mask` is below the wavefront size, so the source lane is in the wavefront @device_override @inline KI.shfl_xor(val::T, mask::Integer) where {T <: ShuffleTypes} = bpermute_lane(val, lane_id() ⊻ (mask % Cint)) +# the shuffles with a `width` use KernelInterface's fallbacks, a `ds_bpermute` from a lane +# computed with a few integer operations, like `Device.shfl` etc. (which use `activelane`). + # `ballot` only sets the bits of active lanes, i.e. of the work-items of the sub-group, and # its result is uniform. `wavefrontsize`, which selects the 32- or 64-bit ballot, is folded # to the wavefront size the kernel is compiled for (see `fold_wavefrontsize!`). From 8e0e202c443bd9ca0f6dd53f956de673a075191b Mon Sep 17 00:00:00 2001 From: Valentin Churavy Date: Sun, 4 Oct 2026 15:51:41 +0200 Subject: [PATCH 4/4] Use Julia's min and max for floats The device overrides of `Base.min` and `Base.max` for floats called OCML's `__ocml_min`/`__ocml_max`, which, like C's `fmin`/`fmax`, return the other argument for a NaN, while Julia's `min` and `max` return NaN. Drop them: Julia's own definitions (`llvm.minimum`/`llvm.maximum` on Julia 1.12+, arithmetic before) compile for AMDGPU. Assisted-by: Claude Code (Opus 5.5) --- src/device/gcn/math.jl | 6 ------ test/device/math.jl | 11 +++++++++++ 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/src/device/gcn/math.jl b/src/device/gcn/math.jl index df1cc231f..d45b2e6f5 100644 --- a/src/device/gcn/math.jl +++ b/src/device/gcn/math.jl @@ -60,12 +60,6 @@ for jltype in (Float64, Float32, Float16) @eval @device_override Base.fma(x::$jltype, y::$jltype, z::$jltype) = ccall( $("extern __ocml_fma_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype, $jltype), x, y, z) - @eval @device_override Base.min(x::$jltype, y::$jltype) = ccall( - $("extern __ocml_min_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype), x, y) - - @eval @device_override Base.max(x::$jltype, y::$jltype) = ccall( - $("extern __ocml_max_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype), x, y) - @eval @device_override Base.copysign(x::$jltype, y::$jltype) = ccall( $("extern __ocml_copysign_$(fntypes[jltype])"), llvmcall, $jltype, ($jltype, $jltype), x, y) diff --git a/test/device/math.jl b/test/device/math.jl index 618eada48..a2e433e85 100644 --- a/test/device/math.jl +++ b/test/device/math.jl @@ -20,6 +20,17 @@ using Base.FastMath end end +@testset "min/max" begin + # Julia's semantics: NaN propagates, and -0.0 < 0.0 + a = [NaN, 1, -0.0, 0.0, 2, NaN, Inf, -Inf] + b = [1, NaN, 0.0, -0.0, 3, NaN, NaN, 1] + for T in (Float16, Float32, Float64) + x, y = T.(a), T.(b) + @test isequal(Array(max.(ROCArray(x), ROCArray(y))), max.(x, y)) + @test isequal(Array(min.(ROCArray(x), ROCArray(y))), min.(x, y)) + end +end + @testset "Fast min/max" begin function ker!(x) x[1] = @fastmath max(x[1], zero(eltype(x)))