Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .buildkite/pipeline.yml
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ steps:
# Julia 1.10 does not support [sources], so dev the in-tree
# SPIRVIntrinsics; Pkg.test then carries it into the test sandbox.
Pkg.develop(path="lib/intrinsics")
Pkg.develop("KernelAbstractions")

println("+++ :julia: Running tests")
Pkg.test(; coverage=true, test_args=`--platform=cuda`)'
Expand Down Expand Up @@ -48,6 +49,7 @@ steps:
# against the registry, which has no SPIRVIntrinsics 1. Pkg.test
# then carries the in-tree copy into the test sandbox.
Pkg.develop(path="lib/intrinsics")
Pkg.develop("KernelAbstractions")
Pkg.add("{{matrix.pocl}}_jll")
Pkg.add("InteractiveUtils")

Expand Down
6 changes: 5 additions & 1 deletion .github/workflows/Test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -142,7 +142,11 @@ jobs:
using Pkg
# Julia 1.10 does not support [sources], so dev the in-tree
# SPIRVIntrinsics; Pkg.test then carries it into the test sandbox.
Pkg.develop(path="lib/intrinsics")'
Pkg.develop(path="lib/intrinsics")
Pkg.develop("KernelAbstractions")
# Pkg < 1.12 drops a developed weak dependency from the project, breaking the
# extension; restore the project, keeping the developed package in the manifest.
VERSION < v"1.12" && run(`git checkout Project.toml`)'

- name: Test OpenCL.jl
uses: julia-actions/julia-runtest@v1
Expand Down
14 changes: 12 additions & 2 deletions Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@ Adapt = "79e6a3ab-5dfb-504d-930d-738a2a938a0e"
GPUArrays = "0c68f7d7-f131-5f86-a1c3-88cf8149b2d7"
GPUCompiler = "61eb1bfa-7361-4325-ad38-22787b887f55"
GPUToolbox = "096a3bc2-3ced-46d0-87f4-dd12716f4bfc"
KernelAbstractions = "63c18a36-062a-441e-b654-da1e3ab1ce7c"
KernelInterface = "4ee993da-d684-4d17-a7dd-4e58e78d92bf"
LLVM = "929cbde3-209d-540e-8aea-75f648917ca0"
LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e"
OpenCL_jll = "6cb37087-e8b6-5417-8430-1f242f1e46e4"
Expand All @@ -26,15 +26,22 @@ SPIRV_Tools_jll = "6ac6d60f-d740-5983-97d7-a4482c0689f4"
StaticArrays = "90137ffa-7385-5640-81b9-e52037218182"
spirv2clc_jll = "f0274c0c-8c8a-59f1-85b7-f7d60330c5fb"

[weakdeps]
KernelAbstractions = "63c18a36-062a-441e-b654-da1e3ab1ce7c"

[sources]
SPIRVIntrinsics = {path = "lib/intrinsics"}

[extensions]
KernelAbstractionsExt = "KernelAbstractions"

[compat]
Adapt = "4"
GPUArrays = "11.2.1"
GPUCompiler = "2.9"
GPUToolbox = "3.1"
KernelAbstractions = "0.9.38"
KernelAbstractions = "0.9, 0.10"
KernelInterface = "0.2.3"
LLVM = "9.6"
LinearAlgebra = "1"
OpenCL_jll = "=2024.10.24"
Expand All @@ -50,3 +57,6 @@ SPIRV_Tools_jll = "2025.1"
StaticArrays = "1"
julia = "1.10"
spirv2clc_jll = "0.2.0"

[extras]
KernelAbstractions = "63c18a36-062a-441e-b654-da1e3ab1ce7c"
113 changes: 113 additions & 0 deletions ext/KernelAbstractionsExt.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,113 @@
module KernelAbstractionsExt

using OpenCL
using OpenCL: @device_override, method_table

import KernelAbstractions as KA

import StaticArrays

import Adapt

Adapt.adapt_storage(::OpenCLBackend, a::Array) = Adapt.adapt(CLArray, a)
Adapt.adapt_storage(::OpenCLBackend, a::CLArray) = a
Adapt.adapt_storage(::KA.CPU, a::CLArray) = convert(Array, a)

# `@Const` applies `constify` inside the kernel, where arguments have already been
# converted to device arrays, so the rule has to be registered for `CLDeviceArray`
# rather than for `CLArray`.
Adapt.adapt_storage(::KA.ConstAdaptor, a::CLDeviceArray) = Base.Experimental.Const(a)

## Kernel Launch

function KA.mkcontext(kernel::KA.Kernel{OpenCLBackend}, _ndrange, iterspace)
KA.CompilerMetadata{KA.ndrange(kernel), KA.DynamicCheck}(_ndrange, iterspace)
end
function KA.mkcontext(kernel::KA.Kernel{OpenCLBackend}, I, _ndrange, iterspace,
::Dynamic) where Dynamic
KA.CompilerMetadata{KA.ndrange(kernel), Dynamic}(I, _ndrange, iterspace)
end

function KA.launch_config(kernel::KA.Kernel{OpenCLBackend}, ndrange, workgroupsize)
if ndrange isa Integer
ndrange = (ndrange,)
end
if workgroupsize isa Integer
workgroupsize = (workgroupsize, )
end

# partition checked that the ndrange's agreed
if KA.ndrange(kernel) <: KA.StaticSize
ndrange = nothing
end

iterspace, dynamic = if KA.workgroupsize(kernel) <: KA.DynamicSize &&
workgroupsize === nothing
# use ndrange as preliminary workgroupsize for autotuning
KA.partition(kernel, ndrange, ndrange)
else
KA.partition(kernel, ndrange, workgroupsize)
end

return ndrange, workgroupsize, iterspace, dynamic
end

# the number of indices along each dimension of `ndrange`, which may contain ranges
extents(ndrange) = size(CartesianIndices(ndrange))

function threads_to_workgroupsize(threads, ndrange)
total = 1
return map(ndrange) do n
x = max(min(div(threads, total), n), 1)
total *= x
return x
end
end

function (obj::KA.Kernel{OpenCLBackend})(args...; ndrange=nothing, workgroupsize=nothing)
obj.backend.platform === cl.platform() || OpenCL.OpenCLKernels.platform_mismatch_warning(obj.backend.platform, cl.platform())

ndrange, workgroupsize, iterspace, dynamic =
KA.launch_config(obj, ndrange, workgroupsize)
# nothing to launch (or compile) for an empty ndrange
length(KA.blocks(iterspace)) == 0 && return nothing

# this might not be the final context, since we may tune the workgroupsize
ctx = KA.mkcontext(obj, ndrange, iterspace)
kernel = @opencl launch=false obj.f(ctx, args...)

# figure out the optimal workgroupsize automatically
if KA.workgroupsize(obj) <: KA.DynamicSize && workgroupsize === nothing
wg_info = cl.work_group_info(kernel.fun, cl.device())
wg_size_nd = threads_to_workgroupsize(wg_info.size, extents(ndrange))
iterspace, dynamic = KA.partition(obj, ndrange, wg_size_nd)
ctx = KA.mkcontext(obj, ndrange, iterspace)
end

groups = length(KA.blocks(iterspace))
items = length(KA.workitems(iterspace))

# Launch kernel
global_size = groups * items
local_size = items
kernel(ctx, args...; global_size, local_size)

return nothing
end

@device_override @inline function KA.__validindex(ctx)
if KA.__dynamic_checkbounds(ctx)
I = KA.__index_Global_Cartesian(ctx)
return I in KA.__ndrange(ctx)
else
return true
end
end

@device_override @inline function KA.Scratchpad(ctx, ::Type{T}, ::Val{Dims}) where {T, Dims}
StaticArrays.MArray{KA.__size(Dims), T}(undef)
end

KA.argconvert(::KA.Kernel{OpenCLBackend}, arg) = OpenCL.kernel_convert(arg)

end
3 changes: 2 additions & 1 deletion src/OpenCL.jl
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@ using GPUArrays
using Random
using Preferences

import KernelAbstractions: KernelAbstractions
import KernelInterface

using Core: LLVMPtr

Expand Down Expand Up @@ -52,4 +52,5 @@ include("random.jl")
include("OpenCLKernels.jl")
import .OpenCLKernels: OpenCLBackend
export OpenCLBackend

end
Loading
Loading