diff --git a/Project.toml b/Project.toml index 0cf5fc004..b7ee65576 100644 --- a/Project.toml +++ b/Project.toml @@ -42,7 +42,7 @@ Adapt = "0.4, 1.0, 2.0, 3.0, 4" Atomix = "1.2.1" EnzymeCore = "0.7, 0.8.1" GPUCompiler = "2.7" -KernelInterface = "0.3" +KernelInterface = "0.4" LLVM = "9.9" LinearAlgebra = "1.6" MacroTools = "0.5" diff --git a/docs/src/kernelinterface.md b/docs/src/kernelinterface.md index 1f6fa0f5e..ca72776c2 100644 --- a/docs/src/kernelinterface.md +++ b/docs/src/kernelinterface.md @@ -268,12 +268,24 @@ optional methods where it can do better than the fallback. In particular: [`adapt(backend, x)`](@ref Adapt.adapt_storage(::Backend, ::Any)) moves data to the backend, preferably by delegating to its array type: `Adapt.adapt_storage(::NewBackend, x) = adapt(NewArray, x)`. -3. Implement [`kernel_function`](@ref), returning a [`Kernel`](@ref) that holds the - backend value it was given, and [`launch`](@ref), which receives an already validated - `NTuple{3, Int}` of work-groups and of work-items. For CUDA.jl, the latter is +3. Implement [`kernel_function`](@ref), which receives the unconverted callable, and + returns a [`Kernel`](@ref) that holds the backend value it was given and keeps that + callable alive. Also implement [`launch`](@ref), which receives an already validated + `NTuple{3, Int}` of work-groups and of work-items, and the arguments as a tuple. Pass + that tuple on to the native launcher rather than splatting it: Julia doesn't turn a + splat of more than 32 elements into a direct call, so kernels with many arguments would + be slow to launch. For the PoCL backend, whose kernels hold the compiled kernel and the + callable, `launch` is ```julia - KI.launch(k::KI.Kernel{CUDABackend}, groups::Dims{3}, items::Dims{3}, args::Vararg{Any, N}; kwargs...) where {N} = - k.kern(args...; threads = items, blocks = groups, kwargs...) + function KI.launch(k::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) + f = k.kern.f + event = GC.@preserve f args begin + event = POCL.launch_tuple(k.kern.kernel, args; local_size = items, global_size = groups .* items) + wait(event) + end + cl.clReleaseEvent(event) + return nothing + end ``` 4. Compute the typed index queries with `% T`, not `T(x)`: a checked conversion leaves an error branch in every kernel. diff --git a/lib/KernelInterface/Project.toml b/lib/KernelInterface/Project.toml index e34a91b0e..6fc175c32 100644 --- a/lib/KernelInterface/Project.toml +++ b/lib/KernelInterface/Project.toml @@ -1,7 +1,7 @@ name = "KernelInterface" uuid = "4ee993da-d684-4d17-a7dd-4e58e78d92bf" authors = ["Valentin Churavy and contributors"] -version = "0.3.0" +version = "0.4.0-dev" [compat] julia = "1.10" diff --git a/lib/KernelInterface/src/launch.jl b/lib/KernelInterface/src/launch.jl index 97809596f..99d181b81 100644 --- a/lib/KernelInterface/src/launch.jl +++ b/lib/KernelInterface/src/launch.jl @@ -48,23 +48,30 @@ struct Kernel{B, Kern} kern::Kern end -# `Vararg{Any, N}` makes Julia specialize on the arguments, which are only passed through -function (kernel::Kernel)( - args::Vararg{Any, N}; numgroups = (), workgroupsize = (), ndrange = (), +# The arguments are passed on as a tuple: Julia doesn't turn a splat of more than 32 +# elements into a direct call, and a method with both varargs and keyword arguments splats +# them into its body. So the keyword method is defined explicitly, as Base does for +# `invokelatest`. `Vararg{Any, N}` makes Julia specialize on the arguments. +(kernel::Kernel)(args::Vararg{Any, N}) where {N} = call_kernel(kernel, args) +Core.kwcall(kwargs::NamedTuple, kernel::Kernel, args::Vararg{Any, N}) where {N} = + call_kernel(kernel, args; kwargs...) + +function call_kernel( + kernel::Kernel, args::Tuple; numgroups = (), workgroupsize = (), ndrange = (), max_work_group_size::Integer = typemax(Int), kwargs... - ) where {N} + ) groups, items = launch_geometry(kernel, numgroups, workgroupsize, ndrange, max_work_group_size) any(iszero, groups) && return nothing - launch(kernel, groups, items, args...; kwargs...) + launch(kernel, groups, items, args; kwargs...) return nothing end """ - launch(kernel::Kernel, groups::Dims{3}, items::Dims{3}, args...; kwargs...) + launch(kernel::Kernel, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) Launch `kernel` with `groups` work-groups of `items` work-items each, passing the host-side -arguments `args`. This is what calling a [`Kernel`](@ref) does after validating and -normalizing the launch geometry; users call the kernel instead. +arguments `args`, a tuple. This is what calling a [`Kernel`](@ref) does after validating +and normalizing the launch geometry; users call the kernel instead. `groups` and `items` are positive, `items` fits [`max_work_group_dims`](@ref) and [`max_work_group_size`](@ref)`(kernel)`, and `groups .* items` doesn't overflow `Int`. @@ -73,15 +80,19 @@ normalizing the launch geometry; users call the kernel instead. !!! note Backend implementations **must** implement: ``` - launch(kernel::Kernel{<:NewBackend}, groups::Dims{3}, items::Dims{3}, args...; kwargs...) + launch(kernel::Kernel{<:NewBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) ``` It converts `args` with [`argconvert`](@ref) (or lets its native launcher do so), and queues the launch on the calling task's queue; it doesn't have to wait for the kernel to - complete. Declare the arguments as `args::Vararg{Any, N}` (with `where {N}`): Julia - doesn't specialize a method on `args...` that it only passes through, which makes every - launch dispatch dynamically. It must throw for keywords it does not support, and may - throw for a number of work-groups the device cannot launch, or for a geometry that + complete. To keep launches with many arguments cheap, it should pass `args` on as a + tuple rather than splatting it: Julia doesn't turn a splat of more than 32 elements into + a direct call. It must throw for keywords it does not support, and may throw for a + number of work-groups the device cannot launch, or for a geometry that backend-specific compiler options of the kernel don't allow. + +!!! compat "KernelInterface 0.4" + Before KernelInterface 0.4, `launch` received the arguments as varargs, + `launch(kernel, groups, items, args...; kwargs...)`. """ function launch end @@ -351,23 +362,34 @@ function argconvert end """ kernel_function(backend, f::F, tt::TT=Tuple{}; name=nothing, kwargs...)::Kernel -Compile the function `f` for arguments of the (device-side) types `tt`, for the active +Compile the callable `f` for arguments of the (device-side) types `tt`, for the active device of `backend`, returning a [`Kernel`](@ref). For a higher-level interface, use [`KernelInterface.@launch`](@ref). +`f` is the host-side callable, not converted with [`argconvert`](@ref): the backend converts +it. For a closure, that matters: it can capture arrays, which its converted form only holds +pointers to. + Keyword arguments: - `name`: override the name that the kernel will have in the generated code. Other keyword arguments are backend-specific compiler options (e.g. `maxthreads` for CUDA.jl); backends throw an error for options they don't support. -The returned kernel doesn't keep any arguments alive: they are passed again at launch. +The returned kernel keeps `f` alive, but not the arguments: they are passed again at +launch. !!! note Backend implementations **must** implement: ``` kernel_function(backend::NewBackend, f::F, tt::TT=Tuple{}; name=nothing, kwargs...) where {F,TT} ``` + It converts `f` with [`argconvert`](@ref) to compile it, and the returned `Kernel` has + to keep `f` itself alive for as long as it can be launched, since the converted `f` + may only hold pointers to the arrays `f` captures. A backend that needs to know about + those arrays at launch, e.g. to declare them to the device, can convert `f` again for + every launch, as it does for the arguments. + The returned `Kernel` stores `backend` itself (not a new default backend), so that options it carries apply to the launch. Kernels must execute with sub-group width [`sub_group_size(backend)`](@ref sub_group_size) if the backend supports sub-groups. @@ -378,6 +400,13 @@ The returned kernel doesn't keep any arguments alive: they are passed again at l """ function kernel_function end +# `Tuple{map(x -> Core.Typeof(argconvert(backend, x)), args)...}`, without `map`, which +# isn't type stable for 32 or more elements +@inline @generated function argument_types(backend, args::Tuple) + types = (:(Core.Typeof(argconvert(backend, args[$i]))) for i in 1:fieldcount(args)) + return :(Tuple{$(types...)}) +end + const MACRO_KWARGS = [:launch] const LAUNCH_KWARGS = [:numgroups, :workgroupsize, :ndrange, :max_work_group_size] @@ -386,8 +415,8 @@ const LAUNCH_KWARGS = [:numgroups, :workgroupsize, :ndrange, :max_work_group_siz Compile `f(args...)` for `backend` and launch it, like `@cuda` or `@metal` do. -`f` and the arguments are converted with [`argconvert`](@ref) and compiled with -[`kernel_function`](@ref), and the resulting [`Kernel`](@ref) is called with the launch +`f` is compiled with [`kernel_function`](@ref) for the types of the arguments converted +with [`argconvert`](@ref), and the resulting [`Kernel`](@ref) is called with the launch keywords `numgroups`, `workgroupsize`, `ndrange` and `max_work_group_size`, whose meaning is documented there. The arguments are kept alive while the launch is being queued. @@ -452,7 +481,7 @@ macro launch(backend, ex...) # FIXME: macro hygiene wrt. escaping kwarg values (this broke with 1.5) # we esc() the whole thing now, necessitating gensyms... - @gensym backend_var f_var kernel_f kernel_args kernel_tt kernel + @gensym backend_var f_var kernel_tt kernel # convert the arguments, call the compiler and launch the kernel # while keeping the original arguments alive @@ -462,10 +491,8 @@ macro launch(backend, ex...) $backend_var = $backend $f_var = $f GC.@preserve $(vars...) $f_var begin - $kernel_f = $argconvert($backend_var, $f_var) - $kernel_args = Base.map(x -> $argconvert($backend_var, x), ($(var_exprs...),)) - $kernel_tt = Tuple{Base.map(Core.Typeof, $kernel_args)...} - $kernel = $kernel_function($backend_var, $kernel_f, $kernel_tt; $(compiler_kwargs...)) + $kernel_tt = $argument_types($backend_var, ($(var_exprs...),)) + $kernel = $kernel_function($backend_var, $f_var, $kernel_tt; $(compiler_kwargs...)) if $launch $kernel($(var_exprs...); $(call_kwargs...)) end diff --git a/lib/KernelInterface/test/interface.jl b/lib/KernelInterface/test/interface.jl index 3cd9a11e6..2e5366689 100644 --- a/lib/KernelInterface/test/interface.jl +++ b/lib/KernelInterface/test/interface.jl @@ -238,6 +238,13 @@ function sub_group_barrier_kernel(scratch, out, ::Val{N}) where {N} return end +# a kernel whose callable captures an array, compiled but not launched yet +function captured_array_kernel(backend, AT, out) + a = AT(Int32[42]) + kernel = KI.@launch backend launch = false (() -> (@inbounds out[1] = a[1]; nothing))() + return kernel, WeakRef(a) +end + function interface_testsuite(backend::KI.Backend, AT) @testset "Launch parameters" begin # unequal group counts and sizes in every dimension, so that confusing them shows @@ -527,6 +534,19 @@ function interface_testsuite(backend::KI.Backend, AT) end end + # The converted callable only holds pointers to the arrays it captures, so the kernel + # has to keep the original alive (and a backend may need it at launch). + @testset "Captured arrays" begin + out = AT(Int32[0]) + kernel, captured = captured_array_kernel(backend, AT, out) + GC.gc(true) + @test captured.value !== nothing + garbage = [AT(fill(Int32(7), 1)) for _ in 1:100] + kernel() + KI.synchronize(backend) + @test Array(out) == Int32[42] + end + @testset "Local memory and barriers" begin N = 32 groups = 3 @@ -693,7 +713,7 @@ function contract_testsuite(backend::KI.Backend, AT) @test hasmethod(KI.copyto!, Tuple{B, AT, Array}) @test hasmethod(KI.argconvert, Tuple{B, Any}) @test hasmethod(KI.kernel_function, Tuple{B, Any, Type}) - @test hasmethod(KI.launch, Tuple{KI.Kernel{B}, Dims{3}, Dims{3}}) + @test hasmethod(KI.launch, Tuple{KI.Kernel{B}, Dims{3}, Dims{3}, Tuple}) @test hasmethod(KI.max_work_group_size, Tuple{B}) @test hasmethod(KI.max_work_group_size, Tuple{KI.Kernel{B}}) @test hasmethod(KI.max_work_group_dims, Tuple{B}) diff --git a/lib/KernelInterface/test/runtests.jl b/lib/KernelInterface/test/runtests.jl index 956417597..a8e5b8def 100644 --- a/lib/KernelInterface/test/runtests.jl +++ b/lib/KernelInterface/test/runtests.jl @@ -255,7 +255,7 @@ KI.argconvert(::MockBackend, arg) = arg function KI.kernel_function(backend::MockBackend, f, tt = Tuple{}; name = nothing, kwargs...) return KI.Kernel(backend, MockKernel(f, tt, name, Dict(kwargs), [])) end -function KI.launch(kernel::KI.Kernel{MockBackend}, groups::Dims{3}, items::Dims{3}, args...; kwargs...) +function KI.launch(kernel::KI.Kernel{MockBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) push!(kernel.kern.launches, (; groups, items, args, kwargs = Dict(kwargs))) return :ignored end @@ -276,9 +276,20 @@ function KI.launch_configuration( push!(kernel.backend.queries, (; nitems, max_work_group_size)) return (; workgroupsize = min(96, max_work_group_size)) end -KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, args...) = +KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) = push!(kernel.kern, (groups, items)) +# a callable that KernelInterface mustn't convert: the backend does +struct HostCallable end +(::HostCallable)(x) = nothing +KI.argconvert(::MockBackend, ::HostCallable) = error("only the backend should convert the callable") + +# ... and one that does nothing, to measure the overhead of launching +struct NullBackend <: KI.Backend end +KI.max_work_group_size(::KI.Kernel{NullBackend}) = 1024 +KI.max_work_group_dims(::NullBackend) = (1024, 1024, 64) +KI.launch(::KI.Kernel{NullBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) = nothing + @testset "launch geometry" begin kernel = KI.kernel_function(MockBackend(), identity, Tuple{Int}) function geometry(; kwargs...) @@ -345,6 +356,13 @@ KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, kernel(1; ndrange = 4, stream = :mine) @test last(kernel.kern.launches).kwargs == Dict(:stream => :mine) + # The arguments reach the backend as one tuple, whatever their number, and a single + # tuple-valued argument stays one argument. + kernel((1, 2); ndrange = 4) + @test last(kernel.kern.launches).args == ((1, 2),) + kernel(ntuple(identity, 40)...; ndrange = 4) + @test last(kernel.kern.launches).args == ntuple(identity, 40) + # Auto-sizing uses the backend's recommendation, not the limit, and tells it both the # size of the launch and the cap. occupancy = KI.Kernel(OccupancyBackend(), []) @@ -411,6 +429,17 @@ function counted_backend() return MockBackend() end +# Julia doesn't turn a splat of more than 32 elements into a direct call, so launching with +# many arguments allocates unless they're passed on as a tuple +@testset "many arguments" begin + kernel = KI.Kernel(NullBackend(), nothing) + @eval launch_few(k) = k($((1:4)...); numgroups = 2, workgroupsize = 4) + @eval launch_many(k) = k($((1:40)...); numgroups = 2, workgroupsize = 4) + launch_few(kernel) + launch_many(kernel) + @test @allocated(launch_many(kernel)) <= @allocated(launch_few(kernel)) +end + @testset "@launch" begin backend = MockBackend() @@ -439,6 +468,9 @@ end optioned = KI.@launch backend ndrange = 4 maxthreads = 32 dummy(1, 2.0) @test isempty(only(optioned.kern.launches).kwargs) + # The callable is compiled unconverted. + @test (KI.@launch backend launch = false HostCallable()(1)).kern.f isa HostCallable + # Splatted arguments are supported. splatted = KI.@launch backend launch = false dummy((1, 2.0)...) @test splatted.kern.tt == Tuple{Int, Float64} diff --git a/src/backend_launch.jl b/src/backend_launch.jl index 38b2c7e90..fbd5cfd8d 100644 --- a/src/backend_launch.jl +++ b/src/backend_launch.jl @@ -74,7 +74,15 @@ end argconvert(kernel::Kernel{<:KI.Backend}, arg) = KI.argconvert(backend(kernel), arg) -function (obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}; ndrange = nothing, workgroupsize = nothing) where {N} +# The arguments are passed on as a tuple: Julia doesn't turn a splat of more than 32 +# elements into a direct call, and a method with both varargs and keyword arguments splats +# them into its body. So the keyword method is defined explicitly, as Base does for +# `invokelatest`. +(obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}) where {N} = launch_tuple(obj, args) +Core.kwcall(kwargs::NamedTuple, obj::Kernel{<:KI.Backend}, args::Vararg{Any, N}) where {N} = + launch_tuple(obj, args; kwargs...) + +function launch_tuple(obj::Kernel, args::Tuple; ndrange = nothing, workgroupsize = nothing) ndrange, workgroupsize, iterspace, dynamic = launch_config(obj, ndrange, workgroupsize) # nothing to launch (or compile) for an empty ndrange any(iszero, size(blocks(iterspace))) && return nothing @@ -84,21 +92,19 @@ function (obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}; ndrange = nothing, wo launch = select_launch(obj, workgroupsize, iterspace) if launch === NDLaunch{Int32}() # the common case, specialized statically - launch_kernel(obj, NDLaunch{Int32}(), ndrange, workgroupsize, iterspace, args...) + launch_kernel(obj, NDLaunch{Int32}(), ndrange, workgroupsize, iterspace, args) else - launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args...) + launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args) end return nothing end -function launch_kernel( - obj::Kernel, launch, ndrange, _workgroupsize, iterspace, args::Vararg{Any, N} - ) where {N} +function launch_kernel(obj::Kernel, launch, ndrange, _workgroupsize, iterspace, args::Tuple) b = backend(obj) # this might not be the final context, since we may tune the workgroupsize ctx = mkcontext(obj, ndrange, iterspace, launch) - kernel = compile(obj, ctx, args...) + kernel = compile(obj, ctx, args) # tune the workgroup size, keeping the context type (and thus the kernel) the same if workgroupsize(obj) <: DynamicSize && _workgroupsize === nothing @@ -112,16 +118,30 @@ function launch_kernel( groups = size(blocks(iterspace)) items = size(workitems(iterspace)) if launch isa NDLaunch - kernel(ctx, args...; numgroups = groups, workgroupsize = items) + call_kernel(kernel, ctx, args, groups, items) else - kernel(ctx, args...; numgroups = prod(groups), workgroupsize = prod(items)) + call_kernel(kernel, ctx, args, prod(groups), prod(items)) end return nothing end -@inline function compile(obj::Kernel, ctx, args::Vararg{Any, N}) where {N} +@inline function compile(obj::Kernel, ctx, args::Tuple) b = backend(obj) - f = KI.argconvert(b, obj.f) - tt = Tuple{Core.Typeof(KI.argconvert(b, ctx)), map(arg -> Core.Typeof(KI.argconvert(b, arg)), args)...} - return KI.kernel_function(b, f, tt; compiler_options(obj)...) + tt = argument_types(b, ctx, args) + return KI.kernel_function(b, obj.f, tt; compiler_options(obj)...) +end + +# The helpers below avoid splatting the arguments, and `map`, which isn't type stable for 32 +# or more elements. + +# `Tuple{map(x -> Core.Typeof(KI.argconvert(backend, x)), (ctx, args...))...}` +@inline @generated function argument_types(backend, ctx, args::Tuple) + types = (:(Core.Typeof(KI.argconvert(backend, args[$i]))) for i in 1:fieldcount(args)) + return :(Tuple{Core.Typeof(KI.argconvert(backend, ctx)), $(types...)}) +end + +# `kernel(ctx, args...; numgroups, workgroupsize)` +@inline @generated function call_kernel(kernel::KI.Kernel, ctx, args::Tuple, numgroups, workgroupsize) + argexprs = (:(args[$i]) for i in 1:fieldcount(args)) + return :(kernel(ctx, $(argexprs...); numgroups, workgroupsize)) end diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 48a1633c0..b2f2c6603 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -132,27 +132,40 @@ KI.supports_atomics(::POCLBackend) = true KI.argconvert(::POCLBackend, arg) = clconvert(arg) +# a compiled kernel, and the callable it was compiled from. the compiled kernel only holds +# pointers to the arrays the callable captures, so the callable has to be kept alive. +struct POCLKernel{K, F} + kernel::K + f::F +end + function KI.kernel_function(backend::POCLBackend, f::F, tt::TT = Tuple{}; name = nothing, kwargs...) where {F, TT} # fix the sub-group width, as `KI.sub_group_size` promises sub_group_size = device_limits().sub_group_size - kern = if sub_group_size > 0 - clfunction(f, tt; name, sub_group_size, kwargs...) + kernel = if sub_group_size > 0 + clfunction(clconvert(f), tt; name, sub_group_size, kwargs...) else - clfunction(f, tt; name, kwargs...) + clfunction(clconvert(f), tt; name, kwargs...) end + kern = POCLKernel(kernel, f) return KI.Kernel{POCLBackend, typeof(kern)}(backend, kern) end -function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Vararg{Any, N}) where {N} - # POCL launches synchronously, see the implementation note on `synchronize` - event = obj.kern(args...; local_size = items, global_size = groups .* items) - wait(event) +function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) + # the kernel only gets pointers to the arrays in `args` and captured by `f`, so keep + # them alive until it completes. POCL launches synchronously, see the implementation + # note on `synchronize` + f = obj.kern.f + event = GC.@preserve f args begin + event = POCL.launch_tuple(obj.kern.kernel, args; local_size = items, global_size = groups .* items) + wait(event) + end cl.clReleaseEvent(event) return nothing end function KI.max_work_group_size(kernel::KI.Kernel{<:POCLBackend})::Int - wginfo = cl.work_group_info(kernel.kern.fun, device()) + wginfo = cl.work_group_info(kernel.kern.kernel.fun, device()) return Int(wginfo.size) end # querying the device allocates, so cache the limits that every launch needs diff --git a/src/pocl/compiler/execution.jl b/src/pocl/compiler/execution.jl index 7e0e20de2..897debaf8 100644 --- a/src/pocl/compiler/execution.jl +++ b/src/pocl/compiler/execution.jl @@ -52,7 +52,7 @@ macro opencl(ex...) # FIXME: macro hygiene wrt. escaping kwarg values (this broke with 1.5) # we esc() the whole thing now, necessitating gensyms... - @gensym f_var kernel_f kernel_args kernel_tt kernel + @gensym f_var kernel_f kernel_tt kernel # convert the arguments, call the compiler and launch the kernel # while keeping the original arguments alive @@ -62,8 +62,7 @@ macro opencl(ex...) $f_var = $f GC.@preserve $(vars...) $f_var begin $kernel_f = $clconvert($f_var) - $kernel_args = map($clconvert, ($(var_exprs...),)) - $kernel_tt = Tuple{map(Core.Typeof, $kernel_args)...} + $kernel_tt = $argument_types(($(var_exprs...),)) $kernel = $clfunction($kernel_f, $kernel_tt; $(compiler_kwargs...)) if $launch $kernel($(var_exprs...); $(call_kwargs...)) @@ -153,6 +152,13 @@ function clconvert(arg, pointers::Union{Nothing, Vector{Ptr{Cvoid}}} = nothing) return adapt(KernelAdaptor(pointers), arg) end +# `Tuple{map(x -> Core.Typeof(clconvert(x)), args)...}`, without `map`, which isn't type +# stable for 32 or more elements +@inline @generated function argument_types(args::Tuple) + types = (:(Core.Typeof(clconvert(args[$i]))) for i in 1:fieldcount(args)) + return :(Tuple{$(types...)}) +end + ## abstract kernel functionality @@ -160,12 +166,21 @@ abstract type AbstractKernel{F, TT} end pass_arg(@nospecialize dt) = !(GPUCompiler.isghosttype(dt) || Core.Compiler.isconstType(dt)) -@inline @generated function (kernel::AbstractKernel{F, TT})( - args::Vararg{Any, N}; - global_size = (1,), local_size = nothing - ) where {F, TT, N} +# The arguments are passed on as a tuple: Julia doesn't turn a splat of more than 32 +# elements into a direct call, and a method with both varargs and keyword arguments splats +# them into its body. So the keyword method is defined explicitly. +(kernel::AbstractKernel)(args::Vararg{Any, N}) where {N} = launch_tuple(kernel, args) +Core.kwcall(kwargs::NamedTuple, kernel::AbstractKernel, args::Vararg{Any, N}) where {N} = + launch_tuple(kernel, args; kwargs...) + +@inline launch_tuple(kernel::AbstractKernel, args::Tuple; global_size = (1,), local_size = nothing) = + launch_converted(kernel, args, global_size, local_size) + +@inline @generated function launch_converted( + kernel::AbstractKernel{F, TT}, args::Tuple, global_size, local_size + ) where {F, TT} sig = Tuple{F, TT.parameters...} # Base.signature_type with a function type - args = (:(kernel.f), (:(clconvert(args[$i])) for i in 1:length(args))...) + args = (:(kernel.f), (:(clconvert(args[$i])) for i in 1:fieldcount(args))...) # filter out ghost arguments that shouldn't be passed to_pass = map(pass_arg, sig.parameters) @@ -186,8 +201,11 @@ pass_arg(@nospecialize dt) = !(GPUCompiler.isghosttype(dt) || Core.Compiler.isco # finalize types call_tt = Base.to_tuple_type(call_t) + # the converted arguments only hold pointers to the arrays in `args` return quote - $cl.clcall(kernel.fun, $call_tt, $(call_args...); global_size, local_size, kernel.rng_state) + GC.@preserve args begin + $cl.clcall(kernel.fun, $call_tt, ($(call_args...),); global_size, local_size, kernel.rng_state) + end end end diff --git a/src/pocl/nanoOpenCL.jl b/src/pocl/nanoOpenCL.jl index 34450bbdc..cec2e0deb 100644 --- a/src/pocl/nanoOpenCL.jl +++ b/src/pocl/nanoOpenCL.jl @@ -1293,11 +1293,10 @@ function set_arg!(k::Kernel, idx::Integer, arg::T) where {T} return k end -set_args!(k::Kernel, args::Vararg{Any, N}) where {N} = _set_args!(k, 1, args...) -@inline _set_args!(k::Kernel, i::Int) = nothing -@inline function _set_args!(k::Kernel, i::Int, arg, args::Vararg{Any, N}) where {N} - set_arg!(k, i, arg) - return _set_args!(k, i + 1, args...) +# one call per argument: a splat of more than 32 arguments isn't a direct call +@inline @generated function set_args!(k::Kernel, args::Tuple) + calls = (:(set_arg!(k, $i, args[$i])) for i in 1:fieldcount(args)) + return :($(calls...); nothing) end # work sizes padded to the three dimensions OpenCL devices support @@ -1389,31 +1388,32 @@ function enqueue_kernel( end function call( - k::Kernel, args::Vararg{Any, N}; global_size = (1,), local_size = nothing, + k::Kernel, args::Tuple; global_size = (1,), local_size = nothing, global_work_offset = nothing, svm_pointers::Union{Nothing, Vector{Ptr{Cvoid}}} = nothing, rng_state = false - ) where {N} - set_args!(k, args...) + ) + set_args!(k, args) if svm_pointers !== nothing && !isempty(svm_pointers) clSetKernelExecInfo( k, CL_KERNEL_EXEC_INFO_SVM_PTRS, sizeof(svm_pointers), svm_pointers ) end - return enqueue_kernel(k, global_size, local_size; global_work_offset, rng_state, nargs = N) + return enqueue_kernel(k, global_size, local_size; global_work_offset, rng_state, nargs = length(args)) end # convert the argument values to match the kernel's signature (specified by the user) # (this mimics `lower-ccall` in julia-syntax.scm) -@inline @generated function convert_arguments(f::Function, ::Type{tt}, args...) where {tt} +@inline @generated function convert_arguments(f::Function, ::Type{tt}, args::Tuple) where {tt} types = tt.parameters + nargs = fieldcount(args) ex = quote end - converted_args = Vector{Symbol}(undef, length(args)) - arg_ptrs = Vector{Symbol}(undef, length(args)) - for i in 1:length(args) + converted_args = Vector{Symbol}(undef, nargs) + arg_ptrs = Vector{Symbol}(undef, nargs) + for i in 1:nargs converted_args[i] = gensym() arg_ptrs[i] = gensym() push!(ex.args, :($(converted_args[i]) = Base.cconvert($(types[i]), args[$i]))) @@ -1424,7 +1424,7 @@ end ex.args, ( quote GC.@preserve $(converted_args...) begin - f($(arg_ptrs...)) + f(($(arg_ptrs...),)) end end ).args @@ -1433,14 +1433,13 @@ end return ex end -clcall(f::F, types::Tuple, args::Vararg{Any, N}; kwargs...) where {N, F} = - clcall(f, _to_tuple_type(types), args...; kwargs...) +# the arguments are passed as a tuple, see `set_args!` +clcall(f::F, types::Tuple, args::Tuple; kwargs...) where {F} = + clcall(f, _to_tuple_type(types), args; kwargs...) -function clcall(k::Kernel, types::Type{T}, args::Vararg{Any, N}; kwargs...) where {T, N} - call_closure = function (converted_args::Vararg{Any, N}) - return call(k, converted_args...; kwargs...) - end - return convert_arguments(call_closure, types, args...) +function clcall(k::Kernel, types::Type{T}, args::Tuple; kwargs...) where {T} + call_closure = converted_args -> call(k, converted_args; kwargs...) + return convert_arguments(call_closure, types, args) end struct KernelWorkGroupInfo diff --git a/test/codegen_checks.jl b/test/codegen_checks.jl index 59dff1502..6db27c38c 100644 --- a/test/codegen_checks.jl +++ b/test/codegen_checks.jl @@ -175,7 +175,7 @@ end @check "define spir_kernel void @{{.*}}gpu_codegen_global_linear" @check "udiv i32" @device_code_llvm debuginfo = :none KernelAbstractions.launch_kernel( - kernel, KernelAbstractions.LinearLaunch{Int32}(), ndrange, workgroupsize, iterspace, B + kernel, KernelAbstractions.LinearLaunch{Int32}(), ndrange, workgroupsize, iterspace, (B,) ) KernelAbstractions.synchronize(backend) end diff --git a/test/launch.jl b/test/launch.jl index 1b89c2474..ac8e62e82 100644 --- a/test/launch.jl +++ b/test/launch.jl @@ -38,6 +38,13 @@ end @inbounds A[I] = lmem[i] end +# more arguments than Julia splats efficiently (32) +const MANY_ARGS = [Symbol(:x, i) for i in 1:40] +@eval @kernel function launch_many!(A, $(MANY_ARGS...)) + I = @index(Global, Linear) + @inbounds A[I] = $(foldl((a, b) -> :($a + $b), MANY_ARGS)) +end + default_launcher(kernel, args...; ndrange, workgroupsize = nothing) = kernel(args...; ndrange, workgroupsize) @@ -71,7 +78,7 @@ function check_indices(launcher, backend, AT, kernel, ndrange; workgroupsize = n return true end -function launch_testsuite(backend, AT; launcher = default_launcher) +function launch_testsuite(backend, AT; launcher = default_launcher, skip_tests = Set{String}()) @testset "index layout" begin shapes = Tuple[(), (7,), (37,), (5, 7), (33, 3), (3, 5, 7), (2, 3, 4, 5)] @testset "$shape, workgroupsize=$wgs" for shape in shapes, @@ -125,6 +132,14 @@ function launch_testsuite(backend, AT; launcher = default_launcher) end end + # back ends that limit the number of kernel arguments (Metal: 31 buffers) can skip this + @conditional_testset "many arguments" skip_tests begin + A = AT(zeros(Int, 5)) + launcher(launch_many!(backend()), A, 1:40...; ndrange = length(A)) + synchronize(backend()) + @test all(==(sum(1:40)), Array(A)) + end + @testset "synchronize with padding lanes" begin for (shape, wgs) in (((37,), (8,)), ((7, 6), (4, 4)), ((5, 3, 3), (2, 2, 2))) A = AT(zeros(Int, shape)) diff --git a/test/runtests.jl b/test/runtests.jl index ef7cb2b2e..3beb38cbf 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -70,6 +70,34 @@ if "cl_khr_fp16" in POCL.device().extensions end end +# Julia doesn't turn a splat of more than 32 elements into a direct call, so a launch with +# many arguments allocates unless every layer passes them on as a tuple +@testset "POCL launch with many arguments" begin + xs = [Symbol(:x, i) for i in 1:40] + mod = @eval module $(gensym()) + using KernelAbstractions + @kernel function few!(A, x1, x2, x3, x4) + I = @index(Global, Linear) + @inbounds A[I] = x1 + x2 + x3 + x4 + end + @kernel function many!(A, $(xs...)) + I = @index(Global, Linear) + @inbounds A[I] = $(foldl((a, b) -> :($a + $b), xs)) + end + # the arguments are written out, since splatting them here would allocate too + launch_few(k, A) = k(A, $((1:4)...); ndrange = length(A)) + launch_many(k, A) = k(A, $((1:40)...); ndrange = length(A)) + end + + A = zeros(Int, 16) + few = mod.few!(CPU(), 16) + many = mod.many!(CPU(), 16) + mod.launch_few(few, A) + mod.launch_many(many, A) + @test all(==(sum(1:40)), A) + @test @allocated(mod.launch_many(many, A)) <= @allocated(mod.launch_few(few, A)) +end + @testset "POCL compilation cache" begin mod = @eval module $(gensym()) @noinline child() = return @@ -203,7 +231,7 @@ end CartesianIndices((2, 2)), nothing, TransposedMapping() ) A = zeros(Int, 5, 7) - KernelAbstractions.launch_kernel(kernel, launch, CartesianIndices(A), nothing, iterspace, A) + KernelAbstractions.launch_kernel(kernel, launch, CartesianIndices(A), nothing, iterspace, (A,)) @test A == LinearIndices(A) end @testset "custom iteration space, $launch" for launch in (nothing, KA.LinearLaunch{Int}(), KA.NDLaunch{Int}()) @@ -212,7 +240,7 @@ end iterspace = KA.NDRange{2, KA.StaticSize{(2, 2)}, KA.StaticSize{(4, 4)}}(nothing, ItemOffsets((1, 2))) ndrange = CartesianIndices((2:8, 3:7)) A = zeros(Int, 9, 8) - KernelAbstractions.launch_kernel(kernel, launch, ndrange, nothing, iterspace, A) + KernelAbstractions.launch_kernel(kernel, launch, ndrange, nothing, iterspace, (A,)) @test A[ndrange] == LinearIndices(ndrange) A[ndrange] .= 0 @test all(iszero, A) @@ -225,7 +253,7 @@ end ndrange, workgroupsize, iterspace, _ = KA.launch_config(kernel, ndrange, workgroupsize) # an N-d launch is limited to three dimensions l = launch isa KA.NDLaunch && ndims(iterspace) > 3 ? KA.LinearLaunch{Int}() : launch - KernelAbstractions.launch_kernel(kernel, l, ndrange, workgroupsize, iterspace, args...) + KernelAbstractions.launch_kernel(kernel, l, ndrange, workgroupsize, iterspace, args) end @testset "$launch" begin Testsuite.launch_testsuite(CPU, Array; launcher) diff --git a/test/testsuite.jl b/test/testsuite.jl index 9a0aa0048..2e3d2ae10 100644 --- a/test/testsuite.jl +++ b/test/testsuite.jl @@ -82,7 +82,7 @@ function testsuite(backend, backend_str, backend_mod, AT, DAT; skip_tests = Set{ end @conditional_testset "Launch" skip_tests begin - launch_testsuite(backend, AT) + launch_testsuite(backend, AT; skip_tests) end @conditional_testset "copyto!" skip_tests begin