From 06a9c4cc21f7272b1c3c9df335bb64db0b6a6e14 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 30 Sep 2026 16:24:02 +0200 Subject: [PATCH 1/5] PoCL: pass kernel arguments on as tuples Each layer of PoCL's launch path took the kernel arguments as varargs and splatted them into the next: the kernel object's call, `clcall`, `call`, and `set_args!`, which recursed with one splat per argument. Julia doesn't turn a splat of more than 32 elements into a direct call, and a method with both varargs and keyword arguments splats them into its body, so launching a kernel with many arguments went through `Core._apply_iterate` several times. Pass the arguments along as one tuple instead, and generate the per-argument code. The kernel object keeps its call syntax, with its keyword method defined explicitly, and `launch_tuple` takes the tuple directly. `@opencl` builds the argument types without `map`, which isn't type stable for 32 or more elements. --- src/pocl/backend.jl | 2 +- src/pocl/compiler/execution.jl | 33 +++++++++++++++++++-------- src/pocl/nanoOpenCL.jl | 41 +++++++++++++++++----------------- 3 files changed, 45 insertions(+), 31 deletions(-) diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 48a1633c0..132036cbc 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -145,7 +145,7 @@ end function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Vararg{Any, N}) where {N} # POCL launches synchronously, see the implementation note on `synchronize` - event = obj.kern(args...; local_size = items, global_size = groups .* items) + event = POCL.launch_tuple(obj.kern, args; local_size = items, global_size = groups .* items) wait(event) cl.clReleaseEvent(event) return nothing diff --git a/src/pocl/compiler/execution.jl b/src/pocl/compiler/execution.jl index 7e0e20de2..95ce2796d 100644 --- a/src/pocl/compiler/execution.jl +++ b/src/pocl/compiler/execution.jl @@ -52,7 +52,7 @@ macro opencl(ex...) # FIXME: macro hygiene wrt. escaping kwarg values (this broke with 1.5) # we esc() the whole thing now, necessitating gensyms... - @gensym f_var kernel_f kernel_args kernel_tt kernel + @gensym f_var kernel_f kernel_tt kernel # convert the arguments, call the compiler and launch the kernel # while keeping the original arguments alive @@ -62,8 +62,7 @@ macro opencl(ex...) $f_var = $f GC.@preserve $(vars...) $f_var begin $kernel_f = $clconvert($f_var) - $kernel_args = map($clconvert, ($(var_exprs...),)) - $kernel_tt = Tuple{map(Core.Typeof, $kernel_args)...} + $kernel_tt = $argument_types(($(var_exprs...),)) $kernel = $clfunction($kernel_f, $kernel_tt; $(compiler_kwargs...)) if $launch $kernel($(var_exprs...); $(call_kwargs...)) @@ -153,6 +152,13 @@ function clconvert(arg, pointers::Union{Nothing, Vector{Ptr{Cvoid}}} = nothing) return adapt(KernelAdaptor(pointers), arg) end +# `Tuple{map(x -> Core.Typeof(clconvert(x)), args)...}`, without `map`, which isn't type +# stable for 32 or more elements +@inline @generated function argument_types(args::Tuple) + types = (:(Core.Typeof(clconvert(args[$i]))) for i in 1:fieldcount(args)) + return :(Tuple{$(types...)}) +end + ## abstract kernel functionality @@ -160,12 +166,21 @@ abstract type AbstractKernel{F, TT} end pass_arg(@nospecialize dt) = !(GPUCompiler.isghosttype(dt) || Core.Compiler.isconstType(dt)) -@inline @generated function (kernel::AbstractKernel{F, TT})( - args::Vararg{Any, N}; - global_size = (1,), local_size = nothing - ) where {F, TT, N} +# The arguments are passed on as a tuple: Julia doesn't turn a splat of more than 32 +# elements into a direct call, and a method with both varargs and keyword arguments splats +# them into its body. So the keyword method is defined explicitly. +(kernel::AbstractKernel)(args::Vararg{Any, N}) where {N} = launch_tuple(kernel, args) +Core.kwcall(kwargs::NamedTuple, kernel::AbstractKernel, args::Vararg{Any, N}) where {N} = + launch_tuple(kernel, args; kwargs...) + +@inline launch_tuple(kernel::AbstractKernel, args::Tuple; global_size = (1,), local_size = nothing) = + launch_converted(kernel, args, global_size, local_size) + +@inline @generated function launch_converted( + kernel::AbstractKernel{F, TT}, args::Tuple, global_size, local_size + ) where {F, TT} sig = Tuple{F, TT.parameters...} # Base.signature_type with a function type - args = (:(kernel.f), (:(clconvert(args[$i])) for i in 1:length(args))...) + args = (:(kernel.f), (:(clconvert(args[$i])) for i in 1:fieldcount(args))...) # filter out ghost arguments that shouldn't be passed to_pass = map(pass_arg, sig.parameters) @@ -187,7 +202,7 @@ pass_arg(@nospecialize dt) = !(GPUCompiler.isghosttype(dt) || Core.Compiler.isco call_tt = Base.to_tuple_type(call_t) return quote - $cl.clcall(kernel.fun, $call_tt, $(call_args...); global_size, local_size, kernel.rng_state) + $cl.clcall(kernel.fun, $call_tt, ($(call_args...),); global_size, local_size, kernel.rng_state) end end diff --git a/src/pocl/nanoOpenCL.jl b/src/pocl/nanoOpenCL.jl index 34450bbdc..cec2e0deb 100644 --- a/src/pocl/nanoOpenCL.jl +++ b/src/pocl/nanoOpenCL.jl @@ -1293,11 +1293,10 @@ function set_arg!(k::Kernel, idx::Integer, arg::T) where {T} return k end -set_args!(k::Kernel, args::Vararg{Any, N}) where {N} = _set_args!(k, 1, args...) -@inline _set_args!(k::Kernel, i::Int) = nothing -@inline function _set_args!(k::Kernel, i::Int, arg, args::Vararg{Any, N}) where {N} - set_arg!(k, i, arg) - return _set_args!(k, i + 1, args...) +# one call per argument: a splat of more than 32 arguments isn't a direct call +@inline @generated function set_args!(k::Kernel, args::Tuple) + calls = (:(set_arg!(k, $i, args[$i])) for i in 1:fieldcount(args)) + return :($(calls...); nothing) end # work sizes padded to the three dimensions OpenCL devices support @@ -1389,31 +1388,32 @@ function enqueue_kernel( end function call( - k::Kernel, args::Vararg{Any, N}; global_size = (1,), local_size = nothing, + k::Kernel, args::Tuple; global_size = (1,), local_size = nothing, global_work_offset = nothing, svm_pointers::Union{Nothing, Vector{Ptr{Cvoid}}} = nothing, rng_state = false - ) where {N} - set_args!(k, args...) + ) + set_args!(k, args) if svm_pointers !== nothing && !isempty(svm_pointers) clSetKernelExecInfo( k, CL_KERNEL_EXEC_INFO_SVM_PTRS, sizeof(svm_pointers), svm_pointers ) end - return enqueue_kernel(k, global_size, local_size; global_work_offset, rng_state, nargs = N) + return enqueue_kernel(k, global_size, local_size; global_work_offset, rng_state, nargs = length(args)) end # convert the argument values to match the kernel's signature (specified by the user) # (this mimics `lower-ccall` in julia-syntax.scm) -@inline @generated function convert_arguments(f::Function, ::Type{tt}, args...) where {tt} +@inline @generated function convert_arguments(f::Function, ::Type{tt}, args::Tuple) where {tt} types = tt.parameters + nargs = fieldcount(args) ex = quote end - converted_args = Vector{Symbol}(undef, length(args)) - arg_ptrs = Vector{Symbol}(undef, length(args)) - for i in 1:length(args) + converted_args = Vector{Symbol}(undef, nargs) + arg_ptrs = Vector{Symbol}(undef, nargs) + for i in 1:nargs converted_args[i] = gensym() arg_ptrs[i] = gensym() push!(ex.args, :($(converted_args[i]) = Base.cconvert($(types[i]), args[$i]))) @@ -1424,7 +1424,7 @@ end ex.args, ( quote GC.@preserve $(converted_args...) begin - f($(arg_ptrs...)) + f(($(arg_ptrs...),)) end end ).args @@ -1433,14 +1433,13 @@ end return ex end -clcall(f::F, types::Tuple, args::Vararg{Any, N}; kwargs...) where {N, F} = - clcall(f, _to_tuple_type(types), args...; kwargs...) +# the arguments are passed as a tuple, see `set_args!` +clcall(f::F, types::Tuple, args::Tuple; kwargs...) where {F} = + clcall(f, _to_tuple_type(types), args; kwargs...) -function clcall(k::Kernel, types::Type{T}, args::Vararg{Any, N}; kwargs...) where {T, N} - call_closure = function (converted_args::Vararg{Any, N}) - return call(k, converted_args...; kwargs...) - end - return convert_arguments(call_closure, types, args...) +function clcall(k::Kernel, types::Type{T}, args::Tuple; kwargs...) where {T} + call_closure = converted_args -> call(k, converted_args; kwargs...) + return convert_arguments(call_closure, types, args) end struct KernelWorkGroupInfo From 71861ce61a631d45faa2f0158c1caa34caafa4e4 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 30 Sep 2026 16:24:26 +0200 Subject: [PATCH 2/5] KernelInterface 0.4: pass the kernel arguments to `launch` as a tuple The back-end hook was `launch(kernel, groups, items, args...; kwargs...)`. A method with both varargs and keyword arguments splats the arguments into its body, and Julia doesn't turn a splat of more than 32 elements into a direct call, so a launch with many arguments allocated in every back end, whatever the caller did. Avoiding that meant writing a `Core.kwcall` method by hand in each back end. Make the hook take the arguments as one tuple instead: launch(kernel::Kernel{<:NewBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) A back end can then pass the tuple on to its own launcher. This is breaking, so this is KernelInterface 0.4. No back end depends on 0.3 yet. Calling a `Kernel` keeps its syntax. Its keyword method is defined explicitly, so that it can pass the arguments on as a tuple, and `@launch` builds the argument types without `map`, which isn't type stable for 32 or more elements. --- Project.toml | 2 +- docs/src/kernelinterface.md | 13 +++++-- lib/KernelInterface/Project.toml | 2 +- lib/KernelInterface/src/launch.jl | 49 ++++++++++++++++++--------- lib/KernelInterface/test/interface.jl | 2 +- lib/KernelInterface/test/runtests.jl | 28 +++++++++++++-- src/pocl/backend.jl | 2 +- 7 files changed, 73 insertions(+), 25 deletions(-) diff --git a/Project.toml b/Project.toml index 0cf5fc004..b7ee65576 100644 --- a/Project.toml +++ b/Project.toml @@ -42,7 +42,7 @@ Adapt = "0.4, 1.0, 2.0, 3.0, 4" Atomix = "1.2.1" EnzymeCore = "0.7, 0.8.1" GPUCompiler = "2.7" -KernelInterface = "0.3" +KernelInterface = "0.4" LLVM = "9.9" LinearAlgebra = "1.6" MacroTools = "0.5" diff --git a/docs/src/kernelinterface.md b/docs/src/kernelinterface.md index 1f6fa0f5e..0db16a339 100644 --- a/docs/src/kernelinterface.md +++ b/docs/src/kernelinterface.md @@ -270,10 +270,17 @@ optional methods where it can do better than the fallback. In particular: `Adapt.adapt_storage(::NewBackend, x) = adapt(NewArray, x)`. 3. Implement [`kernel_function`](@ref), returning a [`Kernel`](@ref) that holds the backend value it was given, and [`launch`](@ref), which receives an already validated - `NTuple{3, Int}` of work-groups and of work-items. For CUDA.jl, the latter is + `NTuple{3, Int}` of work-groups and of work-items, and the arguments as a tuple. Pass + that tuple on to the native launcher rather than splatting it: Julia doesn't turn a + splat of more than 32 elements into a direct call, so kernels with many arguments would + be slow to launch. For the PoCL backend, `launch` is ```julia - KI.launch(k::KI.Kernel{CUDABackend}, groups::Dims{3}, items::Dims{3}, args::Vararg{Any, N}; kwargs...) where {N} = - k.kern(args...; threads = items, blocks = groups, kwargs...) + function KI.launch(k::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) + event = POCL.launch_tuple(k.kern, args; local_size = items, global_size = groups .* items) + wait(event) + cl.clReleaseEvent(event) + return nothing + end ``` 4. Compute the typed index queries with `% T`, not `T(x)`: a checked conversion leaves an error branch in every kernel. diff --git a/lib/KernelInterface/Project.toml b/lib/KernelInterface/Project.toml index e34a91b0e..6fc175c32 100644 --- a/lib/KernelInterface/Project.toml +++ b/lib/KernelInterface/Project.toml @@ -1,7 +1,7 @@ name = "KernelInterface" uuid = "4ee993da-d684-4d17-a7dd-4e58e78d92bf" authors = ["Valentin Churavy and contributors"] -version = "0.3.0" +version = "0.4.0-dev" [compat] julia = "1.10" diff --git a/lib/KernelInterface/src/launch.jl b/lib/KernelInterface/src/launch.jl index 97809596f..490ce23ea 100644 --- a/lib/KernelInterface/src/launch.jl +++ b/lib/KernelInterface/src/launch.jl @@ -48,23 +48,30 @@ struct Kernel{B, Kern} kern::Kern end -# `Vararg{Any, N}` makes Julia specialize on the arguments, which are only passed through -function (kernel::Kernel)( - args::Vararg{Any, N}; numgroups = (), workgroupsize = (), ndrange = (), +# The arguments are passed on as a tuple: Julia doesn't turn a splat of more than 32 +# elements into a direct call, and a method with both varargs and keyword arguments splats +# them into its body. So the keyword method is defined explicitly, as Base does for +# `invokelatest`. `Vararg{Any, N}` makes Julia specialize on the arguments. +(kernel::Kernel)(args::Vararg{Any, N}) where {N} = call_kernel(kernel, args) +Core.kwcall(kwargs::NamedTuple, kernel::Kernel, args::Vararg{Any, N}) where {N} = + call_kernel(kernel, args; kwargs...) + +function call_kernel( + kernel::Kernel, args::Tuple; numgroups = (), workgroupsize = (), ndrange = (), max_work_group_size::Integer = typemax(Int), kwargs... - ) where {N} + ) groups, items = launch_geometry(kernel, numgroups, workgroupsize, ndrange, max_work_group_size) any(iszero, groups) && return nothing - launch(kernel, groups, items, args...; kwargs...) + launch(kernel, groups, items, args; kwargs...) return nothing end """ - launch(kernel::Kernel, groups::Dims{3}, items::Dims{3}, args...; kwargs...) + launch(kernel::Kernel, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) Launch `kernel` with `groups` work-groups of `items` work-items each, passing the host-side -arguments `args`. This is what calling a [`Kernel`](@ref) does after validating and -normalizing the launch geometry; users call the kernel instead. +arguments `args`, a tuple. This is what calling a [`Kernel`](@ref) does after validating +and normalizing the launch geometry; users call the kernel instead. `groups` and `items` are positive, `items` fits [`max_work_group_dims`](@ref) and [`max_work_group_size`](@ref)`(kernel)`, and `groups .* items` doesn't overflow `Int`. @@ -73,15 +80,19 @@ normalizing the launch geometry; users call the kernel instead. !!! note Backend implementations **must** implement: ``` - launch(kernel::Kernel{<:NewBackend}, groups::Dims{3}, items::Dims{3}, args...; kwargs...) + launch(kernel::Kernel{<:NewBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) ``` It converts `args` with [`argconvert`](@ref) (or lets its native launcher do so), and queues the launch on the calling task's queue; it doesn't have to wait for the kernel to - complete. Declare the arguments as `args::Vararg{Any, N}` (with `where {N}`): Julia - doesn't specialize a method on `args...` that it only passes through, which makes every - launch dispatch dynamically. It must throw for keywords it does not support, and may - throw for a number of work-groups the device cannot launch, or for a geometry that + complete. To keep launches with many arguments cheap, it should pass `args` on as a + tuple rather than splatting it: Julia doesn't turn a splat of more than 32 elements into + a direct call. It must throw for keywords it does not support, and may throw for a + number of work-groups the device cannot launch, or for a geometry that backend-specific compiler options of the kernel don't allow. + +!!! compat "KernelInterface 0.4" + Before KernelInterface 0.4, `launch` received the arguments as varargs, + `launch(kernel, groups, items, args...; kwargs...)`. """ function launch end @@ -378,6 +389,13 @@ The returned kernel doesn't keep any arguments alive: they are passed again at l """ function kernel_function end +# `Tuple{map(x -> Core.Typeof(argconvert(backend, x)), args)...}`, without `map`, which +# isn't type stable for 32 or more elements +@inline @generated function argument_types(backend, args::Tuple) + types = (:(Core.Typeof(argconvert(backend, args[$i]))) for i in 1:fieldcount(args)) + return :(Tuple{$(types...)}) +end + const MACRO_KWARGS = [:launch] const LAUNCH_KWARGS = [:numgroups, :workgroupsize, :ndrange, :max_work_group_size] @@ -452,7 +470,7 @@ macro launch(backend, ex...) # FIXME: macro hygiene wrt. escaping kwarg values (this broke with 1.5) # we esc() the whole thing now, necessitating gensyms... - @gensym backend_var f_var kernel_f kernel_args kernel_tt kernel + @gensym backend_var f_var kernel_f kernel_tt kernel # convert the arguments, call the compiler and launch the kernel # while keeping the original arguments alive @@ -463,8 +481,7 @@ macro launch(backend, ex...) $f_var = $f GC.@preserve $(vars...) $f_var begin $kernel_f = $argconvert($backend_var, $f_var) - $kernel_args = Base.map(x -> $argconvert($backend_var, x), ($(var_exprs...),)) - $kernel_tt = Tuple{Base.map(Core.Typeof, $kernel_args)...} + $kernel_tt = $argument_types($backend_var, ($(var_exprs...),)) $kernel = $kernel_function($backend_var, $kernel_f, $kernel_tt; $(compiler_kwargs...)) if $launch $kernel($(var_exprs...); $(call_kwargs...)) diff --git a/lib/KernelInterface/test/interface.jl b/lib/KernelInterface/test/interface.jl index 3cd9a11e6..76aaadd03 100644 --- a/lib/KernelInterface/test/interface.jl +++ b/lib/KernelInterface/test/interface.jl @@ -693,7 +693,7 @@ function contract_testsuite(backend::KI.Backend, AT) @test hasmethod(KI.copyto!, Tuple{B, AT, Array}) @test hasmethod(KI.argconvert, Tuple{B, Any}) @test hasmethod(KI.kernel_function, Tuple{B, Any, Type}) - @test hasmethod(KI.launch, Tuple{KI.Kernel{B}, Dims{3}, Dims{3}}) + @test hasmethod(KI.launch, Tuple{KI.Kernel{B}, Dims{3}, Dims{3}, Tuple}) @test hasmethod(KI.max_work_group_size, Tuple{B}) @test hasmethod(KI.max_work_group_size, Tuple{KI.Kernel{B}}) @test hasmethod(KI.max_work_group_dims, Tuple{B}) diff --git a/lib/KernelInterface/test/runtests.jl b/lib/KernelInterface/test/runtests.jl index 956417597..15e9d4ad8 100644 --- a/lib/KernelInterface/test/runtests.jl +++ b/lib/KernelInterface/test/runtests.jl @@ -255,7 +255,7 @@ KI.argconvert(::MockBackend, arg) = arg function KI.kernel_function(backend::MockBackend, f, tt = Tuple{}; name = nothing, kwargs...) return KI.Kernel(backend, MockKernel(f, tt, name, Dict(kwargs), [])) end -function KI.launch(kernel::KI.Kernel{MockBackend}, groups::Dims{3}, items::Dims{3}, args...; kwargs...) +function KI.launch(kernel::KI.Kernel{MockBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) push!(kernel.kern.launches, (; groups, items, args, kwargs = Dict(kwargs))) return :ignored end @@ -276,9 +276,15 @@ function KI.launch_configuration( push!(kernel.backend.queries, (; nitems, max_work_group_size)) return (; workgroupsize = min(96, max_work_group_size)) end -KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, args...) = +KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) = push!(kernel.kern, (groups, items)) +# ... and one that does nothing, to measure the overhead of launching +struct NullBackend <: KI.Backend end +KI.max_work_group_size(::KI.Kernel{NullBackend}) = 1024 +KI.max_work_group_dims(::NullBackend) = (1024, 1024, 64) +KI.launch(::KI.Kernel{NullBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple; kwargs...) = nothing + @testset "launch geometry" begin kernel = KI.kernel_function(MockBackend(), identity, Tuple{Int}) function geometry(; kwargs...) @@ -345,6 +351,13 @@ KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, kernel(1; ndrange = 4, stream = :mine) @test last(kernel.kern.launches).kwargs == Dict(:stream => :mine) + # The arguments reach the backend as one tuple, whatever their number, and a single + # tuple-valued argument stays one argument. + kernel((1, 2); ndrange = 4) + @test last(kernel.kern.launches).args == ((1, 2),) + kernel(ntuple(identity, 40)...; ndrange = 4) + @test last(kernel.kern.launches).args == ntuple(identity, 40) + # Auto-sizing uses the backend's recommendation, not the limit, and tells it both the # size of the launch and the cap. occupancy = KI.Kernel(OccupancyBackend(), []) @@ -411,6 +424,17 @@ function counted_backend() return MockBackend() end +# Julia doesn't turn a splat of more than 32 elements into a direct call, so launching with +# many arguments allocates unless they're passed on as a tuple +@testset "many arguments" begin + kernel = KI.Kernel(NullBackend(), nothing) + @eval launch_few(k) = k($((1:4)...); numgroups = 2, workgroupsize = 4) + @eval launch_many(k) = k($((1:40)...); numgroups = 2, workgroupsize = 4) + launch_few(kernel) + launch_many(kernel) + @test @allocated(launch_many(kernel)) <= @allocated(launch_few(kernel)) +end + @testset "@launch" begin backend = MockBackend() diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 132036cbc..8df99604c 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -143,7 +143,7 @@ function KI.kernel_function(backend::POCLBackend, f::F, tt::TT = Tuple{}; name = return KI.Kernel{POCLBackend, typeof(kern)}(backend, kern) end -function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Vararg{Any, N}) where {N} +function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) # POCL launches synchronously, see the implementation note on `synchronize` event = POCL.launch_tuple(obj.kern, args; local_size = items, global_size = groups .* items) wait(event) From 4543f43f23f70dd7790134f8250c5c6d479940a8 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 30 Sep 2026 16:24:40 +0200 Subject: [PATCH 3/5] Launch @kernel kernels with many arguments without splatting The generic launch splatted the kernel arguments from the `Kernel` call down to the `KI.Kernel` call, and mapped over them to compute the argument types. With more than 32 arguments neither is a direct call, so on PoCL a kernel with 40 arguments allocated 25.7 KB per launch, against 32 bytes with 4 arguments. Pass them along as a tuple, as `KI.Kernel` now does, and generate the code that computes the argument types and calls the `KI.Kernel`. A 40-argument launch now allocates as much as a 4-argument one. CUDA.jl#3309 did the same for CUDA's own launch of KA kernels, which the generic launch replaces. --- src/backend_launch.jl | 43 +++++++++++++++++++++++++++++++----------- test/codegen_checks.jl | 2 +- test/launch.jl | 17 ++++++++++++++++- test/runtests.jl | 34 ++++++++++++++++++++++++++++++--- test/testsuite.jl | 2 +- 5 files changed, 81 insertions(+), 17 deletions(-) diff --git a/src/backend_launch.jl b/src/backend_launch.jl index 38b2c7e90..791443200 100644 --- a/src/backend_launch.jl +++ b/src/backend_launch.jl @@ -74,7 +74,15 @@ end argconvert(kernel::Kernel{<:KI.Backend}, arg) = KI.argconvert(backend(kernel), arg) -function (obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}; ndrange = nothing, workgroupsize = nothing) where {N} +# The arguments are passed on as a tuple: Julia doesn't turn a splat of more than 32 +# elements into a direct call, and a method with both varargs and keyword arguments splats +# them into its body. So the keyword method is defined explicitly, as Base does for +# `invokelatest`. +(obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}) where {N} = launch_tuple(obj, args) +Core.kwcall(kwargs::NamedTuple, obj::Kernel{<:KI.Backend}, args::Vararg{Any, N}) where {N} = + launch_tuple(obj, args; kwargs...) + +function launch_tuple(obj::Kernel, args::Tuple; ndrange = nothing, workgroupsize = nothing) ndrange, workgroupsize, iterspace, dynamic = launch_config(obj, ndrange, workgroupsize) # nothing to launch (or compile) for an empty ndrange any(iszero, size(blocks(iterspace))) && return nothing @@ -84,21 +92,19 @@ function (obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}; ndrange = nothing, wo launch = select_launch(obj, workgroupsize, iterspace) if launch === NDLaunch{Int32}() # the common case, specialized statically - launch_kernel(obj, NDLaunch{Int32}(), ndrange, workgroupsize, iterspace, args...) + launch_kernel(obj, NDLaunch{Int32}(), ndrange, workgroupsize, iterspace, args) else - launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args...) + launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args) end return nothing end -function launch_kernel( - obj::Kernel, launch, ndrange, _workgroupsize, iterspace, args::Vararg{Any, N} - ) where {N} +function launch_kernel(obj::Kernel, launch, ndrange, _workgroupsize, iterspace, args::Tuple) b = backend(obj) # this might not be the final context, since we may tune the workgroupsize ctx = mkcontext(obj, ndrange, iterspace, launch) - kernel = compile(obj, ctx, args...) + kernel = compile(obj, ctx, args) # tune the workgroup size, keeping the context type (and thus the kernel) the same if workgroupsize(obj) <: DynamicSize && _workgroupsize === nothing @@ -112,16 +118,31 @@ function launch_kernel( groups = size(blocks(iterspace)) items = size(workitems(iterspace)) if launch isa NDLaunch - kernel(ctx, args...; numgroups = groups, workgroupsize = items) + call_kernel(kernel, ctx, args, groups, items) else - kernel(ctx, args...; numgroups = prod(groups), workgroupsize = prod(items)) + call_kernel(kernel, ctx, args, prod(groups), prod(items)) end return nothing end -@inline function compile(obj::Kernel, ctx, args::Vararg{Any, N}) where {N} +@inline function compile(obj::Kernel, ctx, args::Tuple) b = backend(obj) f = KI.argconvert(b, obj.f) - tt = Tuple{Core.Typeof(KI.argconvert(b, ctx)), map(arg -> Core.Typeof(KI.argconvert(b, arg)), args)...} + tt = argument_types(b, ctx, args) return KI.kernel_function(b, f, tt; compiler_options(obj)...) end + +# The helpers below avoid splatting the arguments, and `map`, which isn't type stable for 32 +# or more elements. + +# `Tuple{map(x -> Core.Typeof(KI.argconvert(backend, x)), (ctx, args...))...}` +@inline @generated function argument_types(backend, ctx, args::Tuple) + types = (:(Core.Typeof(KI.argconvert(backend, args[$i]))) for i in 1:fieldcount(args)) + return :(Tuple{Core.Typeof(KI.argconvert(backend, ctx)), $(types...)}) +end + +# `kernel(ctx, args...; numgroups, workgroupsize)` +@inline @generated function call_kernel(kernel::KI.Kernel, ctx, args::Tuple, numgroups, workgroupsize) + argexprs = (:(args[$i]) for i in 1:fieldcount(args)) + return :(kernel(ctx, $(argexprs...); numgroups, workgroupsize)) +end diff --git a/test/codegen_checks.jl b/test/codegen_checks.jl index 59dff1502..6db27c38c 100644 --- a/test/codegen_checks.jl +++ b/test/codegen_checks.jl @@ -175,7 +175,7 @@ end @check "define spir_kernel void @{{.*}}gpu_codegen_global_linear" @check "udiv i32" @device_code_llvm debuginfo = :none KernelAbstractions.launch_kernel( - kernel, KernelAbstractions.LinearLaunch{Int32}(), ndrange, workgroupsize, iterspace, B + kernel, KernelAbstractions.LinearLaunch{Int32}(), ndrange, workgroupsize, iterspace, (B,) ) KernelAbstractions.synchronize(backend) end diff --git a/test/launch.jl b/test/launch.jl index 1b89c2474..ac8e62e82 100644 --- a/test/launch.jl +++ b/test/launch.jl @@ -38,6 +38,13 @@ end @inbounds A[I] = lmem[i] end +# more arguments than Julia splats efficiently (32) +const MANY_ARGS = [Symbol(:x, i) for i in 1:40] +@eval @kernel function launch_many!(A, $(MANY_ARGS...)) + I = @index(Global, Linear) + @inbounds A[I] = $(foldl((a, b) -> :($a + $b), MANY_ARGS)) +end + default_launcher(kernel, args...; ndrange, workgroupsize = nothing) = kernel(args...; ndrange, workgroupsize) @@ -71,7 +78,7 @@ function check_indices(launcher, backend, AT, kernel, ndrange; workgroupsize = n return true end -function launch_testsuite(backend, AT; launcher = default_launcher) +function launch_testsuite(backend, AT; launcher = default_launcher, skip_tests = Set{String}()) @testset "index layout" begin shapes = Tuple[(), (7,), (37,), (5, 7), (33, 3), (3, 5, 7), (2, 3, 4, 5)] @testset "$shape, workgroupsize=$wgs" for shape in shapes, @@ -125,6 +132,14 @@ function launch_testsuite(backend, AT; launcher = default_launcher) end end + # back ends that limit the number of kernel arguments (Metal: 31 buffers) can skip this + @conditional_testset "many arguments" skip_tests begin + A = AT(zeros(Int, 5)) + launcher(launch_many!(backend()), A, 1:40...; ndrange = length(A)) + synchronize(backend()) + @test all(==(sum(1:40)), Array(A)) + end + @testset "synchronize with padding lanes" begin for (shape, wgs) in (((37,), (8,)), ((7, 6), (4, 4)), ((5, 3, 3), (2, 2, 2))) A = AT(zeros(Int, shape)) diff --git a/test/runtests.jl b/test/runtests.jl index ef7cb2b2e..3beb38cbf 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -70,6 +70,34 @@ if "cl_khr_fp16" in POCL.device().extensions end end +# Julia doesn't turn a splat of more than 32 elements into a direct call, so a launch with +# many arguments allocates unless every layer passes them on as a tuple +@testset "POCL launch with many arguments" begin + xs = [Symbol(:x, i) for i in 1:40] + mod = @eval module $(gensym()) + using KernelAbstractions + @kernel function few!(A, x1, x2, x3, x4) + I = @index(Global, Linear) + @inbounds A[I] = x1 + x2 + x3 + x4 + end + @kernel function many!(A, $(xs...)) + I = @index(Global, Linear) + @inbounds A[I] = $(foldl((a, b) -> :($a + $b), xs)) + end + # the arguments are written out, since splatting them here would allocate too + launch_few(k, A) = k(A, $((1:4)...); ndrange = length(A)) + launch_many(k, A) = k(A, $((1:40)...); ndrange = length(A)) + end + + A = zeros(Int, 16) + few = mod.few!(CPU(), 16) + many = mod.many!(CPU(), 16) + mod.launch_few(few, A) + mod.launch_many(many, A) + @test all(==(sum(1:40)), A) + @test @allocated(mod.launch_many(many, A)) <= @allocated(mod.launch_few(few, A)) +end + @testset "POCL compilation cache" begin mod = @eval module $(gensym()) @noinline child() = return @@ -203,7 +231,7 @@ end CartesianIndices((2, 2)), nothing, TransposedMapping() ) A = zeros(Int, 5, 7) - KernelAbstractions.launch_kernel(kernel, launch, CartesianIndices(A), nothing, iterspace, A) + KernelAbstractions.launch_kernel(kernel, launch, CartesianIndices(A), nothing, iterspace, (A,)) @test A == LinearIndices(A) end @testset "custom iteration space, $launch" for launch in (nothing, KA.LinearLaunch{Int}(), KA.NDLaunch{Int}()) @@ -212,7 +240,7 @@ end iterspace = KA.NDRange{2, KA.StaticSize{(2, 2)}, KA.StaticSize{(4, 4)}}(nothing, ItemOffsets((1, 2))) ndrange = CartesianIndices((2:8, 3:7)) A = zeros(Int, 9, 8) - KernelAbstractions.launch_kernel(kernel, launch, ndrange, nothing, iterspace, A) + KernelAbstractions.launch_kernel(kernel, launch, ndrange, nothing, iterspace, (A,)) @test A[ndrange] == LinearIndices(ndrange) A[ndrange] .= 0 @test all(iszero, A) @@ -225,7 +253,7 @@ end ndrange, workgroupsize, iterspace, _ = KA.launch_config(kernel, ndrange, workgroupsize) # an N-d launch is limited to three dimensions l = launch isa KA.NDLaunch && ndims(iterspace) > 3 ? KA.LinearLaunch{Int}() : launch - KernelAbstractions.launch_kernel(kernel, l, ndrange, workgroupsize, iterspace, args...) + KernelAbstractions.launch_kernel(kernel, l, ndrange, workgroupsize, iterspace, args) end @testset "$launch" begin Testsuite.launch_testsuite(CPU, Array; launcher) diff --git a/test/testsuite.jl b/test/testsuite.jl index 9a0aa0048..2e3d2ae10 100644 --- a/test/testsuite.jl +++ b/test/testsuite.jl @@ -82,7 +82,7 @@ function testsuite(backend, backend_str, backend_mod, AT, DAT; skip_tests = Set{ end @conditional_testset "Launch" skip_tests begin - launch_testsuite(backend, AT) + launch_testsuite(backend, AT; skip_tests) end @conditional_testset "copyto!" skip_tests begin From c26da5a962221551ec96a0624a278b1266f8dd28 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 30 Sep 2026 16:42:06 +0200 Subject: [PATCH 4/5] PoCL: keep kernel arguments alive while launching Converting an `Array` argument for PoCL yields a device array that only holds a pointer, and nothing kept the original arguments alive after they were converted. The garbage collector could therefore free an array while its remaining arguments were converted, or while the kernel ran. A launch through KernelAbstractions happens to keep them rooted, but calling a PoCL kernel directly doesn't: with a forced collection during argument conversion, the array passed to the kernel was finalized before the launch was queued. Preserve the arguments while converting them and queuing the launch, and in `KI.launch`, which waits for the kernel, until it completes. The latter matters once waiting no longer blocks the garbage collector. --- src/pocl/backend.jl | 9 ++++++--- src/pocl/compiler/execution.jl | 5 ++++- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 8df99604c..31f045e1b 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -144,9 +144,12 @@ function KI.kernel_function(backend::POCLBackend, f::F, tt::TT = Tuple{}; name = end function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) - # POCL launches synchronously, see the implementation note on `synchronize` - event = POCL.launch_tuple(obj.kern, args; local_size = items, global_size = groups .* items) - wait(event) + # the kernel only gets pointers to the arrays in `args`, so keep them alive until it + # completes. POCL launches synchronously, see the implementation note on `synchronize` + event = GC.@preserve args begin + event = POCL.launch_tuple(obj.kern, args; local_size = items, global_size = groups .* items) + wait(event) + end cl.clReleaseEvent(event) return nothing end diff --git a/src/pocl/compiler/execution.jl b/src/pocl/compiler/execution.jl index 95ce2796d..897debaf8 100644 --- a/src/pocl/compiler/execution.jl +++ b/src/pocl/compiler/execution.jl @@ -201,8 +201,11 @@ Core.kwcall(kwargs::NamedTuple, kernel::AbstractKernel, args::Vararg{Any, N}) wh # finalize types call_tt = Base.to_tuple_type(call_t) + # the converted arguments only hold pointers to the arrays in `args` return quote - $cl.clcall(kernel.fun, $call_tt, ($(call_args...),); global_size, local_size, kernel.rng_state) + GC.@preserve args begin + $cl.clcall(kernel.fun, $call_tt, ($(call_args...),); global_size, local_size, kernel.rng_state) + end end end From b2a4ee1c1f74277fb92a0e4ccb750881b9377987 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 30 Sep 2026 19:21:46 +0200 Subject: [PATCH 5/5] KernelInterface: compile kernels from the unconverted callable `KI.@launch` and KernelAbstractions' generic launch converted the callable with `argconvert` before passing it to `kernel_function`. For a closure, the converted callable only holds pointers to the arrays it captures, so nothing kept those arrays alive: a kernel compiled with `launch=false` could read freed memory once the garbage collector had run. A back end also couldn't see the captured arrays at launch, which Metal needs to declare them to the command encoder, as `@metal` has done since Metal.jl#992. `kernel_function` now receives the original callable and converts it itself, and the kernel keeps the original alive. PoCL's kernels keep it next to the compiled kernel, and root it while launching. KernelInterface's testsuite checks that a kernel whose closure captures an array still works after a collection. --- docs/src/kernelinterface.md | 15 ++++++++++----- lib/KernelInterface/src/launch.jl | 24 +++++++++++++++++------- lib/KernelInterface/test/interface.jl | 20 ++++++++++++++++++++ lib/KernelInterface/test/runtests.jl | 8 ++++++++ src/backend_launch.jl | 3 +-- src/pocl/backend.jl | 26 ++++++++++++++++++-------- 6 files changed, 74 insertions(+), 22 deletions(-) diff --git a/docs/src/kernelinterface.md b/docs/src/kernelinterface.md index 0db16a339..ca72776c2 100644 --- a/docs/src/kernelinterface.md +++ b/docs/src/kernelinterface.md @@ -268,16 +268,21 @@ optional methods where it can do better than the fallback. In particular: [`adapt(backend, x)`](@ref Adapt.adapt_storage(::Backend, ::Any)) moves data to the backend, preferably by delegating to its array type: `Adapt.adapt_storage(::NewBackend, x) = adapt(NewArray, x)`. -3. Implement [`kernel_function`](@ref), returning a [`Kernel`](@ref) that holds the - backend value it was given, and [`launch`](@ref), which receives an already validated +3. Implement [`kernel_function`](@ref), which receives the unconverted callable, and + returns a [`Kernel`](@ref) that holds the backend value it was given and keeps that + callable alive. Also implement [`launch`](@ref), which receives an already validated `NTuple{3, Int}` of work-groups and of work-items, and the arguments as a tuple. Pass that tuple on to the native launcher rather than splatting it: Julia doesn't turn a splat of more than 32 elements into a direct call, so kernels with many arguments would - be slow to launch. For the PoCL backend, `launch` is + be slow to launch. For the PoCL backend, whose kernels hold the compiled kernel and the + callable, `launch` is ```julia function KI.launch(k::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) - event = POCL.launch_tuple(k.kern, args; local_size = items, global_size = groups .* items) - wait(event) + f = k.kern.f + event = GC.@preserve f args begin + event = POCL.launch_tuple(k.kern.kernel, args; local_size = items, global_size = groups .* items) + wait(event) + end cl.clReleaseEvent(event) return nothing end diff --git a/lib/KernelInterface/src/launch.jl b/lib/KernelInterface/src/launch.jl index 490ce23ea..99d181b81 100644 --- a/lib/KernelInterface/src/launch.jl +++ b/lib/KernelInterface/src/launch.jl @@ -362,23 +362,34 @@ function argconvert end """ kernel_function(backend, f::F, tt::TT=Tuple{}; name=nothing, kwargs...)::Kernel -Compile the function `f` for arguments of the (device-side) types `tt`, for the active +Compile the callable `f` for arguments of the (device-side) types `tt`, for the active device of `backend`, returning a [`Kernel`](@ref). For a higher-level interface, use [`KernelInterface.@launch`](@ref). +`f` is the host-side callable, not converted with [`argconvert`](@ref): the backend converts +it. For a closure, that matters: it can capture arrays, which its converted form only holds +pointers to. + Keyword arguments: - `name`: override the name that the kernel will have in the generated code. Other keyword arguments are backend-specific compiler options (e.g. `maxthreads` for CUDA.jl); backends throw an error for options they don't support. -The returned kernel doesn't keep any arguments alive: they are passed again at launch. +The returned kernel keeps `f` alive, but not the arguments: they are passed again at +launch. !!! note Backend implementations **must** implement: ``` kernel_function(backend::NewBackend, f::F, tt::TT=Tuple{}; name=nothing, kwargs...) where {F,TT} ``` + It converts `f` with [`argconvert`](@ref) to compile it, and the returned `Kernel` has + to keep `f` itself alive for as long as it can be launched, since the converted `f` + may only hold pointers to the arrays `f` captures. A backend that needs to know about + those arrays at launch, e.g. to declare them to the device, can convert `f` again for + every launch, as it does for the arguments. + The returned `Kernel` stores `backend` itself (not a new default backend), so that options it carries apply to the launch. Kernels must execute with sub-group width [`sub_group_size(backend)`](@ref sub_group_size) if the backend supports sub-groups. @@ -404,8 +415,8 @@ const LAUNCH_KWARGS = [:numgroups, :workgroupsize, :ndrange, :max_work_group_siz Compile `f(args...)` for `backend` and launch it, like `@cuda` or `@metal` do. -`f` and the arguments are converted with [`argconvert`](@ref) and compiled with -[`kernel_function`](@ref), and the resulting [`Kernel`](@ref) is called with the launch +`f` is compiled with [`kernel_function`](@ref) for the types of the arguments converted +with [`argconvert`](@ref), and the resulting [`Kernel`](@ref) is called with the launch keywords `numgroups`, `workgroupsize`, `ndrange` and `max_work_group_size`, whose meaning is documented there. The arguments are kept alive while the launch is being queued. @@ -470,7 +481,7 @@ macro launch(backend, ex...) # FIXME: macro hygiene wrt. escaping kwarg values (this broke with 1.5) # we esc() the whole thing now, necessitating gensyms... - @gensym backend_var f_var kernel_f kernel_tt kernel + @gensym backend_var f_var kernel_tt kernel # convert the arguments, call the compiler and launch the kernel # while keeping the original arguments alive @@ -480,9 +491,8 @@ macro launch(backend, ex...) $backend_var = $backend $f_var = $f GC.@preserve $(vars...) $f_var begin - $kernel_f = $argconvert($backend_var, $f_var) $kernel_tt = $argument_types($backend_var, ($(var_exprs...),)) - $kernel = $kernel_function($backend_var, $kernel_f, $kernel_tt; $(compiler_kwargs...)) + $kernel = $kernel_function($backend_var, $f_var, $kernel_tt; $(compiler_kwargs...)) if $launch $kernel($(var_exprs...); $(call_kwargs...)) end diff --git a/lib/KernelInterface/test/interface.jl b/lib/KernelInterface/test/interface.jl index 76aaadd03..2e5366689 100644 --- a/lib/KernelInterface/test/interface.jl +++ b/lib/KernelInterface/test/interface.jl @@ -238,6 +238,13 @@ function sub_group_barrier_kernel(scratch, out, ::Val{N}) where {N} return end +# a kernel whose callable captures an array, compiled but not launched yet +function captured_array_kernel(backend, AT, out) + a = AT(Int32[42]) + kernel = KI.@launch backend launch = false (() -> (@inbounds out[1] = a[1]; nothing))() + return kernel, WeakRef(a) +end + function interface_testsuite(backend::KI.Backend, AT) @testset "Launch parameters" begin # unequal group counts and sizes in every dimension, so that confusing them shows @@ -527,6 +534,19 @@ function interface_testsuite(backend::KI.Backend, AT) end end + # The converted callable only holds pointers to the arrays it captures, so the kernel + # has to keep the original alive (and a backend may need it at launch). + @testset "Captured arrays" begin + out = AT(Int32[0]) + kernel, captured = captured_array_kernel(backend, AT, out) + GC.gc(true) + @test captured.value !== nothing + garbage = [AT(fill(Int32(7), 1)) for _ in 1:100] + kernel() + KI.synchronize(backend) + @test Array(out) == Int32[42] + end + @testset "Local memory and barriers" begin N = 32 groups = 3 diff --git a/lib/KernelInterface/test/runtests.jl b/lib/KernelInterface/test/runtests.jl index 15e9d4ad8..a8e5b8def 100644 --- a/lib/KernelInterface/test/runtests.jl +++ b/lib/KernelInterface/test/runtests.jl @@ -279,6 +279,11 @@ end KI.launch(kernel::KI.Kernel{OccupancyBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) = push!(kernel.kern, (groups, items)) +# a callable that KernelInterface mustn't convert: the backend does +struct HostCallable end +(::HostCallable)(x) = nothing +KI.argconvert(::MockBackend, ::HostCallable) = error("only the backend should convert the callable") + # ... and one that does nothing, to measure the overhead of launching struct NullBackend <: KI.Backend end KI.max_work_group_size(::KI.Kernel{NullBackend}) = 1024 @@ -463,6 +468,9 @@ end optioned = KI.@launch backend ndrange = 4 maxthreads = 32 dummy(1, 2.0) @test isempty(only(optioned.kern.launches).kwargs) + # The callable is compiled unconverted. + @test (KI.@launch backend launch = false HostCallable()(1)).kern.f isa HostCallable + # Splatted arguments are supported. splatted = KI.@launch backend launch = false dummy((1, 2.0)...) @test splatted.kern.tt == Tuple{Int, Float64} diff --git a/src/backend_launch.jl b/src/backend_launch.jl index 791443200..fbd5cfd8d 100644 --- a/src/backend_launch.jl +++ b/src/backend_launch.jl @@ -127,9 +127,8 @@ end @inline function compile(obj::Kernel, ctx, args::Tuple) b = backend(obj) - f = KI.argconvert(b, obj.f) tt = argument_types(b, ctx, args) - return KI.kernel_function(b, f, tt; compiler_options(obj)...) + return KI.kernel_function(b, obj.f, tt; compiler_options(obj)...) end # The helpers below avoid splatting the arguments, and `map`, which isn't type stable for 32 diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 31f045e1b..b2f2c6603 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -132,22 +132,32 @@ KI.supports_atomics(::POCLBackend) = true KI.argconvert(::POCLBackend, arg) = clconvert(arg) +# a compiled kernel, and the callable it was compiled from. the compiled kernel only holds +# pointers to the arrays the callable captures, so the callable has to be kept alive. +struct POCLKernel{K, F} + kernel::K + f::F +end + function KI.kernel_function(backend::POCLBackend, f::F, tt::TT = Tuple{}; name = nothing, kwargs...) where {F, TT} # fix the sub-group width, as `KI.sub_group_size` promises sub_group_size = device_limits().sub_group_size - kern = if sub_group_size > 0 - clfunction(f, tt; name, sub_group_size, kwargs...) + kernel = if sub_group_size > 0 + clfunction(clconvert(f), tt; name, sub_group_size, kwargs...) else - clfunction(f, tt; name, kwargs...) + clfunction(clconvert(f), tt; name, kwargs...) end + kern = POCLKernel(kernel, f) return KI.Kernel{POCLBackend, typeof(kern)}(backend, kern) end function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, args::Tuple) - # the kernel only gets pointers to the arrays in `args`, so keep them alive until it - # completes. POCL launches synchronously, see the implementation note on `synchronize` - event = GC.@preserve args begin - event = POCL.launch_tuple(obj.kern, args; local_size = items, global_size = groups .* items) + # the kernel only gets pointers to the arrays in `args` and captured by `f`, so keep + # them alive until it completes. POCL launches synchronously, see the implementation + # note on `synchronize` + f = obj.kern.f + event = GC.@preserve f args begin + event = POCL.launch_tuple(obj.kern.kernel, args; local_size = items, global_size = groups .* items) wait(event) end cl.clReleaseEvent(event) @@ -155,7 +165,7 @@ function KI.launch(obj::KI.Kernel{POCLBackend}, groups::Dims{3}, items::Dims{3}, end function KI.max_work_group_size(kernel::KI.Kernel{<:POCLBackend})::Int - wginfo = cl.work_group_info(kernel.kern.fun, device()) + wginfo = cl.work_group_info(kernel.kern.kernel.fun, device()) return Int(wginfo.size) end # querying the device allocates, so cache the limits that every launch needs