From e840bc40721fe1f58e1253e663f96e39bdeea916 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Wed, 30 Sep 2026 09:34:55 +0200 Subject: [PATCH 1/2] Launch kernels with a static ndrange and a tuned workgroup size These couldn't be launched at all: `launch_config` dropped the static ndrange before partitioning with it as the preliminary workgroup size, and `partition` made the number of blocks static, so the context type changed when the workgroup size was tuned after compiling the kernel. The number of blocks is now only static if the workgroup size is too. `launch_config` also partitions with the given ndrange, so that it is checked against the static one. --- src/KernelAbstractions.jl | 9 +++++++-- src/pocl/backend.jl | 18 ++++++++++-------- test/launch.jl | 7 +++++++ 3 files changed, 24 insertions(+), 10 deletions(-) diff --git a/src/KernelAbstractions.jl b/src/KernelAbstractions.jl index c0c7229bc..a5e0ab529 100644 --- a/src/KernelAbstractions.jl +++ b/src/KernelAbstractions.jl @@ -563,13 +563,18 @@ last (possibly partial) workgroup. Primarily used by backend implementations and @assert ndrange !== nothing blocks, workgroupsize, dynamic = NDIteration.partition(extents(ndrange), workgroupsize) - if static_ndrange <: StaticSize + # the number of blocks is only static if the workgroup size is too: a backend that tunes + # the workgroup size would otherwise change the type of the kernel's context + if static_ndrange <: StaticSize && static_workgroupsize <: StaticSize static_blocks = StaticSize{blocks} blocks = nothing - mapping = NDIteration.static_mapping(ndrange) else static_blocks = DynamicSize blocks = CartesianIndices(blocks) + end + if static_ndrange <: StaticSize + mapping = NDIteration.static_mapping(ndrange) + else mapping = NDIteration.dynamic_mapping(ndrange) end diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 4bb67491b..25dd2b835 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -151,18 +151,17 @@ function KA.launch_config(kernel::KA.Kernel{POCLBackend}, ndrange, workgroupsize workgroupsize = (workgroupsize,) end - # partition checked that the ndrange's agreed - if KA.ndrange(kernel) <: KA.StaticSize - ndrange = nothing - end - iterspace, dynamic = if KA.workgroupsize(kernel) <: KA.DynamicSize && workgroupsize === nothing - # use ndrange as preliminary workgroupsize for autotuning - KA.partition(kernel, ndrange, ndrange) + # use the ndrange as preliminary workgroupsize for autotuning + KA.partition(kernel, ndrange, something(ndrange, static_ndrange(kernel))) else + # this also checks that a given ndrange agrees with a static one KA.partition(kernel, ndrange, workgroupsize) end + if KA.ndrange(kernel) <: KA.StaticSize + ndrange = nothing + end return ndrange, workgroupsize, iterspace, dynamic end @@ -184,7 +183,8 @@ function launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args::Var # figure out the optimal workgroupsize automatically if KA.workgroupsize(obj) <: KA.DynamicSize && workgroupsize === nothing wg_info = cl.work_group_info(kernel.fun, device()) - wg_size_nd = KA.launch_workgroupsize(KA.backend(obj), launch, wg_info.size, ndrange) + range = something(ndrange, static_ndrange(obj)) + wg_size_nd = KA.launch_workgroupsize(KA.backend(obj), launch, wg_info.size, range) iterspace, dynamic = KA.partition(obj, ndrange, wg_size_nd) ctx = KA.mkcontext(obj, ndrange, iterspace, launch) end @@ -211,6 +211,8 @@ end pad3(t::Tuple) = (t..., ntuple(_ -> 1, 3 - length(t))...) +static_ndrange(kernel) = KA.ndrange(kernel) <: KA.StaticSize ? KA.get(KA.ndrange(kernel)) : nothing + KI.argconvert(::POCLBackend, arg) = clconvert(arg) function KI.kernel_function(backend::POCLBackend, f::F, tt::TT = Tuple{}; name = nothing, kwargs...) where {F, TT} diff --git a/test/launch.jl b/test/launch.jl index 1c1988fde..1b89c2474 100644 --- a/test/launch.jl +++ b/test/launch.jl @@ -91,6 +91,13 @@ function launch_testsuite(backend, AT; launcher = default_launcher) @testset "static ndrange" begin @test check_indices(launcher, backend(), AT, launch_indices!(backend(), (4, 2), (9, 5)), (9, 5)) @test check_indices(launcher, backend(), AT, launch_indices!(backend(), (4, 2), (0:8, -2:2)), (0:8, -2:2)) + + # with a tuned workgroup size, over more work-items than fit a workgroup + n = KI.max_work_group_size(backend()) + 1 + kernel = launch_indices!(backend(), KernelAbstractions.DynamicSize(), KernelAbstractions.StaticSize((n, 3))) + @test check_indices(launcher, backend(), AT, kernel, (n, 3)) + # a given ndrange has to agree with the static one + @test_throws ErrorException check_indices(launcher, backend(), AT, kernel, (n, 2)) end @testset "offsets" begin From 063ab6ae0814dde6eb46fdff4f894beac4fbb9a7 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Mon, 28 Sep 2026 11:14:47 +0200 Subject: [PATCH 2/2] Launch @kernel kernels on any KernelInterface backend `(::KA.Kernel{<:KI.Backend})(args...; ndrange, workgroupsize)` now partitions the ndrange, selects the launch, compiles the kernel with `KI.kernel_function`, tunes the workgroup size with `KI.launch_configuration` (passing the number of work-items as `nitems`) and launches it through the `KI.Kernel`, which validates the sizes. Backends no longer copy `mkcontext`, `launch_config` and the launch itself; they customize it with `KI.launch_configuration` and the new `compiler_options` hook, and still implement `Scratchpad` and their Adapt rules. A backend's own `(::KA.Kernel{MyBackend})` method still takes precedence. PoCL uses the generic launch. --- docs/src/api.md | 1 + docs/src/implementations.md | 94 +++++++++++--------------- src/KernelAbstractions.jl | 14 ++-- src/backend_launch.jl | 127 ++++++++++++++++++++++++++++++++++++ src/pocl/backend.jl | 88 ------------------------- test/codegen_checks.jl | 2 +- test/runtests.jl | 21 +++++- 7 files changed, 188 insertions(+), 159 deletions(-) create mode 100644 src/backend_launch.jl diff --git a/docs/src/api.md b/docs/src/api.md index 1be06929d..826cd7455 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -112,4 +112,5 @@ KernelAbstractions.LinearLaunch KernelAbstractions.NDLaunch KernelAbstractions.select_launch KernelAbstractions.launch_workgroupsize +KernelAbstractions.compiler_options ``` diff --git a/docs/src/implementations.md b/docs/src/implementations.md index 91fd68841..172934763 100644 --- a/docs/src/implementations.md +++ b/docs/src/implementations.md @@ -78,63 +78,43 @@ Adapt.adapt_storage(::CUDABackend, x) = adapt(CuArray, x) ## Launching `@kernel` kernels -A kernel written with [`@kernel`](@ref) receives a hidden context, a -`KernelAbstractions.CompilerMetadata` built by the backend's `mkcontext`, from which -[`@index`](@ref) computes its indices. By default (a context without a `launch`), -`@index` assumes that the kernel was launched on a 1-D grid of -`length(blocks(iterspace))` groups of `length(workitems(iterspace))` work-items. It then -decomposes the linear hardware ids into Cartesian positions, which takes integer divisions -when the `ndrange` is not known at compile time, and computes in `Int`. - -A backend **may** launch kernels differently, and pass the `launch` keyword to the -`CompilerMetadata` constructor to tell `@index` how: - -- [`NDLaunch{T}`](@ref KernelAbstractions.NDLaunch): the grid has the shape of the - iteration space (for as many dimensions as the backend's grid has, i.e. up to 3), so - `@index` doesn't need any divisions; -- [`LinearLaunch{T}`](@ref KernelAbstractions.LinearLaunch): the default 1-D grid. - -Either way `@index` computes in `T`, e.g. `Int32`, which is faster on GPUs. The backend -has to implement the typed [`KI.get_group_id`](@ref KernelInterface.get_group_id) and -[`KI.get_local_id`](@ref KernelInterface.get_local_id) queries such that they compute in -`T` too, e.g. without checked conversions. - -[`select_launch`](@ref KernelAbstractions.select_launch) chooses the launch from the -iteration space, whether the workgroup size will be tuned, and the limits of the backend -([`KI.max_work_group_size`](@ref KernelInterface.max_work_group_size), -[`KI.max_work_group_dims`](@ref KernelInterface.max_work_group_dims) and -[`KI.max_num_groups`](@ref KernelInterface.max_num_groups)). It doesn't depend on the -workgroup size a backend tunes afterwards, which keeps the context type (and thus the -compiled kernel) the same before and after tuning, as long as the backend tunes with -[`launch_workgroupsize`](@ref KernelAbstractions.launch_workgroupsize). A launch then -looks like this: - -```julia -function (obj::KA.Kernel{MyBackend})(args...; ndrange = nothing, workgroupsize = nothing) - ndrange, workgroupsize, iterspace, dynamic = KA.launch_config(obj, ndrange, workgroupsize) - launch = KA.select_launch(obj, workgroupsize, iterspace) - ctx = KA.CompilerMetadata{KA.ndrange(obj), KA.DynamicCheck}(ndrange, iterspace; launch) - kernel = compile(obj.f, ctx, args...) - - if KA.workgroupsize(obj) <: KA.DynamicSize && workgroupsize === nothing - threads = max_threads(kernel) # at most `KI.max_work_group_size(backend)` - workgroupsize = KA.launch_workgroupsize(backend, launch, threads, ndrange) - iterspace, dynamic = KA.partition(obj, ndrange, workgroupsize) - ctx = KA.CompilerMetadata{KA.ndrange(obj), KA.DynamicCheck}(ndrange, iterspace; launch) - end - - groups, items = size(KA.blocks(iterspace)), size(KA.workitems(iterspace)) - prod(groups) == 0 && return - if launch isa KA.NDLaunch - run(kernel, ctx, args...; groups, items) # padded to 3 dimensions - else - run(kernel, ctx, args...; groups = prod(groups), items = prod(items)) - end -end -``` - -The POCL backend is an example. Backends that launch on an N-d grid **must not** override -`__validindex` or the `__index_*` functions, which dispatch on the launch. +KernelAbstractions launches [`@kernel`](@ref) kernels on any backend that implements +[KernelInterface](@ref kernelinterface): it partitions the `ndrange`, builds the kernel's +hidden context (a `KernelAbstractions.CompilerMetadata`), compiles the kernel with +[`KI.kernel_function`](@ref KernelInterface.kernel_function), tunes the workgroup size, and +launches it with [`KI.launch`](@ref KernelInterface.launch). A backend doesn't implement any +of that itself, but it needs KernelInterface's typed index queries and an N-d +[`KI.launch`](@ref KernelInterface.launch), and it must not override KernelAbstractions' index +functions (see below). It **may** customize the launch through: + +- [`KI.launch_configuration`](@ref KernelInterface.launch_configuration): the workgroup size + used when the kernel has no static or given one. It receives the number of work-items in + the `ndrange` as `nitems`, e.g. to prefer more workgroups over larger ones. +- [`KernelAbstractions.compiler_options`](@ref): compiler options for a kernel, e.g. a + register hint derived from its static workgroup size. +- `KernelAbstractions.Scratchpad`, which backs [`@private`](@ref) arrays and has to be + implemented (`@device_override`) for a backend's device: e.g. a stack allocation, or a + `StaticArrays.MArray`. +- `Adapt.adapt_storage(::KernelAbstractions.ConstAdaptor, x)` for the backend's device + arrays, which implements [`@Const`](@ref). + +[`@index`](@ref) computes its indices from how a kernel was launched: on a grid with the +shape of the iteration space ([`NDLaunch`](@ref KernelAbstractions.NDLaunch), for up to as +many dimensions as the backend's grid has), which doesn't need any divisions, or on a 1-D +grid ([`LinearLaunch`](@ref KernelAbstractions.LinearLaunch)). Either way it computes in a +narrow index type such as `Int32` when the iteration space fits, which is why the typed +[`KI.get_group_id`](@ref KernelInterface.get_group_id) and +[`KI.get_local_id`](@ref KernelInterface.get_local_id) queries have to compute in that type +too, as KernelInterface specifies. For the same reason, backends **must not** override +`__validindex` or the `__index_*` functions. + +A backend can still implement `(obj::KernelAbstractions.Kernel{MyBackend})(args...; ndrange, workgroupsize)` +to launch kernels itself, e.g. while it is being ported to KernelInterface. That relies on +KernelAbstractions internals: it has to choose the launch with +[`select_launch`](@ref KernelAbstractions.select_launch), pass it to the kernel's context, +and tune the workgroup size with +[`launch_workgroupsize`](@ref KernelAbstractions.launch_workgroupsize), as the generic +launch in `src/backend_launch.jl` does. Packages that customize the iteration space (with a custom `partition` and `expand`) don't need to do anything for these launches: the index functions only compute the global diff --git a/src/KernelAbstractions.jl b/src/KernelAbstractions.jl index a5e0ab529..c9c10a943 100644 --- a/src/KernelAbstractions.jl +++ b/src/KernelAbstractions.jl @@ -472,12 +472,8 @@ synchronize(backend) Use [`workgroupsize`](@ref KernelAbstractions.workgroupsize), [`ndrange`](@ref KernelAbstractions.ndrange), and [`backend`](@ref KernelAbstractions.backend) to inspect a kernel's static configuration. -!!! note - Backend implementations **must** implement: - ``` - (kernel::Kernel{<:NewBackend})(args...; ndrange=nothing, workgroupsize=nothing) - ``` - As well as the on-device functionality. +Kernels are launched on any backend that implements [KernelInterface](@ref kernelinterface); +see the [notes for backend implementations](@ref implementations_notes). """ struct Kernel{Backend, WorkgroupSize <: _Size, NDRange <: _Size, Fun} backend::Backend @@ -616,10 +612,6 @@ function __workitems_iterspace end end end -# for reflection -function mkcontext end -function launch_config end - include("macros.jl") include("spawn.jl") @@ -649,6 +641,8 @@ automatically when a kernel is launched. argconvert(k::Kernel{T}, arg) where {T} = error("Don't know how to convert arguments for Kernel{$T}") +include("backend_launch.jl") + # Enzyme support supports_enzyme(::Backend) = false function __fake_compiler_job end diff --git a/src/backend_launch.jl b/src/backend_launch.jl new file mode 100644 index 000000000..38b2c7e90 --- /dev/null +++ b/src/backend_launch.jl @@ -0,0 +1,127 @@ +### +# Launching `@kernel` kernels on a KernelInterface backend +# +# Every backend implementing KernelInterface launches `@kernel` kernels with the methods +# below. Backends customize them through the hooks documented in `implementations.md` +# (`compiler_options`, and KernelInterface's `launch_configuration`) instead of +# reimplementing the launch. +### + +""" + mkcontext(kernel::Kernel, ndrange, iterspace, [launch]) + +The hidden context argument for launching `kernel` over `ndrange`, partitioned as +`iterspace`, with the launch configuration `launch` (see [`select_launch`](@ref)). +""" +mkcontext(kernel::Kernel, _ndrange, iterspace) = + CompilerMetadata{ndrange(kernel), DynamicCheck}(_ndrange, iterspace) +mkcontext(kernel::Kernel, _ndrange, iterspace, launch) = + CompilerMetadata{ndrange(kernel), DynamicCheck}(_ndrange, iterspace; launch) +mkcontext(kernel::Kernel, I, _ndrange, iterspace, ::Dynamic) where {Dynamic} = + CompilerMetadata{ndrange(kernel), Dynamic}(I, _ndrange, iterspace) + +""" + launch_config(kernel::Kernel, ndrange, workgroupsize) + +Normalize the launch arguments of `kernel`, and partition the `ndrange`. Returns the +`ndrange` (`nothing` if it's static), the `workgroupsize` (`nothing` if it will be tuned), +the iteration space and whether it needs bounds checks. If the workgroup size will be +tuned, the iteration space is preliminary: it uses the `ndrange` as the workgroup size. +""" +function launch_config(kernel::Kernel, _ndrange, _workgroupsize) + if _ndrange isa Integer + _ndrange = (_ndrange,) + end + if _workgroupsize isa Integer + _workgroupsize = (_workgroupsize,) + end + + iterspace, dynamic = if workgroupsize(kernel) <: DynamicSize && _workgroupsize === nothing + # use the ndrange as preliminary workgroupsize for autotuning + partition(kernel, _ndrange, something(_ndrange, static_ndrange(kernel))) + else + # this also checks that a given ndrange agrees with a static one + partition(kernel, _ndrange, _workgroupsize) + end + if ndrange(kernel) <: StaticSize + _ndrange = nothing + end + + return _ndrange, _workgroupsize, iterspace, dynamic +end + +""" + compiler_options(kernel::Kernel)::NamedTuple + +Backend-specific compiler options for compiling `kernel` with +[`KI.kernel_function`](@ref KernelInterface.kernel_function), e.g. a hint derived from its +static workgroup size (CUDA.jl passes `maxthreads`). Backends **may** implement this for +their backend type; the default is no options. +""" +compiler_options(::Kernel) = (;) + +static_ndrange(kernel::Kernel) = ndrange(kernel) <: StaticSize ? get(ndrange(kernel)) : nothing + +# the product of `dims`, saturated at `typemax(Int)` +function saturated_prod(dims::Dims) + n = 1 + for d in dims + n, overflow = Base.mul_with_overflow(n, d) + overflow && return typemax(Int) + end + return n +end + +argconvert(kernel::Kernel{<:KI.Backend}, arg) = KI.argconvert(backend(kernel), arg) + +function (obj::Kernel{<:KI.Backend})(args::Vararg{Any, N}; ndrange = nothing, workgroupsize = nothing) where {N} + ndrange, workgroupsize, iterspace, dynamic = launch_config(obj, ndrange, workgroupsize) + # nothing to launch (or compile) for an empty ndrange + any(iszero, size(blocks(iterspace))) && return nothing + + # launch on an N-d grid, computing indices in 32 bits, if possible. this doesn't depend + # on the tuned workgroup size, so the context (and thus the kernel) doesn't either. + launch = select_launch(obj, workgroupsize, iterspace) + if launch === NDLaunch{Int32}() + # the common case, specialized statically + launch_kernel(obj, NDLaunch{Int32}(), ndrange, workgroupsize, iterspace, args...) + else + launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args...) + end + return nothing +end + +function launch_kernel( + obj::Kernel, launch, ndrange, _workgroupsize, iterspace, args::Vararg{Any, N} + ) where {N} + b = backend(obj) + + # this might not be the final context, since we may tune the workgroupsize + ctx = mkcontext(obj, ndrange, iterspace, launch) + kernel = compile(obj, ctx, args...) + + # tune the workgroup size, keeping the context type (and thus the kernel) the same + if workgroupsize(obj) <: DynamicSize && _workgroupsize === nothing + range = something(ndrange, static_ndrange(obj)) + threads = KI.launch_configuration(kernel; nitems = saturated_prod(extents(range))).workgroupsize + iterspace, _ = partition(obj, ndrange, launch_workgroupsize(b, launch, threads, range)) + ctx = mkcontext(obj, ndrange, iterspace, launch) + end + + # launching through the `KI.Kernel` validates the sizes against the kernel's limits + groups = size(blocks(iterspace)) + items = size(workitems(iterspace)) + if launch isa NDLaunch + kernel(ctx, args...; numgroups = groups, workgroupsize = items) + else + kernel(ctx, args...; numgroups = prod(groups), workgroupsize = prod(items)) + end + return nothing +end + +@inline function compile(obj::Kernel, ctx, args::Vararg{Any, N}) where {N} + b = backend(obj) + f = KI.argconvert(b, obj.f) + tt = Tuple{Core.Typeof(KI.argconvert(b, ctx)), map(arg -> Core.Typeof(KI.argconvert(b, arg)), args)...} + return KI.kernel_function(b, f, tt; compiler_options(obj)...) +end diff --git a/src/pocl/backend.jl b/src/pocl/backend.jl index 25dd2b835..48a1633c0 100644 --- a/src/pocl/backend.jl +++ b/src/pocl/backend.jl @@ -130,89 +130,6 @@ KI.supports_atomics(::POCLBackend) = true ## Kernel Launch -function KA.mkcontext(kernel::KA.Kernel{POCLBackend}, _ndrange, iterspace) - return KA.CompilerMetadata{KA.ndrange(kernel), KA.DynamicCheck}(_ndrange, iterspace) -end -function KA.mkcontext(kernel::KA.Kernel{POCLBackend}, _ndrange, iterspace, launch) - return KA.CompilerMetadata{KA.ndrange(kernel), KA.DynamicCheck}(_ndrange, iterspace; launch) -end -function KA.mkcontext( - kernel::KA.Kernel{POCLBackend}, I, _ndrange, iterspace, - ::Dynamic - ) where {Dynamic} - return KA.CompilerMetadata{KA.ndrange(kernel), Dynamic}(I, _ndrange, iterspace) -end - -function KA.launch_config(kernel::KA.Kernel{POCLBackend}, ndrange, workgroupsize) - if ndrange isa Integer - ndrange = (ndrange,) - end - if workgroupsize isa Integer - workgroupsize = (workgroupsize,) - end - - iterspace, dynamic = if KA.workgroupsize(kernel) <: KA.DynamicSize && - workgroupsize === nothing - # use the ndrange as preliminary workgroupsize for autotuning - KA.partition(kernel, ndrange, something(ndrange, static_ndrange(kernel))) - else - # this also checks that a given ndrange agrees with a static one - KA.partition(kernel, ndrange, workgroupsize) - end - if KA.ndrange(kernel) <: KA.StaticSize - ndrange = nothing - end - - return ndrange, workgroupsize, iterspace, dynamic -end - -function (obj::KA.Kernel{POCLBackend})(args::Vararg{Any, N}; ndrange = nothing, workgroupsize = nothing) where {N} - ndrange, workgroupsize, iterspace, dynamic = - KA.launch_config(obj, ndrange, workgroupsize) - # the launch doesn't depend on the tuned workgroup size, so neither does the context - launch = KA.select_launch(obj, workgroupsize, iterspace) - launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args...) - return nothing -end - -function launch_kernel(obj, launch, ndrange, workgroupsize, iterspace, args::Vararg{Any, N}) where {N} - # this might not be the final context, since we may tune the workgroupsize - ctx = KA.mkcontext(obj, ndrange, iterspace, launch) - kernel = @opencl launch = false obj.f(ctx, args...) - - # figure out the optimal workgroupsize automatically - if KA.workgroupsize(obj) <: KA.DynamicSize && workgroupsize === nothing - wg_info = cl.work_group_info(kernel.fun, device()) - range = something(ndrange, static_ndrange(obj)) - wg_size_nd = KA.launch_workgroupsize(KA.backend(obj), launch, wg_info.size, range) - iterspace, dynamic = KA.partition(obj, ndrange, wg_size_nd) - ctx = KA.mkcontext(obj, ndrange, iterspace, launch) - end - - groups = size(KA.blocks(iterspace)) - items = size(KA.workitems(iterspace)) - if prod(groups) == 0 - return nothing - end - - # Launch kernel - if launch isa KA.NDLaunch - local_size = pad3(items) - global_size = local_size .* pad3(groups) - else - local_size = prod(items) - global_size = prod(groups) * local_size - end - event = kernel(ctx, args...; global_size, local_size) - wait(event) - cl.clReleaseEvent(event) - return nothing -end - -pad3(t::Tuple) = (t..., ntuple(_ -> 1, 3 - length(t))...) - -static_ndrange(kernel) = KA.ndrange(kernel) <: KA.StaticSize ? KA.get(KA.ndrange(kernel)) : nothing - KI.argconvert(::POCLBackend, arg) = clconvert(arg) function KI.kernel_function(backend::POCLBackend, f::F, tt::TT = Tuple{}; name = nothing, kwargs...) where {F, TT} @@ -345,9 +262,4 @@ end POCL._print(args...) end - -## Other - -KA.argconvert(::KA.Kernel{POCLBackend}, arg) = clconvert(arg) - end diff --git a/test/codegen_checks.jl b/test/codegen_checks.jl index e70881b32..59dff1502 100644 --- a/test/codegen_checks.jl +++ b/test/codegen_checks.jl @@ -174,7 +174,7 @@ end @test @filecheck begin @check "define spir_kernel void @{{.*}}gpu_codegen_global_linear" @check "udiv i32" - @device_code_llvm debuginfo = :none KernelAbstractions.POCL.POCLKernels.launch_kernel( + @device_code_llvm debuginfo = :none KernelAbstractions.launch_kernel( kernel, KernelAbstractions.LinearLaunch{Int32}(), ndrange, workgroupsize, iterspace, B ) KernelAbstractions.synchronize(backend) diff --git a/test/runtests.jl b/test/runtests.jl index 07499b704..ef7cb2b2e 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -178,6 +178,21 @@ end Testsuite.select_launch_testsuite() end +@kernel function fill_index!(A) + I = @index(Global, Linear) + @inbounds A[I] = I +end + +@testset "generic launch" begin + # workgroup sizes are validated against the kernel's limits + limit = KernelAbstractions.KI.max_work_group_size(CPU()) + @test_throws ArgumentError fill_index!(CPU())(zeros(Int, 2limit); ndrange = 2limit, workgroupsize = 2limit) + + # iteration spaces with more work-items than an `Int` can count are rejected, instead + # of launching nothing because the number of workgroups overflowed + @test_throws ArgumentError fill_index!(CPU())(zeros(Int, 1); ndrange = (2^22, 2^22, 2^22, 1), workgroupsize = 1) +end + # the shared testsuite only covers the launch configuration POCL selects @testset "POCL launch configurations" begin KA = KernelAbstractions @@ -188,7 +203,7 @@ end CartesianIndices((2, 2)), nothing, TransposedMapping() ) A = zeros(Int, 5, 7) - POCL.POCLKernels.launch_kernel(kernel, launch, CartesianIndices(A), nothing, iterspace, A) + KernelAbstractions.launch_kernel(kernel, launch, CartesianIndices(A), nothing, iterspace, A) @test A == LinearIndices(A) end @testset "custom iteration space, $launch" for launch in (nothing, KA.LinearLaunch{Int}(), KA.NDLaunch{Int}()) @@ -197,7 +212,7 @@ end iterspace = KA.NDRange{2, KA.StaticSize{(2, 2)}, KA.StaticSize{(4, 4)}}(nothing, ItemOffsets((1, 2))) ndrange = CartesianIndices((2:8, 3:7)) A = zeros(Int, 9, 8) - POCL.POCLKernels.launch_kernel(kernel, launch, ndrange, nothing, iterspace, A) + KernelAbstractions.launch_kernel(kernel, launch, ndrange, nothing, iterspace, A) @test A[ndrange] == LinearIndices(ndrange) A[ndrange] .= 0 @test all(iszero, A) @@ -210,7 +225,7 @@ end ndrange, workgroupsize, iterspace, _ = KA.launch_config(kernel, ndrange, workgroupsize) # an N-d launch is limited to three dimensions l = launch isa KA.NDLaunch && ndims(iterspace) > 3 ? KA.LinearLaunch{Int}() : launch - POCL.POCLKernels.launch_kernel(kernel, l, ndrange, workgroupsize, iterspace, args...) + KernelAbstractions.launch_kernel(kernel, l, ndrange, workgroupsize, iterspace, args...) end @testset "$launch" begin Testsuite.launch_testsuite(CPU, Array; launcher)