Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -19,38 +19,49 @@ Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7"
Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c"
Random123 = "74087812-796a-5b5d-8853-05524746bad3"
RandomNumbers = "e6cf234a-135c-5ec9-84dd-332b85af5143"
ScopedValues = "7e506255-f358-4e82-b7e4-beb19740aa63"
SPIRVIntrinsics = "71d1d633-e7e8-4a92-83a1-de8814b09ba8"
SPIRV_LLVM_Backend_jll = "4376b9bf-cff8-51b6-bb48-39421dff0d0c"
SPIRV_Tools_jll = "6ac6d60f-d740-5983-97d7-a4482c0689f4"
pocl_standalone_jll = "54f56a70-6062-5590-a942-1226658f6c83"

[weakdeps]
AMDGPU = "21141c5a-9bdb-4563-92ae-f87d6854732e"
IntelITT = "c9b2f978-7543-4802-ae44-75068f23ee64"
LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e"
NVTX = "5da4648a-3479-48b8-97b9-01cb529c0a1f"
SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf"
StaticArrays = "90137ffa-7385-5640-81b9-e52037218182"

[sources]
KernelInterface = {path = "lib/KernelInterface"}

[extensions]
IntelITTExt = "IntelITT"
LinearAlgebraExt = "LinearAlgebra"
NVTXExt = "NVTX"
ROCTXExt = "AMDGPU"
SparseArraysExt = "SparseArrays"
StaticArraysExt = "StaticArrays"

[compat]
AMDGPU = "2"
Adapt = "0.4, 1.0, 2.0, 3.0, 4"
Atomix = "1.2.1"
GPUCompiler = "2.10"
GPUToolbox = "3.3.2"
IntelITT = "0.2"
KernelInterface = "0.4"
LLVM = "10"
LinearAlgebra = "1.6"
MacroTools = "0.5"
NVTX = "0.3, 1"
PrecompileTools = "1"
Printf = "<0.0.1, 1"
Random = "1"
Random123 = "1.7.1"
RandomNumbers = "1.6.0"
ScopedValues = "1.3"
SPIRVIntrinsics = "1.1.3"
SPIRV_LLVM_Backend_jll = "23"
SPIRV_Tools_jll = "2024.4, 2025.1"
Expand Down
1 change: 1 addition & 0 deletions docs/make.jl
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,7 @@ function main()
"Extras" => [
"extras/unrolling.md",
"extras/pocl_debugging.md",
"extras/profiling.md",
], # Extras
"Notes for implementations" => "implementations.md",
], # pages
Expand Down
114 changes: 114 additions & 0 deletions docs/src/extras/profiling.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
# Profiling

KernelAbstractions can put named ranges on the timeline of a tracing profiler, such as
NVIDIA Nsight Systems or Intel VTune, so that you can see which part of your program a
stretch of kernels belongs to. Annotations are cheap when no profiler is listening: a
single atomic load, and the label isn't even built.
Comment on lines +3 to +6

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
KernelAbstractions can put named ranges on the timeline of a tracing profiler, such as
NVIDIA Nsight Systems or Intel VTune, so that you can see which part of your program a
stretch of kernels belongs to. Annotations are cheap when no profiler is listening: a
single atomic load, and the label isn't even built.
KernelAbstractions can put named ranges on the timeline of a tracing profiler, such as
NVIDIA Nsight Systems or Intel VTune, so that you can see which part of your program a
stretch of kernels belongs to. Annotations are cheap when no profiler is listening.


## Annotating code

Wrap code in [`@profiling_range`](@ref):

```julia
@profiling_range "volume integral" begin
volume_integral!(du, u, backend)
end
```

Ranges can be grouped with a `domain`, which maps to an NVTX or ITT domain:

```julia
@profiling_range "time step $i" domain = "Trixi" begin
step!(integrator)
end
```

Kernel launches are annotated with the kernel's name automatically. For an instantaneous
event, use [`profiling_mark`](@ref), and for ranges that don't follow the structure of the
code, [`profiling_range_start`](@ref KernelAbstractions.profiling_range_start) and
[`profiling_range_end`](@ref KernelAbstractions.profiling_range_end).

## Built-in profiler

To see where time goes without an external profiler, run code under
[`KernelAbstractions.@profile`](@ref KernelAbstractions.@profile). It records the ranges
and kernel launches of an expression, and summarizes them:

```julia-repl
julia> KernelAbstractions.@profile for i in 1:10
@profiling_range "step" begin
mul2(backend)(A; ndrange = length(A))
add(backend)(A, B; ndrange = length(A))
end
end
Profiled 6.04 ms, recording 30 ranges.

Time (%) Total time Calls Avg time Min time Max time Name
──────── ────────── ───── ──────── ──────── ──────── ────
92.6 % 5.59 ms 10 559 µs 298 µs 2.9 ms step
57.6 % 3.48 ms 10 348 µs 109 µs 2.48 ms mul2
37.6 % 2.27 ms 10 227 µs 179 µs 400 µs add
```

Kernel launches synchronize their backend while profiling, so that their ranges measure the
kernel rather than its launch; pass `synchronize = false` to measure launches. Pass
`trace = true` to list every range in order instead. The first call of a kernel includes
its compilation, so profile a warmed-up run.

`@profile` records the task running the expression and the tasks it spawns, e.g. with
[`KernelAbstractions.@spawn`](@ref KernelAbstractions.@spawn), but not other tasks, so
profiles can run concurrently. Wait for spawned tasks within the expression, e.g. with
`@sync`, as what they record after it returns is lost.

## Profilers

Ranges are recorded on the host threads of the process, which is how NVTX, ITT and
roctx work: it is the profiler that attributes the device work launched within a range to
it. So which profiler records the ranges depends on what the process runs under, not on the
backend: running the CPU backend under Nsight Systems gives NVTX ranges, and a GPU backend
under VTune gives ITT tasks.
Comment on lines +65 to +69

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
Ranges are recorded on the host threads of the process, which is how NVTX, ITT and
roctx work: it is the profiler that attributes the device work launched within a range to
it. So which profiler records the ranges depends on what the process runs under, not on the
backend: running the CPU backend under Nsight Systems gives NVTX ranges, and a GPU backend
under VTune gives ITT tasks.
Ranges are recorded on the host threads of the process, which is how NVTX, ITT and
roctx work: it is the profiler that attributes the device work launched within a range to
it.


Ranges go to every registered [`Tracer`](@ref KernelAbstractions.Tracer). These come with
KernelAbstractions, and only register themselves when their profiler is attached:

- **Nsight Systems**: load [NVTX.jl](https://github.com/JuliaGPU/NVTX.jl) (CUDA.jl loads it
too) and run under `nsys profile --trace=nvtx,...`.
- **rocprof**: load [AMDGPU.jl](https://github.com/JuliaGPU/AMDGPU.jl) and run under
`rocprofv3 --marker-trace` (or the legacy `rocprof --roctx-trace`). Ranges are recorded
with roctx, which has no domains, so a `domain` other than `"KernelAbstractions"` prefixes
the label.
- **Intel VTune**: load [IntelITT.jl](https://github.com/JuliaPerf/IntelITT.jl) and run
under VTune.
- **NVTXT**, a text format that Nsight Systems imports, to trace without a profiler: set
`JULIA_KA_NVTXT=1` to write `ka-<pid>.nvtxt` to the working directory, or set it to a
path, in which `%p` is replaced by the process id. Then
```sh
ImportNvtxt --cmd create --nvtxt ka-1234.nvtxt -o report.nsys-rep
```
To trace only part of a program, register an [`NVTXTTracer`](@ref
KernelAbstractions.NVTXTTracer) yourself:
```julia
tracer = KernelAbstractions.register_tracer!(KernelAbstractions.NVTXTTracer("trace.nvtxt"))
run_simulation()
KernelAbstractions.unregister_tracer!(tracer)
close(tracer)
```

Other profilers are supported by subtyping
[`Tracer`](@ref KernelAbstractions.Tracer).

## API

```@docs
@profiling_range
profiling_mark
KernelAbstractions.@profile
KernelAbstractions.ProfileResults
KernelAbstractions.profiling_active
KernelAbstractions.profiling_range_start
KernelAbstractions.profiling_range_end
KernelAbstractions.Tracer
KernelAbstractions.register_tracer!
KernelAbstractions.unregister_tracer!
KernelAbstractions.NVTXTTracer
```
35 changes: 35 additions & 0 deletions ext/IntelITTExt.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
module IntelITTExt

import KernelAbstractions as KA
import IntelITT

# forwards ranges to Intel VTune as ITT tasks, one ITT domain per domain
struct ITTTracer <: KA.Tracer
domains::IdDict{Symbol, IntelITT.Domain}
lock::ReentrantLock
end
ITTTracer() = ITTTracer(IdDict{Symbol, IntelITT.Domain}(), ReentrantLock())

domain(tracer::ITTTracer, name::Symbol) =
@lock tracer.lock get!(() -> IntelITT.Domain(String(name)), tracer.domains, name)

function KA.trace_range_start(tracer::ITTTracer, label, domain_name)
# overlapped tasks may end on another thread, and needn't nest
task = IntelITT.Task(domain(tracer, domain_name), String(label))
IntelITT.start(task)
return task
end

KA.trace_range_end(::ITTTracer, task::IntelITT.Task) = (IntelITT.stop(task); nothing)

const TRACER = Ref{ITTTracer}()

function __init__()
# only under a collector, as otherwise every annotation would be wasted work
if IntelITT.isactive()
TRACER[] = KA.register_tracer!(ITTTracer())
end
return
end

end # module
53 changes: 53 additions & 0 deletions ext/NVTXExt.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
module NVTXExt

import KernelAbstractions as KA
import NVTX

# forwards ranges to Nsight Systems as NVTX ranges, one NVTX domain per domain. NVTX ranges
# annotate host threads; Nsight Systems projects the GPU work launched within them itself.
struct NVTXDomain
domain::NVTX.Domain
# `Symbol` labels are fixed in the code, so they are registered with NVTX once, which
# makes recording them cheaper. Other labels are passed as they are.
strings::IdDict{Symbol, NVTX.StringHandle}
end

struct NVTXTracer <: KA.Tracer
domains::IdDict{Symbol, NVTXDomain}
lock::ReentrantLock
end
NVTXTracer() = NVTXTracer(IdDict{Symbol, NVTXDomain}(), ReentrantLock())

function domain(tracer::NVTXTracer, name::Symbol)
return @lock tracer.lock get!(tracer.domains, name) do
NVTXDomain(NVTX.Domain(String(name)), IdDict{Symbol, NVTX.StringHandle}())
end
end

message(::NVTXTracer, d::NVTXDomain, label::String) = label
message(tracer::NVTXTracer, d::NVTXDomain, label::Symbol) =
@lock tracer.lock get!(() -> NVTX.StringHandle(d.domain, String(label)), d.strings, label)

# process ranges, rather than push/pop, since they may end on another thread
function KA.trace_range_start(tracer::NVTXTracer, label, domain_name)
d = domain(tracer, domain_name)
return NVTX.range_start(d.domain; message = message(tracer, d, label))
end
KA.trace_range_end(::NVTXTracer, id::NVTX.RangeId) = (NVTX.range_end(id); nothing)
function KA.trace_mark(tracer::NVTXTracer, label, domain_name)
d = domain(tracer, domain_name)
NVTX.mark(d.domain; message = message(tracer, d, label))
return nothing
end

const TRACER = Ref{NVTXTracer}()

function __init__()
# only under Nsight, as otherwise every annotation would be wasted work
if NVTX.isactive()
TRACER[] = KA.register_tracer!(NVTXTracer())
end
return
end

end # module
77 changes: 77 additions & 0 deletions ext/ROCTXExt.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
module ROCTXExt

import KernelAbstractions as KA
import AMDGPU
using Base.Libc: Libdl

# forwards ranges to rocprof as roctx ranges. like NVTX, roctx annotates host threads, and
# rocprof attributes the GPU work launched within them. roctx has no domains, so ranges in a
# domain other than `:KernelAbstractions` are prefixed with it.
struct ROCTXTracer <: KA.Tracer
range_start::Ptr{Cvoid}
range_stop::Ptr{Cvoid}
mark::Ptr{Cvoid}
end

"""
ROCTXTracer(library::AbstractString)

A tracer that calls the roctx API in `library`, i.e. `librocprofiler-sdk-roctx` for
`rocprofv3`, or `libroctx64` for the legacy `rocprof`.
"""
function ROCTXTracer(library::AbstractString)
handle = Libdl.dlopen(library)
return ROCTXTracer(
Libdl.dlsym(handle, :roctxRangeStartA), Libdl.dlsym(handle, :roctxRangeStop),
Libdl.dlsym(handle, :roctxMarkA)
)
end

# a `Symbol` is passed to C as its name, without allocating
roctx_message(label, domain) = domain === KA.DEFAULT_DOMAIN ? label : string(domain, ": ", label)

# process ranges, rather than push/pop, since they may end on another thread
KA.trace_range_start(tracer::ROCTXTracer, label, domain) =
ccall(tracer.range_start, UInt64, (Cstring,), roctx_message(label, domain))
KA.trace_range_end(tracer::ROCTXTracer, id::UInt64) =
(ccall(tracer.range_stop, Cvoid, (UInt64,), id); nothing)
KA.trace_mark(tracer::ROCTXTracer, label, domain) =
(ccall(tracer.mark, Cvoid, (Cstring,), roctx_message(label, domain)); nothing)

# rocprofv3 loads its tool through rocprofiler-register, and only intercepts the roctx of
# the rocprofiler-sdk; the legacy rocprof loads its tool into HSA, and intercepts libroctx64
function roctx_libraries()
if haskey(ENV, "ROCP_TOOL_LIBRARIES")
return ["librocprofiler-sdk-roctx", "libroctx64"]
elseif haskey(ENV, "HSA_TOOLS_LIB")
return ["libroctx64", "librocprofiler-sdk-roctx"]
else
return String[]
end
end

function rocm_libdir()
rocm_path = try
AMDGPU.ROCmDiscovery.find_roc_path()
catch
get(ENV, "ROCM_PATH", "/opt/rocm")
end
return joinpath(rocm_path, "lib")
end

const TRACER = Ref{ROCTXTracer}()

function __init__()
# only under rocprof, as otherwise every annotation would be wasted work
names = roctx_libraries()
isempty(names) && return
library = Libdl.find_library(names, [rocm_libdir()])
if isempty(library)
@warn "Running under rocprof, but roctx wasn't found; KernelAbstractions' ranges won't be recorded" names
return
end
TRACER[] = KA.register_tracer!(ROCTXTracer(library))
return
end

end # module
8 changes: 8 additions & 0 deletions src/KernelAbstractions.jl
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ export @index, @groupsize, @ndrange
export @print
export Backend, CPU
export synchronize, get_backend, allocate
export @profiling_range, profiling_mark

import PrecompileTools

Expand Down Expand Up @@ -742,6 +743,13 @@ number as its compute units: `KernelAbstractions.POCL.device().max_compute_units
"""
const CPU = POCLBackend

include("profiling.jl")
include("profiler.jl")
include("precompile.jl")

function __init__()
init_profiling()
return
end

end #module
29 changes: 29 additions & 0 deletions src/backend_launch.jl
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,35 @@ Core.kwcall(kwargs::NamedTuple, obj::Kernel{<:KI.Backend}, args::Vararg{Any, N})
launch_tuple(obj, args; kwargs...)

function launch_tuple(obj::Kernel, args::Tuple; ndrange = nothing, workgroupsize = nothing)
profiling_active() && return launch_traced(obj, args, ndrange, workgroupsize)
return launch_untraced(obj, args, ndrange, workgroupsize)
end

# Inferred and compiled once for all kernels, rather than as part of every kernel's launch,
# which would add to the compilation of every new kernel. A traced launch pays for a dynamic
# dispatch to `launch_untraced` instead.
Base.@nospecializeinfer @noinline function launch_traced(
@nospecialize(obj::Kernel), @nospecialize(args::Tuple), @nospecialize(ndrange), @nospecialize(workgroupsize)
)
id = start_launch_range(kernel_label(obj.f))
try
launch_untraced(obj, args, ndrange, workgroupsize)
synchronize_launch(id, backend(obj))
finally
profiling_range_end(id)
end
return nothing
end

@noinline start_launch_range(label::Symbol) = profiling_range_start(label)

# for a profiler that measures kernels rather than launches
@noinline function synchronize_launch(id, backend)
synchronizes_launches(id) && KI.synchronize(backend)
return nothing
end

function launch_untraced(obj::Kernel, args::Tuple, ndrange, workgroupsize)
ndrange, workgroupsize, iterspace, dynamic = launch_config(obj, ndrange, workgroupsize)
# nothing to launch (or compile) for an empty ndrange
any(iszero, size(blocks(iterspace))) && return nothing
Expand Down
Loading
Loading