Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,7 @@ GPUArrays = "11.5.14"
GPUCompiler = "2.9"
GPUToolbox = "3"
KernelAbstractions = "0.9.2"
LLVM = "9"
LLVM = "10"
LLVMDowngrader_jll = "0.11"
Libdl = "1"
LinearAlgebra = "1"
Expand Down
2 changes: 1 addition & 1 deletion src/AMDGPU.jl
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ using GPUCompiler
using GPUArrays
using GPUArrays: allowscalar
using Libdl
using LLVM, LLVM.Interop
using LLVM
using Preferences
using Printf

Expand Down
2 changes: 1 addition & 1 deletion src/compiler/Compiler.jl
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@ module Compiler
import Core: LLVMPtr

using ..GPUCompiler
using ..LLVM
using ..LLVM, ..LLVM.IR, ..LLVM.Build, ..LLVM.Passes
using Printf

import GPUToolbox: @gcsafe_ccall
Expand Down
76 changes: 37 additions & 39 deletions src/compiler/codegen.jl
Original file line number Diff line number Diff line change
Expand Up @@ -98,7 +98,7 @@ function GPUCompiler.finish_module!(
fold_wavefrontsize!(mod, job.config.params.wavefrontsize64)

# Set kernel target cpu and features.
if LLVM.callconv(entry) == LLVM.API.LLVMAMDGPUKERNELCallConv
if entry.callconv == LLVM.CallConv.AMDGPUKERNEL
target_cpu_attr = StringAttribute("target-cpu", job.config.target.dev_isa)
target_features_attr = StringAttribute("target-features", job.config.target.features)
atomic_attr = StringAttribute("amdgpu-unsafe-fp-atomics", "true")
Expand All @@ -109,7 +109,7 @@ function GPUCompiler.finish_module!(
# grid dimensions are read from it (see device/gcn/indexing.jl),
implicitarg_attr = StringAttribute("amdgpu-implicitarg-num-bytes", "256")

attrs = LLVM.function_attributes(entry)
attrs = entry.function_attributes
push!(attrs, target_cpu_attr)
push!(attrs, target_features_attr)
push!(attrs, atomic_attr)
Expand All @@ -125,19 +125,17 @@ function GPUCompiler.finish_module!(
# And GPUCompiler fails to inline all functions without forcing
# always-inline attributes on them. Add them here.
target_fns = ("signal_exception", "report_exception", "malloc", "__throw_")
inline_attr = EnumAttribute("alwaysinline")
noinline_attr = EnumAttribute("noinline")

for fn in LLVM.functions(mod)
do_inline = any(occursin.(target_fns, LLVM.name(fn)))
for fn in mod.functions
do_inline = any(occursin.(target_fns, fn.name))
if job.config.params.unsafe_fp_atomics || do_inline
attrs = LLVM.function_attributes(fn)
attrs = fn.function_attributes

if do_inline && inline_attr ∉ collect(attrs)
if do_inline && !haskey(attrs, :alwaysinline)
# the two are mutually exclusive, and some matched functions are
# `@noinline` in Base (e.g. `_throw_boundserror_indices` on Julia 1.14)
delete!(attrs, noinline_attr)
push!(attrs, inline_attr)
delete!(attrs, :noinline)
push!(attrs, EnumAttribute(:alwaysinline))
end
end
end
Expand All @@ -146,16 +144,16 @@ function GPUCompiler.finish_module!(
# native hardware atomics (e.g. global_atomic_add_f32) instead of a CAS loop.
# Mirrors Clang's setTargetAtomicMetadata; unsafe_fp_atomics is the opt-in.
if job.config.params.unsafe_fp_atomics
fp_binops = (LLVM.API.LLVMAtomicRMWBinOpFAdd, LLVM.API.LLVMAtomicRMWBinOpFSub,
LLVM.API.LLVMAtomicRMWBinOpFMax, LLVM.API.LLVMAtomicRMWBinOpFMin)
fp_binops = (LLVM.AtomicRMWBinOp.FAdd, LLVM.AtomicRMWBinOp.FSub,
LLVM.AtomicRMWBinOp.FMax, LLVM.AtomicRMWBinOp.FMin)
empty_md = MDNode(Metadata[])
for fn in LLVM.functions(mod), bb in LLVM.blocks(fn), inst in LLVM.instructions(bb)
for fn in mod.functions, bb in fn.blocks, inst in bb.instructions
inst isa LLVM.AtomicRMWInst || continue
op = LLVM.binop(inst)
op = inst.binop
op ∈ fp_binops || continue
md = LLVM.metadata(inst)
md = inst.metadata
md["amdgpu.no.fine.grained.memory"] = empty_md
if op == LLVM.API.LLVMAtomicRMWBinOpFAdd && LLVM.value_type(inst) == LLVM.FloatType()
if op == LLVM.AtomicRMWBinOp.FAdd && inst.value_type isa LLVM.FloatType
md["amdgpu.ignore.denormal.mode"] = empty_md
end
end
Expand All @@ -167,12 +165,11 @@ end
# LLVM only folds `llvm.amdgcn.wavefrontsize` during instruction selection, which then
# fails on branches for the other wavefront size (e.g. in `ballot`), so fold it here.
function fold_wavefrontsize!(mod::LLVM.Module, wavefrontsize64::Bool)
haskey(LLVM.functions(mod), "llvm.amdgcn.wavefrontsize") || return
f = LLVM.functions(mod)["llvm.amdgcn.wavefrontsize"]
ws = ConstantInt(LLVM.return_type(LLVM.function_type(f)), wavefrontsize64 ? 64 : 32)
for use in collect(LLVM.uses(f))
call = LLVM.user(use)::LLVM.CallInst
LLVM.replace_uses!(call, ws)
f = get(mod.functions, "llvm.amdgcn.wavefrontsize", nothing)
f === nothing && return
ws = ConstantInt(f.function_type.return_type, wavefrontsize64 ? 64 : 32)
for call in collect(f.users)
LLVM.replace_uses!(call::LLVM.CallInst, ws)
LLVM.erase!(call)
end
return
Expand Down Expand Up @@ -370,23 +367,33 @@ function find_global_hostcalls(mod::LLVM.Module)
:malloc_hostcall, :free_hostcall, :print_hostcall, :printf_hostcall)

global_hostcalls = Symbol[]
for gbl in LLVM.globals(mod), gbl_name in global_hostcall_names
occursin("__$gbl_name", LLVM.name(gbl)) || continue
for gbl in mod.globals, gbl_name in global_hostcall_names
occursin("__$gbl_name", gbl.name) || continue
push!(global_hostcalls, gbl_name)
end
return global_hostcalls
end

function hipcompile(@nospecialize(job::CompilerJob))
obj, meta = JuliaContext() do ctx
GPUCompiler.compile(:obj, job)
# the IR in `meta` is ours: inspect it in here, and dispose of it so that it does not leak
obj, entry, late_hostcalls, extinit_globals, relocations = JuliaContext() do ctx
obj, meta = GPUCompiler.compile(:obj, job)
@dispose ir=meta.ir begin
# Filter out extinit global from `relocations` that :patch strategy emits.
relocated = Set(rec.name for rec in meta.relocations.records)
extinit_globals = [gv.name for gv in ir.globals
if gv.externally_initialized && gv.name ∉ relocated]

obj, meta.entry.name, find_global_hostcalls(ir), extinit_globals,
meta.relocations
end
end

# Collect early-detected hostcalls written by link_libraries! on this task.
# Falls back gracefully to empty if link_libraries! was not called.
global_hostcalls = pop!(task_local_storage(), :amdgpu_early_hostcalls, Symbol[])
# Late global hostcalls detection.
append!(global_hostcalls, find_global_hostcalls(meta.ir))
append!(global_hostcalls, late_hostcalls)

if !isempty(global_hostcalls)
@info """Global hostcalls detected!
Expand All @@ -398,14 +405,6 @@ function hipcompile(@nospecialize(job::CompilerJob))
"""
end

entry = LLVM.name(meta.entry)

# Filter out extinit global from `relocations` that :patch strategy emits.
relocations = meta.relocations
relocated = Set(rec.name for rec in relocations.records)
extinit_globals = filter(collect(LLVM.globals(meta.ir))) do gv
isextinit(gv) && LLVM.name(gv) ∉ relocated
end .|> LLVM.name
if !isempty(extinit_globals)
@warn """
HIP backend does not support setting extinit globals.
Expand Down Expand Up @@ -471,15 +470,14 @@ function GPUCompiler.finish_ir!(
job, mod, entry)
job.config.kernel || return entry

name = LLVM.name(entry)
tm = GPUCompiler.llvm_machine(job.config.target)
name = entry.name
# The textual pass name is only registered since LLVM 18; it's a pure
# optimization, so skip it on older LLVM (e.g. Julia 1.10's LLVM 15).
if LLVM.version() >= v"18"
@dispose pb=NewPMPassBuilder() begin
@dispose tm=GPUCompiler.llvm_machine(job.config.target) pb=PassBuilder() begin
add!(pb, "amdgpu-attributor")
run!(pb, mod, tm)
end
end
return functions(mod)[name]
return mod.functions[name]
end
38 changes: 14 additions & 24 deletions src/compiler/device_libs.jl
Original file line number Diff line number Diff line change
Expand Up @@ -61,22 +61,22 @@ Names of all symbols `mod` references but does not define.
"""
function undefined_symbols(mod::LLVM.Module)
undefined = Set{String}()
for f in LLVM.functions(mod)
LLVM.isdeclaration(f) && push!(undefined, LLVM.name(f))
for f in mod.functions
LLVM.isdeclaration(f) && push!(undefined, f.name)
end
for g in LLVM.globals(mod)
LLVM.isdeclaration(g) && push!(undefined, LLVM.name(g))
for g in mod.globals
LLVM.isdeclaration(g) && push!(undefined, g.name)
end
return undefined
end

function defined_symbols(mod::LLVM.Module)
defined = Set{String}()
for f in LLVM.functions(mod)
LLVM.isdeclaration(f) || push!(defined, LLVM.name(f))
for f in mod.functions
LLVM.isdeclaration(f) || push!(defined, f.name)
end
for g in LLVM.globals(mod)
LLVM.isdeclaration(g) || push!(defined, LLVM.name(g))
for g in mod.globals
LLVM.isdeclaration(g) || push!(defined, g.name)
end
return defined
end
Expand Down Expand Up @@ -140,30 +140,20 @@ function load_and_link!(
end
end

inline_attr = EnumAttribute("alwaysinline")
noinline_attr = EnumAttribute("noinline")

for f in LLVM.functions(lib)
fn_name = LLVM.name(f)
for f in lib.functions
fn_name = f.name

# FIXME: We should be able to inline this, that we can't means
# we are inserting calls to it late.
startswith(fn_name, "__ockl_hsa_signal") && continue

attrs = function_attributes(f)
inline = true
for attr in collect(attrs)
if kind(attr) == kind(noinline_attr)
inline = false
break
end
end
inline && push!(attrs, inline_attr)
attrs = f.function_attributes
haskey(attrs, :noinline) || push!(attrs, EnumAttribute(:alwaysinline))
end

# override triple and datalayout to avoid warnings
triple!(lib, triple(mod))
datalayout!(lib, datalayout(mod))
lib.triple = mod.triple
lib.datalayout = mod.datalayout
LLVM.link!(mod, lib)
return true
end
39 changes: 10 additions & 29 deletions src/compiler/zeroinit_lds.jl
Original file line number Diff line number Diff line change
@@ -1,31 +1,14 @@
# Calculate the size of an LLVM type.
llvmsize(::LLVM.LLVMHalf) = sizeof(Float16)
llvmsize(::LLVM.LLVMFloat) = sizeof(Float32)
llvmsize(::LLVM.LLVMDouble) = sizeof(Float64)
function llvmsize(::LLVM.IntegerType)
div(Int(intwidth(GenericValue(LLVM.Int128Type(), -1))), 8)
end

llvmsize(ty::LLVM.ArrayType) = length(ty) * llvmsize(eltype(ty))
llvmsize(ty::LLVM.StructType) = ispacked(ty) ?
sum(llvmsize(elem) for elem in elements(ty)) :
8 * length(elements(ty)) # FIXME: Properly determine non-packed sizing
llvmsize(ty::LLVM.PointerType) = div(Sys.WORD_SIZE, 8)
llvmsize(ty::LLVM.VectorType) = size(ty)
llvmsize(ty) = error("Unknown size for type: $ty, typeof: $(typeof(ty))")

function zeroinit_lds!(mod::LLVM.Module, entry::LLVM.Function)
if LLVM.callconv(entry) != LLVM.API.LLVMAMDGPUKERNELCallConv
if entry.callconv != LLVM.CallConv.AMDGPUKERNEL
return entry
end

to_init = []
for gbl in LLVM.globals(mod)
if startswith(LLVM.name(gbl), "__zeroinit")
vt = value_type(gbl)
as = LLVM.addrspace(vt)
for gbl in mod.globals
if startswith(gbl.name, "__zeroinit")
as = gbl.value_type.addrspace
if as == AMDGPU.Device.AS.Local
sz = llvmsize(global_value_type(gbl))
sz = LLVM.storage_size(mod.datalayout, gbl.global_value_type)
push!(to_init, (gbl, sz))
end
end
Expand All @@ -34,21 +17,19 @@ function zeroinit_lds!(mod::LLVM.Module, entry::LLVM.Function)

@dispose builder=IRBuilder() begin
# Make these the first operations we do.
block = first(LLVM.blocks(entry))
instruction = first(LLVM.instructions(block))
position!(builder, instruction)
instruction = first(entry.entry.instructions)
position!(builder, LLVM.before(instruction))

# Use memset to clear all values to 0.
for (gbl, sz) in to_init
sz == 0 && continue
LLVM.memset!(builder, gbl,
ConstantInt(UInt8(0)), ConstantInt(sz), LLVM.alignment(gbl))
ConstantInt(UInt8(0)), ConstantInt(sz), gbl.alignment)
end

# Synchronize the workgroup to prevent races.
sync_ft = LLVM.FunctionType(LLVM.VoidType())
sync_f = LLVM.Function(mod, LLVM.Intrinsic("llvm.amdgcn.s.barrier"))
call!(builder, sync_ft, sync_f)
sync_f = LLVM.Function(mod, Intrinsic("llvm.amdgcn.s.barrier"))
call!(builder, sync_f.function_type, sync_f)
end
return entry
end
3 changes: 1 addition & 2 deletions src/device/Device.jl
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ module Device

using ..BFloat16s
using ..GPUCompiler
using ..LLVM
using ..LLVM, ..LLVM.IR, ..LLVM.Build
using ..LLVM.Interop

import ..Adapt
Expand All @@ -19,7 +19,6 @@ import .AMDGPU: aligned_sizeof
import ..UnsafeAtomics

include("addrspaces.jl")
include("globals.jl")
include("strings.jl")
include("exceptions.jl")
include("gcn.jl")
Expand Down
Loading
Loading