diff --git a/Project.toml b/Project.toml index 8bc9180ef..ddf3cde39 100644 --- a/Project.toml +++ b/Project.toml @@ -63,7 +63,7 @@ GPUArrays = "11.5.14" GPUCompiler = "2.9" GPUToolbox = "3" KernelAbstractions = "0.9.2" -LLVM = "9" +LLVM = "10" LLVMDowngrader_jll = "0.11" Libdl = "1" LinearAlgebra = "1" diff --git a/src/AMDGPU.jl b/src/AMDGPU.jl index 636b3bf27..b609462a6 100644 --- a/src/AMDGPU.jl +++ b/src/AMDGPU.jl @@ -7,7 +7,7 @@ using GPUCompiler using GPUArrays using GPUArrays: allowscalar using Libdl -using LLVM, LLVM.Interop +using LLVM using Preferences using Printf diff --git a/src/compiler/Compiler.jl b/src/compiler/Compiler.jl index 89f391d38..3c5a0570c 100644 --- a/src/compiler/Compiler.jl +++ b/src/compiler/Compiler.jl @@ -3,7 +3,7 @@ module Compiler import Core: LLVMPtr using ..GPUCompiler -using ..LLVM +using ..LLVM, ..LLVM.IR, ..LLVM.Build, ..LLVM.Passes using Printf import GPUToolbox: @gcsafe_ccall diff --git a/src/compiler/codegen.jl b/src/compiler/codegen.jl index 120bf5246..a65ba4657 100644 --- a/src/compiler/codegen.jl +++ b/src/compiler/codegen.jl @@ -98,7 +98,7 @@ function GPUCompiler.finish_module!( fold_wavefrontsize!(mod, job.config.params.wavefrontsize64) # Set kernel target cpu and features. - if LLVM.callconv(entry) == LLVM.API.LLVMAMDGPUKERNELCallConv + if entry.callconv == LLVM.CallConv.AMDGPUKERNEL target_cpu_attr = StringAttribute("target-cpu", job.config.target.dev_isa) target_features_attr = StringAttribute("target-features", job.config.target.features) atomic_attr = StringAttribute("amdgpu-unsafe-fp-atomics", "true") @@ -109,7 +109,7 @@ function GPUCompiler.finish_module!( # grid dimensions are read from it (see device/gcn/indexing.jl), implicitarg_attr = StringAttribute("amdgpu-implicitarg-num-bytes", "256") - attrs = LLVM.function_attributes(entry) + attrs = entry.function_attributes push!(attrs, target_cpu_attr) push!(attrs, target_features_attr) push!(attrs, atomic_attr) @@ -125,19 +125,17 @@ function GPUCompiler.finish_module!( # And GPUCompiler fails to inline all functions without forcing # always-inline attributes on them. Add them here. target_fns = ("signal_exception", "report_exception", "malloc", "__throw_") - inline_attr = EnumAttribute("alwaysinline") - noinline_attr = EnumAttribute("noinline") - for fn in LLVM.functions(mod) - do_inline = any(occursin.(target_fns, LLVM.name(fn))) + for fn in mod.functions + do_inline = any(occursin.(target_fns, fn.name)) if job.config.params.unsafe_fp_atomics || do_inline - attrs = LLVM.function_attributes(fn) + attrs = fn.function_attributes - if do_inline && inline_attr ∉ collect(attrs) + if do_inline && !haskey(attrs, :alwaysinline) # the two are mutually exclusive, and some matched functions are # `@noinline` in Base (e.g. `_throw_boundserror_indices` on Julia 1.14) - delete!(attrs, noinline_attr) - push!(attrs, inline_attr) + delete!(attrs, :noinline) + push!(attrs, EnumAttribute(:alwaysinline)) end end end @@ -146,16 +144,16 @@ function GPUCompiler.finish_module!( # native hardware atomics (e.g. global_atomic_add_f32) instead of a CAS loop. # Mirrors Clang's setTargetAtomicMetadata; unsafe_fp_atomics is the opt-in. if job.config.params.unsafe_fp_atomics - fp_binops = (LLVM.API.LLVMAtomicRMWBinOpFAdd, LLVM.API.LLVMAtomicRMWBinOpFSub, - LLVM.API.LLVMAtomicRMWBinOpFMax, LLVM.API.LLVMAtomicRMWBinOpFMin) + fp_binops = (LLVM.AtomicRMWBinOp.FAdd, LLVM.AtomicRMWBinOp.FSub, + LLVM.AtomicRMWBinOp.FMax, LLVM.AtomicRMWBinOp.FMin) empty_md = MDNode(Metadata[]) - for fn in LLVM.functions(mod), bb in LLVM.blocks(fn), inst in LLVM.instructions(bb) + for fn in mod.functions, bb in fn.blocks, inst in bb.instructions inst isa LLVM.AtomicRMWInst || continue - op = LLVM.binop(inst) + op = inst.binop op ∈ fp_binops || continue - md = LLVM.metadata(inst) + md = inst.metadata md["amdgpu.no.fine.grained.memory"] = empty_md - if op == LLVM.API.LLVMAtomicRMWBinOpFAdd && LLVM.value_type(inst) == LLVM.FloatType() + if op == LLVM.AtomicRMWBinOp.FAdd && inst.value_type isa LLVM.FloatType md["amdgpu.ignore.denormal.mode"] = empty_md end end @@ -167,12 +165,11 @@ end # LLVM only folds `llvm.amdgcn.wavefrontsize` during instruction selection, which then # fails on branches for the other wavefront size (e.g. in `ballot`), so fold it here. function fold_wavefrontsize!(mod::LLVM.Module, wavefrontsize64::Bool) - haskey(LLVM.functions(mod), "llvm.amdgcn.wavefrontsize") || return - f = LLVM.functions(mod)["llvm.amdgcn.wavefrontsize"] - ws = ConstantInt(LLVM.return_type(LLVM.function_type(f)), wavefrontsize64 ? 64 : 32) - for use in collect(LLVM.uses(f)) - call = LLVM.user(use)::LLVM.CallInst - LLVM.replace_uses!(call, ws) + f = get(mod.functions, "llvm.amdgcn.wavefrontsize", nothing) + f === nothing && return + ws = ConstantInt(f.function_type.return_type, wavefrontsize64 ? 64 : 32) + for call in collect(f.users) + LLVM.replace_uses!(call::LLVM.CallInst, ws) LLVM.erase!(call) end return @@ -370,23 +367,33 @@ function find_global_hostcalls(mod::LLVM.Module) :malloc_hostcall, :free_hostcall, :print_hostcall, :printf_hostcall) global_hostcalls = Symbol[] - for gbl in LLVM.globals(mod), gbl_name in global_hostcall_names - occursin("__$gbl_name", LLVM.name(gbl)) || continue + for gbl in mod.globals, gbl_name in global_hostcall_names + occursin("__$gbl_name", gbl.name) || continue push!(global_hostcalls, gbl_name) end return global_hostcalls end function hipcompile(@nospecialize(job::CompilerJob)) - obj, meta = JuliaContext() do ctx - GPUCompiler.compile(:obj, job) + # the IR in `meta` is ours: inspect it in here, and dispose of it so that it does not leak + obj, entry, late_hostcalls, extinit_globals, relocations = JuliaContext() do ctx + obj, meta = GPUCompiler.compile(:obj, job) + @dispose ir=meta.ir begin + # Filter out extinit global from `relocations` that :patch strategy emits. + relocated = Set(rec.name for rec in meta.relocations.records) + extinit_globals = [gv.name for gv in ir.globals + if gv.externally_initialized && gv.name ∉ relocated] + + obj, meta.entry.name, find_global_hostcalls(ir), extinit_globals, + meta.relocations + end end # Collect early-detected hostcalls written by link_libraries! on this task. # Falls back gracefully to empty if link_libraries! was not called. global_hostcalls = pop!(task_local_storage(), :amdgpu_early_hostcalls, Symbol[]) # Late global hostcalls detection. - append!(global_hostcalls, find_global_hostcalls(meta.ir)) + append!(global_hostcalls, late_hostcalls) if !isempty(global_hostcalls) @info """Global hostcalls detected! @@ -398,14 +405,6 @@ function hipcompile(@nospecialize(job::CompilerJob)) """ end - entry = LLVM.name(meta.entry) - - # Filter out extinit global from `relocations` that :patch strategy emits. - relocations = meta.relocations - relocated = Set(rec.name for rec in relocations.records) - extinit_globals = filter(collect(LLVM.globals(meta.ir))) do gv - isextinit(gv) && LLVM.name(gv) ∉ relocated - end .|> LLVM.name if !isempty(extinit_globals) @warn """ HIP backend does not support setting extinit globals. @@ -471,15 +470,14 @@ function GPUCompiler.finish_ir!( job, mod, entry) job.config.kernel || return entry - name = LLVM.name(entry) - tm = GPUCompiler.llvm_machine(job.config.target) + name = entry.name # The textual pass name is only registered since LLVM 18; it's a pure # optimization, so skip it on older LLVM (e.g. Julia 1.10's LLVM 15). if LLVM.version() >= v"18" - @dispose pb=NewPMPassBuilder() begin + @dispose tm=GPUCompiler.llvm_machine(job.config.target) pb=PassBuilder() begin add!(pb, "amdgpu-attributor") run!(pb, mod, tm) end end - return functions(mod)[name] + return mod.functions[name] end diff --git a/src/compiler/device_libs.jl b/src/compiler/device_libs.jl index 49dcd9c58..af5594bd6 100644 --- a/src/compiler/device_libs.jl +++ b/src/compiler/device_libs.jl @@ -61,22 +61,22 @@ Names of all symbols `mod` references but does not define. """ function undefined_symbols(mod::LLVM.Module) undefined = Set{String}() - for f in LLVM.functions(mod) - LLVM.isdeclaration(f) && push!(undefined, LLVM.name(f)) + for f in mod.functions + LLVM.isdeclaration(f) && push!(undefined, f.name) end - for g in LLVM.globals(mod) - LLVM.isdeclaration(g) && push!(undefined, LLVM.name(g)) + for g in mod.globals + LLVM.isdeclaration(g) && push!(undefined, g.name) end return undefined end function defined_symbols(mod::LLVM.Module) defined = Set{String}() - for f in LLVM.functions(mod) - LLVM.isdeclaration(f) || push!(defined, LLVM.name(f)) + for f in mod.functions + LLVM.isdeclaration(f) || push!(defined, f.name) end - for g in LLVM.globals(mod) - LLVM.isdeclaration(g) || push!(defined, LLVM.name(g)) + for g in mod.globals + LLVM.isdeclaration(g) || push!(defined, g.name) end return defined end @@ -140,30 +140,20 @@ function load_and_link!( end end - inline_attr = EnumAttribute("alwaysinline") - noinline_attr = EnumAttribute("noinline") - - for f in LLVM.functions(lib) - fn_name = LLVM.name(f) + for f in lib.functions + fn_name = f.name # FIXME: We should be able to inline this, that we can't means # we are inserting calls to it late. startswith(fn_name, "__ockl_hsa_signal") && continue - attrs = function_attributes(f) - inline = true - for attr in collect(attrs) - if kind(attr) == kind(noinline_attr) - inline = false - break - end - end - inline && push!(attrs, inline_attr) + attrs = f.function_attributes + haskey(attrs, :noinline) || push!(attrs, EnumAttribute(:alwaysinline)) end # override triple and datalayout to avoid warnings - triple!(lib, triple(mod)) - datalayout!(lib, datalayout(mod)) + lib.triple = mod.triple + lib.datalayout = mod.datalayout LLVM.link!(mod, lib) return true end diff --git a/src/compiler/zeroinit_lds.jl b/src/compiler/zeroinit_lds.jl index 9aa712a56..66066696a 100644 --- a/src/compiler/zeroinit_lds.jl +++ b/src/compiler/zeroinit_lds.jl @@ -1,31 +1,14 @@ -# Calculate the size of an LLVM type. -llvmsize(::LLVM.LLVMHalf) = sizeof(Float16) -llvmsize(::LLVM.LLVMFloat) = sizeof(Float32) -llvmsize(::LLVM.LLVMDouble) = sizeof(Float64) -function llvmsize(::LLVM.IntegerType) - div(Int(intwidth(GenericValue(LLVM.Int128Type(), -1))), 8) -end - -llvmsize(ty::LLVM.ArrayType) = length(ty) * llvmsize(eltype(ty)) -llvmsize(ty::LLVM.StructType) = ispacked(ty) ? - sum(llvmsize(elem) for elem in elements(ty)) : - 8 * length(elements(ty)) # FIXME: Properly determine non-packed sizing -llvmsize(ty::LLVM.PointerType) = div(Sys.WORD_SIZE, 8) -llvmsize(ty::LLVM.VectorType) = size(ty) -llvmsize(ty) = error("Unknown size for type: $ty, typeof: $(typeof(ty))") - function zeroinit_lds!(mod::LLVM.Module, entry::LLVM.Function) - if LLVM.callconv(entry) != LLVM.API.LLVMAMDGPUKERNELCallConv + if entry.callconv != LLVM.CallConv.AMDGPUKERNEL return entry end to_init = [] - for gbl in LLVM.globals(mod) - if startswith(LLVM.name(gbl), "__zeroinit") - vt = value_type(gbl) - as = LLVM.addrspace(vt) + for gbl in mod.globals + if startswith(gbl.name, "__zeroinit") + as = gbl.value_type.addrspace if as == AMDGPU.Device.AS.Local - sz = llvmsize(global_value_type(gbl)) + sz = LLVM.storage_size(mod.datalayout, gbl.global_value_type) push!(to_init, (gbl, sz)) end end @@ -34,21 +17,19 @@ function zeroinit_lds!(mod::LLVM.Module, entry::LLVM.Function) @dispose builder=IRBuilder() begin # Make these the first operations we do. - block = first(LLVM.blocks(entry)) - instruction = first(LLVM.instructions(block)) - position!(builder, instruction) + instruction = first(entry.entry.instructions) + position!(builder, LLVM.before(instruction)) # Use memset to clear all values to 0. for (gbl, sz) in to_init sz == 0 && continue LLVM.memset!(builder, gbl, - ConstantInt(UInt8(0)), ConstantInt(sz), LLVM.alignment(gbl)) + ConstantInt(UInt8(0)), ConstantInt(sz), gbl.alignment) end # Synchronize the workgroup to prevent races. - sync_ft = LLVM.FunctionType(LLVM.VoidType()) - sync_f = LLVM.Function(mod, LLVM.Intrinsic("llvm.amdgcn.s.barrier")) - call!(builder, sync_ft, sync_f) + sync_f = LLVM.Function(mod, Intrinsic("llvm.amdgcn.s.barrier")) + call!(builder, sync_f.function_type, sync_f) end return entry end diff --git a/src/device/Device.jl b/src/device/Device.jl index 3ac9463ac..c797e36a9 100644 --- a/src/device/Device.jl +++ b/src/device/Device.jl @@ -2,7 +2,7 @@ module Device using ..BFloat16s using ..GPUCompiler -using ..LLVM +using ..LLVM, ..LLVM.IR, ..LLVM.Build using ..LLVM.Interop import ..Adapt @@ -19,7 +19,6 @@ import .AMDGPU: aligned_sizeof import ..UnsafeAtomics include("addrspaces.jl") -include("globals.jl") include("strings.jl") include("exceptions.jl") include("gcn.jl") diff --git a/src/device/gcn/indexing.jl b/src/device/gcn/indexing.jl index d6ec22292..d4d26c872 100644 --- a/src/device/gcn/indexing.jl +++ b/src/device/gcn/indexing.jl @@ -8,32 +8,15 @@ function _range_metadata(::Type{T}, range) where T return MDNode([ConstantInt(lo), ConstantInt(hi)]) end -@device_function @generated function _index(::Val{fname}, ::Val{name}, ::Val{range}) where {fname, name, range} - @dispose ctx=Context() begin - T_int32 = LLVM.Int32Type() - - # create function - llvm_f, _ = create_function(T_int32) - mod = LLVM.parent(llvm_f) - - # generate IR - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - # call the indexing intrinsic - intr_typ = LLVM.FunctionType(T_int32) - intr = LLVM.Function(mod, "llvm.amdgcn.$fname.id.$name", intr_typ) - idx = call!(builder, intr_typ, intr) - - # attach range metadata - md = _range_metadata(UInt32, range) - md === nothing || (metadata(idx)[LLVM.MD_range] = md) - ret!(builder, idx) - end - - call_function(llvm_f, UInt32) - end +@device_function @llvmgenerated builder function _index(::Val{fname}, ::Val{name}, ::Val{range})::UInt32 where {fname, name, range} + # call the indexing intrinsic + intr = LLVM.Function(current_module(builder), Intrinsic("llvm.amdgcn.$fname.id.$name")) + idx = call!(builder, intr.function_type, intr) + + # attach range metadata + md = _range_metadata(UInt32, range) + md === nothing || (idx.metadata[MD_range] = md) + idx end # Workgroup/grid dimensions come from the *hidden kernel arguments* (code object @@ -43,50 +26,31 @@ end const _hidden_block_count_offset = 0 # u32 × 3 const _hidden_group_size_offset = 12 # u16 × 3 -@device_function @generated function _dim(::Val{offset}, ::Type{T}, ::Val{range}) where {offset, T, range} - @dispose ctx=Context() begin - T_int8 = LLVM.Int8Type() - T_int32 = LLVM.Int32Type() - - _as = convert(Int, AS.Constant) - T_ptr_i8 = LLVM.PointerType(T_int8, _as) - - T_T = convert(LLVMType, T) - T_ptr_T = LLVM.PointerType(T_T, _as) - - # create function - llvm_f, _ = create_function(T_int32) - mod = LLVM.parent(llvm_f) - - # generate IR - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - # get the implicit (hidden) kernel argument pointer - intr_typ = LLVM.FunctionType(T_ptr_i8) - intr = LLVM.Function(mod, "llvm.amdgcn.implicitarg.ptr", intr_typ) - ptr = call!(builder, intr_typ, intr) - - # load the field - idx_ptr_i8 = inbounds_gep!(builder, T_int8, ptr, [ConstantInt(offset)]) - idx_ptr_T = bitcast!(builder, idx_ptr_i8, T_ptr_T) - idx_T = load!(builder, T_T, idx_ptr_T) - # the hidden block is at least 4-byte aligned; tell LLVM so the - # backend keeps merged accesses SMEM-selectable - alignment!(idx_T, gcd(4, offset)) - idx = zext!(builder, idx_T, T_int32) - - # attach range metadata; the hidden arguments never change during - # a dispatch - md = _range_metadata(T, range) - md === nothing || (metadata(idx_T)[LLVM.MD_range] = md) - metadata(idx_T)[LLVM.MD_invariant_load] = MDNode(LLVM.Metadata[]) - ret!(builder, idx) - end - - call_function(llvm_f, UInt32) - end +@device_function @llvmgenerated builder function _dim(::Val{offset}, ::Type{T}, ::Val{range})::UInt32 where {offset, T, range} + T_int8 = LLVM.Int8Type() + T_int32 = LLVM.Int32Type() + + T_T = convert(LLVMType, T) + T_ptr_T = LLVM.PointerType(T_T, convert(Int, AS.Constant)) + + # get the implicit (hidden) kernel argument pointer + intr = LLVM.Function(current_module(builder), Intrinsic("llvm.amdgcn.implicitarg.ptr")) + ptr = call!(builder, intr.function_type, intr) + + # load the field + idx_ptr_i8 = inbounds_gep!(builder, T_int8, ptr, [ConstantInt(offset)]) + idx_ptr_T = bitcast!(builder, idx_ptr_i8, T_ptr_T) + # the hidden block is at least 4-byte aligned; tell LLVM so the + # backend keeps merged accesses SMEM-selectable + idx_T = load!(builder, T_T, idx_ptr_T; align=gcd(4, offset)) + idx = zext!(builder, idx_T, T_int32) + + # attach range metadata; the hidden arguments never change during + # a dispatch + md = _range_metadata(T, range) + md === nothing || (idx_T.metadata[MD_range] = md) + idx_T.metadata[MD_invariant_load] = MDNode(Metadata[]) + idx end # TODO: look these up for the current device/queue diff --git a/src/device/gcn/memory_static.jl b/src/device/gcn/memory_static.jl index b57937911..4962630f6 100644 --- a/src/device/gcn/memory_static.jl +++ b/src/device/gcn/memory_static.jl @@ -1,54 +1,41 @@ "Allocates on-device memory statically from the specified address space." -@generated function alloc_special( +@llvmgenerated builder function alloc_special( ::Val{id}, ::Type{T}, ::Val{as}, ::Val{len}, ::Val{zeroinit} = Val{false}(), -) where {id,T,as,len,zeroinit} - @dispose ctx=Context() begin - eltyp = convert(LLVMType, T) +)::LLVMPtr{T,as} where {id,T,as,len,zeroinit} + eltyp = convert(LLVMType, T) - # old versions of GPUArrays invoke _shmem with an integer id; make sure those are unique - if !isa(id, String) || !isa(id, Symbol) - id = "alloc_special_$id" - end - if zeroinit - id = "__zeroinit_" * id - end - - T_ptr_i8 = convert(LLVMType, LLVMPtr{T,as}) - - # create a function - llvm_f, _ = create_function(T_ptr_i8) + # old versions of GPUArrays invoke _shmem with an integer id; make sure those are unique + name = id + if !isa(id, String) || !isa(id, Symbol) + name = "alloc_special_$id" + end + if zeroinit + name = "__zeroinit_" * name + end - # create the global variable - mod = LLVM.parent(llvm_f) - gv_typ = LLVM.ArrayType(eltyp, len) - gv = GlobalVariable(mod, gv_typ, string(id), as) - if len > 0 - if as == AS.Local - linkage!(gv, LLVM.API.LLVMExternalLinkage) - # NOTE: Backend doesn't support initializer for local AS - elseif as == AS.Private - linkage!(gv, LLVM.API.LLVMInternalLinkage) - initializer!(gv, null(gv_typ)) - end + T_ptr_i8 = convert(LLVMType, LLVMPtr{T,as}) + + # create the global variable + gv_typ = LLVM.ArrayType(eltyp, len) + gv = GlobalVariable(current_module(builder), gv_typ, name, as) + if len > 0 + if as == AS.Local + gv.linkage = LLVM.Linkage.External + # NOTE: Backend doesn't support initializer for local AS + elseif as == AS.Private + gv.linkage = LLVM.Linkage.Internal + gv.initializer = null(gv_typ) end + end - # By requesting a larger-than-datatype alignment, - # we might be able to vectorize. - # TODO: Make the alignment configurable - alignment!(gv, Base.max(32, Base.datatype_alignment(T))) - - # generate IR - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - ptr_with_as = gep!(builder, gv_typ, gv, [ConstantInt(0), ConstantInt(0)]) - ptr = bitcast!(builder, ptr_with_as, T_ptr_i8) - ret!(builder, ptr) - end + # By requesting a larger-than-datatype alignment, + # we might be able to vectorize. + # TODO: Make the alignment configurable + gv.alignment = Base.max(32, Base.datatype_alignment(T)) - call_function(llvm_f, LLVMPtr{T,as}) - end + # generate IR + ptr_with_as = gep!(builder, gv_typ, gv, [ConstantInt(0), ConstantInt(0)]) + bitcast!(builder, ptr_with_as, T_ptr_i8) end @inline alloc_local(id, T, len, zeroinit=false) = @@ -95,53 +82,16 @@ macro ROCDynamicLocalArray(T, dims, zeroinit=true, offset=0) end # TODO: Support various types of len -@inline @generated function memcpy!(dest_ptr::LLVMPtr{UInt8,DestAS}, src_ptr::LLVMPtr{UInt8,SrcAS}, len::LT) where {DestAS,SrcAS,LT<:Union{Int64,UInt64}} - @dispose ctx=Context() begin - T_nothing = LLVM.VoidType() - T_pint8_dest = convert(LLVMType, dest_ptr) - T_pint8_src = convert(LLVMType, src_ptr) - T_int64 = convert(LLVMType, len) - T_int1 = LLVM.Int1Type() - - llvm_f, _ = create_function(T_nothing, [T_pint8_dest, T_pint8_src, T_int64]) - mod = LLVM.parent(llvm_f) - T_intr = LLVM.FunctionType(T_nothing, [T_pint8_dest, T_pint8_src, T_int64, T_int1]) - intr = LLVM.Function(mod, "llvm.memcpy.p$(DestAS)i8.p$(SrcAS)i8.i64", T_intr) - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - dest_ptr_i8 = parameters(llvm_f)[1] - src_ptr_i8 = parameters(llvm_f)[2] - call!(builder, T_intr, intr, [dest_ptr_i8, src_ptr_i8, parameters(llvm_f)[3], ConstantInt(T_int1, 0)]) - ret!(builder) - end - call_function(llvm_f, Nothing, Tuple{LLVMPtr{UInt8,DestAS},LLVMPtr{UInt8,SrcAS},LT}, :dest_ptr, :src_ptr, :len) - end +# NOTE: these shadow LLVM.Build's `memcpy!` and `memset!`, so use those qualified. +@llvmgenerated builder function memcpy!(dest_ptr::LLVMPtr{UInt8,DestAS}, src_ptr::LLVMPtr{UInt8,SrcAS}, len::LT)::Nothing where {DestAS,SrcAS,LT<:Union{Int64,UInt64}} + LLVM.memcpy!(builder, dest_ptr, src_ptr, len) + nothing end memcpy!(dest_ptr::LLVMPtr{T,DestAS}, src_ptr::LLVMPtr{T,SrcAS}, len::Integer) where {T,DestAS,SrcAS} = memcpy!(reinterpret(LLVMPtr{UInt8,DestAS}, dest_ptr), reinterpret(LLVMPtr{UInt8,SrcAS}, src_ptr), UInt64(len)) -@inline @generated function memset!(dest_ptr::LLVMPtr{UInt8,DestAS}, value::UInt8, len::LT) where {DestAS,LT<:Union{Int64,UInt64}} - @dispose ctx=Context() begin - T_nothing = LLVM.VoidType() - T_pint8_dest = convert(LLVMType, dest_ptr) - T_int8 = convert(LLVMType, value) - T_int64 = convert(LLVMType, len) - T_int1 = LLVM.Int1Type() - - llvm_f, _ = create_function(T_nothing, [T_pint8_dest, T_int8, T_int64]) - mod = LLVM.parent(llvm_f) - T_intr = LLVM.FunctionType(T_nothing, [T_pint8_dest, T_int8, T_int64, T_int1]) - intr = LLVM.Function(mod, "llvm.memset.p$(DestAS)i8.i64", T_intr) - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - call!(builder, T_intr, intr, [parameters(llvm_f)[1], parameters(llvm_f)[2], parameters(llvm_f)[3], ConstantInt(T_int1, 0)]) - ret!(builder) - end - call_function(llvm_f, Nothing, Tuple{LLVMPtr{UInt8,DestAS},UInt8,LT}, :dest_ptr, :value, :len) - end +@llvmgenerated builder function memset!(dest_ptr::LLVMPtr{UInt8,DestAS}, value::UInt8, len::LT)::Nothing where {DestAS,LT<:Union{Int64,UInt64}} + LLVM.memset!(builder, dest_ptr, value, len) + nothing end memset!(dest_ptr::LLVMPtr{T,DestAS}, value::UInt8, len::Integer) where {T,DestAS} = memset!(convert(LLVMPtr{UInt8,DestAS}, dest_ptr), value, UInt64(len)) diff --git a/src/device/gcn/synchronization.jl b/src/device/gcn/synchronization.jl index 6d8b682dc..8378b3707 100644 --- a/src/device/gcn/synchronization.jl +++ b/src/device/gcn/synchronization.jl @@ -1,17 +1,15 @@ +@llvmgenerated builder function _fence(::Val{ordering}, ::Val{scope})::Nothing where {ordering, scope} + fence!(builder, parse(LLVM.AtomicOrdering.T, String(ordering)); scope=String(scope)) + nothing +end + for ord in UnsafeAtomics.Internal.orderings for sync in (AMDGPU.syncscope_agent, AMDGPU.syncscope_workgroup) - @eval @device_function function UnsafeAtomics.fence(::$(typeof(ord)), ::$(typeof(sync))) - Base.llvmcall( - $(""" - define void @fence() #0 { - entry: - fence $sync $ord - ret void - } - attributes #0 = { alwaysinline } - """, "fence"), Nothing, Tuple{}) - end + ordering = Val(UnsafeAtomics.Internal.llvm_ordering(ord)) + scope = Val(UnsafeAtomics.Internal.llvm_syncscope(sync)) + @eval @device_function UnsafeAtomics.fence(::$(typeof(ord)), ::$(typeof(sync))) = + _fence($ordering, $scope) end end diff --git a/src/device/globals.jl b/src/device/globals.jl deleted file mode 100644 index 125489965..000000000 --- a/src/device/globals.jl +++ /dev/null @@ -1,44 +0,0 @@ -# copied from CUDAnative PR 422 - -# Gets a pointer to a global with a particular name. If the global -# does not exist yet, then it is declared in the global memory address -# space. -@inline @generated function get_global_pointer(::Val{global_name}, ::Type{T})::LLVMPtr{T} where {global_name, T} - @dispose ctx=Context() begin - T_global = convert(LLVMType, T) - T_result = convert(LLVMType, Ptr{T}) - - # Create a thunk that computes a pointer to the global. - llvm_f, _ = create_function(T_result) - mod = LLVM.parent(llvm_f) - - # Figure out if the global has been defined already. - global_set = LLVM.globals(mod) - global_name_string = String(global_name) - if haskey(global_set, global_name_string) - global_var = global_set[global_name_string] - else - # If the global hasn't been defined already, then we'll define - # it in the global address space, i.e., address space one. - global_var = GlobalVariable(mod, T_global, global_name_string, 1) - linkage!(global_var, LLVM.API.LLVMExternalLinkage) - extinit!(global_var, true) - #set_used!(mod, global_var) - end - - # Generate IR that computes the global's address. - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - # Cast the global variable's type to the result type. - result = ptrtoint!(builder, global_var, T_result) - ret!(builder, result) - end - - # Call the function. - quote - $(LLVMPtr{T, AS.Global})($(call_function(llvm_f, Ptr{T}))) - end - end -end diff --git a/src/device/random.jl b/src/device/random.jl index cce7d1cbf..5f8d7749b 100644 --- a/src/device/random.jl +++ b/src/device/random.jl @@ -9,37 +9,19 @@ import RandomNumbers # global state -@inline @generated function emit_global_random_values(::Val{name}) where name - @dispose ctx=Context() begin - T_val = convert(LLVMType, UInt32) - T_ptr = convert(LLVMType, LLVMPtr{UInt32,AS.Local}) - - # define function and get LLVM module - llvm_f, _ = create_function(T_ptr) - mod = LLVM.parent(llvm_f) - - # create a global memory global variable - T_global = LLVM.ArrayType(T_val, 32) - gv = GlobalVariable(mod, T_global, "__zeroinit_global_random_$(name)", AS.Local) - linkage!(gv, LLVM.API.LLVMExternalLinkage) - - # TODO: we need alwaysinline to ensure we don't access LDS in a non-kernel function - push!(function_attributes(llvm_f), EnumAttribute("alwaysinline")) - - # generate IR - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) - - untyped_ptr = bitcast!(builder, ptr, T_ptr) - - ret!(builder, untyped_ptr) - end - - call_function(llvm_f, LLVMPtr{UInt32,AS.Local}) - end +# NOTE: the generated function is `alwaysinline`, which ensures we don't access LDS in a +# non-kernel function +@llvmgenerated builder function emit_global_random_values(::Val{name})::LLVMPtr{UInt32,AS.Local} where name + T_val = convert(LLVMType, UInt32) + T_ptr = convert(LLVMType, LLVMPtr{UInt32,AS.Local}) + + # create a global memory global variable + T_global = LLVM.ArrayType(T_val, 32) + gv = GlobalVariable(current_module(builder), T_global, "__zeroinit_global_random_$(name)", AS.Local) + gv.linkage = LLVM.Linkage.External + + ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) + bitcast!(builder, ptr, T_ptr) end # shared memory with the actual seed, per warp, loaded lazily or overridden by calling `seed!` @@ -188,48 +170,30 @@ end # copied from Base because we don't support its global tables # a hacky method of exposing constant tables as constant GPU memory -function emit_constant_array(name::Symbol, data::AbstractArray{T}) where {T} - @dispose ctx=Context() begin - T_val = convert(LLVMType, T) - T_ptr = convert(LLVMType, LLVMPtr{T,AS.Constant}) - - # define function and get LLVM module - llvm_f, _ = create_function(T_ptr) - mod = LLVM.parent(llvm_f) - - # create a global memory global variable - # TODO: global_var alignment? - T_global = LLVM.ArrayType(T_val, length(data)) - # XXX: why can't we use a single name like emit_shmem - gv = GlobalVariable(mod, T_global, "gpu_$(name)_data", AS.Constant) - alignment!(gv, 16) - linkage!(gv, LLVM.API.LLVMInternalLinkage) - initializer!(gv, ConstantArray(data)) - - # generate IR - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) - - untyped_ptr = bitcast!(builder, ptr, T_ptr) - - ret!(builder, untyped_ptr) - end - - call_function(llvm_f, LLVMPtr{T,AS.Constant}) - end +@llvmgenerated builder function emit_constant_array(::Val{name}, ::Type{T})::LLVMPtr{T,AS.Constant} where {name, T} + data = getfield(Random, name)::AbstractArray{T} + T_val = convert(LLVMType, T) + T_ptr = convert(LLVMType, LLVMPtr{T,AS.Constant}) + + # create a global memory global variable + # TODO: global_var alignment? + T_global = LLVM.ArrayType(T_val, length(data)) + # XXX: why can't we use a single name like emit_shmem + gv = GlobalVariable(current_module(builder), T_global, "gpu_$(name)_data", AS.Constant) + gv.alignment = 16 + gv.linkage = LLVM.Linkage.Internal + gv.initializer = ConstantArray(data) + + ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) + bitcast!(builder, ptr, T_ptr) end for var in [:ki, :wi, :fi, :ke, :we, :fe] val = getfield(Random, var) gpu_var = Symbol("gpu_$var") arr_typ = :(ROCDeviceArray{$(eltype(val)),$(ndims(val)),AS.Constant}) - @eval @inline @generated function $gpu_var() - ptr = emit_constant_array($(QuoteNode(var)), $val) - Expr(:call, $arr_typ, $(size(val)), ptr) - end + @eval @inline $gpu_var() = + $arr_typ($(size(val)), emit_constant_array($(Val(var)), $(eltype(val)))) end ## randn diff --git a/src/device/runtime.jl b/src/device/runtime.jl index 0084d40a2..ebcb9b748 100644 --- a/src/device/runtime.jl +++ b/src/device/runtime.jl @@ -4,31 +4,12 @@ using Core: LLVMPtr @inline @generated kernel_state() = GPUCompiler.kernel_state_value(AMDGPU.KernelState) -@generated function llvm_atomic_cas(ptr::LLVMPtr{T,A}, cmp::T, val::T) where {T, A} - @dispose ctx=Context() begin - T_val = convert(LLVMType, T) - T_ptr = convert(LLVMType, ptr) - - T_typed_ptr = LLVM.PointerType(T_val, A) - llvm_f, _ = create_function(T_val, [T_ptr, T_val, T_val]) - - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - typed_ptr = bitcast!(builder, parameters(llvm_f)[1], T_typed_ptr) - res = atomic_cmpxchg!( - builder, typed_ptr, parameters(llvm_f)[2], - parameters(llvm_f)[3], - LLVM.API.LLVMAtomicOrderingAcquireRelease, - LLVM.API.LLVMAtomicOrderingAcquire, - #=single threaded=# false) - - rv = extract_value!(builder, res, 0) - ret!(builder, rv) - end - call_function(llvm_f, T, Tuple{LLVMPtr{T,A}, T, T}, :ptr, :cmp, :val) - end +@llvmgenerated builder function llvm_atomic_cas(ptr::LLVMPtr{T,A}, cmp::T, val::T)::T where {T, A} + T_typed_ptr = LLVM.PointerType(convert(LLVMType, T), A) + typed_ptr = bitcast!(builder, ptr, T_typed_ptr) + res = atomic_cmpxchg!(builder, typed_ptr, cmp, val, + LLVM.AtomicOrdering.AcquireRelease, LLVM.AtomicOrdering.Acquire) + extract_value!(builder, res, 0) end function output_context() diff --git a/src/device/strings.jl b/src/device/strings.jl index af71e2429..7ddbc2c54 100644 --- a/src/device/strings.jl +++ b/src/device/strings.jl @@ -1,66 +1,7 @@ ## Device-side string utilities -@inline @generated function alloc_string(::Val{sym}) where sym - str = String(sym) - @dispose ctx=Context() begin - T_pint8 = LLVM.PointerType(LLVM.Int8Type(), AS.Global) - llvm_f, _ = create_function(T_pint8) - - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - str_ptr = globalstring_ptr!(builder, str) - ptr = addrspacecast!(builder, str_ptr, T_pint8) - ret!(builder, ptr) - end - call_function(llvm_f, LLVMPtr{UInt8, AS.Global}) - end -end - -@generated function string_length(ex::Union{Ptr,LLVMPtr}) - @dispose ctx=Context() begin - T_ex = convert(LLVMType, ex) - T_ex_ptr = LLVM.PointerType(T_ex) - T_i8 = LLVM.Int8Type() - T_i8_ptr = LLVM.PointerType(T_i8) - T_i64 = LLVM.Int64Type() - llvm_f, _ = create_function(T_i64, [T_ex]) - mod = LLVM.parent(llvm_f) - - @dispose builder=IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - check = BasicBlock(llvm_f, "check") - done = BasicBlock(llvm_f, "done") - - position!(builder, entry) - init_offset = ConstantInt(0) - input_ptr = if T_ex isa LLVM.PointerType - parameters(llvm_f)[1] - else - inttoptr!(builder, parameters(llvm_f)[1], T_ex_ptr) - end - if LLVM.addrspace(value_type(input_ptr)) != LLVM.addrspace(T_ex_ptr) - input_ptr = addrspacecast!(builder, input_ptr, T_ex_ptr) - end - input_ptr = bitcast!(builder, input_ptr, T_i8_ptr) - br!(builder, check) - - position!(builder, check) - offset = phi!(builder, T_i64) - next_offset = add!(builder, offset, ConstantInt(1)) - append!(LLVM.incoming(offset), [(init_offset, entry), (next_offset, check)]) - - ptr = gep!(builder, T_i8, input_ptr, [offset]) - value = load!(builder, T_i8, ptr) - cond = icmp!(builder, LLVM.API.LLVMIntEQ, value, ConstantInt(0x0)) - br!(builder, cond, done, check) - - position!(builder, done) - ret!(builder, offset) - end - call_function(llvm_f, Int64, Tuple{ex}, :ex) - end +@llvmgenerated builder function alloc_string(::Val{sym})::LLVMPtr{UInt8,AS.Global} where sym + globalstring_ptr!(builder, String(sym); addrspace=AS.Global) end @inline strlen(::Val{S}) where S = length(String(S)) diff --git a/src/runtime/hip-execution.jl b/src/runtime/hip-execution.jl index f52272bfa..a5bf0907d 100644 --- a/src/runtime/hip-execution.jl +++ b/src/runtime/hip-execution.jl @@ -41,10 +41,7 @@ end # filter out ghost arguments that shouldn't be passed. predicate = dt -> GPUCompiler.isghosttype(dt) || Core.Compiler.isconstType(dt) - # Note: Define a single LLVM context, otherwise it is created per every param. - to_pass = LLVM.Context() do _ - map(!predicate, sig.parameters) - end + to_pass = map(!predicate, sig.parameters) call_t = Type[x[1] for x in zip(sig.parameters, to_pass) if x[2]] call_args = Union{Expr,Symbol}[x[1] for x in zip(args, to_pass) if x[2]] diff --git a/test/core/codegen.jl b/test/core/codegen.jl index e7943491e..002614939 100644 --- a/test/core/codegen.jl +++ b/test/core/codegen.jl @@ -1,6 +1,7 @@ using Test using AMDGPU import GPUCompiler +import LLVM using AMDGPU: Device, ROCArray, @roc using AMDGPU.Device: sync_workgroup, workitemIdx, workgroupIdx, workgroupDim using KernelAbstractions: @atomic @@ -118,8 +119,11 @@ end kernel=true, name=nothing, always_inline=true) tt = Tuple{AMDGPU.Device.ROCDeviceVector{Float32, AMDGPU.Device.AS.Global}} job = GPUCompiler.CompilerJob(GPUCompiler.methodinstance(typeof(oob_kern!), tt), config) - asm, _ = GPUCompiler.JuliaContext() do _ - GPUCompiler.compile(:asm, job) + asm = GPUCompiler.JuliaContext() do _ + asm, meta = GPUCompiler.compile(:asm, job) + # the IR belongs to us: dispose of it, or it leaks along with the context + LLVM.dispose(meta.ir) + asm end # the exception path signals through an atomic compare-and-swap @test occursin("cmpswap", asm)