Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -43,4 +43,4 @@ Serialization = "1"
TOML = "1"
Tracy = "0.1.4"
UUIDs = "1"
julia = "1.10"
julia = "1.10 - 1.13"
31 changes: 31 additions & 0 deletions src/irgen.jl
Original file line number Diff line number Diff line change
Expand Up @@ -190,13 +190,20 @@ function lower_throw!(@nospecialize(job::CompilerJob), mod::LLVM.Module)
"jl_eof_error" => "EOF error",
]

# Julia's codegen replaces an `llvmcall` of an intrinsic it doesn't know (e.g. one
# removed from LLVM) by a run-time `jl_error`. report those at compile time instead.
errors = IRError[]

for f in functions(mod)
fn = LLVM.name(f)
for (throw_fn, name) in throw_functions
occursin(throw_fn, fn) || continue

for use in uses(f)
call = user(use)::LLVM.CallInst
if is_unknown_intrinsic_error(call)
push!(errors, (UNKNOWN_INTRINSIC, backtrace(call), nothing))
end

# replace the throw with a PTX-compatible exception
@dispose builder=IRBuilder() begin
Expand Down Expand Up @@ -232,6 +239,7 @@ function lower_throw!(@nospecialize(job::CompilerJob), mod::LLVM.Module)
end

end
isempty(errors) || throw(InvalidIRError(job, errors))
return changed
end

Expand Down Expand Up @@ -359,6 +367,29 @@ function inline_unreachable_control_flow!(@nospecialize(job::CompilerJob), mod::
return changed
end

# demote unordered atomic loads and stores to plain ones
#
# Julia marks accesses to heap references `unordered` so that a read racing with the GC, or
# with another thread's write, cannot observe a torn pointer. There is no device GC and no
# such race for GPUCompiler to protect against, so the ordering carries no meaning here, but
# not every back-end can express it: SPIR-V's OpAtomicLoad/OpAtomicStore only take scalar
# integer or floating-point operands, so the Khronos translator turns an `unordered` load of a
# pointer into an invalid pointer-typed atomic that consumers reject (Intel's compiler fails
# with an undefined `__spirv_AtomicLoad(long**, int, int)`), and AIR has no atomic load or
# store instructions at all. Run after optimization, where dropping the ordering cannot
# enable new transformations; stronger orderings are left intact.
function demote_unordered_atomics!(mod::LLVM.Module)
changed = false
for f in functions(mod), bb in blocks(f), inst in instructions(bb)
(inst isa LLVM.LoadInst || inst isa LLVM.StoreInst) || continue
is_atomic(inst) || continue
ordering(inst) == LLVM.API.LLVMAtomicOrderingUnordered || continue
ordering!(inst, LLVM.API.LLVMAtomicOrderingNotAtomic)
changed = true
end
return changed
end

# lower `trap` to a clean return to get rid of `unreachable` and `noreturn`
#
# this is for compatibility with back-ends that don't support (SPIR-V) or have
Expand Down
79 changes: 79 additions & 0 deletions src/metal.jl
Original file line number Diff line number Diff line change
Expand Up @@ -503,6 +503,9 @@ function lower_air!(@nospecialize(job::CompilerJob{MetalCompilerTarget}), mod::L
# Metal.malloc uses.
rewrite_generic_null_selects!(mod)

# AIR does not support LLVM atomic load/store instructions (see `demote_unordered_atomics!`)
demote_unordered_atomics!(mod)

# strip device-side `trap`s and rewrite `unreachable` into clean returns (#433, #370). this
# runs post-`optimize!`, after the trap has finished serving as the optimizer guard; the pass
# force-inlines throwing functions into the kernel first so the rewrite is sound, then scrubs
Expand Down Expand Up @@ -1810,6 +1813,82 @@ function lower_llvm_intrinsics!(@nospecialize(job::CompilerJob), fun::LLVM.Funct
end
end

# floating-point class tests, which LLVM forms out of combined comparisons (e.g., of
# `isfinite` and `iszero`) but AIR does not support: test the value's bits instead
if intr == LLVM.Intrinsic("llvm.is.fpclass")
x, test = arguments(call)
typ = value_type(x)

# XXX: LLVM C API doesn't have getPrimitiveSizeInBits
jltyp = if typ == LLVM.HalfType()
Float16
elseif typ == LLVM.FloatType()
Float32
elseif typ == LLVM.DoubleType()
Float64
else
error("Unsupported is.fpclass type: $typ")
end
mask = convert(Int, test)

@dispose builder=IRBuilder() begin
position!(builder, call)
debuglocation!(builder, call)

ityp = LLVM.IntType(8*sizeof(jltyp))
bits = bitcast!(builder, x, ityp)
magnitude = and!(builder, bits, LLVM.ConstantInt(ityp, ~Base.sign_mask(jltyp)))

# tests for the classes of the magnitude, emitted when needed
inf = Base.exponent_mask(jltyp)
qnan = inf | (Base.significand_mask(jltyp) + one(inf)) >> 1
normal = reinterpret(Unsigned, floatmin(jltyp))
compare(pred, lhs, rhs) = icmp!(builder, pred, lhs, LLVM.ConstantInt(ityp, rhs))
test_nan() = compare(LLVM.API.LLVMIntUGT, magnitude, inf)
test_qnan() = compare(LLVM.API.LLVMIntUGE, magnitude, qnan)
test_inf() = compare(LLVM.API.LLVMIntEQ, magnitude, inf)
test_normal() = compare(LLVM.API.LLVMIntULT,
sub!(builder, magnitude, LLVM.ConstantInt(ityp, normal)),
inf - normal)
test_subnormal() = compare(LLVM.API.LLVMIntULT,
sub!(builder, magnitude, LLVM.ConstantInt(ityp, 1)),
normal - 1)
test_zero() = compare(LLVM.API.LLVMIntEQ, magnitude, 0)
negative = nothing
function with_sign(test, neg)
if negative === nothing
negative = compare(LLVM.API.LLVMIntSLT, bits, 0)
end
and!(builder, test, neg ? negative : not!(builder, negative))
end

# combine the tested classes, as encoded by the `FPClassTest` mask bits
terms = LLVM.Value[]
tested(bit) = mask & (1 << bit) != 0
if tested(0) && tested(1)
push!(terms, test_nan())
elseif tested(0)
push!(terms, and!(builder, test_nan(), not!(builder, test_qnan())))
elseif tested(1)
push!(terms, test_qnan())
end
for (test, negbit, posbit) in ((test_inf, 2, 9), (test_normal, 3, 8),
(test_subnormal, 4, 7), (test_zero, 5, 6))
if tested(negbit) && tested(posbit)
push!(terms, test())
elseif tested(negbit) || tested(posbit)
push!(terms, with_sign(test(), tested(negbit)))
end
end
new_value = isempty(terms) ? LLVM.ConstantInt(LLVM.Int1Type(), 0) :
foldl((a, b) -> or!(builder, a, b), terms)

replace_uses!(call, new_value)
erase!(call)
changed = true
end
end

# copysign
if intr == LLVM.Intrinsic("llvm.copysign")
arg0, arg1 = operands(call)
Expand Down
6 changes: 6 additions & 0 deletions src/optim.jl
Original file line number Diff line number Diff line change
Expand Up @@ -361,6 +361,12 @@ function buildIntrinsicLoweringPipeline(mpm, @nospecialize(job::CompilerJob), op
if uses_julia_runtime(job) && VERSION >= v"1.11.0-DEV.208"
add!(fpm, FinalLowerGCPass())
end
# codegen emits atomic modifications (e.g. `@atomic x.f += 1`) as calls to the
# `julia.atomicmodify` pseudo-intrinsic, which Julia expands after GC lowering
# (JuliaLang/julia#57010)
@static if VERSION >= v"1.13.0-DEV.321"
add!(fpm, ExpandAtomicModifyPass())
end
end
if uses_julia_runtime(job) && VERSION < v"1.11.0-DEV.208"
add!(mpm, FinalLowerGCPass())
Expand Down
4 changes: 4 additions & 0 deletions src/spirv.jl
Original file line number Diff line number Diff line change
Expand Up @@ -103,6 +103,10 @@ function finish_ir!(job::CompilerJob{SPIRVCompilerTarget}, mod::LLVM.Module,
# OpUnreachable (UB if reached), which PoCL and friends handle poorly.
lower_unreachable_control_flow!(job, mod)

# SPIR-V cannot express atomic loads and stores of pointers, which is what Julia's
# `unordered` heap-reference accesses become; the orderings serve no purpose on device
demote_unordered_atomics!(mod)

# convert the kernel state argument to a byval reference
if job.config.kernel
state = kernel_state_type(job)
Expand Down
30 changes: 29 additions & 1 deletion src/validation.jl
Original file line number Diff line number Diff line change
Expand Up @@ -135,6 +135,7 @@ const CCALL_FUNCTION = "call to an external C function"
const LAZY_FUNCTION = "call to a lazy-initialized function"
const DELAYED_BINDING = "use of an undefined name"
const DYNAMIC_CALL = "dynamic function invocation"
const UNKNOWN_INTRINSIC = "call to an unknown LLVM intrinsic"

function Base.showerror(io::IO, err::InvalidIRError)
print(io, "InvalidIRError: compiling ", err.job.source, " resulted in invalid LLVM IR")
Expand Down Expand Up @@ -222,14 +223,41 @@ function check_ir!(job, errors::Vector{IRError}, inst::LLVM.LoadInst)
return errors
end

# the contents of a constant string global, or `nothing`
function constant_string(val::LLVM.Value)
while val isa LLVM.ConstantExpr
val = first(operands(val))
end
val isa LLVM.GlobalVariable || return nothing
init = initializer(val)
init === nothing && return nothing
LLVM.API.LLVMIsConstantString(init) == 1 || return nothing
len = Ref{Csize_t}()
ptr = LLVM.API.LLVMGetAsString(init, len)
return rstrip(unsafe_string(convert(Ptr{UInt8}, ptr), len[]), '\0')
end

# Julia's codegen replaces an `llvmcall` of an intrinsic it doesn't know, e.g. one that was
# removed from LLVM, with a call to `jl_error`, deferring the error to run time.
function is_unknown_intrinsic_error(call::LLVM.CallInst)
dest = called_operand(call)
dest isa LLVM.Function || return false
LLVM.name(dest) in ("jl_error", "ijl_error") || return false
args = arguments(call)
length(args) == 1 || return false
return constant_string(args[1]) == "llvmcall only supports intrinsic calls"
end

function check_ir!(job, errors::Vector{IRError}, inst::LLVM.CallInst)
bt = backtrace(inst)
dest = called_operand(inst)
if isa(dest, LLVM.Function)
fn = LLVM.name(dest)

# some special handling for runtime functions that we don't implement
if fn == "jl_get_binding_or_error" || fn == "ijl_get_binding_or_error"
if is_unknown_intrinsic_error(inst)
push!(errors, (UNKNOWN_INTRINSIC, bt, nothing))
elseif fn == "jl_get_binding_or_error" || fn == "ijl_get_binding_or_error"
try
m, sym = arguments(inst)
sym = first(operands(sym::ConstantExpr))::ConstantInt
Expand Down
97 changes: 97 additions & 0 deletions test/metal.jl
Original file line number Diff line number Diff line change
Expand Up @@ -1001,6 +1001,50 @@ end
end
end

@testset "floating-point class lowering" begin
# LLVM combines floating-point comparisons into llvm.is.fpclass, which AIR lacks, so it is
# expanded into tests on the value's bits. Check these against Julia's classification by
# lowering calls on constant values (which the IR builder folds).
km = @eval module $(gensym())
f() = return
end
job, _ = Metal.create_job(km.f, Tuple{})

# the `FPClassTest` bit of a value
function fpclass(x::T) where {T}
if isnan(x)
quiet = reinterpret(Unsigned, x) & (Base.significand_mask(T) + 1) >> 1 != 0
return quiet ? 1 : 0
end
bit = isinf(x) ? 9 : iszero(x) ? 6 : issubnormal(x) ? 7 : 8
return signbit(x) ? 11 - bit : bit
end

Context() do ctx
for (T, typ, suffix) in ((Float16, "half", "f16"), (Float32, "float", "f32"),
(Float64, "double", "f64"))
snan = reinterpret(T, Base.exponent_mask(T) | one(Base.exponent_mask(T)))
values = T[0, -0.0, nextfloat(zero(T)), -nextfloat(zero(T)), prevfloat(floatmin(T)),
floatmin(T), -floatmin(T), 1, -1, floatmax(T), -floatmax(T),
Inf, -Inf, NaN, -T(NaN), snan, -snan]
for x in values, mask in [0:37:1023; (1 .<< (0:9)); 0x3ff]
val = "bitcast (i$(8*sizeof(T)) $(reinterpret(Signed, x)) to $typ)"
ir = """
declare i1 @llvm.is.fpclass.$suffix($typ, i32)
define i1 @f() {
%r = call i1 @llvm.is.fpclass.$suffix($typ $val, i32 $mask)
ret i1 %r
}"""
mod = parse(LLVM.Module, ir)
f = functions(mod)["f"]
GPUCompiler.lower_llvm_intrinsics!(job, f)
ret = only(filter(i -> i isa LLVM.RetInst, collect(instructions(entry(f)))))
@test convert(Bool, operands(ret)[1]) == (mask & (1 << fpclass(x)) != 0)
end
end
end
end

@testset "fast-math min/max lowering" begin
# Base min/max -> llvm.minimum/maximum lower to a NaN-propagating wrapper (air.minimum/
# maximum, which falls back to air.fmin/fmax). @fastmath/fastmath set `nnan`, so the relaxed
Expand Down Expand Up @@ -1225,6 +1269,59 @@ end
end
end

@testset "unordered atomic demotion" begin
# Demote unordered LLVM loads/stores, but preserve stronger atomics.
Context() do ctx
ir = """
define void @f(i8** %p, i8** %q, i64* %r) {
entry:
%a = load atomic i8*, i8** %p unordered, align 8
store atomic i8* %a, i8** %q unordered, align 8
%b = load atomic i64, i64* %r monotonic, align 8
store atomic i64 %b, i64* %r release, align 8
store i8* %a, i8** %p, align 8
ret void
}
"""
mod = parse(LLVM.Module, ir)
insts() = [i for f in functions(mod) for bb in blocks(f) for i in instructions(bb)]
memops() = filter(i -> i isa LLVM.LoadInst || i isa LLVM.StoreInst, insts())

@test count(is_atomic, memops()) == 4
@test GPUCompiler.demote_unordered_atomics!(mod)
atomics = filter(is_atomic, memops())
@test length(atomics) == 2
@test all(i -> ordering(i) != LLVM.API.LLVMAtomicOrderingUnordered, atomics)
@test !occursin("unordered", string(mod))
@test (verify(mod); true)
# idempotent
@test !GPUCompiler.demote_unordered_atomics!(mod)
end

# end-to-end: Julia's own `:unordered` accesses (as codegen emits for heap-reference
# fields) must reach the AIR as plain loads and stores
function kernel(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1})
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int}, p), :unordered)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int}, q), x, :unordered)
return
end
source = methodinstance(typeof(kernel), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}},
Base.get_world_counter())
target = MetalCompilerTarget(; macos=v"12.2", metal=v"3.0", air=v"3.0")
config = CompilerConfig(target, Metal.CompilerParams(); kernel=true)
job = CompilerJob(source, config)

# precondition: the accesses survive optimization as unordered atomics
ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true))
@test occursin(r"load atomic .* unordered", ir)
@test occursin(r"store atomic .* unordered", ir)

air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true))
@test !occursin(r"(load|store) atomic", air)
@test occursin(r"load i64", air)
@test occursin(r"store i64", air)
end

# byval lowering must strip the (non-IPO-safe) Julia const-region metadata off loads derived
# from the materialized argument; check the helper walks gep/addrspacecast chains and removes it.
@testset "const-region metadata stripping for materialized args" begin
Expand Down
32 changes: 32 additions & 0 deletions test/native.jl
Original file line number Diff line number Diff line change
Expand Up @@ -362,6 +362,25 @@ end
end
end

@testset "atomic field modifications" begin
# from Julia 1.13, `@atomic x.f += 1` is a call to the `julia.atomicmodify`
# pseudo-intrinsic, which has to be expanded (JuliaLang/julia#57010)
mod = @eval module $(gensym())
mutable struct Counter
@atomic n::Int
end
function increment(c::Counter)
@atomic c.n += 1
return
end
end

@test @filecheck implicit_check_not=["julia.atomicmodify"] begin
@check "{{atomicrmw add|cmpxchg}}"
Native.code_llvm(mod.increment, Tuple{mod.Counter})
end
end

@testset "tracked pointers" begin
mod = @eval module $(gensym())
function kernel(a)
Expand Down Expand Up @@ -622,6 +641,19 @@ end
end
end

@testset "unknown intrinsics" begin
# Julia's codegen lowers these to a run-time error, which we report at compile time
mod = @eval module $(gensym())
kernel() = (ccall("llvm.nonexistent.intrinsic", llvmcall, Cvoid, ()); return)
end

@test_throws_message(InvalidIRError,
Native.code_execution(mod.kernel, Tuple{})) do msg
occursin(GPUCompiler.UNKNOWN_INTRINSIC, msg) &&
occursin(r"\[\d+\] kernel", msg)
end
end

@testset "invalid LLVM IR (ccall)" begin
mod = @eval module $(gensym())
function foobar(p)
Expand Down
Loading
Loading