From 03d602c21f894b4ca5a787edf1d293c9feeb49d6 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Thu, 1 Oct 2026 09:16:40 +0200 Subject: [PATCH 1/5] Port the POCL back-end to LLVM.jl 10 Import the IR and Build vocabularies next to LLVM.Interop, since `using LLVM` no longer exports the API, and use properties instead of the removed accessor functions (metadata, function attributes, module functions, linkage, initializer, alignment, the entry block and the subprogram of a function, ...). create_function and call_function were removed: the additional kernel arguments, the constant tables for randn and the invariant loads now generate their IR with generate_llvmcall. Insertion points are explicit now: the block that initializes the RNG state is created LLVM.before the entry block, and the builder positioned LLVM.at_end of it. The builder gets the line-0 location of the kernel's subprogram for the instructions it inserts; the old `debuglocation!(builder, first(instructions(top_bb)))` copied that location onto the first instruction of the original entry block (or cleared its location without a subprogram), which wasn't the intent. --- Project.toml | 2 +- src/pocl/compiler/compilation.jl | 25 +++++++++++----------- src/pocl/device/array.jl | 36 ++++++++++++-------------------- src/pocl/device/random.jl | 26 +++++++---------------- src/pocl/device/runtime.jl | 27 +++++++----------------- src/pocl/pocl.jl | 3 +-- 6 files changed, 42 insertions(+), 77 deletions(-) diff --git a/Project.toml b/Project.toml index 06ce1adbb..42557f6db 100644 --- a/Project.toml +++ b/Project.toml @@ -45,7 +45,7 @@ EnzymeCore = "0.7, 0.8.1" GPUCompiler = "2.10" GPUToolbox = "3.3.2" KernelInterface = "0.4" -LLVM = "9.9" +LLVM = "10" LinearAlgebra = "1.6" MacroTools = "0.5" PrecompileTools = "1" diff --git a/src/pocl/compiler/compilation.jl b/src/pocl/compiler/compilation.jl index 14bc038b7..86511d2d9 100644 --- a/src/pocl/compiler/compilation.jl +++ b/src/pocl/compiler/compilation.jl @@ -63,12 +63,12 @@ function GPUCompiler.finish_module!( sg_size = job.config.params.sub_group_size if sg_size !== nothing - metadata(entry)["intel_reqd_sub_group_size"] = MDNode([ConstantInt(Int32(sg_size))]) + entry.metadata["intel_reqd_sub_group_size"] = MDNode([ConstantInt(Int32(sg_size))]) end # if this kernel uses our RNG, we should prime the shared state. # XXX: these transformations should really happen at the Julia IR level... - if haskey(functions(mod), "julia.opencl.random_keys") && job.config.kernel + if haskey(mod.functions, "julia.opencl.random_keys") && job.config.kernel # insert call to `initialize_rng_state` f = initialize_rng_state ft = typeof(f) @@ -82,16 +82,15 @@ function GPUCompiler.finish_module!( GPUCompiler.deferred_codegen_jobs[id] = job # generate IR for calls to `deferred_codegen` and the resulting function pointer - top_bb = first(blocks(entry)) - bb = BasicBlock(top_bb, "initialize_rng") + top_bb = entry.entry + bb = BasicBlock(LLVM.before(top_bb), "initialize_rng") @dispose builder = IRBuilder() begin - position!(builder, bb) - subprogram = LLVM.subprogram(entry) + position!(builder, LLVM.at_end(bb)) + subprogram = entry.subprogram if subprogram !== nothing loc = DILocation(0, 0, subprogram) - debuglocation!(builder, loc) + builder.debug_location = loc end - debuglocation!(builder, first(instructions(top_bb))) # call the `deferred_codegen` marker function T_ptr = if LLVM.version() >= v"17" @@ -103,8 +102,8 @@ function GPUCompiler.finish_module!( end T_id = convert(LLVMType, Int) deferred_codegen_ft = LLVM.FunctionType(T_ptr, [T_id]) - deferred_codegen = if haskey(functions(mod), "deferred_codegen") - functions(mod)["deferred_codegen"] + deferred_codegen = if haskey(mod.functions, "deferred_codegen") + mod.functions["deferred_codegen"] else LLVM.Function(mod, "deferred_codegen", deferred_codegen_ft) end @@ -119,7 +118,7 @@ function GPUCompiler.finish_module!( br!(builder, top_bb) # note the use of the device-side RNG in this kernel - push!(function_attributes(entry), StringAttribute("julia.opencl.rng", "")) + push!(entry.function_attributes, StringAttribute("julia.opencl.rng", "")) end # XXX: put some of the above behind GPUCompiler abstractions @@ -243,8 +242,8 @@ function compile_to_obj(@nospecialize(job::CompilerJob)) return JuliaContext() do ctx obj, meta = invoke_frozen(GPUCompiler.compile, :obj, job) - entry = LLVM.name(meta.entry) - device_rng = StringAttribute("julia.opencl.rng", "") in collect(function_attributes(meta.entry)) + entry = meta.entry.name + device_rng = StringAttribute("julia.opencl.rng", "") in collect(meta.entry.function_attributes) (; obj, entry, device_rng) end diff --git a/src/pocl/device/array.jl b/src/pocl/device/array.jl index eca9ea7d6..1bccb98dd 100644 --- a/src/pocl/device/array.jl +++ b/src/pocl/device/array.jl @@ -162,34 +162,24 @@ end @inline @generated function unsafe_invariant_load(ptr::LLVMPtr{T, AS}, i::I, ::Val{align}) where {T, AS, I, align} sizeof(T) == 0 && return T.instance ispow2(align) || return :(error("unsafe_invariant_load: alignment must be a power of 2, got $($align)")) - return @dispose ctx = Context() begin + return generate_llvmcall(T, Tuple{LLVMPtr{T, AS}, I}, :ptr, :(i - one(I))) do builder, ptr, idx eltyp = convert(LLVMType, T) - T_idx = convert(LLVMType, I) - T_ptr = convert(LLVMType, ptr) T_typed_ptr = LLVM.PointerType(eltyp, AS) - llvm_f, _ = create_function(eltyp, LLVMType[T_ptr, T_idx]) - - @dispose builder = IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - base = if supports_typed_pointers(ctx) - bitcast!(builder, parameters(llvm_f)[1], T_typed_ptr) - else - parameters(llvm_f)[1] - end - gep = inbounds_gep!(builder, eltyp, base, [parameters(llvm_f)[2]]) - ld = load!(builder, eltyp, gep) - if AS != 0 - metadata(ld)[LLVM.MD_tbaa] = tbaa_addrspace(AS) - end - metadata(ld)[LLVM.MD_invariant_load] = MDNode(LLVM.Metadata[]) - alignment!(ld, align) - - ret!(builder, ld) + base = if supports_typed_pointers(LLVM.context()) + bitcast!(builder, ptr, T_typed_ptr) + else + ptr + end + gep = inbounds_gep!(builder, eltyp, base, [idx]) + ld = load!(builder, eltyp, gep) + if AS != 0 + ld.metadata[LLVM.MD_tbaa] = tbaa_addrspace(AS) end + ld.metadata[LLVM.MD_invariant_load] = MDNode(LLVM.Metadata[]) + ld.alignment = align - call_function(llvm_f, T, Tuple{LLVMPtr{T, AS}, I}, :ptr, :(i - one(I))) + ld end end diff --git a/src/pocl/device/random.jl b/src/pocl/device/random.jl index 4360b25f9..7c506dd11 100644 --- a/src/pocl/device/random.jl +++ b/src/pocl/device/random.jl @@ -162,36 +162,26 @@ end # a hacky method of exposing constant tables as constant GPU memory function emit_constant_array(name::Symbol, data::AbstractArray{T}) where {T} - return @dispose ctx = Context() begin + return generate_llvmcall(LLVMPtr{T, AS.UniformConstant}, Tuple{}) do builder T_val = convert(LLVMType, T) T_ptr = convert(LLVMType, LLVMPtr{T, AS.UniformConstant}) - # define function and get LLVM module - llvm_f, _ = create_function(T_ptr) - mod = LLVM.parent(llvm_f) + # get LLVM module + mod = current_module(builder) # create a global memory global variable # TODO: global_var alignment? T_global = LLVM.ArrayType(T_val, length(data)) # XXX: why can't we use a single name like emit_shmem gv = GlobalVariable(mod, T_global, "gpu_$(name)_data", AS.UniformConstant) - linkage!(gv, LLVM.API.LLVMInternalLinkage) - initializer!(gv, ConstantArray(data)) - alignment!(gv, 16) + gv.linkage = LLVM.API.LLVMInternalLinkage + gv.initializer = ConstantArray(data) + gv.alignment = 16 # generate IR - @dispose builder = IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) + ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) - ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) - - untyped_ptr = bitcast!(builder, ptr, T_ptr) - - ret!(builder, untyped_ptr) - end - - call_function(llvm_f, LLVMPtr{T, AS.UniformConstant}) + untyped_ptr = bitcast!(builder, ptr, T_ptr) end end diff --git a/src/pocl/device/runtime.jl b/src/pocl/device/runtime.jl index f99bde239..f20a72e03 100644 --- a/src/pocl/device/runtime.jl +++ b/src/pocl/device/runtime.jl @@ -45,40 +45,27 @@ end # then get propagated across function calls to the caller. function additional_arg_intr(mod::LLVM.Module, T_state, name) - state_intr = if haskey(functions(mod), "julia.opencl.$name") - functions(mod)["julia.opencl.$name"] + state_intr = if haskey(mod.functions, "julia.opencl.$name") + mod.functions["julia.opencl.$name"] else LLVM.Function(mod, "julia.opencl.$name", LLVM.FunctionType(T_state)) end - push!(function_attributes(state_intr), EnumAttribute("readnone", 0)) + push!(state_intr.function_attributes, EnumAttribute("readnone", 0)) return state_intr end # run-time equivalent function additional_arg_value(state, name) - return @dispose ctx = Context() begin + return generate_llvmcall(state, Tuple{}) do builder T_state = convert(LLVMType, state) - # create function - llvm_f, _ = create_function(T_state) - mod = LLVM.parent(llvm_f) - # get intrinsic - state_intr = additional_arg_intr(mod, T_state, name) - state_intr_ft = function_type(state_intr) + state_intr = additional_arg_intr(current_module(builder), T_state, name) + state_intr_ft = state_intr.function_type # generate IR - @dispose builder = IRBuilder() begin - entry = BasicBlock(llvm_f, "entry") - position!(builder, entry) - - val = call!(builder, state_intr_ft, state_intr, Value[], name) - - ret!(builder, val) - end - - call_function(llvm_f, state) + call!(builder, state_intr_ft, state_intr, Value[], name) end end diff --git a/src/pocl/pocl.jl b/src/pocl/pocl.jl index bdb6240c2..a3acdd7a5 100644 --- a/src/pocl/pocl.jl +++ b/src/pocl/pocl.jl @@ -84,8 +84,7 @@ function queue() end using GPUCompiler -using LLVM, LLVM.Interop -import LLVM: LLVM, MDNode, ConstantInt, metadata +using LLVM, LLVM.IR, LLVM.Build, LLVM.Interop using SPIRV_LLVM_Backend_jll, SPIRV_Tools_jll using Adapt From 91a70959d9f9490d3602912488bce0017cb7fb34 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Thu, 1 Oct 2026 09:18:33 +0200 Subject: [PATCH 2/5] Use LLVM.jl's lookups and enumerations instead of the C API Look up or declare functions with get! on the module's function view, check for the RNG attribute by its kind instead of collecting and comparing attributes, use the scoped linkage enumeration instead of LLVM.API, and the public metadata kinds. The intrinsics for the additional kernel arguments now only get their readnone attribute when they're declared, rather than every time they are looked up. --- src/pocl/compiler/compilation.jl | 6 ++---- src/pocl/device/array.jl | 4 ++-- src/pocl/device/random.jl | 2 +- src/pocl/device/runtime.jl | 11 ++++------- 4 files changed, 9 insertions(+), 14 deletions(-) diff --git a/src/pocl/compiler/compilation.jl b/src/pocl/compiler/compilation.jl index 86511d2d9..1d480da81 100644 --- a/src/pocl/compiler/compilation.jl +++ b/src/pocl/compiler/compilation.jl @@ -102,9 +102,7 @@ function GPUCompiler.finish_module!( end T_id = convert(LLVMType, Int) deferred_codegen_ft = LLVM.FunctionType(T_ptr, [T_id]) - deferred_codegen = if haskey(mod.functions, "deferred_codegen") - mod.functions["deferred_codegen"] - else + deferred_codegen = get!(mod.functions, "deferred_codegen") do LLVM.Function(mod, "deferred_codegen", deferred_codegen_ft) end fptr = call!(builder, deferred_codegen_ft, deferred_codegen, [ConstantInt(id)]) @@ -243,7 +241,7 @@ function compile_to_obj(@nospecialize(job::CompilerJob)) obj, meta = invoke_frozen(GPUCompiler.compile, :obj, job) entry = meta.entry.name - device_rng = StringAttribute("julia.opencl.rng", "") in collect(meta.entry.function_attributes) + device_rng = haskey(meta.entry.function_attributes, "julia.opencl.rng") (; obj, entry, device_rng) end diff --git a/src/pocl/device/array.jl b/src/pocl/device/array.jl index 1bccb98dd..68ba5f7ed 100644 --- a/src/pocl/device/array.jl +++ b/src/pocl/device/array.jl @@ -174,9 +174,9 @@ end gep = inbounds_gep!(builder, eltyp, base, [idx]) ld = load!(builder, eltyp, gep) if AS != 0 - ld.metadata[LLVM.MD_tbaa] = tbaa_addrspace(AS) + ld.metadata[MD_tbaa] = tbaa_addrspace(AS) end - ld.metadata[LLVM.MD_invariant_load] = MDNode(LLVM.Metadata[]) + ld.metadata[MD_invariant_load] = MDNode(LLVM.Metadata[]) ld.alignment = align ld diff --git a/src/pocl/device/random.jl b/src/pocl/device/random.jl index 7c506dd11..ba08a11fb 100644 --- a/src/pocl/device/random.jl +++ b/src/pocl/device/random.jl @@ -174,7 +174,7 @@ function emit_constant_array(name::Symbol, data::AbstractArray{T}) where {T} T_global = LLVM.ArrayType(T_val, length(data)) # XXX: why can't we use a single name like emit_shmem gv = GlobalVariable(mod, T_global, "gpu_$(name)_data", AS.UniformConstant) - gv.linkage = LLVM.API.LLVMInternalLinkage + gv.linkage = LLVM.Linkage.Internal gv.initializer = ConstantArray(data) gv.alignment = 16 diff --git a/src/pocl/device/runtime.jl b/src/pocl/device/runtime.jl index f20a72e03..e3809c93d 100644 --- a/src/pocl/device/runtime.jl +++ b/src/pocl/device/runtime.jl @@ -45,14 +45,11 @@ end # then get propagated across function calls to the caller. function additional_arg_intr(mod::LLVM.Module, T_state, name) - state_intr = if haskey(mod.functions, "julia.opencl.$name") - mod.functions["julia.opencl.$name"] - else - LLVM.Function(mod, "julia.opencl.$name", LLVM.FunctionType(T_state)) + return get!(mod.functions, "julia.opencl.$name") do + state_intr = LLVM.Function(mod, "julia.opencl.$name", LLVM.FunctionType(T_state)) + push!(state_intr.function_attributes, EnumAttribute(:readnone)) + state_intr end - push!(state_intr.function_attributes, EnumAttribute("readnone", 0)) - - return state_intr end # run-time equivalent From 30500349b9aba5a298fbb8ad8f4d9bb4eaf8dfb9 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Thu, 1 Oct 2026 09:24:59 +0200 Subject: [PATCH 3/5] Generate IR with @llvmgenerated and use LLVM.jl's helpers The accessor for the additional kernel arguments, the invariant load and the constant tables for randn were @generated functions around generate_llvmcall: define them with @llvmgenerated instead, which derives the LLVM signature from the Julia one. The invariant load's checks move to a wrapper, and its generator uses an unconditional typed bitcast (which folds away with opaque pointers) and load!'s align keyword. The table accessors become plain functions that pass the table's name and element type to one generator, which reads the (constant) table from Random. Use memory effects instead of the readnone attribute, which is invalid on functions as of LLVM 16, derive the return type of the deferred_codegen declaration from Julia's lowering of Ptr{Cvoid} instead of a version decision tree, and implement NDIteration.assume with LLVM.Interop.assume instead of a copy of its IR. --- src/nditeration.jl | 17 ++--------- src/pocl/compiler/compilation.jl | 9 ++---- src/pocl/device/array.jl | 35 ++++++++++------------- src/pocl/device/random.jl | 49 ++++++++++++++------------------ src/pocl/device/runtime.jl | 19 ++++--------- 5 files changed, 45 insertions(+), 84 deletions(-) diff --git a/src/nditeration.jl b/src/nditeration.jl index c16142b79..30a2238ab 100644 --- a/src/nditeration.jl +++ b/src/nditeration.jl @@ -7,6 +7,7 @@ export DynamicCheck, NoDynamicCheck import Adapt import Base.@pure +import LLVM struct DynamicCheck end struct NoDynamicCheck end @@ -211,21 +212,7 @@ end Assume that the condition `cond` is true. This is a hint to the compiler, possibly enabling it to optimize more aggressively. """ -@inline assume(cond::Bool) = Base.llvmcall( - ( - """ - declare void @llvm.assume(i1) - - define void @entry(i8) #0 { - %cond = icmp eq i8 %0, 1 - call void @llvm.assume(i1 %cond) - ret void - } - - attributes #0 = { alwaysinline }""", "entry", - ), - Nothing, Tuple{Bool}, cond -) +@inline assume(cond::Bool) = LLVM.Interop.assume(cond) @inline function assume_nonzero(CI::CartesianIndices) return ntuple(Val(ndims(CI))) do I diff --git a/src/pocl/compiler/compilation.jl b/src/pocl/compiler/compilation.jl index 1d480da81..c8e4b9251 100644 --- a/src/pocl/compiler/compilation.jl +++ b/src/pocl/compiler/compilation.jl @@ -93,13 +93,8 @@ function GPUCompiler.finish_module!( end # call the `deferred_codegen` marker function - T_ptr = if LLVM.version() >= v"17" - LLVM.PointerType() - elseif VERSION >= v"1.12.0-DEV.225" - LLVM.PointerType(LLVM.Int8Type()) - else - LLVM.Int64Type() - end + # (declared like GPUCompiler's `ccall("extern deferred_codegen", llvmcall, Ptr{Cvoid}, ...)`) + T_ptr = convert(LLVMType, Ptr{Cvoid}) T_id = convert(LLVMType, Int) deferred_codegen_ft = LLVM.FunctionType(T_ptr, [T_id]) deferred_codegen = get!(mod.functions, "deferred_codegen") do diff --git a/src/pocl/device/array.jl b/src/pocl/device/array.jl index 68ba5f7ed..67b814605 100644 --- a/src/pocl/device/array.jl +++ b/src/pocl/device/array.jl @@ -159,28 +159,23 @@ end # There is no SPIR-V equivalent of NVPTX's `ld.global.nc`, so instead of a dedicated # instruction we mark the load `!invariant.load`, which lets LLVM hoist it out of loops # and reorder it across stores to other objects. -@inline @generated function unsafe_invariant_load(ptr::LLVMPtr{T, AS}, i::I, ::Val{align}) where {T, AS, I, align} +@inline function unsafe_invariant_load(ptr::LLVMPtr{T}, i::I, ::Val{align}) where {T, I, align} sizeof(T) == 0 && return T.instance - ispow2(align) || return :(error("unsafe_invariant_load: alignment must be a power of 2, got $($align)")) - return generate_llvmcall(T, Tuple{LLVMPtr{T, AS}, I}, :ptr, :(i - one(I))) do builder, ptr, idx - eltyp = convert(LLVMType, T) - T_typed_ptr = LLVM.PointerType(eltyp, AS) - - base = if supports_typed_pointers(LLVM.context()) - bitcast!(builder, ptr, T_typed_ptr) - else - ptr - end - gep = inbounds_gep!(builder, eltyp, base, [idx]) - ld = load!(builder, eltyp, gep) - if AS != 0 - ld.metadata[MD_tbaa] = tbaa_addrspace(AS) - end - ld.metadata[MD_invariant_load] = MDNode(LLVM.Metadata[]) - ld.alignment = align - - ld + ispow2(align) || error("unsafe_invariant_load: alignment must be a power of 2, got ", align) + return _unsafe_invariant_load(ptr, i - one(I), Val(align)) +end +@llvmgenerated builder function _unsafe_invariant_load( + ptr::LLVMPtr{T, AS}, i::Integer, ::Val{align} + )::T where {T, AS, align} + eltyp = convert(LLVMType, T) + # `LLVMPtr` is an `i8*` with typed pointers (with opaque pointers, this cast folds away) + ptr = bitcast!(builder, ptr, LLVM.PointerType(eltyp, AS)) + ld = load!(builder, eltyp, inbounds_gep!(builder, eltyp, ptr, [i]); align) + if AS != 0 + ld.metadata[MD_tbaa] = tbaa_addrspace(AS) end + ld.metadata[MD_invariant_load] = MDNode(LLVM.Metadata[]) + return ld end @device_function @inline function const_arrayref(A::CLDeviceArray{T}, index::Integer) where {T} diff --git a/src/pocl/device/random.jl b/src/pocl/device/random.jl index ba08a11fb..b3f8f52f0 100644 --- a/src/pocl/device/random.jl +++ b/src/pocl/device/random.jl @@ -159,40 +159,33 @@ function Random.rand(rng::Philox2x32{R}, ::Type{UInt64}) where {R} end -# a hacky method of exposing constant tables as constant GPU memory - -function emit_constant_array(name::Symbol, data::AbstractArray{T}) where {T} - return generate_llvmcall(LLVMPtr{T, AS.UniformConstant}, Tuple{}) do builder - T_val = convert(LLVMType, T) - T_ptr = convert(LLVMType, LLVMPtr{T, AS.UniformConstant}) - - # get LLVM module - mod = current_module(builder) - - # create a global memory global variable - # TODO: global_var alignment? - T_global = LLVM.ArrayType(T_val, length(data)) - # XXX: why can't we use a single name like emit_shmem - gv = GlobalVariable(mod, T_global, "gpu_$(name)_data", AS.UniformConstant) - gv.linkage = LLVM.Linkage.Internal - gv.initializer = ConstantArray(data) - gv.alignment = 16 - - # generate IR - ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) - - untyped_ptr = bitcast!(builder, ptr, T_ptr) - end +# a hacky method of exposing constant tables as constant GPU memory: the table `Random.$name` +# becomes an internal global in the constant address space. its contents are embedded in the +# cached IR, so Random's tables are assumed to be immutable. +@llvmgenerated builder function emit_constant_array( + ::Val{name}, ::Type{T} + )::LLVMPtr{T, AS.UniformConstant} where {name, T} + data = getfield(Random, name)::AbstractArray{T} + + # create a global memory global variable + # TODO: global_var alignment? + T_global = LLVM.ArrayType(convert(LLVMType, T), length(data)) + # XXX: why can't we use a single name like emit_shmem + gv = GlobalVariable(current_module(builder), T_global, "gpu_$(name)_data", AS.UniformConstant) + gv.linkage = LLVM.Linkage.Internal + gv.initializer = ConstantArray(data) + gv.alignment = 16 + + ptr = gep!(builder, T_global, gv, [ConstantInt(0), ConstantInt(0)]) + return bitcast!(builder, ptr, convert(LLVMType, LLVMPtr{T, AS.UniformConstant})) end for var in [:ki, :wi, :fi, :ke, :we, :fe] val = getfield(Random, var) gpu_var = Symbol("gpu_$var") arr_typ = :(CLDeviceArray{$(eltype(val)), $(ndims(val)), AS.UniformConstant}) - @eval @inline @generated function $gpu_var() - ptr = emit_constant_array($(QuoteNode(var)), $val) - return Expr(:call, $arr_typ, $(size(val)), ptr) - end + @eval @inline $gpu_var() = + $arr_typ($(size(val)), emit_constant_array(Val($(QuoteNode(var))), $(eltype(val)))) end ## randn diff --git a/src/pocl/device/runtime.jl b/src/pocl/device/runtime.jl index e3809c93d..965d8f015 100644 --- a/src/pocl/device/runtime.jl +++ b/src/pocl/device/runtime.jl @@ -47,26 +47,17 @@ end function additional_arg_intr(mod::LLVM.Module, T_state, name) return get!(mod.functions, "julia.opencl.$name") do state_intr = LLVM.Function(mod, "julia.opencl.$name", LLVM.FunctionType(T_state)) - push!(state_intr.function_attributes, EnumAttribute(:readnone)) + state_intr.memory_effects = MemoryEffects(:none) state_intr end end # run-time equivalent -function additional_arg_value(state, name) - return generate_llvmcall(state, Tuple{}) do builder - T_state = convert(LLVMType, state) - - # get intrinsic - state_intr = additional_arg_intr(current_module(builder), T_state, name) - state_intr_ft = state_intr.function_type - - # generate IR - call!(builder, state_intr_ft, state_intr, Value[], name) - end +@llvmgenerated builder function additional_arg_value(::Type{T}, ::Val{name})::T where {T, name} + state_intr = additional_arg_intr(current_module(builder), convert(LLVMType, T), name) + call!(builder, state_intr.function_type, state_intr, Value[], String(name)) end for name in [:random_keys, :random_counters] - @eval @inline @generated $name() = - additional_arg_value(LLVMPtr{UInt32, AS.Workgroup}, $(String(name))) + @eval @inline $name() = additional_arg_value(LLVMPtr{UInt32, AS.Workgroup}, Val($(QuoteNode(name)))) end From 06656b6f8e735b7d52c15b9c52bbc2600279f5bc Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Thu, 1 Oct 2026 13:03:36 +0200 Subject: [PATCH 4/5] Dispose of the compiled IR once it has been inspected GPUCompiler now hands the module that Julia's code generator produced to the caller of compile: on Julia 1.11 and later it moves the module out of the native code, and on 1.10 (LLVM 15, which can't do that) it returns the module for the caller to consume or dispose of. compile_to_obj only read the entry point's name and attributes, and never disposed of the module. That leaked it: up to Julia 1.13, the native-code descriptor is never freed and keeps its thread-safe module, and with it the context, alive, so the module isn't freed together with the context either. memcheck also reported every compiled module. Dispose of the IR with @dispose inside the JuliaContext block, once the entry point has been inspected, like GPUCompiler's own callers do, so that it is also disposed of when the inspection throws. --- src/pocl/compiler/compilation.jl | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/src/pocl/compiler/compilation.jl b/src/pocl/compiler/compilation.jl index c8e4b9251..11d0c9670 100644 --- a/src/pocl/compiler/compilation.jl +++ b/src/pocl/compiler/compilation.jl @@ -235,10 +235,12 @@ function compile_to_obj(@nospecialize(job::CompilerJob)) return JuliaContext() do ctx obj, meta = invoke_frozen(GPUCompiler.compile, :obj, job) - entry = meta.entry.name - device_rng = haskey(meta.entry.function_attributes, "julia.opencl.rng") - - (; obj, entry, device_rng) + # we own the IR: inspect it, then dispose of it + @dispose ir = meta.ir begin + entry = meta.entry.name + device_rng = haskey(meta.entry.function_attributes, "julia.opencl.rng") + (; obj, entry, device_rng) + end end end From d3696fcd93a9652f5145f6711f8d927ff47f4b27 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Thu, 1 Oct 2026 10:02:03 +0200 Subject: [PATCH 5/5] Widen the index of invariant loads to Int before the GEP The index was decremented in its own type and passed to getelementptr as is, which sign-extends narrower indices, so e.g. a UInt32 index of 0x80000001 addressed a negative offset. Convert it to Int in Julia first, like LLVM.jl's unsafe_load does. --- src/pocl/device/array.jl | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/src/pocl/device/array.jl b/src/pocl/device/array.jl index 67b814605..10459a4ec 100644 --- a/src/pocl/device/array.jl +++ b/src/pocl/device/array.jl @@ -159,13 +159,16 @@ end # There is no SPIR-V equivalent of NVPTX's `ld.global.nc`, so instead of a dedicated # instruction we mark the load `!invariant.load`, which lets LLVM hoist it out of loops # and reorder it across stores to other objects. -@inline function unsafe_invariant_load(ptr::LLVMPtr{T}, i::I, ::Val{align}) where {T, I, align} +# +# like `unsafe_load`, the index is widened to `Int` in Julia, where its signedness is known, +# because `getelementptr` sign-extends narrower indices. +@inline function unsafe_invariant_load(ptr::LLVMPtr{T}, i::Integer, ::Val{align}) where {T, align} sizeof(T) == 0 && return T.instance ispow2(align) || error("unsafe_invariant_load: alignment must be a power of 2, got ", align) - return _unsafe_invariant_load(ptr, i - one(I), Val(align)) + return _unsafe_invariant_load(ptr, Int(i) - 1, Val(align)) end @llvmgenerated builder function _unsafe_invariant_load( - ptr::LLVMPtr{T, AS}, i::Integer, ::Val{align} + ptr::LLVMPtr{T, AS}, i::Int, ::Val{align} )::T where {T, AS, align} eltyp = convert(LLVMType, T) # `LLVMPtr` is an `i8*` with typed pointers (with opaque pointers, this cast folds away)