From 9eb33a1c58f12f6e8dc31bb7fd5f98c7e5649ec2 Mon Sep 17 00:00:00 2001 From: Tim Besard Date: Fri, 25 Sep 2026 07:57:18 +0200 Subject: [PATCH] Drop thrown values on targets that can't throw. Targets that cannot throw report exceptions without inspecting the thrown value. The new Julia IR pass `drop_throw_arguments!` replaces the argument of `throw` with `nothing` on those targets, allowing Julia's DCE to remove exception construction when proven removable. Such dead constructions used to survive to the back-end because Julia does not inline into throw blocks and the constructors allocate through the device allocator (for example, `InexactError`'s boxed arguments, #904, and `DomainError` with a `LazyString`), with GC orderings on the stores into those objects. This supersedes `lower_throw!`'s LLVM-level erasure of the thrown value, which erased the call's direct argument regardless of what it was. --- src/interface.jl | 12 +++++- src/irgen.jl | 27 ++----------- src/jlgen.jl | 46 ++++++++++++++++++++++ test/gcn.jl | 10 ++--- test/helpers/runtime.jl | 11 ++++++ test/metal.jl | 61 ++++++++++++++++++----------- test/native.jl | 85 +++++++++++++++++++++++++++++++++++++++++ test/ptx.jl | 10 ++--- test/spirv.jl | 41 ++++++++++++++++++++ 9 files changed, 246 insertions(+), 57 deletions(-) diff --git a/src/interface.jl b/src/interface.jl index d0187e38..df3805a1 100644 --- a/src/interface.jl +++ b/src/interface.jl @@ -399,7 +399,9 @@ end end # does this target support throwing Julia exceptions with jl_throw? -# if not, calls to throw will be replaced with calls to the GPU runtime +# if not, calls to throw will be replaced with calls to the GPU runtime, which report the +# exception without looking at the thrown value, so that value is not computed at all +# (see `drop_throw_arguments!`) can_throw(@nospecialize(job::CompilerJob)) = uses_julia_runtime(job) # does this target support loading from Julia safepoints? @@ -648,7 +650,13 @@ optimization_params(@nospecialize(job::CompilerJob)) = # # GPUCompiler.julia_ir_passes(job::CompilerJob{MyTarget}) = # (@invoke(GPUCompiler.julia_ir_passes(job::CompilerJob))..., my_pass!) -julia_ir_passes(@nospecialize(job::CompilerJob)) = () +function julia_ir_passes(@nospecialize(job::CompilerJob)) + passes = () + if !can_throw(job) + passes = (passes..., drop_throw_arguments!) + end + return passes +end # how much debuginfo to emit function llvm_debug_info(@nospecialize(job::CompilerJob)) diff --git a/src/irgen.jl b/src/irgen.jl index cbaef211..2dbf5ddd 100644 --- a/src/irgen.jl +++ b/src/irgen.jl @@ -156,13 +156,10 @@ end ## exception handling -# this pass lowers `jl_throw` and friends to GPU-compatible exceptions. -# this isn't strictly necessary, but has a couple of advantages: -# - we can kill off unused exception arguments that otherwise would allocate or invoke -# - we can fake debug information (lacking a stack unwinder) -# -# once we have thorough inference (ie. discarding `@nospecialize` and thus supporting -# exception arguments) and proper debug info to unwind the stack, this pass can go. +# this pass lowers `jl_throw` and friends to GPU-compatible exceptions, reporting the kind of +# exception and faking debug information (lacking a stack unwinder). the thrown values are not +# used: throws in Julia code already throw `nothing` (see `drop_throw_arguments!`), and what +# remains of the arguments of codegen's own throws (e.g. `jl_type_error`) is left to DCE. function lower_throw!(@nospecialize(job::CompilerJob), mod::LLVM.Module) changed = false @tracepoint "lower throw" begin @@ -207,24 +204,8 @@ function lower_throw!(@nospecialize(job::CompilerJob), mod::LLVM.Module) end # remove the call - call_args = arguments(call) erase!(call) - # HACK: kill the exceptions' unused arguments - # this is needed for throwing objects with @nospecialize constructors. - for arg in call_args - # peek through casts - if isa(arg, LLVM.AddrSpaceCastInst) - cast = arg - arg = first(operands(cast)) - isempty(uses(cast)) && erase!(cast) - end - - if isa(arg, LLVM.Instruction) && isempty(uses(arg)) - erase!(arg) - end - end - changed = true end diff --git a/src/jlgen.jl b/src/jlgen.jl index 8fadad85..1b27c00b 100644 --- a/src/jlgen.jl +++ b/src/jlgen.jl @@ -354,6 +354,52 @@ end end end +@static if hasfield(CC.InstructionStream, :stmt) + ir_stmts(ir::CC.IRCode) = ir.stmts.stmt +else + ir_stmts(ir::CC.IRCode) = ir.stmts.inst +end + +# remove the statements a pass made dead: `compact!` removes unused statements whose effects +# allow it, transitively, and `adce_pass!` also dead phi cycles (as Julia's own pipeline does) +function julia_ir_dce!(ir::CC.IRCode, opt::CC.OptimizationState) + ir = CC.compact!(ir) + res = CC.adce_pass!(ir, opt.inlining) + if res isa CC.IRCode # 1.10 + ir = CC.compact!(res, true) + else # 1.11+: `ir => made_changes` + ir, made_changes = res + made_changes && (ir = CC.compact!(ir, true)) + end + return ir +end + +# replace the argument of every `throw` by `nothing` +# +# Targets that cannot throw (see `can_throw`) lower `throw` to an exception report that does +# not look at the thrown value (see `lower_throw!`), so building that value is wasted work. +# Worse, it often survives: Julia does not inline into throw blocks, exception constructors +# allocate (e.g. `InexactError`'s boxed `args` tuple) through the device allocator, which LLVM +# cannot remove, and Julia's GC lowering puts atomic orderings on the stores into such dead +# objects that some back-ends cannot express. Without the use, Julia's DCE removes every +# construction whose effects it proved `removable_if_unused`, which covers the exceptions Base +# throws with lazily-built messages. Constructions that are not proven removable, like eagerly +# building a message string, stay. +function drop_throw_arguments!(::CC.AbstractInterpreter, opt::CC.OptimizationState, + ir::CC.IRCode) + stmts = ir_stmts(ir) + changed = false + for i in eachindex(stmts) + stmt = stmts[i] + (stmt isa Expr && stmt.head === :call && length(stmt.args) == 2) || continue + stmt.args[2] === nothing && continue + CC.singleton_type(CC.argextype(stmt.args[1], ir)) === Core.throw || continue + stmts[i] = Expr(:call, stmt.args[1], nothing) + changed = true + end + return changed ? julia_ir_dce!(ir, opt) : ir +end + ## driving inference and walking callees diff --git a/test/gcn.jl b/test/gcn.jl index 0c667959..3e904824 100644 --- a/test/gcn.jl +++ b/test/gcn.jl @@ -526,19 +526,19 @@ end @testset "float boxes" begin mod = @eval module $(gensym()) function kernel(a,b) - # Int32(a) may fail, boxing the Float32 for the @nospecialize ctor + # Int32(a) may fail, throwing an `InexactError`, whose `@nospecialize` + # constructor would box the Float32 c = Int32(a) unsafe_store!(b, c) return end end + # the exception object isn't constructed, as nothing looks at it @test @filecheck begin @check_label "define void @{{(julia|j)_kernel_[0-9]+}}" - # 1.10 boxes through jl_box_float32; 1.11+ specializes the constructor - # and boxes through the GC pool allocator instead - @check cond=(VERSION < v"1.11-") "jl_box_float32" - @check cond=(VERSION >= v"1.11-") "gpu_gc_pool_alloc" + @check_not "jl_box_float32" + @check_not "gpu_gc_pool_alloc" GCN.code_llvm(mod.kernel, Tuple{Float32,Ptr{Float32}}; dump_module=true) end GCN.code_native(devnull, mod.kernel, Tuple{Float32,Ptr{Float32}}) diff --git a/test/helpers/runtime.jl b/test/helpers/runtime.jl index c35cbb04..5207afc3 100644 --- a/test/helpers/runtime.jl +++ b/test/helpers/runtime.jl @@ -7,3 +7,14 @@ module TestRuntime report_exception_name(ex) = return report_exception_frame(idx, func, file, line) = return end + +# a runtime with an external allocator, keeping allocations visible to LLVM (with a +# constant-null allocator, allocations and everything that follows them fold away) +module ExternalAllocatorRuntime + malloc(sz) = ccall("extern test_malloc", llvmcall, Ptr{Nothing}, (Csize_t,), sz) + signal_exception() = return + report_oom(sz) = return + report_exception(ex) = return + report_exception_name(ex) = return + report_exception_frame(idx, func, file, line) = return +end diff --git a/test/metal.jl b/test/metal.jl index 2a35f40c..26a78e65 100644 --- a/test/metal.jl +++ b/test/metal.jl @@ -1690,36 +1690,53 @@ end end end -@testset "Bool conversion exception allocation" begin +@testset "exception allocations" begin + # a thrown exception's construction is dead once `throw` is lowered, but it used to survive + # (e.g. as an un-inlined constructor call), allocating through the device allocator, with + # Julia's GC orderings on the stores into it (`unordered`, `release`) that AIR cannot + # express (#904, Metal.jl#955) mod = @eval module $(gensym()) using ..GPUCompiler - module Runtime - # Keep allocation visible to LLVM; a constant-null allocator erases - # the heap-reference stores which exposed #904. - malloc(sz) = ccall("extern test_malloc", llvmcall, Ptr{Nothing}, (Csize_t,), sz) - signal_exception() = return - report_oom(sz) = return - report_exception(ex) = return - report_exception_name(ex) = return - report_exception_frame(idx, func, file, line) = return - end + import ..ExternalAllocatorRuntime struct Params <: GPUCompiler.AbstractCompilerParams end - GPUCompiler.runtime_module(::CompilerJob{<:Any,Params}) = Runtime - function kernel(out, x) + GPUCompiler.runtime_module(::CompilerJob{<:Any,Params}) = ExternalAllocatorRuntime + GPUCompiler.isintrinsic(job::CompilerJob{MetalCompilerTarget,Params}, fn::String) = + fn == "test_malloc" || + @invoke GPUCompiler.isintrinsic(job::CompilerJob{MetalCompilerTarget}, fn::String) + + # `InexactError` boxes its arguments in a tuple (#904) + function bool(out, x) unsafe_store!(out, Bool(x)) return end + + # `DomainError` with a lazily-built message (Metal.jl#955) + function domain(out, x) + x < 0 && throw(DomainError(x, LazyString("log1p was called with ", x))) + unsafe_store!(out, x) + return + end + + # throws emitted by codegen + function tuple_index(out, t, i) + unsafe_store!(out, t[i]) + return + end end - source = methodinstance(typeof(mod.kernel), Tuple{Core.LLVMPtr{Bool,1},Int}, - Base.get_world_counter()) - target = MetalCompilerTarget(; macos=v"12.2", metal=v"3.0", air=v"3.0") - job = CompilerJob(source, CompilerConfig(target, mod.Params(); kernel=true)) - ir = sprint(io -> GPUCompiler.code_llvm(io, job; dump_module=true)) - if VERSION >= v"1.12-" - @test occursin(r"store atomic .* unordered", ir) + + for (f, tt) in ((mod.bool, Tuple{Core.LLVMPtr{Bool,1},Int}), + (mod.domain, Tuple{Core.LLVMPtr{Float32,1},Float32}), + (mod.tuple_index, Tuple{Core.LLVMPtr{Int,1},NTuple{3,Int},Int})) + source = methodinstance(typeof(f), tt, Base.get_world_counter()) + target = MetalCompilerTarget(; macos=v"12.2", metal=v"3.0", air=v"3.0") + job = CompilerJob(source, CompilerConfig(target, mod.Params(); kernel=true)) + + @test @filecheck implicit_check_not=["{{(load|store) atomic|atomicrmw|cmpxchg}}", "call {{.*}}@test_malloc"] begin + @check "define void @_Z" + @check "call void @llvm.trap()" + GPUCompiler.code_llvm(stdout, job; dump_module=true) + end end - air = sprint(io -> GPUCompiler.code_native(io, job; dump_module=true)) - @test !occursin(r"(load|store) atomic", air) end @testset "atomic demotion" begin diff --git a/test/native.jl b/test/native.jl index 9dc37ca2..0a166cd2 100644 --- a/test/native.jl +++ b/test/native.jl @@ -937,6 +937,91 @@ end @test :callee in mod.seen end +@testset "throw arguments" begin + mod = @eval module $(gensym()) + # an exception object that Julia can prove removable + removable(x) = x > 1 ? throw(InexactError(:removable, Int, x)) : x + # a message with side effects, which have to stay + eager(x) = x > 1 ? throw(ArgumentError(string("bad value ", x))) : x + end + # resolve callees like the pass does (on Julia 1.14, they aren't `GlobalRef`s anymore) + CC = Core.Compiler + is_throw(ci, stmt) = Meta.isexpr(stmt, :call) && + CC.singleton_type(CC.argextype(stmt.args[1], ci, CC.VarState[])) === Core.throw + throws(ci) = filter(stmt -> is_throw(ci, stmt), ci.code) + invokes(ci, name) = count(stmt -> Meta.isexpr(stmt, :invoke) && + occursin(name, string(stmt.args[2])), ci.code) + + # with the Julia runtime, exceptions are thrown as usual + ci, _ = only(Native.code_typed(mod.removable, Tuple{Int}; jlruntime=true)) + @test !isempty(throws(ci)) + @test all(stmt -> stmt.args[2] isa Core.SSAValue, throws(ci)) + @test invokes(ci, "InexactError") == 1 + + # without it, the thrown value is unused, and so is its construction + ci, _ = only(Native.code_typed(mod.removable, Tuple{Int}; jlruntime=false)) + @test !isempty(throws(ci)) + @test all(stmt -> stmt.args[2] === nothing, throws(ci)) + @test invokes(ci, "InexactError") == 0 + + # unless the construction isn't known to be free of side effects + ci, _ = only(Native.code_typed(mod.eager, Tuple{Int}; jlruntime=false)) + @test !isempty(throws(ci)) + @test all(stmt -> stmt.args[2] === nothing, throws(ci)) + @test invokes(ci, "string") + invokes(ci, "print_to_string") >= 1 +end + +@testset "throw arguments: dead code" begin + mod = @eval module $(gensym()) + # a constructor with a side effect + struct Logged <: Exception + code::Int + @noinline function Logged(p::Ptr{Int}, code::Int) + unsafe_store!(p, code) + new(code) + end + end + function side_effect(p::Ptr{Int}, x::Int) + x > 1 && throw(Logged(p, x)) + return x + end + + # exception values swapped around a loop (a dead phi cycle, only removed by ADCE) + function loop_carried(x::Int) + a = InexactError(:loop_carried, Int, x) + b = InexactError(:loop_carried, Int, -x) + i = 0 + while i < x + i += 1 + a, b = b, a + end + x > 100 && throw(a) + return i + end + + # a `@nospecialize` helper, whose inferred source Julia < 1.12 used to discard + @noinline throw_nospecialize(@nospecialize(x)) = throw(ArgumentError("nospecialize")) + nospecialize(x::Int) = x > 1 ? throw_nospecialize(x) : x + end + invokes(ci, name) = count(stmt -> Meta.isexpr(stmt, :invoke) && + occursin(name, string(stmt.args[2])), ci.code) + + ci, _ = only(Native.code_typed(mod.side_effect, Tuple{Ptr{Int}, Int}; jlruntime=false)) + @test invokes(ci, "Logged") == 1 + + ci, _ = only(Native.code_typed(mod.loop_carried, Tuple{Int}; jlruntime=false)) + @test invokes(ci, "InexactError") == 0 + @test !any(stmt -> stmt isa Core.PhiNode && + any(v -> v isa Core.SSAValue && ci.ssavaluetypes[v.id] <: InexactError, + stmt.values), ci.code) + + @test @filecheck begin + @check_label "define {{.*}} @{{(julia|j)_nospecialize_[0-9]+}}" + @check_not "ArgumentError" + Native.code_llvm(mod.nospecialize, Tuple{Int}; jlruntime=false, dump_module=true) + end +end + @testset "function attributes" begin mod = @eval module $(gensym()) @inline function convergent_barrier() diff --git a/test/ptx.jl b/test/ptx.jl index 2a7734a4..69c3f160 100644 --- a/test/ptx.jl +++ b/test/ptx.jl @@ -519,19 +519,19 @@ end @testset "float boxes" begin mod = @eval module $(gensym()) function kernel(a,b) - # Int32(a) may fail, boxing the Float32 for the @nospecialize ctor + # Int32(a) may fail, throwing an `InexactError`, whose `@nospecialize` + # constructor would box the Float32 c = Int32(a) unsafe_store!(b, c) return end end + # the exception object isn't constructed, as nothing looks at it @test @filecheck begin @check_label "define void @{{(julia|j)_kernel_[0-9]+}}" - # 1.10 boxes through jl_box_float32; 1.11+ specializes the constructor - # and boxes through the GC pool allocator instead - @check cond=(VERSION < v"1.11-") "jl_box_float32" - @check cond=(VERSION >= v"1.11-") "gpu_gc_pool_alloc" + @check_not "jl_box_float32" + @check_not "gpu_gc_pool_alloc" PTX.code_llvm(mod.kernel, Tuple{Float32,Ptr{Float32}}; dump_module=true) end PTX.code_native(devnull, mod.kernel, Tuple{Float32,Ptr{Float32}}) diff --git a/test/spirv.jl b/test/spirv.jl index 8fd930ac..785d009c 100644 --- a/test/spirv.jl +++ b/test/spirv.jl @@ -322,6 +322,47 @@ end end end +@testset "exception allocations" begin + # a thrown exception's construction is dead once `throw` is lowered, but it used to survive + # (e.g. as an un-inlined constructor call), allocating through the device allocator, with + # Julia's GC orderings on the stores into it, which SPIR-V cannot express for pointers + # (#924) + mod = @eval module $(gensym()) + using ..GPUCompiler + import ..ExternalAllocatorRuntime + struct Params <: GPUCompiler.AbstractCompilerParams end + GPUCompiler.runtime_module(::CompilerJob{<:Any,Params}) = ExternalAllocatorRuntime + GPUCompiler.isintrinsic(job::CompilerJob{SPIRVCompilerTarget,Params}, fn::String) = + fn == "test_malloc" || + @invoke GPUCompiler.isintrinsic(job::CompilerJob{SPIRVCompilerTarget}, fn::String) + + # `InexactError` boxes its arguments in a tuple + function bool(out, x) + unsafe_store!(out, Bool(x)) + return + end + + # `DomainError` with a lazily-built message + function domain(out, x) + x < 0 && throw(DomainError(x, LazyString("log1p was called with ", x))) + unsafe_store!(out, x) + return + end + end + + for (f, tt) in ((mod.bool, Tuple{Core.LLVMPtr{Bool,1},Int}), + (mod.domain, Tuple{Core.LLVMPtr{Float32,1},Float32})) + source = methodinstance(typeof(f), tt, Base.get_world_counter()) + target = SPIRVCompilerTarget(; backend, validate=true) + job = CompilerJob(source, CompilerConfig(target, mod.Params(); kernel=true)) + + @test @filecheck implicit_check_not=["{{(load|store) atomic|atomicrmw|cmpxchg}}", "call {{.*}}@test_malloc"] begin + @check "define spir_kernel void @_Z" + GPUCompiler.code_llvm(stdout, job; dump_module=true) + end + end +end + @testset "atomic demotion" begin # Julia's `unordered` heap-reference accesses and `release` type-tag stores cannot be # expressed in SPIR-V when they involve pointers (OpAtomicLoad/OpAtomicStore take scalars