Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 0 additions & 28 deletions src/irgen.jl
Original file line number Diff line number Diff line change
Expand Up @@ -343,34 +343,6 @@ function inline_unreachable_control_flow!(@nospecialize(job::CompilerJob), mod::
return changed
end

# demote LLVM atomic loads and stores to plain ones
#
# Julia marks accesses to heap references `unordered` so that a read racing with the GC, or
# with another thread's write, cannot observe a torn pointer, and stores the type tag of a
# freshly allocated object with `release` ordering so that no other thread can observe the
# object before its header. There is no device GC and no such race for GPUCompiler to
# protect against, so these orderings carry no meaning here, but not every back-end can
# express them: SPIR-V's OpAtomicLoad/OpAtomicStore only take scalar integer or
# floating-point operands, so the Khronos translator turns an atomic access of a pointer
# into an invalid pointer-typed atomic that consumers reject (Intel's compiler fails with an
# undefined `__spirv_AtomicLoad(long**, int, int)`), and AIR has no atomic load or store
# instructions at all: Apple's back-end aborts on them (`XPC_ERROR_CONNECTION_INTERRUPTED`
# from the driver; the macOS 26 AGX compiler reports `unable to legalize instruction:
# store release (p0)` for a type-tag store through the generic pointer the device allocator
# returns). Run after optimization, where dropping the ordering cannot enable new
# transformations. Device-side atomics proper go through target intrinsics, not these
# instructions, so every remaining one is such Julia bookkeeping and gets demoted.
function demote_atomics!(mod::LLVM.Module)
changed = false
for f in functions(mod), bb in blocks(f), inst in instructions(bb)
(inst isa LLVM.LoadInst || inst isa LLVM.StoreInst) || continue
is_atomic(inst) || continue
ordering!(inst, LLVM.API.LLVMAtomicOrderingNotAtomic)
changed = true
end
return changed
end

# lower `trap` to a clean return to get rid of `unreachable` and `noreturn`
#
# this is for compatibility with back-ends that don't support (SPIR-V) or have
Expand Down
3 changes: 0 additions & 3 deletions src/metal.jl
Original file line number Diff line number Diff line change
Expand Up @@ -659,9 +659,6 @@ function lower_air!(@nospecialize(job::CompilerJob{MetalCompilerTarget}), mod::L
# Metal.malloc uses.
rewrite_generic_null_selects!(mod)

# AIR does not support LLVM atomic load/store instructions (see `demote_atomics!`)
demote_atomics!(mod)

# the macOS 27 back-end rejects bare LLVM fences (Metal.jl#968)
lower_fences!(job, mod)

Expand Down
4 changes: 0 additions & 4 deletions src/spirv.jl
Original file line number Diff line number Diff line change
Expand Up @@ -115,10 +115,6 @@ function finish_ir!(job::CompilerJob{SPIRVCompilerTarget}, mod::LLVM.Module,
# OpUnreachable (UB if reached), which PoCL and friends handle poorly.
lower_unreachable_control_flow!(job, mod)

# SPIR-V cannot express atomic loads and stores of pointers, which is what Julia's
# heap-reference accesses and type-tag stores are; the orderings serve no purpose on device
demote_atomics!(mod)

# the SPIR-V back-ends lower `llvm.minimum`/`llvm.maximum` to NaN-ignoring `fmin`/`fmax`
lower_minimum_maximum!(mod)

Expand Down
81 changes: 12 additions & 69 deletions test/metal.jl
Original file line number Diff line number Diff line change
Expand Up @@ -1739,7 +1739,6 @@ end
end
end


@testset "unsupported allocations" begin
# without a garbage collector, device code can only allocate objects that do not
# reference other objects
Expand Down Expand Up @@ -1835,77 +1834,21 @@ end
end
end

@testset "atomic demotion" begin
# Demote every LLVM atomic load/store, whatever its ordering or value type.
Context() do ctx
ir = """
define void @f(i8** %p, i8** %q, i64* %r) {
entry:
%a = load atomic i8*, i8** %p unordered, align 8
store atomic i8* %a, i8** %q unordered, align 8
%t = load atomic i8*, i8** %p acquire, align 8
store atomic i8* %t, i8** %q release, align 8
%b = load atomic i64, i64* %r monotonic, align 8
store atomic i64 %b, i64* %r release, align 8
store i8* %a, i8** %p, align 8
ret void
}
"""
mod = parse(LLVM.Module, ir)
insts() = [i for f in functions(mod) for bb in blocks(f) for i in instructions(bb)]
memops() = filter(i -> i isa LLVM.LoadInst || i isa LLVM.StoreInst, insts())

@test count(is_atomic, memops()) == 6
@test GPUCompiler.demote_atomics!(mod)
@test count(is_atomic, memops()) == 0
@test !occursin("atomic", string(mod))
@test (verify(mod); true)
# idempotent
@test !GPUCompiler.demote_atomics!(mod)
end

# end-to-end: Julia's own `:unordered` accesses (as codegen emits for heap-reference
# fields) must reach the AIR as plain loads and stores
function kernel(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1})
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int}, p), :unordered)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int}, q), x, :unordered)
@testset "LLVM atomics are not demoted" begin
# atomics in user code (e.g. UnsafeAtomics' `load`/`store!`) must reach the back-end
function kernel(p::Core.LLVMPtr{Int32,1}, q::Core.LLVMPtr{Int32,1})
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int32}, p), :acquire)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int32}, q), x, :release)
return
end
source = methodinstance(typeof(kernel), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}},
Base.get_world_counter())
target = MetalCompilerTarget(; macos=v"12.2", metal=v"3.0", air=v"3.0")
config = CompilerConfig(target, Metal.CompilerParams(); kernel=true)
job = CompilerJob(source, config)

# precondition: the accesses survive optimization as unordered atomics
ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true))
@test occursin(r"load atomic .* unordered", ir)
@test occursin(r"store atomic .* unordered", ir)

air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true))
@test !occursin(r"(load|store) atomic", air)
@test occursin(r"load i64", air)
@test occursin(r"store i64", air)

# likewise for the `release` store of a pointer that codegen emits for an allocated
# object's type tag. Only Julia 1.12+ lowers `Ptr` values to LLVM pointers, which is
# what makes this the case Apple's back-end cannot legalize.
@static if VERSION >= v"1.12"
function tagged(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1})
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Ptr{Int}}, p), :acquire)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int}}, q), x, :release)
return
end
source = methodinstance(typeof(tagged), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}},
Base.get_world_counter())
job = CompilerJob(source, config)

ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true))
@test occursin(r"load atomic ptr.* acquire", ir)
@test occursin(r"store atomic ptr.* release", ir)

air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true))
@test !occursin(r"(load|store) atomic", air)
@test @filecheck begin
@check "load atomic i32"
@check_same "acquire"
@check "store atomic i32"
@check_same "release"
Metal.code_native(kernel, Tuple{Core.LLVMPtr{Int32,1}, Core.LLVMPtr{Int32,1}};
kernel=true, dump_module=true)
end
end

Expand Down
31 changes: 13 additions & 18 deletions test/spirv.jl
Original file line number Diff line number Diff line change
Expand Up @@ -417,37 +417,32 @@ end
end
end


@testset "atomic demotion" begin
# Julia's `unordered` heap-reference accesses and `release` type-tag stores cannot be
# expressed in SPIR-V when they involve pointers (OpAtomicLoad/OpAtomicStore take scalars
# only): the translator would emit an invalid pointer-typed atomic. They carry no meaning
# without a device GC, so `demote_atomics!` turns them into plain accesses.
@testset "LLVM atomics" begin
# atomics in user code (e.g. UnsafeAtomics' `load`/`store!`, Atomix' `get`/`set!`) must
# reach the back-end
mod = @eval module $(gensym())
function kernel(p::Ptr{Ptr{Int}}, q::Ptr{Ptr{Int}})
x = Core.Intrinsics.atomic_pointerref(p, :unordered)
Core.Intrinsics.atomic_pointerset(q, x, :unordered)
y = Core.Intrinsics.atomic_pointerref(p, :acquire)
Core.Intrinsics.atomic_pointerset(q, y, :release)
function kernel(p::Ptr{Int32}, q::Ptr{Int32})
x = Core.Intrinsics.atomic_pointerref(p, :acquire)
Core.Intrinsics.atomic_pointerset(q, x, :release)
return
end
end
tt = Tuple{Ptr{Ptr{Int}}, Ptr{Ptr{Int}}}
tt = Tuple{Ptr{Int32}, Ptr{Int32}}

@test @filecheck begin
@check_label "define spir_kernel void @_Z6kernel"
@check_not "load atomic"
@check_not "store atomic"
@check "ret void"
@check "load atomic i32"
@check_same "acquire"
@check "store atomic i32"
@check_same "release"
SPIRV.code_llvm(mod.kernel, tt; backend, kernel=true)
end

# the SPIR-V is validated by the helper, so an invalid pointer-typed atomic would fail here
@test @filecheck begin
@check "OpEntryPoint Kernel %[[KERNEL:[^ ]+]]"
@check "%[[KERNEL]] = OpFunction %void None"
@check_not "OpAtomicLoad"
@check_not "OpAtomicStore"
@check "OpAtomicLoad"
@check "OpAtomicStore"
SPIRV.code_native(mod.kernel, tt; backend, kernel=true)
end
end
Expand Down
Loading