Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 14 additions & 18 deletions src/irgen.jl
Original file line number Diff line number Diff line change
Expand Up @@ -343,28 +343,24 @@ function inline_unreachable_control_flow!(@nospecialize(job::CompilerJob), mod::
return changed
end

# demote LLVM atomic loads and stores to plain ones
# demote `unordered` LLVM atomic loads and stores to plain ones
#
# Julia marks accesses to heap references `unordered` so that a read racing with the GC, or
# with another thread's write, cannot observe a torn pointer, and stores the type tag of a
# freshly allocated object with `release` ordering so that no other thread can observe the
# object before its header. There is no device GC and no such race for GPUCompiler to
# protect against, so these orderings carry no meaning here, but not every back-end can
# express them: SPIR-V's OpAtomicLoad/OpAtomicStore only take scalar integer or
# floating-point operands, so the Khronos translator turns an atomic access of a pointer
# into an invalid pointer-typed atomic that consumers reject (Intel's compiler fails with an
# undefined `__spirv_AtomicLoad(long**, int, int)`), and AIR has no atomic load or store
# instructions at all: Apple's back-end aborts on them (`XPC_ERROR_CONNECTION_INTERRUPTED`
# from the driver; the macOS 26 AGX compiler reports `unable to legalize instruction:
# store release (p0)` for a type-tag store through the generic pointer the device allocator
# returns). Run after optimization, where dropping the ordering cannot enable new
# transformations. Device-side atomics proper go through target intrinsics, not these
# instructions, so every remaining one is such Julia bookkeeping and gets demoted.
function demote_atomics!(mod::LLVM.Module)
# Julia marks accesses to heap references `unordered`, so that a read racing with the GC or
# with another thread's write cannot observe a torn pointer. There is no device GC, and
# aligned accesses don't tear, so the ordering carries no meaning here. Not every back-end can
# express it, though: SPIR-V's OpAtomicLoad/OpAtomicStore only take scalar integer or
# floating-point operands, so the Khronos translator turns an atomic access of a pointer into
# an invalid pointer-typed atomic, and Metal only has atomics on device and threadgroup memory,
# while these accesses are often of objects passed by reference (e.g. an immutable struct
# with a `DataType` field, or with a reference to a mutable object). Rejecting allocations of
# objects with references (see `check_allocation!`) does not remove them, as they don't
# involve an allocation. Other orderings are left alone, because they are user atomics (e.g.
# UnsafeAtomics' `load`/`store!`) that the back-end must see.
function demote_unordered_atomics!(mod::LLVM.Module)
changed = false
for f in functions(mod), bb in blocks(f), inst in instructions(bb)
(inst isa LLVM.LoadInst || inst isa LLVM.StoreInst) || continue
is_atomic(inst) || continue
is_atomic(inst) && ordering(inst) == LLVM.API.LLVMAtomicOrderingUnordered || continue
ordering!(inst, LLVM.API.LLVMAtomicOrderingNotAtomic)
changed = true
end
Expand Down
7 changes: 4 additions & 3 deletions src/metal.jl
Original file line number Diff line number Diff line change
Expand Up @@ -597,6 +597,10 @@ function finish_ir!(@nospecialize(job::CompilerJob{MetalCompilerTarget}), mod::L
entry::LLVM.Function)
entry_fn = LLVM.name(entry)

# Julia's `unordered` heap-reference accesses are pointer-sized, and often of thread
# memory, neither of which AIR has atomics for (see `demote_unordered_atomics!`)
demote_unordered_atomics!(mod)

# convert the kernel state argument to a reference
if job.config.kernel && kernel_state_type(job) !== Nothing
entry = kernel_state_to_reference!(job, mod, entry)
Expand Down Expand Up @@ -659,9 +663,6 @@ function lower_air!(@nospecialize(job::CompilerJob{MetalCompilerTarget}), mod::L
# Metal.malloc uses.
rewrite_generic_null_selects!(mod)

# AIR does not support LLVM atomic load/store instructions (see `demote_atomics!`)
demote_atomics!(mod)

# the macOS 27 back-end rejects bare LLVM fences (Metal.jl#968)
lower_fences!(job, mod)

Expand Down
4 changes: 2 additions & 2 deletions src/spirv.jl
Original file line number Diff line number Diff line change
Expand Up @@ -116,8 +116,8 @@ function finish_ir!(job::CompilerJob{SPIRVCompilerTarget}, mod::LLVM.Module,
lower_unreachable_control_flow!(job, mod)

# SPIR-V cannot express atomic loads and stores of pointers, which is what Julia's
# heap-reference accesses and type-tag stores are; the orderings serve no purpose on device
demote_atomics!(mod)
# `unordered` heap-reference accesses are (see `demote_unordered_atomics!`)
demote_unordered_atomics!(mod)

# the SPIR-V back-ends lower `llvm.minimum`/`llvm.maximum` to NaN-ignoring `fmin`/`fmax`
lower_minimum_maximum!(mod)
Expand Down
88 changes: 19 additions & 69 deletions test/metal.jl
Original file line number Diff line number Diff line change
Expand Up @@ -1806,7 +1806,6 @@ end
end
end


@testset "unsupported allocations" begin
# without a garbage collector, device code can only allocate objects that do not
# reference other objects
Expand Down Expand Up @@ -1902,77 +1901,28 @@ end
end
end

@testset "atomic demotion" begin
# Demote every LLVM atomic load/store, whatever its ordering or value type.
Context() do ctx
ir = """
define void @f(i8** %p, i8** %q, i64* %r) {
entry:
%a = load atomic i8*, i8** %p unordered, align 8
store atomic i8* %a, i8** %q unordered, align 8
%t = load atomic i8*, i8** %p acquire, align 8
store atomic i8* %t, i8** %q release, align 8
%b = load atomic i64, i64* %r monotonic, align 8
store atomic i64 %b, i64* %r release, align 8
store i8* %a, i8** %p, align 8
ret void
}
"""
mod = parse(LLVM.Module, ir)
insts() = [i for f in functions(mod) for bb in blocks(f) for i in instructions(bb)]
memops() = filter(i -> i isa LLVM.LoadInst || i isa LLVM.StoreInst, insts())

@test count(is_atomic, memops()) == 6
@test GPUCompiler.demote_atomics!(mod)
@test count(is_atomic, memops()) == 0
@test !occursin("atomic", string(mod))
@test (verify(mod); true)
# idempotent
@test !GPUCompiler.demote_atomics!(mod)
end

# end-to-end: Julia's own `:unordered` accesses (as codegen emits for heap-reference
# fields) must reach the AIR as plain loads and stores
function kernel(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1})
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int}, p), :unordered)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int}, q), x, :unordered)
@testset "LLVM atomics" begin
# atomics in user code (e.g. UnsafeAtomics' `load`/`store!`) must reach the back-end,
# while Julia's `unordered` heap-reference accesses, which AIR cannot express when they
# are of pointers or outside device and threadgroup memory, become plain ones
function kernel(p::Core.LLVMPtr{Int32,1}, q::Core.LLVMPtr{Int32,1})
r = reinterpret(Ptr{Ptr{Int32}}, p)
y = Core.Intrinsics.atomic_pointerref(r, :unordered)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int32}}, q), y, :unordered)
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int32}, p), :acquire)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int32}, q), x, :release)
return
end
source = methodinstance(typeof(kernel), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}},
Base.get_world_counter())
target = MetalCompilerTarget(; macos=v"12.2", metal=v"3.0", air=v"3.0")
config = CompilerConfig(target, Metal.CompilerParams(); kernel=true)
job = CompilerJob(source, config)

# precondition: the accesses survive optimization as unordered atomics
ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true))
@test occursin(r"load atomic .* unordered", ir)
@test occursin(r"store atomic .* unordered", ir)

air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true))
@test !occursin(r"(load|store) atomic", air)
@test occursin(r"load i64", air)
@test occursin(r"store i64", air)

# likewise for the `release` store of a pointer that codegen emits for an allocated
# object's type tag. Only Julia 1.12+ lowers `Ptr` values to LLVM pointers, which is
# what makes this the case Apple's back-end cannot legalize.
@static if VERSION >= v"1.12"
function tagged(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1})
x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Ptr{Int}}, p), :acquire)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int}}, q), x, :release)
return
end
source = methodinstance(typeof(tagged), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}},
Base.get_world_counter())
job = CompilerJob(source, config)

ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true))
@test occursin(r"load atomic ptr.* acquire", ir)
@test occursin(r"store atomic ptr.* release", ir)

air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true))
@test !occursin(r"(load|store) atomic", air)
@test @filecheck begin
@check_not "unordered"
@check "load atomic i32"
@check_same "acquire"
@check "store atomic i32"
@check_same "release"
@check_not "unordered"
Metal.code_native(kernel, Tuple{Core.LLVMPtr{Int32,1}, Core.LLVMPtr{Int32,1}};
kernel=true, dump_module=true)
end
end

Expand Down
34 changes: 18 additions & 16 deletions test/spirv.jl
Original file line number Diff line number Diff line change
Expand Up @@ -417,27 +417,29 @@ end
end
end


@testset "atomic demotion" begin
# Julia's `unordered` heap-reference accesses and `release` type-tag stores cannot be
# expressed in SPIR-V when they involve pointers (OpAtomicLoad/OpAtomicStore take scalars
# only): the translator would emit an invalid pointer-typed atomic. They carry no meaning
# without a device GC, so `demote_atomics!` turns them into plain accesses.
@testset "LLVM atomics" begin
# atomics in user code (e.g. UnsafeAtomics' `load`/`store!`, Atomix' `get`/`set!`) must
# reach the back-end, while Julia's `unordered` heap-reference accesses, which SPIR-V
# cannot express when they are of pointers, become plain ones
mod = @eval module $(gensym())
function kernel(p::Ptr{Ptr{Int}}, q::Ptr{Ptr{Int}})
x = Core.Intrinsics.atomic_pointerref(p, :unordered)
Core.Intrinsics.atomic_pointerset(q, x, :unordered)
y = Core.Intrinsics.atomic_pointerref(p, :acquire)
Core.Intrinsics.atomic_pointerset(q, y, :release)
function kernel(p::Ptr{Int32}, q::Ptr{Int32})
y = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Ptr{Int32}}, p), :unordered)
Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int32}}, q), y, :unordered)
x = Core.Intrinsics.atomic_pointerref(p, :acquire)
Core.Intrinsics.atomic_pointerset(q, x, :release)
return
end
end
tt = Tuple{Ptr{Ptr{Int}}, Ptr{Ptr{Int}}}
tt = Tuple{Ptr{Int32}, Ptr{Int32}}

@test @filecheck begin
@check_label "define spir_kernel void @_Z6kernel"
@check_not "load atomic"
@check_not "store atomic"
@check_not "unordered"
@check "load atomic i32"
@check_same "acquire"
@check "store atomic i32"
@check_same "release"
@check_not "unordered"
@check "ret void"
SPIRV.code_llvm(mod.kernel, tt; backend, kernel=true)
end
Expand All @@ -446,8 +448,8 @@ end
@test @filecheck begin
@check "OpEntryPoint Kernel %[[KERNEL:[^ ]+]]"
@check "%[[KERNEL]] = OpFunction %void None"
@check_not "OpAtomicLoad"
@check_not "OpAtomicStore"
@check "OpAtomicLoad"
@check "OpAtomicStore"
SPIRV.code_native(mod.kernel, tt; backend, kernel=true)
end
end
Expand Down
Loading