diff --git a/src/irgen.jl b/src/irgen.jl index 2dbf5ddd..ac253a75 100644 --- a/src/irgen.jl +++ b/src/irgen.jl @@ -343,28 +343,24 @@ function inline_unreachable_control_flow!(@nospecialize(job::CompilerJob), mod:: return changed end -# demote LLVM atomic loads and stores to plain ones +# demote `unordered` LLVM atomic loads and stores to plain ones # -# Julia marks accesses to heap references `unordered` so that a read racing with the GC, or -# with another thread's write, cannot observe a torn pointer, and stores the type tag of a -# freshly allocated object with `release` ordering so that no other thread can observe the -# object before its header. There is no device GC and no such race for GPUCompiler to -# protect against, so these orderings carry no meaning here, but not every back-end can -# express them: SPIR-V's OpAtomicLoad/OpAtomicStore only take scalar integer or -# floating-point operands, so the Khronos translator turns an atomic access of a pointer -# into an invalid pointer-typed atomic that consumers reject (Intel's compiler fails with an -# undefined `__spirv_AtomicLoad(long**, int, int)`), and AIR has no atomic load or store -# instructions at all: Apple's back-end aborts on them (`XPC_ERROR_CONNECTION_INTERRUPTED` -# from the driver; the macOS 26 AGX compiler reports `unable to legalize instruction: -# store release (p0)` for a type-tag store through the generic pointer the device allocator -# returns). Run after optimization, where dropping the ordering cannot enable new -# transformations. Device-side atomics proper go through target intrinsics, not these -# instructions, so every remaining one is such Julia bookkeeping and gets demoted. -function demote_atomics!(mod::LLVM.Module) +# Julia marks accesses to heap references `unordered`, so that a read racing with the GC or +# with another thread's write cannot observe a torn pointer. There is no device GC, and +# aligned accesses don't tear, so the ordering carries no meaning here. Not every back-end can +# express it, though: SPIR-V's OpAtomicLoad/OpAtomicStore only take scalar integer or +# floating-point operands, so the Khronos translator turns an atomic access of a pointer into +# an invalid pointer-typed atomic, and Metal only has atomics on device and threadgroup memory, +# while these accesses are often of objects passed by reference (e.g. an immutable struct +# with a `DataType` field, or with a reference to a mutable object). Rejecting allocations of +# objects with references (see `check_allocation!`) does not remove them, as they don't +# involve an allocation. Other orderings are left alone, because they are user atomics (e.g. +# UnsafeAtomics' `load`/`store!`) that the back-end must see. +function demote_unordered_atomics!(mod::LLVM.Module) changed = false for f in functions(mod), bb in blocks(f), inst in instructions(bb) (inst isa LLVM.LoadInst || inst isa LLVM.StoreInst) || continue - is_atomic(inst) || continue + is_atomic(inst) && ordering(inst) == LLVM.API.LLVMAtomicOrderingUnordered || continue ordering!(inst, LLVM.API.LLVMAtomicOrderingNotAtomic) changed = true end diff --git a/src/metal.jl b/src/metal.jl index 65b3da95..3180b02f 100644 --- a/src/metal.jl +++ b/src/metal.jl @@ -597,6 +597,10 @@ function finish_ir!(@nospecialize(job::CompilerJob{MetalCompilerTarget}), mod::L entry::LLVM.Function) entry_fn = LLVM.name(entry) + # Julia's `unordered` heap-reference accesses are pointer-sized, and often of thread + # memory, neither of which AIR has atomics for (see `demote_unordered_atomics!`) + demote_unordered_atomics!(mod) + # convert the kernel state argument to a reference if job.config.kernel && kernel_state_type(job) !== Nothing entry = kernel_state_to_reference!(job, mod, entry) @@ -659,9 +663,6 @@ function lower_air!(@nospecialize(job::CompilerJob{MetalCompilerTarget}), mod::L # Metal.malloc uses. rewrite_generic_null_selects!(mod) - # AIR does not support LLVM atomic load/store instructions (see `demote_atomics!`) - demote_atomics!(mod) - # the macOS 27 back-end rejects bare LLVM fences (Metal.jl#968) lower_fences!(job, mod) diff --git a/src/spirv.jl b/src/spirv.jl index 3a4515ba..a1486d69 100644 --- a/src/spirv.jl +++ b/src/spirv.jl @@ -116,8 +116,8 @@ function finish_ir!(job::CompilerJob{SPIRVCompilerTarget}, mod::LLVM.Module, lower_unreachable_control_flow!(job, mod) # SPIR-V cannot express atomic loads and stores of pointers, which is what Julia's - # heap-reference accesses and type-tag stores are; the orderings serve no purpose on device - demote_atomics!(mod) + # `unordered` heap-reference accesses are (see `demote_unordered_atomics!`) + demote_unordered_atomics!(mod) # the SPIR-V back-ends lower `llvm.minimum`/`llvm.maximum` to NaN-ignoring `fmin`/`fmax` lower_minimum_maximum!(mod) diff --git a/test/metal.jl b/test/metal.jl index 67c7b11f..d5d5af90 100644 --- a/test/metal.jl +++ b/test/metal.jl @@ -1806,7 +1806,6 @@ end end end - @testset "unsupported allocations" begin # without a garbage collector, device code can only allocate objects that do not # reference other objects @@ -1902,77 +1901,28 @@ end end end -@testset "atomic demotion" begin - # Demote every LLVM atomic load/store, whatever its ordering or value type. - Context() do ctx - ir = """ - define void @f(i8** %p, i8** %q, i64* %r) { - entry: - %a = load atomic i8*, i8** %p unordered, align 8 - store atomic i8* %a, i8** %q unordered, align 8 - %t = load atomic i8*, i8** %p acquire, align 8 - store atomic i8* %t, i8** %q release, align 8 - %b = load atomic i64, i64* %r monotonic, align 8 - store atomic i64 %b, i64* %r release, align 8 - store i8* %a, i8** %p, align 8 - ret void - } - """ - mod = parse(LLVM.Module, ir) - insts() = [i for f in functions(mod) for bb in blocks(f) for i in instructions(bb)] - memops() = filter(i -> i isa LLVM.LoadInst || i isa LLVM.StoreInst, insts()) - - @test count(is_atomic, memops()) == 6 - @test GPUCompiler.demote_atomics!(mod) - @test count(is_atomic, memops()) == 0 - @test !occursin("atomic", string(mod)) - @test (verify(mod); true) - # idempotent - @test !GPUCompiler.demote_atomics!(mod) - end - - # end-to-end: Julia's own `:unordered` accesses (as codegen emits for heap-reference - # fields) must reach the AIR as plain loads and stores - function kernel(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1}) - x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int}, p), :unordered) - Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int}, q), x, :unordered) +@testset "LLVM atomics" begin + # atomics in user code (e.g. UnsafeAtomics' `load`/`store!`) must reach the back-end, + # while Julia's `unordered` heap-reference accesses, which AIR cannot express when they + # are of pointers or outside device and threadgroup memory, become plain ones + function kernel(p::Core.LLVMPtr{Int32,1}, q::Core.LLVMPtr{Int32,1}) + r = reinterpret(Ptr{Ptr{Int32}}, p) + y = Core.Intrinsics.atomic_pointerref(r, :unordered) + Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int32}}, q), y, :unordered) + x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Int32}, p), :acquire) + Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Int32}, q), x, :release) return end - source = methodinstance(typeof(kernel), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}}, - Base.get_world_counter()) - target = MetalCompilerTarget(; macos=v"12.2", metal=v"3.0", air=v"3.0") - config = CompilerConfig(target, Metal.CompilerParams(); kernel=true) - job = CompilerJob(source, config) - - # precondition: the accesses survive optimization as unordered atomics - ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true)) - @test occursin(r"load atomic .* unordered", ir) - @test occursin(r"store atomic .* unordered", ir) - air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true)) - @test !occursin(r"(load|store) atomic", air) - @test occursin(r"load i64", air) - @test occursin(r"store i64", air) - - # likewise for the `release` store of a pointer that codegen emits for an allocated - # object's type tag. Only Julia 1.12+ lowers `Ptr` values to LLVM pointers, which is - # what makes this the case Apple's back-end cannot legalize. - @static if VERSION >= v"1.12" - function tagged(p::Core.LLVMPtr{Int,1}, q::Core.LLVMPtr{Int,1}) - x = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Ptr{Int}}, p), :acquire) - Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int}}, q), x, :release) - return - end - source = methodinstance(typeof(tagged), Tuple{Core.LLVMPtr{Int,1}, Core.LLVMPtr{Int,1}}, - Base.get_world_counter()) - job = CompilerJob(source, config) - - ir = sprint(io->GPUCompiler.code_llvm(io, job; dump_module=true)) - @test occursin(r"load atomic ptr.* acquire", ir) - @test occursin(r"store atomic ptr.* release", ir) - - air = sprint(io->GPUCompiler.code_native(io, job; dump_module=true)) - @test !occursin(r"(load|store) atomic", air) + @test @filecheck begin + @check_not "unordered" + @check "load atomic i32" + @check_same "acquire" + @check "store atomic i32" + @check_same "release" + @check_not "unordered" + Metal.code_native(kernel, Tuple{Core.LLVMPtr{Int32,1}, Core.LLVMPtr{Int32,1}}; + kernel=true, dump_module=true) end end diff --git a/test/spirv.jl b/test/spirv.jl index 28d6a997..5c89775a 100644 --- a/test/spirv.jl +++ b/test/spirv.jl @@ -417,27 +417,29 @@ end end end - -@testset "atomic demotion" begin - # Julia's `unordered` heap-reference accesses and `release` type-tag stores cannot be - # expressed in SPIR-V when they involve pointers (OpAtomicLoad/OpAtomicStore take scalars - # only): the translator would emit an invalid pointer-typed atomic. They carry no meaning - # without a device GC, so `demote_atomics!` turns them into plain accesses. +@testset "LLVM atomics" begin + # atomics in user code (e.g. UnsafeAtomics' `load`/`store!`, Atomix' `get`/`set!`) must + # reach the back-end, while Julia's `unordered` heap-reference accesses, which SPIR-V + # cannot express when they are of pointers, become plain ones mod = @eval module $(gensym()) - function kernel(p::Ptr{Ptr{Int}}, q::Ptr{Ptr{Int}}) - x = Core.Intrinsics.atomic_pointerref(p, :unordered) - Core.Intrinsics.atomic_pointerset(q, x, :unordered) - y = Core.Intrinsics.atomic_pointerref(p, :acquire) - Core.Intrinsics.atomic_pointerset(q, y, :release) + function kernel(p::Ptr{Int32}, q::Ptr{Int32}) + y = Core.Intrinsics.atomic_pointerref(reinterpret(Ptr{Ptr{Int32}}, p), :unordered) + Core.Intrinsics.atomic_pointerset(reinterpret(Ptr{Ptr{Int32}}, q), y, :unordered) + x = Core.Intrinsics.atomic_pointerref(p, :acquire) + Core.Intrinsics.atomic_pointerset(q, x, :release) return end end - tt = Tuple{Ptr{Ptr{Int}}, Ptr{Ptr{Int}}} + tt = Tuple{Ptr{Int32}, Ptr{Int32}} @test @filecheck begin @check_label "define spir_kernel void @_Z6kernel" - @check_not "load atomic" - @check_not "store atomic" + @check_not "unordered" + @check "load atomic i32" + @check_same "acquire" + @check "store atomic i32" + @check_same "release" + @check_not "unordered" @check "ret void" SPIRV.code_llvm(mod.kernel, tt; backend, kernel=true) end @@ -446,8 +448,8 @@ end @test @filecheck begin @check "OpEntryPoint Kernel %[[KERNEL:[^ ]+]]" @check "%[[KERNEL]] = OpFunction %void None" - @check_not "OpAtomicLoad" - @check_not "OpAtomicStore" + @check "OpAtomicLoad" + @check "OpAtomicStore" SPIRV.code_native(mod.kernel, tt; backend, kernel=true) end end