Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .buildkite/pipeline.yml
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,11 @@ steps:
agents:
queue: "oneapi"
commands: |
if [[ "{{matrix.julia}}" == "1.10" ]]; then
# XXX: Julia 1.10 ignores [sources]; develop the unregistered KernelInterface instead
git clone --depth 1 --branch tb/ki-0.3 https://github.com/JuliaGPU/KernelAbstractions.jl ka
julia --project -e 'using Pkg; Pkg.develop(path="ka/lib/KernelInterface")'
fi
julia --project=deps deps/build_ci.jl
if: |
build.message !~ /\[skip [^\]]*(tests|julia)/ &&
Expand Down
6 changes: 6 additions & 0 deletions .github/workflows/docs.yml
Original file line number Diff line number Diff line change
Expand Up @@ -31,5 +31,11 @@ jobs:
with:
version: 'lts'
- uses: julia-actions/cache@v3
# XXX: Julia 1.10 ignores [sources]; develop the unregistered KernelInterface instead
- run: |
git clone --depth 1 --branch tb/ki-0.3 https://github.com/JuliaGPU/KernelAbstractions.jl ka
for project in . docs; do
julia --project=$project -e 'using Pkg; Pkg.develop(path="ka/lib/KernelInterface")'
done
- uses: julia-actions/julia-buildpkg@latest
- run: julia --project=docs/ docs/make.jl
10 changes: 10 additions & 0 deletions .gitlab-ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,15 @@ stages: [build, test, coverage]
script:
- !reference [.aurora-env, script]
- julia --color=yes --project=deps deps/build_local.jl
# XXX: Julia 1.10 ignores [sources]; develop the unregistered KernelInterface instead
# (the checkout travels to the test job as an artifact, like test/Manifest.toml)
- |
if [ "$JULIA_VERSION" = "1.10" ]; then
rm -rf ka
git clone --depth 1 --branch tb/ki-0.3 https://github.com/JuliaGPU/KernelAbstractions.jl ka
julia --color=yes --project=. -e 'using Pkg; Pkg.develop(path="ka/lib/KernelInterface")'
julia --color=yes --project=test -e 'using Pkg; Pkg.develop([PackageSpec(path="."), PackageSpec(path="ka/lib/KernelInterface")])'
fi
# Instantiate (and thereby precompile) both environments here so the 1 h batch job
# spends its walltime on tests, not on Pkg. Manifests are gitignored, so the test
# env must be resolved here with oneAPI dev'ed at the checkout (path "..") — a plain
Expand All @@ -113,6 +122,7 @@ stages: [build, test, coverage]
- LocalPreferences.toml
- test/LocalPreferences.toml
- test/Manifest.toml
- ka/lib/KernelInterface
expire_in: 1 week

.test:lts:
Expand Down
7 changes: 6 additions & 1 deletion Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@ GPUArrays = "0c68f7d7-f131-5f86-a1c3-88cf8149b2d7"
GPUCompiler = "61eb1bfa-7361-4325-ad38-22787b887f55"
GPUToolbox = "096a3bc2-3ced-46d0-87f4-dd12716f4bfc"
KernelAbstractions = "63c18a36-062a-441e-b654-da1e3ab1ce7c"
KernelInterface = "4ee993da-d684-4d17-a7dd-4e58e78d92bf"
LLVM = "929cbde3-209d-540e-8aea-75f648917ca0"
Libdl = "8f399da3-3557-5675-b5ff-fb832c97cbdb"
LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e"
Expand All @@ -32,6 +33,9 @@ oneAPI_Level_Zero_Headers_jll = "f4bc562b-d309-54f8-9efb-476e56f0410d"
oneAPI_Level_Zero_Loader_jll = "13eca655-d68d-5b81-8367-6d99d727ab01"
oneAPI_Support_jll = "b049733a-a71d-5ed3-8eba-7d323ac00b36"

[sources]
KernelInterface = {url = "https://github.com/JuliaGPU/KernelAbstractions.jl", rev = "tb/ki-0.3", subdir = "lib/KernelInterface"}

[compat]
AbstractFFTs = "1.5.0"
AcceleratedKernels = "0.3.1, 0.4"
Expand All @@ -42,11 +46,12 @@ GPUArrays = "11.5.14"
GPUCompiler = "2.9"
GPUToolbox = "3.1"
KernelAbstractions = "0.9.39"
KernelInterface = "0.3"
LLVM = "6, 7, 8, 9"
NEO_jll = "=26.18.38308"
PrecompileTools = "1"
Preferences = "1"
SPIRVIntrinsics = "1"
SPIRVIntrinsics = "1.1.3"
SPIRV_LLVM_Backend_jll = "23"
SPIRV_LLVM_Translator_jll = "23"
SPIRV_Tools_jll = "2025.4.0"
Expand Down
15 changes: 14 additions & 1 deletion lib/level-zero/module.jl
Original file line number Diff line number Diff line change
Expand Up @@ -85,13 +85,17 @@ mutable struct ZeKernel
# Read on every launch by the scratch hedge, so it must not cost an API call.
spill::Int

# cached maxGroupSize, seeded by `properties`; -1 while unqueried, and 0 without the
# MAX_GROUP_SIZE extension. Read on every KernelInterface launch.
max_group_size::Int

function ZeKernel(mod, name)
GC.@preserve name begin
desc_ref = Ref(ze_kernel_desc_t(; pKernelName=pointer(name)))
handle_ref = Ref{ze_kernel_handle_t}()
zeKernelCreate(mod, desc_ref, handle_ref)
end
obj = new(mod, handle_ref[], ReentrantLock(), -1)
obj = new(mod, handle_ref[], ReentrantLock(), -1, -1)

finalizer(obj) do obj
zeKernelDestroy(obj)
Expand Down Expand Up @@ -268,6 +272,8 @@ function properties(kernel::ZeKernel)

props = props_ref[]
kernel.spill = Int(props.spillMemSize)
kernel.max_group_size = max_group_size_props_ref === nothing ? 0 :
Int(max_group_size_props_ref[].maxGroupSize)
return (
numKernelArgs=Int(props.numKernelArgs),
requiredGroupSize=ZeDim3(props.requiredGroupSizeX,
Expand Down Expand Up @@ -295,6 +301,13 @@ function spill_mem_size(kernel::ZeKernel)
return s >= 0 ? s : Int(properties(kernel).spillMemSize)
end

# Cached access to a kernel's maxGroupSize, `missing` without the MAX_GROUP_SIZE extension.
function max_group_size(kernel::ZeKernel)
s = kernel.max_group_size
s < 0 && (s = coalesce(properties(kernel).maxGroupSize, 0))
return s > 0 ? s : missing
end


## execution

Expand Down
1 change: 1 addition & 0 deletions lib/level-zero/oneL0.jl
Original file line number Diff line number Diff line change
Expand Up @@ -124,6 +124,7 @@ end
include("context.jl")
include("cmdqueue.jl")
include("cmdlist.jl")
include("synchronization.jl")
include("fence.jl")
include("event.jl")
include("barrier.jl")
Expand Down
195 changes: 195 additions & 0 deletions lib/level-zero/synchronization.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,195 @@
# cooperative synchronization
#
# `zeCommandListHostSynchronize` and `zeCommandQueueSynchronize` block the calling thread
# until the work has completed, so no other task can run on it in the meantime. As CUDA.jl
# does, first busy-wait on a non-blocking query, which keeps the latency of short
# operations low, and then block in the driver on a separate thread, while the calling
# task waits for that thread without blocking the scheduler.

export nonblocking_synchronize

const SyncObject = Union{ZeImmediateCommandList, ZeCommandQueue}

# with a zero timeout, a synchronization is a query
function check_done(res::ze_result_t)
if res == RESULT_NOT_READY
return false
elseif res == RESULT_SUCCESS
return true
else
throw_api_error(res)
end
end
Base.isdone(list::ZeImmediateCommandList) =
check_done(unchecked_zeCommandListHostSynchronize(list, 0))
Base.isdone(queue::ZeCommandQueue) =
check_done(unchecked_zeCommandQueueSynchronize(queue, 0))

# the blocking synchronization, marked GC-safe so that it doesn't keep the GC from running
gcsafe_synchronize(list::ZeImmediateCommandList) =
@gcsafe_ccall libze_loader.zeCommandListHostSynchronize(
list::ze_command_list_handle_t, typemax(UInt64)::UInt64)::ze_result_t
gcsafe_synchronize(queue::ZeCommandQueue) =
@gcsafe_ccall libze_loader.zeCommandQueueSynchronize(
queue::ze_command_queue_handle_t, typemax(UInt64)::UInt64)::ze_result_t


## bidirectional channel

# custom, unbuffered channel that supports returning a value to the sender
# without the need for a second channel
struct BidirectionalChannel{I,O} <: AbstractChannel{I}
cond_take::Threads.Condition # waiting for data to become available
cond_put::Threads.Condition # waiting for a writeable slot
cond_ret::Threads.Condition # waiting for a data to be returned

function BidirectionalChannel{I,O}() where {I,O}
lock = ReentrantLock()
cond_put = Threads.Condition(lock)
cond_take = Threads.Condition(lock)
cond_ret = Threads.Condition(lock)
return new(cond_take, cond_put, cond_ret)
end
end

Base.put!(c::BidirectionalChannel{I}, v) where {I} = put!(c, convert(I, v))
function Base.put!(c::BidirectionalChannel{I,O}, v::I) where {I,O}
lock(c)
try
# wait for a slot to be available
while isempty(c.cond_take)
Base.wait(c.cond_put)
end

# pass a value to the consumer
notify(c.cond_take, v, false, false)

# wait for a return value to be produced
Base.wait(c.cond_ret)::O
finally
unlock(c)
end
end

function Base.take!(f::Base.Callable, c::BidirectionalChannel{I,O}) where {I,O}
lock(c)
try
# notify the producer that we're ready to accept a value
notify(c.cond_put, nothing, false, false)

# receive a value from the producer
v = Base.wait(c.cond_take)::I

# return a value to the producer
ret = f(v)::O
notify(c.cond_ret, ret, false, false)
finally
unlock(c)
end
end

Base.lock(c::BidirectionalChannel) = lock(c.cond_take)
Base.unlock(c::BidirectionalChannel) = unlock(c.cond_take)


## fast path

# before blocking on a separate thread, which has some overhead, busy-wait on a query of
# the object to synchronize. when this returns true, the object still has to be
# synchronized, but that won't block anymore.
function spinning_synchronization(f, obj)
# fast path
f(obj) && return true

# minimize latency of short operations by busy-waiting,
# initially without even yielding to other tasks
spins = 0
while spins < 256
if spins < 32
ccall(:jl_cpu_pause, Cvoid, ())
# temporary solution before we have gc transition support in codegen.
ccall(:jl_gc_safepoint, Cvoid, ())
else
yield()
end
f(obj) && return true
spins += 1
end

return false
end


## slow path: synchronize on a separate thread

const MAX_SYNC_THREADS = 4
const sync_channels = Array{BidirectionalChannel{SyncObject,ze_result_t}}(undef, MAX_SYNC_THREADS)
const sync_channel_cursor = Threads.Atomic{UInt32}(1)
const sync_channel_lock = Base.ReentrantLock()

function synchronization_worker(data)
i = Int(data)
chan = sync_channels[i]

while true
# wait for work
take!(gcsafe_synchronize, chan)
end
end

@noinline function create_synchronization_worker(i)
lock(sync_channel_lock) do
# test and test-and-set
if isassigned(sync_channels, i)
return
end

# should be safe to assign before threads are running;
# any user will just submit work that makes it block
sync_channels[i] = BidirectionalChannel{SyncObject,ze_result_t}()

# we don't know what the size of uv_thread_t is, so reserve enough space
tid = Ref{NTuple{32, UInt8}}(ntuple(i -> 0, 32))

cb = @cfunction(synchronization_worker, Cvoid, (Ptr{Cvoid},))
err = @ccall uv_thread_create(tid::Ptr{Cvoid}, cb::Ptr{Cvoid}, Ptr{Cvoid}(i)::Ptr{Cvoid})::Cint
err == 0 || Base.uv_error("uv_thread_create", err)
err = @ccall uv_thread_detach(tid::Ptr{Cvoid})::Cint
err == 0 || Base.uv_error("uv_thread_detach", err)
end

return
end

"""
nonblocking_synchronize(list_or_queue)

Wait for the work on an immediate command list or command queue to complete, like
[`synchronize`](@ref), but without blocking the Julia scheduler: other tasks keep running
while this one waits.
"""
function nonblocking_synchronize(obj::SyncObject)
if spinning_synchronization(Base.isdone, obj)
# done, so this doesn't block
synchronize(obj)
return
end

# pick a worker channel: sticky per task, so repeated synchronizations from the
# same task always hit the same, already running worker thread.
tls = task_local_storage()
i = get!(tls, :ZeSyncChannel) do
mod1(Threads.atomic_add!(sync_channel_cursor, UInt32(1)), MAX_SYNC_THREADS)
end::Int
if !isassigned(sync_channels, i)
create_synchronization_worker(i)
end
chan = @inbounds sync_channels[i]

# submit the object to synchronize; unlike with regular channels, this `put!` blocks
# until the worker has synchronized it and returned the result
res = put!(chan, obj)
res == RESULT_SUCCESS || throw_api_error(res)

return
end
4 changes: 2 additions & 2 deletions lib/utils/APIUtils.jl
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
module APIUtils

# helpers that facilitate working with C APIs
using GPUToolbox: @checked, @debug_ccall
export @checked, @debug_ccall
using GPUToolbox: @checked, @debug_ccall, @gcsafe_ccall
export @checked, @debug_ccall, @gcsafe_ccall
include("enum.jl")

end
19 changes: 16 additions & 3 deletions src/compiler/compilation.jl
Original file line number Diff line number Diff line change
@@ -1,6 +1,9 @@
## gpucompiler interface implementation

struct oneAPICompilerParams <: AbstractCompilerParams end
Base.@kwdef struct oneAPICompilerParams <: AbstractCompilerParams
sub_group_size::Union{Nothing,Int} = nothing
end

const oneAPICompilerConfig = CompilerConfig{SPIRVCompilerTarget, oneAPICompilerParams}
const oneAPICompilerJob = CompilerJob{SPIRVCompilerTarget,oneAPICompilerParams}

Expand Down Expand Up @@ -58,6 +61,11 @@ function GPUCompiler.finish_module!(job::oneAPICompilerJob, mod::LLVM.Module,
Tuple{CompilerJob{SPIRVCompilerTarget}, typeof(mod), typeof(entry)},
job, mod, entry)

# Set the subgroup size
if job.config.params.sub_group_size !== nothing
metadata(entry)["intel_reqd_sub_group_size"] = MDNode([ConstantInt(Int32(job.config.params.sub_group_size))])
end

# OpenCL 2.0
push!(metadata(mod)["opencl.ocl.version"],
MDNode([ConstantInt(Int32(2)),
Expand Down Expand Up @@ -273,11 +281,16 @@ function _driver_supports_bfloat16_spirv(dev=device())
end
end

@noinline function _compiler_config(dev; kernel=true, name=nothing, always_inline=false, kwargs...)
@noinline function _compiler_config(dev; kernel=true, name=nothing, always_inline=false, sub_group_size=32, kwargs...)
properties = oneL0.module_properties(dev)
supports_fp16 = properties.fp16flags & oneL0.ZE_DEVICE_MODULE_FLAG_FP16 == oneL0.ZE_DEVICE_MODULE_FLAG_FP16
supports_fp64 = properties.fp64flags & oneL0.ZE_DEVICE_MODULE_FLAG_FP64 == oneL0.ZE_DEVICE_MODULE_FLAG_FP64

if sub_group_size ∉ oneL0.compute_properties(dev).subGroupSizes
@error("$sub_group_size is not a valid sub-group size for this device.")
end


# SPIR-V codegen path. The Aurora LTS NEO/IGC runtime only accepts SPIR-V from the
# Khronos translator; the rolling stack uses the LLVM SPIR-V back-end. GPUCompiler picks
# the tool from the target's `backend` field and loads the JLL lazily, so both can be
Expand Down Expand Up @@ -311,7 +324,7 @@ end
# create GPUCompiler objects
target = SPIRVCompilerTarget(; backend, extensions = extensions_str, supports_fp16, supports_fp64, supports_bfloat16,
driver = :intel, kwargs...)
params = oneAPICompilerParams()
params = oneAPICompilerParams(; sub_group_size)
CompilerConfig(target, params; kernel, name, always_inline)
end

Expand Down
Loading
Loading