Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
58576ba
Add tasking (UFI/JuliaCustomTask) support to C++ wrapper
krasow Sep 19, 2026
3559d52
Add Julia tasking API, UFI runtime, and experimental guard
krasow Sep 19, 2026
4da6088
Add tasking examples, tests, and docs
krasow Sep 19, 2026
8ef31fe
Apply formatter to pre-existing files and update gitignore
krasow Sep 19, 2026
31e7d30
Remove duplicate GPU create_julia_task; tasks.jl handles both backends
krasow Sep 20, 2026
d240753
Fix create_library: UFI async system is initialized once in init_ufi
krasow Sep 20, 2026
4a9f268
Add copyto!(LogicalArray, Array) for Array-to-store copies
krasow Sep 20, 2026
dda7830
Fix UFI deadlock: run poller on a real thread and give dev-CI tests t…
krasow Sep 20, 2026
a0ecdc6
Defer Legate handle frees to the launch thread to fix multithreaded f…
krasow Sep 20, 2026
8192140
Run tasking tests with -t N,1 so the caller uses the interactive thread
krasow Sep 20, 2026
2854028
Build the C++ wrapper with all available cores
krasow Sep 20, 2026
9dfb3de
Support partitioned tasks: pass tile strides, copy tiles to/from dens…
krasow Sep 20, 2026
f779213
Defer all wrapped Legate handle destruction to the launch thread (fix…
krasow Sep 20, 2026
e99100f
Support GPU tasks in the UFI: register CUDAExt, GC-safe get_ptr, prec…
krasow Sep 20, 2026
ce18e09
Support partitioned GPU tasks: gather/scatter strided tiles on the de…
krasow Sep 20, 2026
8098671
Gate get_ptr GC-safe path on active GPU tasking to avoid CPU shutdown…
krasow Sep 21, 2026
1f27a0e
Fix shutdown SIGSEGV on Julia 1.12+/1.13: leak deferred-free singleto…
krasow Sep 21, 2026
e033522
Trace shutdown steps to stderr to localize the 1.12/1.13 shutdown seg…
krasow Sep 21, 2026
9321367
Join UFI poller/workers before legate_finish to fix the 1.12/1.13 shu…
krasow Sep 21, 2026
b179e05
Remove the timer reinforcing shutdown mechanism
krasow Sep 21, 2026
8902688
Add CPU tasking stress test to CI
krasow Sep 21, 2026
54d0aa7
1.10 and 1.11 GC safe ccalls
krasow Sep 21, 2026
6924212
try gc_safe = true on 1.12
krasow Sep 21, 2026
87eaf95
refactor ufi interface
krasow Sep 21, 2026
59eb2da
thread guard for --threads 1,0. This is default config on 1.10 and 1.11
krasow Sep 21, 2026
4662da3
Bloat constraint for halo exchange boundaries (#107)
krasow Sep 25, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ on:
push:
paths:
- 'src/**'
- 'test/**'
- 'scripts/**'
- 'deps/build.jl'
- 'deps/buildtools/**'
Expand All @@ -31,6 +32,7 @@ on:
pull_request:
paths:
- 'src/**'
- 'test/**'
- 'scripts/**'
- 'deps/build.jl'
- 'deps/buildtools/**'
Expand Down Expand Up @@ -102,7 +104,7 @@ jobs:
LEGATE_SHOW_CONFIG: "1"
LEGATE_AUTO_CONFIG: "0"
GPUTESTS: "0" # parsed by runtests.jl
LEGATE_CONFIG: "--cpus 1 --utility 1 --sysmem 4000"
LEGATE_CONFIG: "--cpus 2 --utility 1 --sysmem 4000"
run: |
if julia -e 'exit(VERSION >= v"1.12" ? 0 : 1)'; then
export JULIA_NUM_THREADS=1,0
Expand Down
10 changes: 9 additions & 1 deletion .github/workflows/developer.yml
Original file line number Diff line number Diff line change
Expand Up @@ -79,5 +79,13 @@ jobs:
julia --color=yes -e 'using Pkg; Pkg.build("Legate")'

- name: Perform Test
env:
LEGATE_CONFIG: "--cpus 2 --gpus 0 --utility 1 --sysmem 4000"
run: |
julia --color=yes -e 'using Pkg; Pkg.test("Legate")'
julia --color=yes -e 'using Pkg; Pkg.test("Legate"; julia_args=["--threads=4,1"])'

- name: CPU tasking stress test
env:
LEGATE_CONFIG: "--cpus 4 --gpus 0 --utility 1 --sysmem 4000"
run: |
timeout 180s julia --color=yes --threads=4,1 test/cpu_tasking_stress.jl
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,8 @@ build.log
.vscode
node_modules
package-lock.json
test_crash.sh
*.prof

_doxygen/
_doxygen/**
Expand Down
7 changes: 7 additions & 0 deletions Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,14 @@ legate_jl_wrapper_jll = "6e093c14-372f-5516-b4d7-a6ce14f1038b"
legate_jll = "e95fb1d3-fb9e-51b5-bdb8-1a812408cac9"
libcxxwrap_julia_jll = "3eaa8342-bff7-56a5-9981-c04077f7cee7"

[weakdeps]
CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba"

[extensions]
CUDAExt = "CUDA"

[compat]
CUDA = "6.4"
CUDACore = "6.4"
CxxWrap = "0.17.5"
FunctionWrappers = "1.1.3"
Expand Down
12 changes: 7 additions & 5 deletions deps/build.jl
Original file line number Diff line number Diff line change
Expand Up @@ -28,8 +28,8 @@ function build_cpp_wrapper(
)
@info "liblegatewrapper: Building C++ Wrapper Library"
isdir(install_root) && (rm(install_root; recursive=true); mkdir(install_root))
bld_command = `$(joinpath(repo_root, "scripts/build_cpp_wrapper.sh")) $repo_root $legate_root $install_root $(Threads.nthreads())`
BuildTools.run_build_wrapper_script(
bld_command = `$(joinpath(repo_root, "scripts/build_cpp_wrapper.sh")) $repo_root $legate_root $install_root $(Sys.CPU_THREADS)`
return BuildTools.run_build_wrapper_script(
repo_root, bld_command; cuda_root, cuda_enabled, log_dir=@__DIR__
)
end
Expand All @@ -49,7 +49,7 @@ function build_deps(pkg_root, legate_root; cuda_root=nothing, cuda_enabled=true)
log_dir=@__DIR__, is_compatible=is_supported_version,
)
build_cpp_wrapper(pkg_root, legate_root, install_dir; cuda_root, cuda_enabled)
BuildTools.set_jll_artifact_override(:legate_jl_wrapper_jll, install_dir)
return BuildTools.set_jll_artifact_override(:legate_jl_wrapper_jll, install_dir)
end

function build(::LegatePreferences.JLL)
Expand All @@ -67,7 +67,7 @@ function build(::LegatePreferences.Conda)
end

is_legate_installed(legate_root; throw_errors=true)
build_deps(pkg_root, legate_root)
return build_deps(pkg_root, legate_root)
end

function build(::LegatePreferences.Developer)
Expand All @@ -83,7 +83,9 @@ function build(::LegatePreferences.Developer)
end

build_deps(pkg_root, legate_root; cuda_root, cuda_enabled)
set_preferences!(LegatePreferences, "LEGATE_LIBDIR" => joinpath(legate_root, "lib"); force=true)
return set_preferences!(
LegatePreferences, "LEGATE_LIBDIR" => joinpath(legate_root, "lib"); force=true
)
end

const mode_str = load_preference(LegatePreferences, "legate_mode", LegatePreferences.MODE_JLL)
Expand Down
9 changes: 6 additions & 3 deletions docs/src/example.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,9 +5,12 @@
Legate uses a deferred execution model. When a task is submitted in Julia, it is not executed immediately. Instead, the Legate runtime (C++) records the task, analyzes data dependencies, and schedules execution.

**Interaction Model:**
1. **Submission**: The main Julia thread submits tasks to the runtime. This is non-blocking.
2. **Scheduling**: The Legate runtime manages resources and dependencies.
3. **Execution**: Once ready, Legate signals Julia to execute the task. This happens on a dedicated Julia worker task (thread) that handles incoming requests from the runtime. See more information about Julia thread-safety [here](https://docs.julialang.org/en/v1/manual/calling-c-and-fortran-code/#Thread-safety).
1. **Submission**: The main Julia thread submits tasks to the runtime via non-blocking C calls.
2. **Scheduling**: The Legate runtime manages resources and dependencies in the background.
3. **Execution**:
* **Polling**: A low-overhead timer on the main thread polls for ready tasks from the runtime.
* **Workers**: Tasks are dispatched via a thread-safe `Channel` to a pool of background worker threads (default: `nthreads-1`) to execute concurrently. This ensures that the main thread is never blocked by long-running tasks, and the runtime is not blocked by Julia's GC or other operations.
* **Completion**: Workers signal completion back to the runtime upon finishing. See more information about Julia thread-safety [here](https://docs.julialang.org/en/v1/manual/calling-c-and-fortran-code/#Thread-safety).

## Arguments

Expand Down
20 changes: 20 additions & 0 deletions examples/minimal_task.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
using Legate

Legate.Experimental(true) # tasking is experimental

function task_noop(a)
return nothing
end

Legate.ensure_runtime!()
rt = Legate.get_runtime()
lib = Legate.create_library("test")

my_noop_task = Legate.wrap_task(task_noop, Legate.CPUBackend)
a_noop = Legate.create_array([10], Float32)

task0 = Legate.create_julia_task(rt, lib, my_noop_task)
Legate.add_output(task0, a_noop)
Legate.submit_task(rt, task0)

Legate.wait_ufi()
97 changes: 56 additions & 41 deletions examples/tasking.jl
Original file line number Diff line number Diff line change
@@ -1,65 +1,67 @@
using Legate

# args is a vector that can be expanded
# it is in the order that the inputs and outputs are added
# in1 in2 ... out1 out2 ... scalar1 scalar2 ...
function task_test(args)
a, b, c = args # inputs are first, then outputs
function task_test(a, b, c)
@inbounds @simd for i in eachindex(a)
c[i] = a[i] + b[i]
end
end

# init task
function task_init(args)
a, b, c = args
function task_init(a, b, c)
@inbounds @simd for i in eachindex(a)
a[i] = rand(Float32)
b[i] = rand(Float32)
c[i] = 0.0f0
end
end

# 4 arg task
function task_4arg(args)
in1, in2, out1, out2 = args
function task_4arg(in1, in2, out1, out2)
@inbounds @simd for i in eachindex(in1)
out1[i] = in1[i] * 2
out2[i] = in2[i] + 1
end
end

# Task with Scalar argument
function task_scalar(args)
a, b, scalar = args
function task_scalar(a, b, scalar)
@inbounds @simd for i in eachindex(a)
b[i] = a[i] * scalar
end
end

function task_noop(a)
return nothing
end

function test_driver()
Legate.Experimental(true) # tasking is experimental
N = 1000
rt = Legate.get_runtime()
lib = Legate.create_library("test")

my_task = Legate.wrap_task(task_test)
my_init_task = Legate.wrap_task(task_init)
my_4arg_task = Legate.wrap_task(task_4arg)
my_scalar_task = Legate.wrap_task(task_scalar)

# 1. Init Task (3 args)
a = Legate.create_array([10, 10], Float32)
b = Legate.create_array([10, 10], Float32)
c = Legate.create_array([10, 10], Float32)
d = Legate.create_array([10, 10], Float32) # Extra array for 4-arg test

task = Legate.create_julia_task(rt, lib, my_init_task)
my_task = Legate.wrap_task(task_test, Legate.CPUBackend)
my_init_task = Legate.wrap_task(task_init, Legate.CPUBackend)
my_4arg_task = Legate.wrap_task(task_4arg, Legate.CPUBackend)
my_scalar_task = Legate.wrap_task(task_scalar, Legate.CPUBackend)
my_noop_task = Legate.wrap_task(task_noop, Legate.CPUBackend)

# 0. NOOP Task
a_noop = Legate.create_array([10], Float32)
task0 = Legate.create_julia_task(rt, lib, my_noop_task)
Legate.add_output(task0, a_noop)
Legate.submit_task(rt, task0)

# 1. Initialization Task
a = Legate.create_array([N], Float32)
b = Legate.create_array([N], Float32)
c = Legate.create_array([N], Float32)
d = Legate.create_array([N], Float32)

task1 = Legate.create_julia_task(rt, lib, my_init_task)
init_output_vars = Vector{Legate.Variable}()
push!(init_output_vars, Legate.add_output(task, a))
push!(init_output_vars, Legate.add_output(task, b))
push!(init_output_vars, Legate.add_output(task, c))
Legate.default_alignment(task, Vector{Legate.Variable}(), init_output_vars)

Legate.submit_task(rt, task)
push!(init_output_vars, Legate.add_output(task1, a))
push!(init_output_vars, Legate.add_output(task1, b))
push!(init_output_vars, Legate.add_output(task1, c))
Legate.default_alignment(task1, Vector{Legate.Variable}(), init_output_vars)
Legate.submit_task(rt, task1)

# 2. Compute Task (3 args)
task2 = Legate.create_julia_task(rt, lib, my_task)
Expand All @@ -69,37 +71,50 @@ function test_driver()
push!(input_vars, Legate.add_input(task2, b))
push!(output_vars, Legate.add_output(task2, c))
Legate.default_alignment(task2, input_vars, output_vars)

Legate.submit_task(rt, task2)

# 3. Arbitrary Arg Task (4 args: 2 in, 2 out)
task3 = Legate.create_julia_task(rt, lib, my_4arg_task)
in_vars_4 = Vector{Legate.Variable}()
out_vars_4 = Vector{Legate.Variable}()
# Inputs: a, c
push!(in_vars_4, Legate.add_input(task3, a))
push!(in_vars_4, Legate.add_input(task3, c))
# Outputs: b (reuse), d (new)
push!(out_vars_4, Legate.add_output(task3, b))
push!(out_vars_4, Legate.add_output(task3, d))
Legate.default_alignment(task3, in_vars_4, out_vars_4)

Legate.submit_task(rt, task3)

# 4. Scalar Arg Task (2 args + scalar)
task4 = Legate.create_julia_task(rt, lib, my_scalar_task)
in_vars_s = Vector{Legate.Variable}()
out_vars_s = Vector{Legate.Variable}()
# Input: c (result of task2)
push!(in_vars_s, Legate.add_input(task4, c))
# Output: a (reuse)
push!(out_vars_s, Legate.add_output(task4, a))

# Add user scalar argument (Float32)
Legate.add_scalar(task4, Legate.Scalar(2.5f0))
Legate.default_alignment(task4, in_vars_s, out_vars_s)

Legate.submit_task(rt, task4)

# --- VERIFICATION OF ASYNC EXECUTION ---
@info "Submitting 100 bulk tasks to verify asynchronous polling..."
for i in 1:100
t_bulk = Legate.create_julia_task(rt, lib, my_scalar_task)
v_in = Legate.add_input(t_bulk, c)
v_out = Legate.add_output(t_bulk, a)
Legate.add_scalar(t_bulk, Legate.Scalar(1.0f0))
Legate.default_alignment(t_bulk, [v_in], [v_out])
Legate.submit_task(rt, t_bulk)
end
@info "Bulk submission complete. Waiting for completion..."
Legate.wait_ufi()
@info "Verifying results of bulk run..."
val_a = Array(a)
val_c = Array(c)
if val_a ≈ val_c
@info "Verification successful: Target array matches source."
else
@error "Verification FAILED: Target array does not match source."
error("Parallel verification failed")
end
end

if abspath(PROGRAM_FILE) == @__FILE__
Expand Down
11 changes: 3 additions & 8 deletions ext/CUDAExt/CUDAExt.jl
Original file line number Diff line number Diff line change
@@ -1,13 +1,8 @@
module CUDAExt

# using CUDA
# using Legate
using CUDA
using Legate

# using CxxWrap: CxxWrap
# import Legate: wrap_task, create_julia_task, SUPPORTED_TYPES, JuliaGPUTask, CxxPtr, Runtime,
# Library, create_task, JULIA_CUSTOM_GPU_TASK, add_scalar, Scalar, register_task_function,
# _execute_julia_task, get_code_type, TaskArgumentGPU

# include("ufi.jl")
include("ufi.jl")

end # module CUDAExt
Loading
Loading