From 23b2f60798a786463d26288bec4c86dd37ac126d Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Fri, 14 Aug 2026 14:18:19 -0400 Subject: [PATCH 01/49] Mimic cupynumeric Python random number generation (#176) * Supports random number generation for all relevant types. Includes uniform, normal and exponential random numbers. --------- Co-authored-by: David Krasowska --- .buildkite/jll.pipeline.yml | 6 +- Project.toml | 2 + TODO.md | 2 - docs/make.jl | 1 + docs/src/api.md | 4 +- docs/src/api_initialization.md | 29 ++- docs/src/api_random.md | 45 ++++ docs/src/examples/initialization.md | 13 +- docs/src/index.md | 2 +- lib/cunumeric_jl_wrapper/include/types.h | 3 + lib/cunumeric_jl_wrapper/src/types.cpp | 40 +++ lib/cunumeric_jl_wrapper/src/wrapper.cpp | 31 +++ src/cuNumeric.jl | 7 +- src/ndarray/detail/ndarray.jl | 81 ++++++- src/ndarray/ndarray.jl | 41 ---- src/ndarray/random/bitgenerator.jl | 154 ++++++++++++ src/ndarray/random/generator.jl | 270 +++++++++++++++++++++ src/ndarray/random/random.jl | 218 +++++++++++++++++ test/analysis/type_stability.jl | 5 + test/array/random.jl | 296 +++++++++++++++++++++++ test/runtests.jl | 1 + 21 files changed, 1185 insertions(+), 66 deletions(-) create mode 100644 docs/src/api_random.md create mode 100644 src/ndarray/random/bitgenerator.jl create mode 100644 src/ndarray/random/generator.jl create mode 100644 src/ndarray/random/random.jl create mode 100644 test/array/random.jl diff --git a/.buildkite/jll.pipeline.yml b/.buildkite/jll.pipeline.yml index c76c5f8f3..27656870d 100644 --- a/.buildkite/jll.pipeline.yml +++ b/.buildkite/jll.pipeline.yml @@ -11,9 +11,9 @@ steps: plugins: - JuliaCI/julia#v1: version: "{{matrix.julia}}" - # Developer builds install local wrapper overrides into the depot. - # Keep JLL tests in a separate cache so those overrides cannot leak here. - cache_dir: "${HOME}/.cache/julia-buildkite-plugin-jll" + # Isolate Julia versions and compile-time fusion preferences, and keep + # JLL tests separate from developer wrapper overrides. + cache_dir: "${HOME}/.cache/julia-buildkite-plugin-jll-{{matrix.julia}}-{{matrix.fusion}}" - jquick/pre-hook#v1.2.0: command: | if [ "{{matrix.fusion}}" = "on" ]; then diff --git a/Project.toml b/Project.toml index 8104b9c40..0aa016679 100644 --- a/Project.toml +++ b/Project.toml @@ -19,6 +19,7 @@ OpenBLAS32_jll = "656ef2d0-ae68-5445-9ca0-591084a874a2" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" Preferences = "21216c6a-2e73-6563-6e65-726566657250" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" +StaticArrays = "90137ffa-7385-5640-81b9-e52037218182" StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" cunumeric_jl_wrapper_jll = "49048992-29d2-5fd1-994f-9cecf112d624" cupynumeric_jll = "2862d674-414d-5b0b-a494-b21f8deca547" @@ -39,6 +40,7 @@ OpenBLAS32_jll = "0.3" Pkg = "1" Preferences = "1" Random = "1" +StaticArrays = "1" StatsBase = "0.34" cunumeric_jl_wrapper_jll = "26.6.0" cupynumeric_jll = "26.6.0" diff --git a/TODO.md b/TODO.md index 23b4be98b..6637e87ff 100644 --- a/TODO.md +++ b/TODO.md @@ -4,7 +4,5 @@ - Replace `as_type` with `Base.convert` - Support Ints on methods that takes floats - Programatic manipulation of Legate hardware config (not currently possible) -- Float32 random number generation (not possible in current C++ API) -- Normal random numbers (not possible in current C++ API) - Add Aqua.jl to CI to ensure we didn't pirate any types - Fix CodeCov reports diff --git a/docs/make.jl b/docs/make.jl index 027b38f04..d95c9f4eb 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -60,6 +60,7 @@ makedocs(; ], "Public API" => [ "Initialization" => "api_initialization.md", + "Random" => "api_random.md", "Unary Operations" => "api_unary.md", "Binary Operations" => "api_binary.md", "Linear Algebra" => "linalg.md", diff --git a/docs/src/api.md b/docs/src/api.md index 82345f271..2533dc73b 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -1,9 +1,9 @@ # NDArray Reference -Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). +Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). ```@autodocs Modules = [cuNumeric] Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl"] -Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!)) +Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!, :randn, :randn!, :randexp, :randexp!, :default_rng, :random, :random!)) ``` diff --git a/docs/src/api_initialization.md b/docs/src/api_initialization.md index b6b08ef05..bd8ddacc2 100644 --- a/docs/src/api_initialization.md +++ b/docs/src/api_initialization.md @@ -47,7 +47,32 @@ cuNumeric.rand ## rand! ```@docs -Random.rand!(::NDArray{Float64}) +Random.rand!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) ``` -The backend currently draws `Float64` uniforms. `cuNumeric.rand(Float32, dims...)` converts for you. `rand!` on `NDArray` currently requires `Float64` storage. +## randn + +```@docs +cuNumeric.randn +``` + +## randn! + +```@docs +Random.randn!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) +``` + +## randexp + +```@docs +cuNumeric.randexp +``` + +## randexp! + +```@docs +Random.randexp!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) +``` + +See [Random](./api_random.md) for BitGenerators (`XORWOW`, `MRG32k3a`, +`PHILOX4_32_10`), `Generator`, and `default_rng`. diff --git a/docs/src/api_random.md b/docs/src/api_random.md new file mode 100644 index 000000000..5a9923c86 --- /dev/null +++ b/docs/src/api_random.md @@ -0,0 +1,45 @@ +# Random + +Module-level [`rand`](@ref cuNumeric.rand), [`randn`](@ref cuNumeric.randn), +and [`randexp`](@ref cuNumeric.randexp) are listed under +[Initialization](./api_initialization.md). This page covers the cuPyNumeric RNG +stack: BitGenerators, `Generator`, and `default_rng`. + +Draws go through cuRAND with the [`XORWOW`](@ref cuNumeric.XORWOW) random +number generator by default. We support `Float32` and `Float64` uniforms, +normals, and exponentials (`randexp`), `ComplexF32` / `ComplexF64` uniforms +and normals (independent real/imag parts; no `randexp`), `Bool` coin flips, +and signed `Int16` / `Int32` / `Int64`. Ranged integers use Julia +`rand(1:10, dims...)` (inclusive). There is no native Bool or complex +distribution, so `rand(Bool, …)` draws `Int16` values in `{0,1}` and compares +them to zero, and complex draws two real arrays packed as `re + i*imag`. + +`Random.seed!` is not hooked. Use [`default_rng`](@ref cuNumeric.default_rng) +with an explicit seed (or a specific BitGenerator) when you need a private +stream. Module-level `rand` / `randn` / `randexp` always use a process-global XORWOW +generator. + +## BitGenerators + +```@docs +cuNumeric.BitGenerator +cuNumeric.XORWOW +cuNumeric.MRG32k3a +cuNumeric.PHILOX4_32_10 +``` + +## Generator + +```@docs +cuNumeric.Generator +cuNumeric.default_rng +``` + +```julia +g = cuNumeric.default_rng() # XORWOW, fresh seed +g = cuNumeric.default_rng(1234) # XORWOW, fixed seed +g = cuNumeric.default_rng(cuNumeric.PHILOX4_32_10, 1) +A = cuNumeric.random(g, Float32, (8, 8)) # U[0, 1) +cuNumeric.randn!(g, A) # N(0, 1) +cuNumeric.randexp!(g, A; scale=1) # Exp(scale) +``` diff --git a/docs/src/examples/initialization.md b/docs/src/examples/initialization.md index 6a1ddbe4a..afd453637 100644 --- a/docs/src/examples/initialization.md +++ b/docs/src/examples/initialization.md @@ -19,10 +19,19 @@ Fbool = cuNumeric.falses(2, 3) I = cuNumeric.eye(5) I16 = cuNumeric.eye(Float32, 5) -# Uniform random values (default Float32; backend draws Float64 then converts) +# Uniform / normal random values (native Float32 and Float64) R = cuNumeric.rand(4, 4) R64 = cuNumeric.rand(Float64, 1000) cuNumeric.rand!(R64) # fill an existing Float64 array +N = cuNumeric.randn(Float32, 8, 8) +C = cuNumeric.rand(ComplexF32, 8, 8) # independent real/imag uniforms +E = cuNumeric.randexp(Float64, 1000) # exponential, scale 1 (real only) +I = cuNumeric.rand(0:9, 4, 4) # Int64 in 0:9 (inclusive) +Coin = cuNumeric.rand(Bool, 8) # fair coin flips + +# Private stream / non-default engine (see Random in the Public API) +g = cuNumeric.default_rng(cuNumeric.PHILOX4_32_10, 1234) +P = cuNumeric.random(g, Float32, (4, 4)) ``` Shapes can be passed as separate `Int`s or as a `Tuple` / `Dims`: @@ -33,4 +42,4 @@ cuNumeric.zeros((2, 3)) cuNumeric.ones(Float64, (10, 10)) ``` -For signatures and more detail, see [Initialization](../api_initialization.md) in the Public API. +For signatures and more detail, see [Initialization](../api_initialization.md) and [Random](../api_random.md) in the Public API. diff --git a/docs/src/index.md b/docs/src/index.md index 932cc5b52..fe626eb6a 100644 --- a/docs/src/index.md +++ b/docs/src/index.md @@ -49,7 +49,7 @@ x = unwrap(s) # T, e.g. Float32 **The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. -For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). +For API details see [Initialization](./api_initialization.md), [Random](./api_random.md), and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). ### Kernel Fusion diff --git a/lib/cunumeric_jl_wrapper/include/types.h b/lib/cunumeric_jl_wrapper/include/types.h index 88a18dc91..4b21fde4b 100644 --- a/lib/cunumeric_jl_wrapper/include/types.h +++ b/lib/cunumeric_jl_wrapper/include/types.h @@ -68,3 +68,6 @@ void wrap_binary_ops(jlcxx::Module&); // Linear algebra op codes void wrap_linalg_ops(jlcxx::Module& mod); + +// BitGenerator op codes / enums (mirror cupynumeric_c.h) +void wrap_bitgenerator_ops(jlcxx::Module& mod); diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index 867330ea1..7ea80f70d 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -176,3 +176,43 @@ void wrap_linalg_ops(jlcxx::Module& mod) { mod.set_const("GEEV", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEEV}); } + +void wrap_bitgenerator_ops(jlcxx::Module& mod) { + mod.set_const( + "BITGENERATOR", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_BITGENERATOR}); + + mod.set_const("BITGENOP_CREATE", int32_t(CUPYNUMERIC_BITGENOP_CREATE)); + mod.set_const("BITGENOP_DESTROY", int32_t(CUPYNUMERIC_BITGENOP_DESTROY)); + mod.set_const("BITGENOP_RAND_RAW", int32_t(CUPYNUMERIC_BITGENOP_RAND_RAW)); + mod.set_const("BITGENOP_DISTRIBUTION", + int32_t(CUPYNUMERIC_BITGENOP_DISTRIBUTION)); + + mod.set_const("BITGENTYPE_DEFAULT", uint32_t(CUPYNUMERIC_BITGENTYPE_DEFAULT)); + mod.set_const("BITGENTYPE_XORWOW", uint32_t(CUPYNUMERIC_BITGENTYPE_XORWOW)); + mod.set_const("BITGENTYPE_MRG32K3A", + uint32_t(CUPYNUMERIC_BITGENTYPE_MRG32K3A)); + mod.set_const("BITGENTYPE_MTGP32", uint32_t(CUPYNUMERIC_BITGENTYPE_MTGP32)); + mod.set_const("BITGENTYPE_MT19937", uint32_t(CUPYNUMERIC_BITGENTYPE_MT19937)); + mod.set_const("BITGENTYPE_PHILOX4_32_10", + uint32_t(CUPYNUMERIC_BITGENTYPE_PHILOX4_32_10)); + + mod.set_const("BITGENDIST_INTEGERS_16", + uint32_t(CUPYNUMERIC_BITGENDIST_INTEGERS_16)); + mod.set_const("BITGENDIST_INTEGERS_32", + uint32_t(CUPYNUMERIC_BITGENDIST_INTEGERS_32)); + mod.set_const("BITGENDIST_INTEGERS_64", + uint32_t(CUPYNUMERIC_BITGENDIST_INTEGERS_64)); + mod.set_const("BITGENDIST_UNIFORM_32", + uint32_t(CUPYNUMERIC_BITGENDIST_UNIFORM_32)); + mod.set_const("BITGENDIST_UNIFORM_64", + uint32_t(CUPYNUMERIC_BITGENDIST_UNIFORM_64)); + mod.set_const("BITGENDIST_NORMAL_32", + uint32_t(CUPYNUMERIC_BITGENDIST_NORMAL_32)); + mod.set_const("BITGENDIST_NORMAL_64", + uint32_t(CUPYNUMERIC_BITGENDIST_NORMAL_64)); + mod.set_const("BITGENDIST_EXPONENTIAL_32", + uint32_t(CUPYNUMERIC_BITGENDIST_EXPONENTIAL_32)); + mod.set_const("BITGENDIST_EXPONENTIAL_64", + uint32_t(CUPYNUMERIC_BITGENDIST_EXPONENTIAL_64)); +} diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 562ad7af7..907d3c425 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -18,10 +18,12 @@ * Nader Rahhal */ +#include #include #include #include //needed for return type of toString methods #include +#include #include "accessors.h" #include "cupynumeric.h" @@ -67,6 +69,7 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { wrap_binary_ops(mod); wrap_unary_reds(mod); wrap_linalg_ops(mod); + wrap_bitgenerator_ops(mod); using jlcxx::ParameterList; using jlcxx::Parametric; @@ -96,6 +99,34 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { mod.method("get_lib", &get_lib); mod.method("nda_store_to_ndarray", &nda_store_to_ndarray); + // Legate.jl Scalar has no vector constructors. BITGENERATOR (and similar) + // tasks take fixed-array scalars; these helpers pack them from Julia + // pointers. + mod.method("add_vector_scalar_i64", + [](legate::AutoTask& task, const int64_t* p, int32_t n) { + std::vector v; + if (n > 0) { + v.assign(p, p + n); + } + task.add_scalar_arg(legate::Scalar(std::move(v))); + }); + mod.method("add_vector_scalar_f32", + [](legate::AutoTask& task, const float* p, int32_t n) { + std::vector v; + if (n > 0) { + v.assign(p, p + n); + } + task.add_scalar_arg(legate::Scalar(std::move(v))); + }); + mod.method("add_vector_scalar_f64", + [](legate::AutoTask& task, const double* p, int32_t n) { + std::vector v; + if (n > 0) { + v.assign(p, p + n); + } + task.add_scalar_arg(legate::Scalar(std::move(v))); + }); + auto ndarray_accessor = mod.add_type, TypeVar<2>>>("NDArrayAccessor"); ndarray_accessor diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index bb27774a6..d86f36fdb 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -44,7 +44,9 @@ using LinearAlgebra import LinearAlgebra: mul! using Random -import Random: rand! +import Random: rand!, randn!, randexp! + +using StaticArrays: SVector using StatsBase import StatsBase: var, mean @@ -171,6 +173,9 @@ include("cuda/cuda_ptx_task.jl") include("ndarray/broadcast_fusion.jl") include("ndarray/broadcast.jl") include("ndarray/ndarray.jl") +include("ndarray/random/bitgenerator.jl") +include("ndarray/random/generator.jl") +include("ndarray/random/random.jl") include("ndarray/unary.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 5d4d621c0..386850a1c 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -155,22 +155,79 @@ function nda_full_array(dims::Dims{N}, value::T) where {T,N} return NDArray(ptr, T, Val(N)) end -function nda_random(arr::NDArray, gen_code) - @task_scope "rand!" begin - ccall((:nda_random, libnda), - Cvoid, (NDArray_t, Int32), - arr.ptr, Int32(gen_code)) +# Legacy Float64-only CUPYNUMERIC_RAND wrappers; unused after BitGenerator. +# function nda_random(arr::NDArray, gen_code) +# @task_scope "rand!" begin +# ccall((:nda_random, libnda), +# Cvoid, (NDArray_t, Int32), +# arr.ptr, Int32(gen_code)) +# end +# end +# +# function nda_random_array(dims::Dims{N}) where {N} +# shape = collect(UInt64, dims) +# ptr = @task_scope "rand" begin +# ccall((:nda_random_array, libnda), +# NDArray_t, (Int32, Ptr{UInt64}), +# Int32(N), shape) +# end +# return NDArray(ptr, Float64, Val(N)) #* T is always Float64 cause of cupynumeric +# end + +# Pack an SVector as a Legate fixed-array scalar. Empty vectors pass a null ptr. +function _add_vector_scalar!(add!, task, ::SVector{0,T}) where {T} + add!(task, Ptr{T}(C_NULL), Int32(0)) + return nothing +end +function _add_vector_scalar!(add!, task, v::SVector{N,T}) where {N,T} + ref = Ref(v) + GC.@preserve ref begin + add!(task, Ptr{T}(Base.unsafe_convert(Ptr{SVector{N,T}}, ref)), Int32(N)) end + return nothing end -function nda_random_array(dims::Dims{N}) where {N} - shape = collect(UInt64, dims) - ptr = @task_scope "rand" begin - ccall((:nda_random_array, libnda), - NDArray_t, (Int32, Ptr{UInt64}), - Int32(N), shape) +# Match cupynumeric/_thunk/deferred.py::bitgenerator_distribution via Julia +# Legate tasking. Vector scalars still go through tiny C++ helpers because +# Legate.jl Scalar has no std::vector constructors. +function nda_bitgenerator_distribution!( + arr::NDArray, + handle::Int32, + generator_type::UInt32, + seed::UInt64, + flags::UInt32, + distribution::UInt32, + strides::SVector{N,Int64}, + intparams::SVector{NI,Int64}, + floatparams::SVector{NF,Float32}, + doubleparams::SVector{ND,Float64}, +) where {N,NI,NF,ND} + isempty(arr) && return arr + + @task_scope "bitgenerator" begin + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + task = Legate.create_auto_task(rt, lib, cuNumeric.BITGENERATOR) + + st = cuNumeric.get_store(arr) + Legate.add_output(task, st) + finalize(st) + + Legate.add_scalar(task, Legate.Scalar(Int32(cuNumeric.BITGENOP_DISTRIBUTION))) + Legate.add_scalar(task, Legate.Scalar(handle)) + Legate.add_scalar(task, Legate.Scalar(generator_type)) + Legate.add_scalar(task, Legate.Scalar(seed)) + Legate.add_scalar(task, Legate.Scalar(flags)) + Legate.add_scalar(task, Legate.Scalar(distribution)) + + _add_vector_scalar!(cuNumeric.add_vector_scalar_i64, task, strides) + _add_vector_scalar!(cuNumeric.add_vector_scalar_i64, task, intparams) + _add_vector_scalar!(cuNumeric.add_vector_scalar_f32, task, floatparams) + _add_vector_scalar!(cuNumeric.add_vector_scalar_f64, task, doubleparams) + + Legate.submit_auto_task(rt, task) end - return NDArray(ptr, Float64, Val(N)) #* T is always Float64 cause of cupynumeric + return arr end function nda_get_slice(arr::NDArray{T,N}, slices::Vector{Slice}) where {T,N} diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 7dd0b2779..8a7f9afae 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -654,47 +654,6 @@ function ones() return ones(DEFAULT_FLOAT) end -@doc""" - cuNumeric.rand!(arr::NDArray{Float64}) - -Fill `arr` in-place with uniform random `Float64` values. -""" -Random.rand!(arr::NDArray{Float64}) = cuNumeric.nda_random(arr, 0) -function Random.rand!(arr::NDArray{T}) where {T} - return error("rand! only supports NDArray{Float64} for now. Cast with cuNumeric.as_type.") -end - -# Backend only generates Float64. Same-type path needs no cast; other floats -# convert then eagerly drop the Float64 source so it cannot leak until GC. -@doc""" - cuNumeric.rand([T=Float32,] dims::Int...) - cuNumeric.rand([T=Float32,] dims::Tuple) - -Create a new `NDArray` filled with uniform random values. - -The backend currently supports only `Float64` draws. Other floating types are -converted automatically. - -# Examples -```@repl -cuNumeric.rand(2, 2) -cuNumeric.rand((4, 1)) -A = cuNumeric.zeros(Float64, 2, 2); cuNumeric.rand!(A) -``` -""" -rand(::Type{Float64}, dims::Dims) = cuNumeric.nda_random_array(dims) - -function rand(::Type{T}, dims::Dims) where {T<:AbstractFloat} - arrfp64 = cuNumeric.nda_random_array(dims) - arr = cuNumeric.as_type(arrfp64, T) - destroy!(arrfp64) - return arr -end - -rand(::Type{T}, dims::Int...) where {T<:AbstractFloat} = cuNumeric.rand(T, dims) -rand(dims::Dims) = cuNumeric.rand(DEFAULT_FLOAT, dims) -rand(dims::Int...) = cuNumeric.rand(DEFAULT_FLOAT, dims) - #### OPERATIONS #### @doc""" reshape(arr::NDArray, dims::Dims{N}; copy::Val{C}=Val(false)) where {N,C} diff --git a/src/ndarray/random/bitgenerator.jl b/src/ndarray/random/bitgenerator.jl new file mode 100644 index 000000000..a72240ac0 --- /dev/null +++ b/src/ndarray/random/bitgenerator.jl @@ -0,0 +1,154 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +# Mirrors cupynumeric.random._bitgenerator.BitGenerator and the three +# engines Python actually subclasses: XORWOW, MRG32k3a, PHILOX4_32_10. +# CREATE is lazy: the first DISTRIBUTION task materializes the generator. + +""" + BitGenerator + +Abstract supertype for cuRAND engines used by [`Generator`](@ref cuNumeric.Generator). Concrete +types are [`XORWOW`](@ref cuNumeric.XORWOW) (default), [`MRG32k3a`](@ref cuNumeric.MRG32k3a), and +[`PHILOX4_32_10`](@ref cuNumeric.PHILOX4_32_10). +""" +abstract type BitGenerator end + +const _bitgen_id_lock = ReentrantLock() +const _next_bitgen_id = Ref{Int32}(0) +const _bitgen_zombies = Int32[] + +function _next_bitgenerator_id() + return lock(_bitgen_id_lock) do + _next_bitgen_id[] += Int32(1) + return _next_bitgen_id[] + end +end + +function _record_bitgenerator_zombie(handle::Int32) + handle == 0 && return nothing + return lock(_bitgen_id_lock) do + push!(_bitgen_zombies, handle) + return nothing + end +end + +# C-order strides (NDArray stores are row-major). Julia col-major is only +# handled in Array conversions. DISTRIBUTION writes a dense ptr and ignores this. +function _c_order_strides(shape::NTuple{N,Int}) where {N} + return SVector{N,Int64}( + ntuple(Val(N)) do i + p = Int64(1) + @inbounds for d in (i + 1):N + p *= Int64(shape[d]) + end + return p + end, + ) +end + +_c_order_strides(::Tuple{}) = SVector{0,Int64}() + +const _EMPTY_INT64 = SVector{0,Int64}() +const _EMPTY_FLOAT32 = SVector{0,Float32}() +const _EMPTY_FLOAT64 = SVector{0,Float64}() + +function _make_bitgenerator(::Type{B}, seed, flags) where {B<:BitGenerator} + handle = _next_bitgenerator_id() + seed64 = seed === nothing ? UInt64(time_ns()) : UInt64(seed) + bg = B(handle, seed64, UInt32(flags)) + finalizer(_finalize_bitgenerator!, bg) + return bg +end + +function _finalize_bitgenerator!(bg::BitGenerator) + handle = bg.handle + bg.handle = Int32(0) + _record_bitgenerator_zombie(handle) + return nothing +end + +""" + XORWOW(seed=nothing; flags=0) + +Default cuRAND BitGenerator (xorwow). `seed=nothing` draws from `time_ns()`. +""" +mutable struct XORWOW <: BitGenerator + handle::Int32 + seed::UInt64 + flags::UInt32 +end +function XORWOW(seed::Union{Integer,Nothing}=nothing; flags::Integer=0) + return _make_bitgenerator(XORWOW, seed, flags) +end + +""" + MRG32k3a(seed=nothing; flags=0) + +cuRAND MRG32k3a BitGenerator. +""" +mutable struct MRG32k3a <: BitGenerator + handle::Int32 + seed::UInt64 + flags::UInt32 +end +function MRG32k3a(seed::Union{Integer,Nothing}=nothing; flags::Integer=0) + return _make_bitgenerator(MRG32k3a, seed, flags) +end + +""" + PHILOX4_32_10(seed=nothing; flags=0) + +cuRAND Philox4_32_10 BitGenerator. +""" +mutable struct PHILOX4_32_10 <: BitGenerator + handle::Int32 + seed::UInt64 + flags::UInt32 +end +function PHILOX4_32_10(seed::Union{Integer,Nothing}=nothing; flags::Integer=0) + return _make_bitgenerator(PHILOX4_32_10, seed, flags) +end + +generator_type(::XORWOW) = cuNumeric.BITGENTYPE_XORWOW +generator_type(::MRG32k3a) = cuNumeric.BITGENTYPE_MRG32K3A +generator_type(::PHILOX4_32_10) = cuNumeric.BITGENTYPE_PHILOX4_32_10 + +function _bitgenerator_distribution!( + arr::NDArray, + bg::BitGenerator, + distribution, + intparams::SVector{NI,Int64}, + floatparams::SVector{NF,Float32}, + doubleparams::SVector{ND,Float64}, +) where {NI,NF,ND} + return nda_bitgenerator_distribution!( + arr, + bg.handle, + UInt32(generator_type(bg)), + bg.seed, + bg.flags, + UInt32(distribution), + _c_order_strides(size(arr)), + intparams, + floatparams, + doubleparams, + ) +end diff --git a/src/ndarray/random/generator.jl b/src/ndarray/random/generator.jl new file mode 100644 index 000000000..f05dd06a3 --- /dev/null +++ b/src/ndarray/random/generator.jl @@ -0,0 +1,270 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +""" + Generator(bit_generator) + +cuPyNumeric-style RNG wrapping a [`BitGenerator`](@ref cuNumeric.BitGenerator). Module-level +[`rand`](@ref cuNumeric.rand), [`randn`](@ref cuNumeric.randn), [`randexp`](@ref cuNumeric.randexp) use a process-global +XORWOW generator; pass a different engine to `default_rng` for Philox or MRG32k3a. +""" +struct Generator{B<:BitGenerator} + bit_generator::B +end + +const _RNG_INT_TYPES = Union{Int16,Int32,Int64} +const _static_generator = Ref{Union{Nothing,Generator{XORWOW}}}(nothing) + +""" + default_rng() + default_rng(seed) + default_rng(::Type{<:BitGenerator}, seed=nothing; flags=0) + default_rng(bit_generator) + default_rng(generator) + +Return a [`Generator`](@ref cuNumeric.Generator). The no-argument and integer-seed methods use +[`XORWOW`](@ref cuNumeric.XORWOW). This does **not** replace the process-global generator used +by module-level `rand` / `randn`. + +```julia +g = cuNumeric.default_rng(cuNumeric.PHILOX4_32_10, 1234) +cuNumeric.random(g, Float32, (8, 8)) +``` +""" +function default_rng() + return Generator(XORWOW()) +end + +function default_rng(seed::Integer) + return Generator(XORWOW(seed)) +end + +function default_rng( + ::Type{B}, seed::Union{Integer,Nothing}=nothing; flags::Integer=0 +) where {B<:BitGenerator} + return Generator(B(seed; flags)) +end + +function default_rng(bg::BitGenerator) + return Generator(bg) +end + +function default_rng(g::Generator) + return g +end + +function get_static_generator() + gen = _static_generator[] + if gen === nothing + gen = default_rng() + _static_generator[] = gen + end + return gen::Generator{XORWOW} +end + +function random!(g::Generator, arr::NDArray{Float32}) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_UNIFORM_32, + _EMPTY_INT64, SVector{2,Float32}(0, 1), _EMPTY_FLOAT64, + ) + return arr +end + +function random!(g::Generator, arr::NDArray{Float64}) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_UNIFORM_64, + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{2,Float64}(0, 1), + ) + return arr +end + +# No native Bool distribution. Draw {0,1} as Int16 and compare; that is the +# usual path to NDArray{Bool}. Do not as_type to Bool (CxxWrap.CxxBool). +function random!(g::Generator, arr::NDArray{Bool}) + copyto!(arr, random(g, Bool, size(arr))) + return arr +end + +# as_type first: `float .* Complex` is a widening promotion and is disallowed. +# Does not take ownership of `re` / `imag_part`. +function _pack_complex!( + out::NDArray{Complex{T}}, re::NDArray{T}, imag_part::NDArray{T} +) where {T<:SUPPORTED_FLOAT_TYPES} + CT = Complex{T} + re_c = as_type(re, CT) + im_c = as_type(imag_part, CT) + out .= re_c .+ im_c .* CT(0, 1) + destroy!(re_c) + destroy!(im_c) + return out +end + +# No native complex BitGenerator dist. Independent real/imag uniforms, like +# Julia `rand(Complex{T})` (unit square, not the unit disk). +function random!(g::Generator, arr::NDArray{Complex{T}}) where {T<:SUPPORTED_FLOAT_TYPES} + re = random(g, T, size(arr)) + imag_part = random(g, T, size(arr)) + _pack_complex!(arr, re, imag_part) + destroy!(re) + destroy!(imag_part) + return arr +end + +function random!(g::Generator, arr::NDArray{T}) where {T} + return error( + "random! supports Float32, Float64, ComplexF32, ComplexF64, Bool, Int16, Int32, and Int64 NDArray storage" + ) +end + +function random(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + arr = zeros(T, dims) + random!(g, arr) + return arr +end + +function random(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + arr = zeros(T, dims) + random!(g, arr) + return arr +end + +function random(g::Generator, ::Type{Bool}, dims::Dims) + return integers(g, Int16, dims; low=0, high=2) .!= Int16(0) +end + +function _randn!(g::Generator, arr::NDArray{Float32}; loc::Real=0, scale::Real=1) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_NORMAL_32, + _EMPTY_INT64, SVector{2,Float32}(Float32(loc), Float32(scale)), _EMPTY_FLOAT64, + ) + return arr +end + +function _randn!(g::Generator, arr::NDArray{Float64}; loc::Real=0, scale::Real=1) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_NORMAL_64, + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{2,Float64}(Float64(loc), Float64(scale)), + ) + return arr +end + +function randn!(g::Generator, arr::NDArray{Float32}) + return _randn!(g, arr) +end + +function randn!(g::Generator, arr::NDArray{Float64}) + return _randn!(g, arr) +end + +# Independent N(0, 1/2) real/imag so E[|z|²] = 1, matching Julia `randn(Complex{T})`. +# No loc/scale kwargs: shift or scale in user code (`μ .+ σ .* Z`). +function randn!(g::Generator, arr::NDArray{Complex{T}}) where {T<:SUPPORTED_FLOAT_TYPES} + s = 1 / sqrt(T(2)) + re = zeros(T, size(arr)) + imag_part = zeros(T, size(arr)) + _randn!(g, re; scale=s) + _randn!(g, imag_part; scale=s) + _pack_complex!(arr, re, imag_part) + destroy!(re) + destroy!(imag_part) + return arr +end + +function randn!(g::Generator, arr::NDArray{T}) where {T} + return error( + "randn! only supports Float32, Float64, ComplexF32, and ComplexF64 NDArray storage" + ) +end + +function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + arr = zeros(T, dims) + randn!(g, arr) + return arr +end + +function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + arr = zeros(T, dims) + randn!(g, arr) + return arr +end + +function randexp!(g::Generator, arr::NDArray{Float32}; scale::Real=1) + s = Float32(scale) + s > 0 || throw(ArgumentError("scale must be positive, got $scale")) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_EXPONENTIAL_32, + _EMPTY_INT64, SVector{1,Float32}(s), _EMPTY_FLOAT64, + ) + return arr +end + +function randexp!(g::Generator, arr::NDArray{Float64}; scale::Real=1) + s = Float64(scale) + s > 0 || throw(ArgumentError("scale must be positive, got $scale")) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_EXPONENTIAL_64, + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{1,Float64}(s), + ) + return arr +end + +function randexp!(g::Generator, arr::NDArray{T}; scale::Real=1) where {T} + return error("randexp! only supports Float32 and Float64 NDArray storage") +end + +function randexp( + g::Generator, ::Type{T}, dims::Dims; scale::Real=1 +) where {T<:SUPPORTED_FLOAT_TYPES} + arr = zeros(T, dims) + randexp!(g, arr; scale=scale) + return arr +end + +_integer_distribution(::Type{Int16}) = cuNumeric.BITGENDIST_INTEGERS_16 +_integer_distribution(::Type{Int32}) = cuNumeric.BITGENDIST_INTEGERS_32 +_integer_distribution(::Type{Int64}) = cuNumeric.BITGENDIST_INTEGERS_64 + +function _integer_distribution(::Type{T}) where {T} + return throw(ArgumentError("integer random only supports Int16, Int32, and Int64. Got $T.")) +end + +# Kernel draws [low, high); Julia `a:b` is mapped to low=a, high=b+1. +function integers!( + g::Generator, arr::NDArray{T}; low::Integer, high::Integer +) where {T<:_RNG_INT_TYPES} + Int64(high) <= Int64(low) && throw(ArgumentError("low >= high")) + _bitgenerator_distribution!( + arr, g.bit_generator, _integer_distribution(T), + SVector{2,Int64}(Int64(low), Int64(high)), _EMPTY_FLOAT32, _EMPTY_FLOAT64, + ) + return arr +end + +function integers!(g::Generator, arr::NDArray{T}; low::Integer, high::Integer) where {T} + return throw(ArgumentError("integer random only supports Int16, Int32, and Int64")) +end + +function integers( + g::Generator, ::Type{T}, dims::Dims; low::Integer, high::Integer +) where {T<:_RNG_INT_TYPES} + arr = zeros(T, dims) + integers!(g, arr; low=low, high=high) + return arr +end diff --git a/src/ndarray/random/random.jl b/src/ndarray/random/random.jl new file mode 100644 index 000000000..6608ef9cf --- /dev/null +++ b/src/ndarray/random/random.jl @@ -0,0 +1,218 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +# Module-level convenience API, mirroring cupynumeric.random._random.py. +# All draws go through get_static_generator(). + +@doc""" + Random.rand!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + Random.rand!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + Random.rand!(arr::NDArray{<:Union{Int16,Int32,Int64}}) + Random.rand!(arr::NDArray{Bool}) + +Fill `arr` in-place with uniform random values. + +Floating arrays are filled from `[0, 1)`. Complex arrays draw independent +real/imag uniforms in `[0, 1)` (the unit square). Integer arrays use the full +range of the element type, matching Julia `rand(T)`. `Bool` arrays are fair +coin flips. +""" +function Random.rand!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + return random!(get_static_generator(), arr) +end + +function Random.rand!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return random!(get_static_generator(), arr) +end + +function Random.rand!(arr::NDArray{T}) where {T<:_RNG_INT_TYPES} + return integers!(get_static_generator(), arr; low=typemin(T), high=typemax(T)) +end + +function Random.rand!(arr::NDArray{Bool}) + return random!(get_static_generator(), arr) +end + +function Random.rand!(arr::NDArray{T}) where {T} + return error( + "rand! supports Float32, Float64, ComplexF32, ComplexF64, Bool, Int16, Int32, and Int64 NDArray storage" + ) +end + +@doc""" + cuNumeric.rand([T=Float32,] dims::Int...) + cuNumeric.rand([T=Float32,] dims::Tuple) + cuNumeric.rand(r::AbstractUnitRange, dims...) + +Create a new `NDArray` filled with uniform random values. + +Floating types (`Float32` / `Float64`) are drawn natively in `[0, 1)`. +`ComplexF32` / `ComplexF64` draw independent real/imag uniforms (unit square). +`Bool` is a fair coin flip. Integer types `Int16`, `Int32`, and `Int64` use the +full range of `T`. A unit range (`1:10`) draws inclusive integers, matching +Julia `rand(1:10, dims...)`. + +Uses a process-global [`XORWOW`](@ref cuNumeric.XORWOW) generator. For a different engine or +seed, see [`default_rng`](@ref cuNumeric.default_rng) and [`Generator`](@ref cuNumeric.Generator). + +# Examples +```@repl +cuNumeric.rand(2, 2) +cuNumeric.rand((4, 1)) +cuNumeric.rand(0:9, 4, 4) +cuNumeric.rand(Bool, 8) +A = cuNumeric.zeros(Float32, 2, 2); cuNumeric.rand!(A) +``` +""" +function rand(::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + return random(get_static_generator(), T, dims) +end + +function rand(::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + return random(get_static_generator(), T, dims) +end + +function rand(::Type{T}, dims::Dims) where {T<:_RNG_INT_TYPES} + return integers(get_static_generator(), T, dims; low=typemin(T), high=typemax(T)) +end + +rand(::Type{Bool}, dims::Dims) = random(get_static_generator(), Bool, dims) + +function rand( + ::Type{T}, dims::Int... +) where {T<:Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES,_RNG_INT_TYPES,Bool}} + return cuNumeric.rand(T, dims) +end +rand(dims::Dims) = cuNumeric.rand(DEFAULT_FLOAT, dims) +rand(dims::Int...) = cuNumeric.rand(DEFAULT_FLOAT, dims) + +function _unitrange_bounds(r::AbstractUnitRange{<:Integer}) + lo = Int64(first(r)) + hi = Int64(last(r)) + hi < lo && throw(ArgumentError("empty range $r")) + return lo, hi + Int64(1) # kernel is [low, high) +end + +function rand(r::AbstractUnitRange{T}, dims::Dims) where {T<:_RNG_INT_TYPES} + lo, hi = _unitrange_bounds(r) + return integers(get_static_generator(), T, dims; low=lo, high=hi) +end + +rand(r::AbstractUnitRange{<:_RNG_INT_TYPES}, dims::Int...) = cuNumeric.rand(r, dims) + +function rand(r::AbstractUnitRange{Bool}, dims::Dims) + first(r) == last(r) && return fill(first(r), dims) + first(r) == false && last(r) == true && return cuNumeric.rand(Bool, dims) + return throw(ArgumentError("empty range $r")) +end +rand(r::AbstractUnitRange{Bool}, dims::Int...) = cuNumeric.rand(r, dims) + +@doc""" + Random.randn!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + Random.randn!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + +Fill `arr` in-place with standard normal samples. Real arrays have mean 0 and +variance 1. Complex arrays match Julia `randn(Complex{T})`: independent real +and imag parts with variance `1/2`, so `E[|z|²] = 1`. +""" +function Random.randn!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + return randn!(get_static_generator(), arr) +end + +function Random.randn!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return randn!(get_static_generator(), arr) +end + +function Random.randn!(arr::NDArray{T}) where {T} + return error( + "randn! only supports Float32, Float64, ComplexF32, and ComplexF64 NDArray storage" + ) +end + +@doc""" + cuNumeric.randn([T=Float32,] dims::Int...) + cuNumeric.randn([T=Float32,] dims::Tuple) + +Create a new `NDArray` filled with standard normal samples. Complex types +match Julia `randn(Complex{T})` (`E[|z|²] = 1`). + +Uses a process-global [`XORWOW`](@ref cuNumeric.XORWOW) generator. For a different +engine, pass a [`Generator`](@ref cuNumeric.Generator) to `randn!`. Shift or scale +in user code (`μ .+ σ .* Z`); `loc`/`scale` are not part of the public API. + +# Examples +```@repl +cuNumeric.randn(2, 2) +cuNumeric.randn(Float64, 1000) +``` +""" +function randn(::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + return randn(get_static_generator(), T, dims) +end + +function randn(::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + return randn(get_static_generator(), T, dims) +end + +function randn( + ::Type{T}, dims::Int... +) where {T<:Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES}} + return cuNumeric.randn(T, dims) +end +randn(dims::Dims) = cuNumeric.randn(DEFAULT_FLOAT, dims) +randn(dims::Int...) = cuNumeric.randn(DEFAULT_FLOAT, dims) + +@doc""" + Random.randexp!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + +Fill `arr` in-place with exponential samples of scale 1 (mean 1), matching +Julia `randexp`. +""" +function Random.randexp!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + return randexp!(get_static_generator(), arr) +end + +function Random.randexp!(arr::NDArray{T}) where {T} + return error("randexp! only supports Float32 and Float64 NDArray storage") +end + +@doc""" + cuNumeric.randexp([T=Float32,] dims::Int...) + cuNumeric.randexp([T=Float32,] dims::Tuple) + +Create a new `NDArray` filled with exponential samples of scale 1 (mean 1), +matching Julia `randexp`. + +Uses a process-global [`XORWOW`](@ref cuNumeric.XORWOW) generator. For a +different engine or scale, use `randexp!(generator, arr; scale)`. + +# Examples +```@repl +cuNumeric.randexp(2, 2) +cuNumeric.randexp(Float64, 1000) +``` +""" +function randexp(::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + return randexp(get_static_generator(), T, dims) +end + +randexp(::Type{T}, dims::Int...) where {T<:SUPPORTED_FLOAT_TYPES} = cuNumeric.randexp(T, dims) +randexp(dims::Dims) = cuNumeric.randexp(DEFAULT_FLOAT, dims) +randexp(dims::Int...) = cuNumeric.randexp(DEFAULT_FLOAT, dims) diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index b000bd522..db527d199 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -55,6 +55,11 @@ end @test @inferred(cuNumeric.rand(4, 3)) !== nothing @test @inferred(cuNumeric.rand(Float32, 5)) !== nothing + @test @inferred(cuNumeric.randn(Float64, 5)) !== nothing + @test @inferred(cuNumeric.randexp(Float32, 5)) !== nothing + @test @inferred(cuNumeric.rand(0:3, 3)) !== nothing + @test @inferred(cuNumeric.rand(ComplexF32, 5)) !== nothing + @test @inferred(cuNumeric.randn(ComplexF64, 5)) !== nothing # NDArray from Julia Array (Parent-stable attachment) @test @inferred(cuNumeric.NDArray(rand(10))) !== nothing diff --git a/test/array/random.jl b/test/array/random.jl new file mode 100644 index 000000000..9013534d0 --- /dev/null +++ b/test/array/random.jl @@ -0,0 +1,296 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +#= Purpose of test: random + -- dtype / shape of rand, randn, randexp, and in-place fills + -- complex rand / randn (independent real/imag); randexp throws + -- uniforms land in [0, 1) with mean 1/2 and variance 1/12 + -- normals match loc / scale moments (default and shifted) + -- exponentials match scale moments (mean = var = scale) + -- Monte-Carlo integral of exp(-x^2) recovers √π + -- rand(a:b) is inclusive and a small discrete range has the right mean / var + -- XORWOW / MRG32k3a / PHILOX4_32_10 all draw valid uniforms +=# + +function _host(arr) + return allowscalar() do + return Array(arr) + end +end + +function _moments(arr) + h = _host(arr) + return mean(h), var(h) +end + +@testset verbose = true "constructors" begin + for T in (Float32, Float64) + A = cuNumeric.rand(T, 8, 4) + @test eltype(A) === T + @test size(A) == (8, 4) + @test all(0 .<= _host(A) .< 1) + + B = cuNumeric.zeros(T, 5, 5) + cuNumeric.rand!(B) + @test all(0 .<= _host(B) .< 1) + + N = cuNumeric.randn(T, 32) + @test eltype(N) === T + @test size(N) == (32,) + + C = cuNumeric.zeros(T, 16) + Random.randn!(C) + @test eltype(C) === T + + E = cuNumeric.randexp(T, 16) + @test eltype(E) === T + @test size(E) == (16,) + Random.randexp!(C) + @test eltype(C) === T + end + + @test eltype(cuNumeric.randexp(8)) === Float32 + @test_throws MethodError cuNumeric.randexp(Int32, 4) + @test_throws MethodError cuNumeric.randexp(ComplexF32, 4) + @test_throws MethodError cuNumeric.randexp(ComplexF64, 4) + @test_throws MethodError cuNumeric.rand(Complex{Int64}, 4) + @test_throws MethodError cuNumeric.randn(Complex{Int64}, 4) + @test_throws MethodError cuNumeric.randexp(Complex{Int64}, 4) + + for T in (ComplexF32, ComplexF64) + A = cuNumeric.rand(T, 8, 4) + @test eltype(A) === T + @test size(A) == (8, 4) + h = _host(A) + @test all(0 .<= real.(h) .< 1) + @test all(0 .<= imag.(h) .< 1) + + B = cuNumeric.zeros(T, 5, 5) + cuNumeric.rand!(B) + hb = _host(B) + @test all(0 .<= real.(hb) .< 1) + @test all(0 .<= imag.(hb) .< 1) + + N = cuNumeric.randn(T, 32) + @test eltype(N) === T + @test size(N) == (32,) + + C = cuNumeric.zeros(T, 16) + Random.randn!(C) + @test eltype(C) === T + + @test_throws ErrorException Random.randexp!(C) + g = cuNumeric.default_rng() + @test_throws ErrorException randexp!(g, C) + @test_throws MethodError cuNumeric.randexp(g, T, (4,)) + end + + R = cuNumeric.rand(3, 2) + @test eltype(R) === Float32 + @test size(R) == (3, 2) + + @test_throws MethodError cuNumeric.randn(Int32, 4) + + Coin = cuNumeric.rand(Bool, 8, 4) + @test eltype(Coin) === Bool + @test size(Coin) == (8, 4) + @test all(x -> x == true || x == false, _host(Coin)) + Mask = cuNumeric.falses(16) + cuNumeric.rand!(Mask) + @test eltype(Mask) === Bool +end + +@testset verbose = true "uniform moments" begin + n = 65_536 + for T in (Float32, Float64) + μ, v = _moments(cuNumeric.rand(T, n)) + @test abs(μ - 0.5) < 0.03 + @test abs(v - 1 / 12) < 0.01 + + μ2, v2 = _moments(cuNumeric.rand(T, 128, 128)) + @test abs(μ2 - 0.5) < 0.03 + @test abs(v2 - 1 / 12) < 0.01 + end + + μb, vb = _moments(cuNumeric.rand(Bool, 65_536)) + @test abs(μb - 0.5) < 0.03 + @test abs(vb - 0.25) < 0.02 + + n = 65_536 + for T in (ComplexF32, ComplexF64) + Z = _host(cuNumeric.rand(T, n)) + @test abs(mean(real.(Z)) - 0.5) < 0.03 + @test abs(mean(imag.(Z)) - 0.5) < 0.03 + @test abs(var(real.(Z)) - 1 / 12) < 0.01 + @test abs(var(imag.(Z)) - 1 / 12) < 0.01 + end +end + +@testset verbose = true "normal moments" begin + n = 65_536 + for T in (Float32, Float64) + μ, v = _moments(cuNumeric.randn(T, n)) + @test abs(μ) < 0.05 + @test abs(v - 1) < 0.08 + + g = cuNumeric.default_rng() + A = cuNumeric.zeros(T, n) + cuNumeric._randn!(g, A; loc=T(3), scale=T(2)) + μs, vs = _moments(A) + @test abs(μs - 3) < 0.1 + @test abs(vs - 4) < 0.3 + end + + # Julia randn(Complex): Var(re) = Var(im) = 1/2, E[|z|²] = 1. + n = 65_536 + for T in (ComplexF32, ComplexF64) + Z = _host(cuNumeric.randn(T, n)) + @test abs(mean(real.(Z))) < 0.05 + @test abs(mean(imag.(Z))) < 0.05 + @test abs(var(real.(Z)) - 0.5) < 0.08 + @test abs(var(imag.(Z)) - 0.5) < 0.08 + @test abs(mean(abs2.(Z)) - 1) < 0.08 + end +end + +@testset verbose = true "exponential moments" begin + n = 65_536 + for T in (Float32, Float64) + μ, v = _moments(cuNumeric.randexp(T, n)) + @test abs(μ - 1) < 0.08 + @test abs(v - 1) < 0.12 + @test all(>=(zero(T)), _host(cuNumeric.randexp(T, 1024))) + + g = cuNumeric.default_rng() + A = cuNumeric.zeros(T, n) + randexp!(g, A; scale=T(2)) + μs, vs = _moments(A) + @test abs(μs - 2) < 0.12 + @test abs(vs - 4) < 0.4 + end + + @test_throws ArgumentError randexp!(cuNumeric.default_rng(), cuNumeric.zeros(8); scale=0) +end + +@testset verbose = true "monte carlo" begin + # ∫_{-∞}^{∞} exp(-x^2) dx = √π, truncated to [-10, 10] as in the docs example. + n = 131_072 + for T in (Float32, Float64) + xmax = T(10) + Ω = T(2) * xmax + samples = Ω .* cuNumeric.rand(T, n) .- xmax + integrand = (x) -> @. exp(-x^2) + estimate = unwrap((Ω / n) * sum(integrand(samples))) + @test isapprox(estimate, T(sqrt(π)); atol=T(0.08)) + end + + # Area of the unit disk via darts in [0, 1]^2 → π. + n = 131_072 + for T in (Float32, Float64) + x = _host(cuNumeric.rand(T, n)) + y = _host(cuNumeric.rand(T, n)) + π_estimate = 4 * mean(x .^ 2 .+ y .^ 2 .< one(T)) + @test isapprox(π_estimate, T(π); atol=T(0.05)) + end +end + +@testset verbose = true "integers" begin + A = cuNumeric.rand(0:9, 64) + @test eltype(A) === Int + @test all(0 .<= _host(A) .<= 9) + + B = cuNumeric.rand(Int32(-4):Int32(4), 8, 8) + @test eltype(B) === Int32 + @test size(B) == (8, 8) + @test all(-4 .<= _host(B) .<= 4) + + C = cuNumeric.rand(Int16, 32) + @test eltype(C) === Int16 + @test all(typemin(Int16) .<= _host(C) .< typemax(Int16)) + + D = cuNumeric.zeros(Int32, 16) + cuNumeric.rand!(D) + @test all(typemin(Int32) .<= _host(D) .< typemax(Int32)) + + # singleton range is constant + Z = _host(cuNumeric.rand(0:0, 256)) + @test all(==(0), Z) + F = _host(cuNumeric.rand(Int32(5):Int32(5), 128)) + @test all(==(5), F) + + # discrete uniform on 0:9: mean 4.5, var (n^2-1)/12 with n=10 + n = 65_536 + h = Float64.(_host(cuNumeric.rand(0:9, n))) + @test abs(mean(h) - 4.5) < 0.08 + @test abs(var(h) - (100 - 1) / 12) < 0.2 + + @test_throws ArgumentError cuNumeric.rand(1:0, 4) +end + +@testset verbose = true "default_rng" begin + g = cuNumeric.default_rng() + @test g isa cuNumeric.Generator + A = cuNumeric.random(g, Float32, (4, 4)) + @test eltype(A) === Float32 + @test size(A) == (4, 4) + @test all(0 .<= _host(A) .< 1) + + g2 = cuNumeric.default_rng(1234) + @test g2 isa cuNumeric.Generator + @test g2.bit_generator isa cuNumeric.XORWOW + @test g2.bit_generator.seed == UInt64(1234) + B = cuNumeric.randn(g2, Float64, (32,)) + @test eltype(B) === Float64 + + C = cuNumeric.random(g2, ComplexF32, (8, 8)) + @test eltype(C) === ComplexF32 + @test size(C) == (8, 8) + Ch = _host(C) + @test all(0 .<= real.(Ch) .< 1) + @test all(0 .<= imag.(Ch) .< 1) + + s1 = cuNumeric.get_static_generator() + s2 = cuNumeric.get_static_generator() + @test s1 === s2 + @test s1 !== g2 +end + +@testset verbose = true "bitgenerators" begin + n = 16_384 + for B in (cuNumeric.XORWOW, cuNumeric.MRG32k3a, cuNumeric.PHILOX4_32_10) + @testset "$(B)" begin + g = cuNumeric.default_rng(B, 42) + @test g isa cuNumeric.Generator{B} + @test g.bit_generator isa B + @test g.bit_generator.seed == UInt64(42) + + U = cuNumeric.random(g, Float32, (n,)) + @test eltype(U) === Float32 + Uh = _host(U) + @test all(0 .<= Uh .< 1) + @test abs(mean(Uh) - 0.5) < 0.05 + @test abs(var(Uh) - 1 / 12) < 0.02 + + Nrm = cuNumeric.randn(g, Float64, (n,)) + μ, v = _moments(Nrm) + @test abs(μ) < 0.08 + @test abs(v - 1) < 0.12 + end + end +end diff --git a/test/runtests.jl b/test/runtests.jl index 279554fcd..da06059c3 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -26,6 +26,7 @@ end const init_code = quote using LinearAlgebra using Random + using StatsBase import Random: rand ENV["LEGATE_SKIP_RUNTIME"] = "false" From 3e56672ca8784ce88a7312e16234f5e370f74b3d Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Fri, 14 Aug 2026 14:09:10 -0500 Subject: [PATCH 02/49] leverage DEPOT path --- .buildkite/run_developer_ci.sh | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 537494b28..b2deb7f4f 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -24,12 +24,16 @@ cmake --version rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml -# Dev mode rebuilds the wrapper .so each run, so drop stale overrides and the .ji that -# bake @wrapmodule bindings (cuNumeric/Legate + wrapper JLLs) — a cached .ji would -# mismatch the fresh .so and segfault. libcxxwrap override is kept. -DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" -rm -rf "$DEPOT"/packages/*/*/override \ - "$DEPOT"/compiled/v*/{cuNumeric,Legate,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} +# Per-run writable depot layered over the shared cache (read-only). The cuda queue +# shares ${HOME}/.cache/... across machines, so concurrent runs would otherwise clobber +# each other's wrapper overrides and .ji caches (mismatched .ji vs. fresh .so segfaults). +# Julia writes only to DEPOT_PATH[1], so the temp isolates build output while artifacts, +# packages, and registries are read in place from the shared cache. Discarded next build. +SHARED_DEPOT="${JULIA_DEPOT_PATH:-$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')}" +RUN_DEPOT="$(mktemp -d)" +trap 'rm -rf "$RUN_DEPOT"' EXIT +export JULIA_DEPOT_PATH="$RUN_DEPOT:$SHARED_DEPOT" +echo "Isolated build depot: $RUN_DEPOT (shared cache read-only: $SHARED_DEPOT)" LEGATE_BRANCH_INPUT="${BUILDKITE_MESSAGE:-}" if [[ "${BUILDKITE_PULL_REQUEST:-false}" =~ ^[0-9]+$ ]]; then From 7df9986d77b5a25f892a7c11e983a068ec3988d8 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Fri, 14 Aug 2026 14:51:15 -0500 Subject: [PATCH 03/49] fix libcxxwrap --- .buildkite/run_developer_ci.sh | 50 ++++++++++++++++++++++++++++------ 1 file changed, 41 insertions(+), 9 deletions(-) diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index b2deb7f4f..70d7aaa5a 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -24,16 +24,35 @@ cmake --version rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml -# Per-run writable depot layered over the shared cache (read-only). The cuda queue -# shares ${HOME}/.cache/... across machines, so concurrent runs would otherwise clobber -# each other's wrapper overrides and .ji caches (mismatched .ji vs. fresh .so segfaults). -# Julia writes only to DEPOT_PATH[1], so the temp isolates build output while artifacts, -# packages, and registries are read in place from the shared cache. Discarded next build. -SHARED_DEPOT="${JULIA_DEPOT_PATH:-$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')}" -RUN_DEPOT="$(mktemp -d)" +# Isolate each run in a disposable hardlink-clone of the plugin depot so concurrent PR +# jobs on the shared cuda-queue cache can't clobber each other's wrapper overrides or .ji +# (a .ji baked against a stale wrapper .so segfaults). Clone is near-free on one filesystem +# and shares the read-only artifact store; all build writes stay private and are discarded. +SHARED_DEPOT="${JULIA_DEPOT_PATH:-}"; SHARED_DEPOT="${SHARED_DEPOT%%:*}" +[[ -n "$SHARED_DEPOT" ]] || SHARED_DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" +RUN_DEPOT="$(mktemp -d "${HOME}/.cache/cn-ci-depot.XXXXXX")" trap 'rm -rf "$RUN_DEPOT"' EXIT -export JULIA_DEPOT_PATH="$RUN_DEPOT:$SHARED_DEPOT" -echo "Isolated build depot: $RUN_DEPOT (shared cache read-only: $SHARED_DEPOT)" +cp -al "$SHARED_DEPOT/." "$RUN_DEPOT/" +export JULIA_DEPOT_PATH="$RUN_DEPOT" +echo "Isolated build depot: $RUN_DEPOT (hardlink clone of $SHARED_DEPOT)" + +# Reset the cloned wrapper + libcxxwrap wiring so each rebuilds consistently in the clone. +rm -rf "$RUN_DEPOT"/dev/libcxxwrap_julia_jll \ + "$RUN_DEPOT"/packages/*/*/override \ + "$RUN_DEPOT"/compiled/v*/{CxxWrap,Legate,cuNumeric,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} + +# libcxxwrap is pinned and slow to compile: reuse it from a persistent per-Julia store +# (ABI tracks the Julia version, not fusion). Seed the clone if warm, else build+warm below. +# Bust the store (rm it) if Legate bumps the pinned libcxxwrap commit. +JULIA_MM="$(julia --startup-file=no -e 'print("$(VERSION.major).$(VERSION.minor)")')" +LIBCXXW_STORE="${HOME}/.cache/cunumeric-libcxxwrap/${JULIA_MM}/libcxxwrap_julia_jll" +if [[ -f "$LIBCXXW_STORE/override/lib/libcxxwrap_julia.so" ]]; then + mkdir -p "$RUN_DEPOT/dev" + cp -al "$LIBCXXW_STORE" "$RUN_DEPOT/dev/libcxxwrap_julia_jll" + echo "Seeded libcxxwrap from persistent store: $LIBCXXW_STORE" +else + echo "libcxxwrap store cold for Julia $JULIA_MM; this run will build and warm it." +fi LEGATE_BRANCH_INPUT="${BUILDKITE_MESSAGE:-}" if [[ "${BUILDKITE_PULL_REQUEST:-false}" =~ ^[0-9]+$ ]]; then @@ -83,6 +102,19 @@ julia --color=yes --project=. -e ' Pkg.build("cuNumeric") ' +# Warm the persistent store from this run's fresh build. Only happens on the first cold run +# per Julia version (until the pinned commit changes), so a rare concurrent double-build is +# fine: stage then atomically rename, so the first to finish wins and the rest are no-ops. +BUILT_LIBCXXW="$RUN_DEPOT/dev/libcxxwrap_julia_jll" +if [[ ! -e "$LIBCXXW_STORE" && -f "$BUILT_LIBCXXW/override/lib/libcxxwrap_julia.so" ]]; then + mkdir -p "$(dirname "$LIBCXXW_STORE")" + staging="${LIBCXXW_STORE}.tmp.$$" + rm -rf "$staging" + cp -al "$BUILT_LIBCXXW" "$staging" + mv -T "$staging" "$LIBCXXW_STORE" 2>/dev/null || rm -rf "$staging" + echo "Warmed libcxxwrap store: $LIBCXXW_STORE" +fi + cp LocalPreferences.toml test/LocalPreferences.toml julia --color=yes --project=. -e ' From f663c6525e6476c3873fa8818bddb9617bf4566e Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Fri, 14 Aug 2026 15:55:30 -0500 Subject: [PATCH 04/49] try this --- .buildkite/run_developer_ci.sh | 32 +++++--------------------------- 1 file changed, 5 insertions(+), 27 deletions(-) diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 70d7aaa5a..64954d07a 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -36,24 +36,15 @@ cp -al "$SHARED_DEPOT/." "$RUN_DEPOT/" export JULIA_DEPOT_PATH="$RUN_DEPOT" echo "Isolated build depot: $RUN_DEPOT (hardlink clone of $SHARED_DEPOT)" -# Reset the cloned wrapper + libcxxwrap wiring so each rebuilds consistently in the clone. +# Reset the cloned wrapper + libcxxwrap wiring so each rebuilds fresh in this depot. Both +# wrapper CMakeLists and build_jlcxxwrap derive libcxxwrap's path from DEPOT_PATH[1] and bake +# absolute paths into JlCxx's cmake config + rpath, so a libcxxwrap built in another run's +# depot can't be reused here — it must be rebuilt in place. Dropping the dev override lets +# `using Legate` fall back to the stock JLL until build_jlcxxwrap rebuilds the custom one. rm -rf "$RUN_DEPOT"/dev/libcxxwrap_julia_jll \ "$RUN_DEPOT"/packages/*/*/override \ "$RUN_DEPOT"/compiled/v*/{CxxWrap,Legate,cuNumeric,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} -# libcxxwrap is pinned and slow to compile: reuse it from a persistent per-Julia store -# (ABI tracks the Julia version, not fusion). Seed the clone if warm, else build+warm below. -# Bust the store (rm it) if Legate bumps the pinned libcxxwrap commit. -JULIA_MM="$(julia --startup-file=no -e 'print("$(VERSION.major).$(VERSION.minor)")')" -LIBCXXW_STORE="${HOME}/.cache/cunumeric-libcxxwrap/${JULIA_MM}/libcxxwrap_julia_jll" -if [[ -f "$LIBCXXW_STORE/override/lib/libcxxwrap_julia.so" ]]; then - mkdir -p "$RUN_DEPOT/dev" - cp -al "$LIBCXXW_STORE" "$RUN_DEPOT/dev/libcxxwrap_julia_jll" - echo "Seeded libcxxwrap from persistent store: $LIBCXXW_STORE" -else - echo "libcxxwrap store cold for Julia $JULIA_MM; this run will build and warm it." -fi - LEGATE_BRANCH_INPUT="${BUILDKITE_MESSAGE:-}" if [[ "${BUILDKITE_PULL_REQUEST:-false}" =~ ^[0-9]+$ ]]; then echo "Reading Legate branch override from pull request #$BUILDKITE_PULL_REQUEST" @@ -102,19 +93,6 @@ julia --color=yes --project=. -e ' Pkg.build("cuNumeric") ' -# Warm the persistent store from this run's fresh build. Only happens on the first cold run -# per Julia version (until the pinned commit changes), so a rare concurrent double-build is -# fine: stage then atomically rename, so the first to finish wins and the rest are no-ops. -BUILT_LIBCXXW="$RUN_DEPOT/dev/libcxxwrap_julia_jll" -if [[ ! -e "$LIBCXXW_STORE" && -f "$BUILT_LIBCXXW/override/lib/libcxxwrap_julia.so" ]]; then - mkdir -p "$(dirname "$LIBCXXW_STORE")" - staging="${LIBCXXW_STORE}.tmp.$$" - rm -rf "$staging" - cp -al "$BUILT_LIBCXXW" "$staging" - mv -T "$staging" "$LIBCXXW_STORE" 2>/dev/null || rm -rf "$staging" - echo "Warmed libcxxwrap store: $LIBCXXW_STORE" -fi - cp LocalPreferences.toml test/LocalPreferences.toml julia --color=yes --project=. -e ' From 773fc6f05a78d41d86aa6a4d8ec37d39fba13c7e Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Fri, 14 Aug 2026 18:16:45 -0500 Subject: [PATCH 05/49] cp cache into temp storage --- .buildkite/run_developer_ci.sh | 29 +++++++++++++++-------------- 1 file changed, 15 insertions(+), 14 deletions(-) diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 64954d07a..402db7115 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -24,26 +24,27 @@ cmake --version rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml -# Isolate each run in a disposable hardlink-clone of the plugin depot so concurrent PR -# jobs on the shared cuda-queue cache can't clobber each other's wrapper overrides or .ji -# (a .ji baked against a stale wrapper .so segfaults). Clone is near-free on one filesystem -# and shares the read-only artifact store; all build writes stay private and are discarded. -SHARED_DEPOT="${JULIA_DEPOT_PATH:-}"; SHARED_DEPOT="${SHARED_DEPOT%%:*}" -[[ -n "$SHARED_DEPOT" ]] || SHARED_DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" -RUN_DEPOT="$(mktemp -d "${HOME}/.cache/cn-ci-depot.XXXXXX")" -trap 'rm -rf "$RUN_DEPOT"' EXIT -cp -al "$SHARED_DEPOT/." "$RUN_DEPOT/" -export JULIA_DEPOT_PATH="$RUN_DEPOT" -echo "Isolated build depot: $RUN_DEPOT (hardlink clone of $SHARED_DEPOT)" +# Isolate each run in a disposable hardlink-clone of the plugin cache (CACHE_DEPOT) so +# concurrent PR jobs on the shared cuda-queue cache can't clobber each other's wrapper +# overrides or .ji (a .ji baked against a stale wrapper .so segfaults). The clone (WORK_DEPOT) +# is near-free on one filesystem and shares the read-only artifact store; every build write +# lands in WORK_DEPOT and is discarded on exit, and CACHE_DEPOT is only ever read. +CACHE_DEPOT="${JULIA_DEPOT_PATH:-}"; CACHE_DEPOT="${CACHE_DEPOT%%:*}" +[[ -n "$CACHE_DEPOT" ]] || CACHE_DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" +WORK_DEPOT="$(mktemp -d "${HOME}/.cache/cn-ci-depot.XXXXXX")" +trap 'rm -rf "$WORK_DEPOT"' EXIT +cp -al "$CACHE_DEPOT/." "$WORK_DEPOT/" +export JULIA_DEPOT_PATH="$WORK_DEPOT" +echo "Isolated build depot: $WORK_DEPOT (hardlink clone of $CACHE_DEPOT)" # Reset the cloned wrapper + libcxxwrap wiring so each rebuilds fresh in this depot. Both # wrapper CMakeLists and build_jlcxxwrap derive libcxxwrap's path from DEPOT_PATH[1] and bake # absolute paths into JlCxx's cmake config + rpath, so a libcxxwrap built in another run's # depot can't be reused here — it must be rebuilt in place. Dropping the dev override lets # `using Legate` fall back to the stock JLL until build_jlcxxwrap rebuilds the custom one. -rm -rf "$RUN_DEPOT"/dev/libcxxwrap_julia_jll \ - "$RUN_DEPOT"/packages/*/*/override \ - "$RUN_DEPOT"/compiled/v*/{CxxWrap,Legate,cuNumeric,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} +rm -rf "$WORK_DEPOT"/dev/libcxxwrap_julia_jll \ + "$WORK_DEPOT"/packages/*/*/override \ + "$WORK_DEPOT"/compiled/v*/{CxxWrap,Legate,cuNumeric,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} LEGATE_BRANCH_INPUT="${BUILDKITE_MESSAGE:-}" if [[ "${BUILDKITE_PULL_REQUEST:-false}" =~ ^[0-9]+$ ]]; then From 38e70d29c1f95bf41e027d74e252863508e69793 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sat, 15 Aug 2026 12:26:14 -0400 Subject: [PATCH 06/49] Subtype `AbstractArray` + LinearAlgebra.Diagonal Support (#165) * Add Diagonal / UniformScaling support for NDArray * NDArray <: AbstractArray now --- README.md | 43 +- TODO.md | 56 +- deps/build.jl | 2 +- docs/src/api_initialization.md | 36 +- docs/src/examples/initialization.md | 8 +- docs/src/examples/special_mat.md | 23 + docs/src/linalg.md | 106 +++- src/cuNumeric.jl | 1 + src/ndarray/binary.jl | 14 +- src/ndarray/broadcast.jl | 46 +- src/ndarray/broadcast_fusion.jl | 16 +- src/ndarray/detail/ndarray.jl | 4 +- src/ndarray/diagonal.jl | 557 +++++++++++++++++++ src/ndarray/ndarray.jl | 334 +++++++----- src/ndarray/unary.jl | 11 + src/warnings.jl | 204 ++++++- test/analysis/type_stability.jl | 2 +- test/array/diagonal.jl | 811 ++++++++++++++++++++++++++++ test/array/linalg.jl | 6 +- 19 files changed, 2096 insertions(+), 184 deletions(-) create mode 100644 docs/src/examples/special_mat.md create mode 100644 src/ndarray/diagonal.jl create mode 100644 test/array/diagonal.jl diff --git a/README.md b/README.md index b51c551b3..39dd0896b 100644 --- a/README.md +++ b/README.md @@ -2,26 +2,41 @@ cuNumeric.jl cuNumeric.jl +

+ cuNumeric.jl + cuNumeric.jl +

+[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) [![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) +cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. +cuNumeric.jl requires x86 Linux, an NVIDIA GPU, and Julia >= 1.10. If ARM support is of interest open an issue. cuNumeric.jl requires x86 Linux, an NVIDIA GPU, and Julia >= 1.10. If ARM support is of interest open an issue. ### Quick Start +cuNumeric.jl can be installed with the Julia package manager. Activate your preferred environment and then from the Julia REPL run: + + cuNumeric.jl can be installed with the Julia package manager. Activate your preferred environment and then from the Julia REPL run: ```julia using Pkg Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") +using Pkg +Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") ``` The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). +The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). + ```julia using cuNumeric +using cuNumeric cuNumeric.versioninfo() ``` @@ -34,41 +49,57 @@ For more details, see [Hardware](./configuration/hardware.md). The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it is a drop-in replacement. You can use the same constructors (i.e., `zeros`, `ones`, `rand`), broadcasting, slicing, and linear algebra. Under the hood a few details differ from Base, and knowing them can help you write fast code. -**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be partitioned over GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. +**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be partitioned over GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. Functions like `println` result in data being copied to the host and can also be slow. **Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communite with the Julia runtime. When you need a plain Julia number, call `unwrap`: +**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communite with the Julia runtime. When you need a plain Julia number, call `unwrap` or `only`: ```julia s = sum(A) # NDArray{T,0} x = unwrap(s) # T, e.g. Float32 +x2 = only(s) +``` +s = sum(A) # NDArray{T,0} +x = unwrap(s) # T, e.g. Float32 ``` +**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. **The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. +For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). +### Kernel Fusion ### Kernel Fusion +Nested broadcast expressions fuse into a single kernel by default when on GPU. Prefer `@.` for multi-op elementwise code so every operator is dotted and the expression stays completely fused. Even just forgetting the `.` on unary negation (i.e., `y .= -a .+ b .* c`) will result in unfused code. Use the following pattern instead. Nested broadcast expressions fuse into a single kernel by default when on GPU. Prefer `@.` for multi-op elementwise code so every operator is dotted and the expression stays completely fused. Even just forgetting the `.` on unary negation (i.e., `y .= -a .+ b .* c`) will result in unfused code. Use the following pattern instead. +```julia +y .= @. -a + b * c ```julia y .= @. -a + b * c ``` See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. +See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. + ### Helping the Garbage Collector +Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. +`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. `@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. ```julia @analyze_lifetimes begin result = @. A[1:end, :] + B[1:end, :] C .= @. result * 2.0f0 + result = @. A[1:end, :] + B[1:end, :] + C .= @. result * 2.0f0 end ``` @@ -100,3 +131,11 @@ More worked examples (initialization, Gray-Scott, …) are in the documentation ### Known Limitations - There is no support for `Float16` or `ComplexF16` +- Arrays with 4 or more dimensions might have worse performance +- Maximum array dimension is 6 + + +### Known Deviations from Base Julia +- Reductions return 0D stores instead of scalars +- Slices return views +- `inv` does not throw `SingularException` for singular matrices diff --git a/TODO.md b/TODO.md index 6637e87ff..b1bbe0b09 100644 --- a/TODO.md +++ b/TODO.md @@ -5,4 +5,58 @@ - Support Ints on methods that takes floats - Programatic manipulation of Legate hardware config (not currently possible) - Add Aqua.jl to CI to ensure we didn't pirate any types -- Fix CodeCov reports + +## Base + +Easy `Base` / `AbstractArray` gaps for `NDArray`. Module helpers +(`cuNumeric.reshape` / `transpose` / `unique` / …) often exist, but the +corresponding `Base` methods are missing, so calls fall through to +`AbstractArray` and may scalar-index. Wire up `Base.*` when convenient. + +**P0** +- `Base.reshape`, `Base.vec` +- `fill!` (convertible eltypes) +- `collect` + +**P0 done** +- `iszero` / `isone` (on-device reductions → `Bool` via scalar sync; `isone` square 2D only) + +**P1** +- `transpose` / `adjoint` +- `unique` +- `ones_like` +- `dropdims` + +**P2** +- `extrema`, `mean` +- `count` / nonzero +- `abs2` +- numeric `all` / `any` +- `sum(f, A)` specials + +**P3** +- `copyto!(NDArray, AbstractArray)` +- `floor` / `ceil` / `clamp` +- 2D `permutedims` +- `diff` + +## LinearAlgebra + +Starter list of easy/medium LA gaps. Prefer wiring `LinearAlgebra` entry +points so they do not fall through to scalar-indexing Base paths. + +**BLAS-1 style (dense `NDArray`)** +- `axpy!` / `axpby!` — common scale-and-add; examples use broadcast today +- `scal!` and related in-place scale +- `LinearAlgebra.dot` if not already Base-wired to `nda_dot` + +**Reductions / traces** +- `LinearAlgebra.tr` for dense 2D `NDArray` — `cuNumeric.trace` exists; `tr` is + already wired for `Diagonal{<:NDArray}` + +**Diagonal vs fallthrough (context)** +- Already on-device for `Diagonal`: `mul!` / `lmul!` / `rmul!`, `\` / `/`, + `tr`, `norm` / `opnorm`, many predicates — see `docs/src/linalg.md` +- Still fallthrough / unsupported on `Diagonal` (e.g. `svd`, `pinv`, + `cholesky`, host `AbstractArray` RHS): leave alone unless fixing is cheap; + densify intentionally when needed diff --git a/deps/build.jl b/deps/build.jl index efcc23683..68ac071ce 100644 --- a/deps/build.jl +++ b/deps/build.jl @@ -42,7 +42,7 @@ function build_cpp_wrapper( ) @info "libcunumeric_jl_wrapper: Building C++ Wrapper Library" isdir(install_root) && (rm(install_root; recursive=true); mkdir(install_root)) - bld_command = `$(joinpath(repo_root, "scripts/build_cpp_wrapper.sh")) $repo_root $cupynumeric_loc $legate_loc $blas_loc $install_root 8` + bld_command = `$(joinpath(repo_root, "scripts/build_cpp_wrapper.sh")) $repo_root $cupynumeric_loc $legate_loc $blas_loc $install_root $(Threads.nthreads())` return BuildTools.run_build_wrapper_script( repo_root, bld_command; cuda_root, cuda_enabled, log_dir=@__DIR__ ) diff --git a/docs/src/api_initialization.md b/docs/src/api_initialization.md index bd8ddacc2..fdca35848 100644 --- a/docs/src/api_initialization.md +++ b/docs/src/api_initialization.md @@ -2,49 +2,65 @@ Constructors for new `NDArray`s. Default floating-point type is `Float32`. -## zeros +## Basic Initialization + +### zeros ```@docs cuNumeric.zeros ``` -## ones +### ones ```@docs cuNumeric.ones ``` -## fill +### fill ```@docs cuNumeric.fill ``` -## trues +### trues ```@docs cuNumeric.trues ``` -## falses +### falses ```@docs cuNumeric.falses ``` -## eye +## Special Matrices -```@docs -cuNumeric.eye +### Diagonal + +Construct a `Diagonal` matrix whose elements (diagonal only) are stored in an `NDArray`. +`LinearAlgebra.I` can be used to construct dense identity matrices as well. + +```julia +using LinearAlgebra +using cuNumeric + +D = Diagonal(cuNumeric.ones(Float32, 5)) # preferred for diagonal work +I32 = NDArray{Float32}(I, 5, 5) # dense Float32 identity +Ib = NDArray(I, 5, 5) # Bool identity ``` -## rand +See [Linear Algebra](./linalg.md#diagonal-and-identity) for preferred patterns. + +## Random Numbers + +### rand ```@docs cuNumeric.rand ``` -## rand! +### rand! ```@docs Random.rand!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) diff --git a/docs/src/examples/initialization.md b/docs/src/examples/initialization.md index afd453637..d187828dc 100644 --- a/docs/src/examples/initialization.md +++ b/docs/src/examples/initialization.md @@ -3,6 +3,7 @@ Create `NDArray`s with the usual Julia-style constructors. The default element type is `Float32` unless you pass one. ```julia +using LinearAlgebra using cuNumeric # Zeros / ones / fill @@ -15,9 +16,10 @@ F = cuNumeric.fill(7.5f0, (2, 3)) T = cuNumeric.trues(2, 3) Fbool = cuNumeric.falses(2, 3) -# Identity -I = cuNumeric.eye(5) -I16 = cuNumeric.eye(Float32, 5) +# Identity: prefer Diagonal / I; densify only when needed (no eye) +D = Diagonal(cuNumeric.ones(Float32, 5)) +I32 = NDArray{Float32}(I, 5, 5) # dense identity +Ib = NDArray(I, 5, 5) # Bool identity # Uniform / normal random values (native Float32 and Float64) R = cuNumeric.rand(4, 4) diff --git a/docs/src/examples/special_mat.md b/docs/src/examples/special_mat.md new file mode 100644 index 000000000..56e64ce64 --- /dev/null +++ b/docs/src/examples/special_mat.md @@ -0,0 +1,23 @@ +# Special Matrices + +Currently cuNumeric only supports `LinearAlgebra.Diagonal`. Other special matrix types like `Tridiagonal` and `Symmetric` will follow. + +Diagonal matrices are common, require only storage of the diagonal elements and are often simple to compute operations on (i.e. `LinearAlgebra.inv`). `LinearAlgebra.Diagonal` matrices can be constructed from 1D or 2D `NDArray`s and have certain operations implemented (i.e., `eigen` and `inv`). + + +```julia +using cuNumeric +using LinearAlgebra + +one_dim = cuNumeric.NDArray([1,2,3,4,5]) +two_dim = cuNumeric.rand(5,5) + +D1 = Diagonal(one_dim) +D2 = Diagonal(two_dim) + +evals, evecs = eigen(D1) +D1_inv = inv(D1) + +D1 ./= 2 # stays diagonal +arr = D2 .+ two_dim # densifies because `two_dim` is not guranteed to be diagonal +``` diff --git a/docs/src/linalg.md b/docs/src/linalg.md index 103c2c40b..1052ca4a9 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -81,11 +81,113 @@ supported. These helpers live on `NDArray` and are also listed in the Public API: - `cuNumeric.transpose` -- `cuNumeric.eye` - `cuNumeric.diag` (2D to 1D) - `cuNumeric.trace` +## Diagonal and identity + +Prefer structured `LinearAlgebra` types over materializing a full matrix. + +Wrap a 1D `NDArray` in `Diagonal` for scale / solve / inverse along a diagonal. +Use `LinearAlgebra.I` (`UniformScaling`) for `A + I`, `D + I`, and `A * I`. +`D + I` stays a `Diagonal`; `A + I` returns a dense `NDArray`. + +```julia +using LinearAlgebra +using cuNumeric + +d = cuNumeric.ones(Float32, 64) +D = Diagonal(d) # preferred: keep diagonal structure +A = cuNumeric.rand(Float32, 64, 64) +v = cuNumeric.rand(Float32, 64) + +y = D * v # scale a vector +B = D * A # scale rows +X = D \ A # scale columns by 1 ./ d +Di = D + I # still Diagonal +C = A + I # dense NDArray +``` + +### Supported `Diagonal{<:NDArray}` APIs + +These paths stay on-device (no host densify for the math). RHS / other operands +must be `NDArray` unless noted. + +**Construction / display** + +- `Diagonal(v::NDArray{<:Any,1})` — wrap without copying +- `Diagonal(A::NDArray{<:Any,2})` — `Diagonal(diag(A))` +- `Matrix(D)` / `Matrix{T}(D)` — densify to a host `Matrix` (conversion only) +- `show` — densifies `.diag` for printing only + +**Multiply / divide / inverse** + +- `D * A`, `A * D`, `D * v` for 2D / 1D `NDArray` +- `mul!`, `lmul!`, `rmul!` with `NDArray` +- `D \ B`, `A / D`, `ldiv!`, `rdiv!` with `NDArray` +- `inv(D)` — reciprocal on-device; zeros become Inf (no `SingularException`) +- `det(D)` — 0-dimensional `NDArray` product of the diagonal + +**`NDArray` ± `Diagonal`** + +- `A + D`, `D + A`, `A - D`, `D - A` for square 2D `NDArray` + +**UniformScaling (`I`)** + +- `NDArray{T}(I, m, n)` / `NDArray(I, …)`, `copyto!(A, I)`, `one(A)`, `oneunit(A)` +- `A ± I`, `A * I`, `I * A` +- `D ± I`, `D * I`, `I * D`, `copyto!(D, I)` — `D ± I` stays `Diagonal` + +**Broadcast** + +- Structure- or zero-preserving broadcasts on `Diagonal` (e.g. `D .* c`, + `D .*= c`, `D .+ D`) lower to 1D broadcast on `.diag` +- Densifying out-of-place broadcasts (e.g. `D .+ 1`, `D .+ A`) materialize a + dense `NDArray`, matching Base’s densify-to-`Matrix` behavior +- In-place densifying writes into `Diagonal` (e.g. `D .+= 1`, `D .+= A`) still + throw `ArgumentError` (off-diagonal / densify), matching Base + +**Eigen / reductions / predicates / norms** + +- `eigvals(D)`, `eigen(D)`, `eigvecs(D)` — unsorted; values are a copy of the + diagonal (`NDArray`), vectors are `NDArray` identity. Keyword `sortby` is not + supported on this method. +- `tr`, `sum`, `prod`, `maximum`, `minimum` — 0-dimensional `NDArray` (not a Julia scalar) +- `iszero`, `isone`, `istriu`, `istril`, `ishermitian`, `issymmetric`, `isposdef` — 0-dimensional `NDArray{Bool}` +- `opnorm(D)` / `opnorm(D, p)` for `p ∈ {1, 2, Inf}` — 0-dimensional `NDArray` +- `norm(D)` / `norm(D, p)` for finite `p` (including `±Inf`); off-diagonals are zero — 0-dimensional `NDArray` +- `cond(D)` / `cond(D, p)` for `p ∈ {1, 2, Inf}` — 0-dimensional `NDArray` +- `logdet(D)` for real `Diagonal` only — 0-dimensional `NDArray` + +**Helpers on dense `NDArray`** + +- `cuNumeric.diag` / `LinearAlgebra.diag` (2D → 1D), `cuNumeric.trace` / `LinearAlgebra.tr` (2D square → 0D) +- `iszero(A)` — all elements `== zero(T)` → 0-dimensional `NDArray{Bool}` +- `isone(A)` — square 2D vs `_eye(T, n)` → 0-dimensional `NDArray{Bool}` (non-square → `false`) + +### Unsupported / fallthrough + +Other `LinearAlgebra` operations on `Diagonal{<:NDArray}` (for example `svd`, +`svdvals`, `pinv`, `logabsdet`, complex `logdet`, `kron`, `cholesky`, host +`AbstractArray` RHS for `\` / `/` / `ldiv!` / `rdiv!`, or `eigen(...; sortby)`) +are **not** specially implemented. They fall through to Base and typically fail +with the package’s scalar-indexing error (NDArray does not support scalar +indexing without `@allowscalar`). There are no “not implemented” `ArgumentError` +stubs for these. + +Only densify when you truly need a full identity matrix: + +```julia +E = NDArray{Float32}(I, 64, 64) # dense identity +copyto!(A, I) # fill an existing array +E2 = one(A) # same shape / eltype as A +``` + +Avoid building a dense identity (or densifying `D` with `Matrix(D)`) just to +scale or shift; prefer `Diagonal` and `I` instead. There is no public `eye`. + ## Not available yet -There is no public `cholesky`, `eig`, `lu`, matrix `inv`, or `ldiv!` yet. +There is no public dense-matrix `cholesky`, `eig`, `lu`, matrix `inv`, or +`ldiv!` yet (beyond the `Diagonal` / `NDArray` paths listed above). Elementwise `inv` / `^-1` are unary operations, not matrix inverse. diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index d86f36fdb..e05a6e191 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -168,6 +168,7 @@ const FUSE_BROADCAST_EXPRS = CNPreferences.FUSE_BROADCAST const FUSE_BROADCAST_MIN_OPS = CNPreferences.FUSE_BROADCAST_MIN_OPS # Functionality +include("ndarray/diagonal.jl") include("ndarray/promotion.jl") include("cuda/cuda_ptx_task.jl") include("ndarray/broadcast_fusion.jl") diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 07745f483..5631cfa5f 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -136,8 +136,10 @@ function Base.:(+)(rhs1::NDArray{A,N}, rhs2::NDArray{B,N}) where {A,B,N} return _nda_binary_op_promoted!(out, cuNumeric.ADD, rhs1, rhs2) end -Base.:(*)(val::V, arr::NDArray{A}) where {A,V} = _mul_scalar(__my_promote_type(A, V), val, arr) -Base.:(*)(arr::NDArray{A}, val::V) where {A,V} = val * arr +function Base.:(*)(val::V, arr::NDArray{A}) where {A,V<:Number} + return _mul_scalar(__my_promote_type(A, V), val, arr) +end +Base.:(*)(arr::NDArray{A}, val::V) where {A,V<:Number} = val * arr _mul_scalar(::Type{T}, val, arr::NDArray{T}) where {T} = nda_multiply_scalar(arr, T(val)) function _mul_scalar(::Type{U}, val, arr::NDArray) where {U} @@ -156,14 +158,14 @@ function Base.:(*)(rhs1::NDArray{A,2}, rhs2::NDArray{B,2}) where {A,B} end function Base.:(*)(rhs1::NDArray{Bool,2}, rhs2::NDArray{Bool,2}) - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Boolean arrays") ) end function Base.:(*)(rhs1::NDArray{<:Integer,2}, rhs2::NDArray{<:Integer,2}) #* this is a stupid..... - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Integer arrays") ) end @@ -221,14 +223,14 @@ end function LinearAlgebra.mul!(out::NDArray, rhs1::NDArray{Bool,2}, rhs2::NDArray{Bool,2}) #* Could just promote both inputs to Int32 - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Boolean arrays") ) end function LinearAlgebra.mul!(out::NDArray, rhs1::NDArray{<:Integer,2}, rhs2::NDArray{<:Integer,2}) #* this is a stupid..... - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Integer arrays") ) end diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 2d64c65ff..8df0b0ee2 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -10,7 +10,7 @@ function map_cuda_type(::Type{cuNumeric.NDArrayStyle{N}}) where {N} end # Also can be HostMemory or UnifiedMemory function _nd_forbid_mix() - throw( + return throw( ArgumentError( "Broadcast between NDArray and other array types is not supported. " * "Convert explicitly to a single array type before broadcasting.", @@ -26,6 +26,20 @@ Base.BroadcastStyle(::DefaultArrayStyle{0}, a::NDArrayStyle) = a Base.BroadcastStyle(::NDArrayStyle, ::DefaultArrayStyle) = _nd_forbid_mix() Base.BroadcastStyle(::DefaultArrayStyle, ::NDArrayStyle) = _nd_forbid_mix() +# Like Base Diagonal vs Array: structured Diagonal style wins over dense NDArray +# so D.+A uses StructuredMatrixStyle{Diagonal} (densify to NDArray in diagonal.jl) +# instead of ArrayConflict → host Matrix + scalar indexing. +function Base.BroadcastStyle( + ::LinearAlgebra.StructuredMatrixStyle{<:Diagonal}, ::NDArrayStyle +) + return LinearAlgebra.StructuredMatrixStyle{Diagonal}() +end +function Base.BroadcastStyle( + ::NDArrayStyle, ::LinearAlgebra.StructuredMatrixStyle{<:Diagonal} +) + return LinearAlgebra.StructuredMatrixStyle{Diagonal}() +end + Base.broadcastable(A::NDArray) = A #* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? @@ -37,6 +51,16 @@ Base.similar(arr::NDArray{T}, dims::Base.DimOrInd...) where {T} = similar(arr, T Base.similar(arr::NDArray, ::Type{T}) where {T} = similar(arr, T, size(arr)) #* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? +# Prefer Dims over the axes catch-all: with StaticArrays loaded (GPU CI via CUDA), +# `similar(::Type{<:AbstractArray}, ::Tuple{})` is otherwise ambiguous between +# Base, StaticArrays, and our catch-all (0-d broadcast uses axes `()`). +Base.similar(::Type{NDArray{T}}, dims::Dims{N}) where {T,N} = cuNumeric.zeros(T, dims) +function Base.similar( + ::Type{NDArray{T}}, + shape::Tuple{Union{Integer,Base.OneTo},Vararg{Union{Integer,Base.OneTo}}}, +) where {T} + return cuNumeric.zeros(T, map(Int, Base.to_shape.(shape))) +end Base.similar(::Type{NDArray{T}}, axes) where {T} = cuNumeric.zeros(T, Base.to_shape.(axes)) function Base.similar(bc::Broadcasted{NDArrayStyle{N}}, ::Type{ElType}) where {N,ElType} return similar(NDArray{ElType}, axes(bc)) @@ -55,7 +79,7 @@ end # Get depth of Broadcast tree recursively # Need to call instantiate first -bcast_depth(bc::Base.Broadcast.Broadcasted) = maximum(bcast_depth, bc.args, init=0) + 1; +bcast_depth(bc::Base.Broadcast.Broadcasted) = maximum(bcast_depth, bc.args; init=0) + 1; bcast_depth(::Any) = 0 struct BrokenBroadcast{T} end @@ -63,18 +87,22 @@ Base.convert(::Type{BrokenBroadcast{T}}, x) where {T} = BrokenBroadcast{T}() Base.convert(::Type{BrokenBroadcast{T}}, x::BrokenBroadcast{T}) where {T} = x Base.eltype(::Type{BrokenBroadcast{T}}) where {T} = T +# Use cuNumeric promotion (`__recip_type` for inv, etc.), not Base.combine_eltypes +# — e.g. inv.(Int32) must allocate Float32, not Float64. +@inline function _broadcast_copy_eltype(bc::Broadcasted) + return __checked_promote_op(bc.f, Base.Broadcast.eltypes(bc.args)) +end + function Broadcast.copy(bc::Broadcasted{<:NDArrayStyle{0}}) - ElType = Broadcast.combine_eltypes(bc.f, bc.args) + ElType = _broadcast_copy_eltype(bc) if ElType == Union{} ElType = Nothing end - dest = copyto!(similar(bc, ElType), bc) - #! CHECK THIS DOESNT CAUSE ISSUES DUE TO BLOCKING NATURE - return @allowscalar dest[CartesianIndex()] + return copyto!(similar(bc, ElType), bc) end @inline function Broadcast.copy(bc::Broadcasted{<:NDArrayStyle}) - ElType = Broadcast.combine_eltypes(bc.f, bc.args) + ElType = _broadcast_copy_eltype(bc) if ElType == Union{} || !Base.allocatedinline(ElType) ElType = BrokenBroadcast{ElType} end @@ -162,7 +190,9 @@ end end @inline _copyto_unfused!(dest::NDArray{T}, temp_result::NDArray{T}) where {T} = - _store_broadcast_result!(dest, temp_result) + _store_broadcast_result!( + dest, temp_result + ) @inline function _copyto_unfused!(dest::NDArray{T}, temp_result::NDArray) where {T} promoted = checked_promote_arr(temp_result, T) diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 6611edf11..728deec17 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -270,8 +270,22 @@ Also refuses 0-d destinations: `RunPTXBroadcastTask` only supports dims in end # Same-shaped operands do not need Broadcast's dynamic index projection. +# +# Size-1 dimensions are special: Broadcast marks them `keeps=false` even when the +# leaf shape matches `dest` (e.g. length-1 vectors). Linear `I` is still valid in +# that case because the dimension only has index 1. +@inline function _extruded_ok_for_fusion(x::Base.Broadcast.Extruded) + keeps = x.keeps + for i in eachindex(keeps) + if !keeps[i] && size(x.x, i) != 1 + return false + end + end + return true +end + @inline function _unwrap_fusion_arg(x::Base.Broadcast.Extruded) - if all(x.keeps) + if _extruded_ok_for_fusion(x) return x.x end throw( diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 386850a1c..8b7e67671 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -41,7 +41,7 @@ end get_n_dim(ptr::NDArray_t) = Int(ccall((:nda_array_dim, libnda), Int32, (NDArray_t,), ptr)) -abstract type AbstractNDArray{T<:SUPPORTED_TYPES,N} end +abstract type AbstractNDArray{T<:SUPPORTED_TYPES,N} <: AbstractArray{T,N} end @doc""" The NDArray type represents a multi-dimensional array in cuNumeric. @@ -475,7 +475,7 @@ function nda_trace( (NDArray_t, Int32, Int32, Int32, Legate.LegateTypeAllocated), arr.ptr, offset, a1, a2, legate_type) end - return NDArray(ptr, T, Val(1)) + return NDArray(ptr, T, Val(0)) end # transpose reverses the axes: element type and rank are preserved diff --git a/src/ndarray/diagonal.jl b/src/ndarray/diagonal.jl new file mode 100644 index 000000000..ae1d306ad --- /dev/null +++ b/src/ndarray/diagonal.jl @@ -0,0 +1,557 @@ +###### diag / _eye / trace ###### + +@doc""" + cuNumeric.diag(arr::NDArray; k=0) + +Extract the k-th diagonal from a 2D `NDArray`. +""" +function diag(arr::NDArray; k::Int=0) + return nda_diag(arr, Int32(k)) +end + +LinearAlgebra.diag(arr::NDArray{<:Any,2}, k::Integer=0) = nda_diag(arr, Int32(k)) + +# Internal dense identity used by UniformScaling / Diagonal densify helpers. +# Prefer `LinearAlgebra.I` / `NDArray{T}(I, n, n)` / `one(A)` in user code. +function _eye(::Type{T}, rows::Int) where {T} + return nda_eye(Int32(rows), T) +end +_eye(rows::Int) = _eye(DEFAULT_FLOAT, rows) + +@doc""" + cuNumeric.trace(arr::NDArray; offset=0, a1=0, a2=1) + +Compute the trace (sum of a diagonal) of the `NDArray`. +Returns a 0-dimensional `NDArray`. The accumulator type follows promotions of +other reductions like `sum`. +""" +function trace(arr::NDArray{T,2}; offset::Int=0, a1::Int=0, a2::Int=1) where {T} + LinearAlgebra.checksquare(arr) + T_OUT = Base.promote_op(Base.sum, Vector{T}) + return nda_trace(arr, Int32(offset), Int32(a1), Int32(a2), T_OUT) +end + +function LinearAlgebra.tr(arr::NDArray{<:Any,2}) + return cuNumeric.trace(arr) +end + +###### Diagonal constructors ###### + +const DiagonalNDArray{T} = Diagonal{T,<:NDArray{T,1}} + +function LinearAlgebra.Diagonal(arr::NDArray{T,1}) where {T} + return Diagonal{T,typeof(arr)}(arr) +end + +function LinearAlgebra.Diagonal(arr::NDArray{T,2}) where {T} + return Diagonal(diag(arr)) +end + +# Note: Matrix{T} === Array{T,2}, so do not also define Array{T,2}(...). +# Use Base.zeros — bare `zeros` resolves to cuNumeric.zeros inside this module. +function Base.Matrix{T}(D::DiagonalNDArray) where {T} + dv = Array(D.diag) + n = length(dv) + B = Base.zeros(T, n, n) + @inbounds for i in 1:n + B[i, i] = dv[i] + end + return B +end +Base.Matrix(D::DiagonalNDArray{T}) where {T} = Matrix{T}(D) + +function Base.show(io::IO, D::DiagonalNDArray) + return show(io, Diagonal(Array(D.diag))) +end + +function Base.show(io::IO, ::MIME"text/plain", D::DiagonalNDArray) + # Keep the real Diagonal{T,<:NDArray} in the summary; only densify the + # diagonal vector for Base's ⋅-style body formatting. + summary(io, D) + isempty(D) && return nothing + println(io, ":") + Base.print_array(io, Diagonal(Array(D.diag))) + return nothing +end + +###### Diagonal operators ###### + +@inline _diag_vec(D::DiagonalNDArray) = D.diag +@inline _row_scale(d::NDArray{<:Any,1}) = reshape(d, (length(d), 1)) +@inline _col_scale(d::NDArray{<:Any,1}) = reshape(d, (1, length(d))) + +function Base.:*(D::DiagonalNDArray, A::NDArray{<:Any,2}) + size(A, 1) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return _row_scale(_diag_vec(D)) .* A +end + +function Base.:*(A::NDArray{<:Any,2}, D::DiagonalNDArray) + size(A, 2) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return A .* _col_scale(_diag_vec(D)) +end + +function Base.:*(D::DiagonalNDArray, v::NDArray{<:Any,1}) + length(v) == size(D, 1) || throw( + DimensionMismatch("vector length $(length(v)) does not match diagonal $(size(D,1))") + ) + return _diag_vec(D) .* v +end + +function LinearAlgebra.lmul!(D::DiagonalNDArray, B::NDArray) + return copyto!(B, D * B) +end + +function LinearAlgebra.rmul!(A::NDArray, D::DiagonalNDArray) + return copyto!(A, A * D) +end + +function LinearAlgebra.mul!(C::NDArray, D::DiagonalNDArray, A::NDArray) + return copyto!(C, D * A) +end + +function LinearAlgebra.mul!(C::NDArray, A::NDArray, D::DiagonalNDArray) + return copyto!(C, A * D) +end + +function Base.:\(D::DiagonalNDArray, B::NDArray{<:Any,1}) + length(B) == size(D, 1) || throw( + DimensionMismatch("vector length $(length(B)) does not match diagonal $(size(D,1))") + ) + return B ./ _diag_vec(D) +end + +function Base.:\(D::DiagonalNDArray, B::NDArray{<:Any,2}) + size(B, 1) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(B,1))×$(size(B,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return B ./ _row_scale(_diag_vec(D)) +end + +function Base.:/(A::NDArray{<:Any,2}, D::DiagonalNDArray) + size(A, 2) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return A * inv(D) +end + +function LinearAlgebra.ldiv!(D::DiagonalNDArray, B::NDArray) + return copyto!(B, D \ B) +end + +function LinearAlgebra.rdiv!(A::NDArray, D::DiagonalNDArray) + return copyto!(A, A / D) +end + +function Base.inv(D::DiagonalNDArray{T}) where {T} + # Base Julia checks and throws a SingularException. We cannot do + # that without unwrapping the NDArray to a Julia scalar. + return Diagonal(inv.(_diag_vec(D))) +end + +LinearAlgebra.det(D::DiagonalNDArray) = prod(_diag_vec(D)) +LinearAlgebra.tr(D::DiagonalNDArray{<:Number}) = sum(_diag_vec(D)) +Base.sum(D::DiagonalNDArray) = sum(_diag_vec(D)) + +# Generic `prod(::AbstractMatrix)` walks every entry (scalar-indexing). For n>1 a +# Diagonal has off-diagonal zeros, so the product is zero — match Base, as 0D. +function Base.prod(D::DiagonalNDArray{T}) where {T<:Number} + n = size(D, 1) + n == 0 && return NDArray(one(T)) + n == 1 && return prod(_diag_vec(D)) + return NDArray(zero(T)) +end + +function Base.maximum(D::DiagonalNDArray{T}) where {T<:Number} + maxdiag = maximum(_diag_vec(D)) + size(D, 1) > 1 && return max.(zero(T), maxdiag) + return maxdiag +end + +function Base.minimum(D::DiagonalNDArray{T}) where {T<:Number} + mindiag = minimum(_diag_vec(D)) + size(D, 1) > 1 && return min.(zero(T), mindiag) + return mindiag +end + +Base.iszero(D::DiagonalNDArray) = iszero(_diag_vec(D)) +function Base.isone(D::DiagonalNDArray{T}) where {T} + return all(_diag_vec(D) .== one(T)) +end + +# Base walks `iszero(D.diag)` by scalar iteration; keep on-device via `iszero(D)`. +function LinearAlgebra.istriu(D::DiagonalNDArray, k::Integer=0) + return k <= 0 ? NDArray(true) : iszero(D) +end +function LinearAlgebra.istril(D::DiagonalNDArray, k::Integer=0) + return k >= 0 ? NDArray(true) : iszero(D) +end + +# Real Diagonal is always Hermitian/symmetric in Base; Complex Hermitian needs isreal(diag). +LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Real}) = NDArray(true) +function LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Complex}) + return all(imag(_diag_vec(D)) .== zero(real(eltype(D)))) +end +LinearAlgebra.issymmetric(D::DiagonalNDArray{<:Number}) = NDArray(true) + +# Base `isposdef(D) = all(isposdef, D.diag)` scalar-iterates. +function LinearAlgebra.isposdef(D::DiagonalNDArray{T}) where {T<:Real} + isempty(D) && return NDArray(true) + return all(_diag_vec(D) .> zero(T)) +end +function LinearAlgebra.isposdef(D::DiagonalNDArray{T}) where {T<:Complex} + # isposdef(z) = isreal(z) && real(z) > 0 — keep on-device, no host densify. + d = _diag_vec(D) + return all((imag(d) .== zero(real(T))) .& (real(d) .> zero(real(T)))) +end + +###### Eigen / related ###### + +# Base: eigvals(D::Diagonal{<:Number}) = copy(D.diag). Keep NDArray (package style). +function LinearAlgebra.eigvals(D::DiagonalNDArray{<:Number}; permute::Bool=true, scale::Bool=true) + return copy(_diag_vec(D)) +end + +# Unsorted eigen: values are a copy of the diagonal (NDArray); vectors are NDArray I. +# Keyword `sortby` is not accepted on this override. +function LinearAlgebra.eigen( + D::DiagonalNDArray; + permute::Bool=true, + scale::Bool=true, +) + Td = Base.promote_op(/, eltype(D), eltype(D)) + return Eigen(copy(_diag_vec(D)), _eye(Td, size(D, 1))) +end + +function LinearAlgebra.eigvecs( + D::DiagonalNDArray; + permute::Bool=true, + scale::Bool=true, +) + return eigen(D; permute=permute, scale=scale).vectors +end + +# Real logdet is sum(log.(diag)) on-device. Complex logdet / other Base LinearAlgebra +# ops without overrides fall through and may scalar-index `.diag`. +LinearAlgebra.logdet(D::DiagonalNDArray{<:Real}) = sum(log.(_diag_vec(D))) + +# Operator / entrywise norms from the diagonal only (no host densify). +function LinearAlgebra.opnorm(D::DiagonalNDArray, p::Real=2) + if !(p == 1 || p == 2 || p == Inf) + throw(ArgumentError(lazy"invalid p-norm p=$p. Valid: 1, 2, Inf")) + end + isempty(D) && return NDArray(float(real(zero(eltype(D))))) + return maximum(abs.(_diag_vec(D))) +end + +function LinearAlgebra.norm(D::DiagonalNDArray, p::Real=2) + # Off-diagonals are zero, so the matrix vec-norm equals the diag vec-norm. + d = abs.(_diag_vec(D)) + if p == 2 + return sqrt.(sum(d .^ 2)) + elseif p == 1 + return sum(d) + elseif p == Inf + return isempty(D) ? NDArray(float(real(zero(eltype(D))))) : maximum(d) + elseif p == -Inf + return isempty(D) ? NDArray(float(real(zero(eltype(D))))) : minimum(d) + else + return sum(d .^ p) .^ (one(p) / p) + end +end + +function LinearAlgebra.cond(D::DiagonalNDArray, p::Real=2) + if !(p == 1 || p == 2 || p == Inf) + throw(ArgumentError(lazy"invalid p-norm p=$p. Valid: 1, 2, Inf")) + end + isempty(D) && return NDArray(float(one(real(eltype(D))))) + dabs = abs.(_diag_vec(D)) + return maximum(dabs) ./ minimum(dabs) +end + +function Base.:+(A::NDArray{T,2}, D::DiagonalNDArray) where {T} + size(A, 1) == size(A, 2) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return A + (_row_scale(_diag_vec(D)) .* _eye(eltype(D), size(D, 1))) +end +Base.:+(D::DiagonalNDArray, A::NDArray{<:Any,2}) = A + D + +Base.:-(A::NDArray{<:Any,2}, D::DiagonalNDArray) = A + (-D) +Base.:-(D::DiagonalNDArray, A::NDArray{<:Any,2}) = D + (-A) + +###### UniformScaling (LinearAlgebra.I) ###### + +@inline function _uniformscaling_eye(::Type{R}, n::Integer, λ) where {R} + E = _eye(R, Int(n)) + return isone(λ) ? E : nda_multiply_scalar(E, R(λ)) +end + +function NDArray{T}(J::LinearAlgebra.UniformScaling, dims::Dims{2}) where {T} + A = zeros(T, dims) + copyto!(A, J) + return A +end +function NDArray{T}(J::LinearAlgebra.UniformScaling, m::Integer, n::Integer) where {T} + return NDArray{T}(J, Dims((Int(m), Int(n)))) +end +NDArray(J::LinearAlgebra.UniformScaling{T}, dims::Dims{2}) where {T} = NDArray{T}(J, dims) +function NDArray(J::LinearAlgebra.UniformScaling{T}, m::Integer, n::Integer) where {T} + return NDArray{T}(J, Dims((Int(m), Int(n)))) +end + +function Base.copyto!(A::NDArray{T,2}, J::LinearAlgebra.UniformScaling) where {T} + m, n = size(A) + if iszero(J.λ) + return fill!(A, zero(T)) + elseif m == n + return copyto!(A, _uniformscaling_eye(T, m, J.λ)) + else + fill!(A, zero(T)) + k = min(m, n) + A[1:k, 1:k] = _uniformscaling_eye(T, k, J.λ) + return A + end +end + +function Base.:+(A::NDArray{T,2}, J::LinearAlgebra.UniformScaling) where {T} + LinearAlgebra.checksquare(A) + R = Base.promote_op(+, T, typeof(J.λ)) + return A + _uniformscaling_eye(R, size(A, 1), J.λ) +end +Base.:+(J::LinearAlgebra.UniformScaling, A::NDArray{<:Any,2}) = A + J + +Base.:-(A::NDArray{<:Any,2}, J::LinearAlgebra.UniformScaling) = A + (-J) +function Base.:-(J::LinearAlgebra.UniformScaling, A::NDArray{<:Any,2}) + return (-A) + J +end + +# Scale by λ without promoting the array (A * I must not Bool→Float32 promote). +function Base.:*(A::NDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + return _mul_scalar(T, J.λ, A) +end +function Base.:*(J::LinearAlgebra.UniformScaling, A::NDArray{T}) where {T} + return _mul_scalar(T, J.λ, A) +end + +function Base.one(A::NDArray{T,2}) where {T} + LinearAlgebra.checksquare(A) + return _eye(T, size(A, 1)) +end +function Base.oneunit(A::NDArray{T,2}) where {T} + LinearAlgebra.checksquare(A) + return _eye(T, size(A, 1)) +end + +###### Diagonal ↔ UniformScaling ###### + +# Keep Diagonal structure: D + λI == Diagonal(d .+ λ), not a dense matrix. +function Base.:+(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + R = Base.promote_op(+, T, typeof(J.λ)) + return Diagonal(_diag_vec(D) .+ convert(R, J.λ)) +end +Base.:+(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray) = D + J + +Base.:-(D::DiagonalNDArray, J::LinearAlgebra.UniformScaling) = D + (-J) +function Base.:-(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray{T}) where {T} + R = Base.promote_op(-, typeof(J.λ), T) + return Diagonal(convert(R, J.λ) .- _diag_vec(D)) +end + +function Base.:*(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + return Diagonal(_mul_scalar(T, J.λ, _diag_vec(D))) +end +Base.:*(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray) = D * J + +function Base.copyto!(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + fill!(_diag_vec(D), convert(T, J.λ)) + return D +end + +###### Diagonal broadcast ###### + +# LinearAlgebra's StructuredMatrixStyle{Diagonal} path (structuredbroadcast.jl) +# fills `dest.diag[i]` via `Broadcast._broadcast_getindex(bc, (i,i))`, which +# scalar-indexes the NDArray. CUDA/GPUArrays hit the same Base path and do not +# special-case Diagonal broadcast either. +# +# For structure-preserving / zero-preserving broadcasts we lower to a 1D +# broadcast on `.diag` (NDArrayStyle). Fusion then applies to that vector +# broadcast as usual; there is no separate Diagonal-matrix fusion. +# +# Densifying out-of-place broadcasts (e.g. `D .+ 1`, `D .+ A`) allocate a dense +# NDArray and expand Diagonal leaves to `diag .* I` before the usual NDArray +# `copyto!` path — matching Base, which densifies to Matrix. In-place writes +# into Diagonal that would fill off-diagonals still throw ArgumentError. + +@inline _has_diagonal_ndarray(@nospecialize(_)) = false +@inline _has_diagonal_ndarray(::DiagonalNDArray) = true +@inline function _has_diagonal_ndarray(bc::Broadcast.Broadcasted) + return _has_diagonal_ndarray_args(bc.args) +end +@inline _has_diagonal_ndarray_args(::Tuple{}) = false +@inline function _has_diagonal_ndarray_args(args::Tuple) + return _has_diagonal_ndarray(getfield(args, 1)) || + _has_diagonal_ndarray_args(Base.tail(args)) +end + +# Dense NDArray leaves (not wrapped in Diagonal). Used to avoid scalar-indexing +# when building the in-place densifying off-diagonal ArgumentError. +@inline _has_plain_ndarray(@nospecialize(_)) = false +@inline _has_plain_ndarray(::NDArray) = true +@inline function _has_plain_ndarray(bc::Broadcast.Broadcasted) + return _has_plain_ndarray_args(bc.args) +end +@inline _has_plain_ndarray_args(::Tuple{}) = false +@inline function _has_plain_ndarray_args(args::Tuple) + return _has_plain_ndarray(getfield(args, 1)) || + _has_plain_ndarray_args(Base.tail(args)) +end + +@inline _first_diagonal_ndarray_diag(D::DiagonalNDArray) = D.diag +@inline function _first_diagonal_ndarray_diag(bc::Broadcast.Broadcasted) + return _first_diagonal_ndarray_diag_args(bc.args) +end +@inline function _first_diagonal_ndarray_diag_args(args::Tuple) + a = getfield(args, 1) + return if _has_diagonal_ndarray(a) + _first_diagonal_ndarray_diag(a) + else + _first_diagonal_ndarray_diag_args(Base.tail(args)) + end +end + +# Replace Diagonal leaves with their `.diag` vectors; keep scalars / Refs / etc. +@inline _diag_bc_arg(D::Diagonal) = D.diag +@inline function _diag_bc_arg(bc::Broadcast.Broadcasted) + return Broadcast.broadcasted(bc.f, map(_diag_bc_arg, bc.args)...) +end +@inline _diag_bc_arg(x) = x + +@inline function _diagonal_broadcast_preserves_structure(bc::Broadcast.Broadcasted) + # `+` / `-` on Diagonal+Diagonal are zero-preserving (Base's fzeropreserving), + # not isstructurepreserving. Prefer either so D.+D lowers to `.diag` broadcast. + return LinearAlgebra.isstructurepreserving(bc) || LinearAlgebra.fzeropreserving(bc) +end + +# Materialize Diagonal{NDArray} as a dense matrix (same pattern as A ± D). +@inline function _densify_diagonal_ndarray(D::DiagonalNDArray) + d = _diag_vec(D) + n = length(d) + # nda_eye(1) currently yields an unreadable store; build 1×1 from the diag. + n <= 1 && return reshape(copy(d), (n, n)) + return _row_scale(d) .* _eye(eltype(D), n) +end + +# For densifying Structured → dense NDArray copyto!: expand Diagonal leaves. +@inline _expand_diagonal_ndarray_bc_arg(D::DiagonalNDArray) = _densify_diagonal_ndarray(D) +@inline function _expand_diagonal_ndarray_bc_arg(bc::Broadcast.Broadcasted) + return Broadcast.broadcasted(bc.f, map(_expand_diagonal_ndarray_bc_arg, bc.args)...) +end +@inline _expand_diagonal_ndarray_bc_arg(x) = x + +# Off-diagonal Diagonal getindex uses `diagzero` (no `.diag` read), so evaluating +# an off-diagonal broadcast index is safe without `@allowscalar` when every +# non-Diagonal leaf is a scalar. Dense NDArray leaves must not be indexed. +# Used for in-place densifying rejection only (out-of-place densifies). +function _throw_densifying_diagonal_broadcast(bc::Broadcast.Broadcasted) + axs = axes(bc) + if length(axs) >= 2 && length(axs[1]) >= 2 && length(axs[2]) >= 2 + if !_has_plain_ndarray(bc) + v = @inbounds Broadcast._broadcast_getindex(bc, CartesianIndex(2, 1)) + throw( + ArgumentError( + "cannot set off-diagonal entry (2, 1) to a nonzero value ($v)" + ), + ) + end + throw( + ArgumentError( + "cannot set off-diagonal entry (2, 1) to a nonzero value; " * + "broadcast over Diagonal with NDArray diagonal would densify", + ), + ) + end + return throw( + ArgumentError( + "broadcast over Diagonal with NDArray diagonal is not structure-preserving " * + "and would densify; in-place densifying broadcast is not supported", + ), + ) +end + +function Base.similar( + bc::Broadcast.Broadcasted{LinearAlgebra.StructuredMatrixStyle{Diagonal}}, + ::Type{ElType}, +) where {ElType} + inds = axes(bc) + n = length(inds[1]) + if _has_diagonal_ndarray(bc) + if _diagonal_broadcast_preserves_structure(bc) + d = _first_diagonal_ndarray_diag(bc) + return Diagonal(similar(d, ElType, (n,))) + end + # Match Base: densify out-of-place to a dense array (NDArray, not Matrix). + return similar(NDArray{ElType}, inds) + elseif _diagonal_broadcast_preserves_structure(bc) + return LinearAlgebra.structured_broadcast_alloc(bc, Diagonal, ElType, n) + else + return similar( + convert(Broadcast.Broadcasted{Broadcast.DefaultArrayStyle{ndims(bc)}}, bc), + ElType, + ) + end +end + +@inline function _copyto_diagonal_ndarray!( + dest::DiagonalNDArray, bc::Broadcast.Broadcasted +) + axes(bc) == axes(dest) || Broadcast.throwdm(axes(bc), axes(dest)) + # Lower to NDArrayStyle vector broadcast so `_copyto!` / fusion apply. + copyto!(_diag_vec(dest), Broadcast.instantiate(_diag_bc_arg(bc))) + return dest +end + +# Out-of-place densify: destination is dense NDArray from `similar` above. +function Base.copyto!( + dest::NDArray, + bc::Broadcast.Broadcasted{LinearAlgebra.StructuredMatrixStyle{Diagonal}}, +) + axes(dest) == axes(bc) || Broadcast.throwdm(axes(dest), axes(bc)) + isempty(dest) && return dest + expanded = Broadcast.instantiate(_expand_diagonal_ndarray_bc_arg(bc)) + return _copyto!(dest, expanded) +end + +function Base.copyto!( + dest::DiagonalNDArray, + bc::Broadcast.Broadcasted{<:LinearAlgebra.StructuredMatrixStyle}, +) + if !LinearAlgebra.isvalidstructbc(dest, bc) + # 1×1: Base's generic path only writes the diagonal; no off-diagonals to reject. + size(dest, 1) <= 1 || return _throw_densifying_diagonal_broadcast(bc) + end + return _copyto_diagonal_ndarray!(dest, bc) +end + +# Safety net if a densifying Structured broadcast is converted to Nothing +# (Base's `isvalidstructbc` fallback) before reaching the method above. +function Base.copyto!(dest::DiagonalNDArray, bc::Broadcast.Broadcasted{Nothing}) + if !_diagonal_broadcast_preserves_structure(bc) + size(dest, 1) <= 1 || return _throw_densifying_diagonal_broadcast(bc) + end + return _copyto_diagonal_ndarray!(dest, bc) +end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 8a7f9afae..3f531b40d 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -20,6 +20,8 @@ export unwrap +# See TODO.md (Base / LinearAlgebra sections) for AbstractArray and LA gaps. + @doc""" cuNumeric.transpose(arr::NDArray) @@ -29,39 +31,6 @@ function transpose(arr::NDArray) return nda_transpose(arr) end -@doc""" - cuNumeric.eye([T=Float32,] rows::Int) - -Create a 2D identity `NDArray` of size `rows × rows` with element type `T`. -The default type is Float32 if not specified. -""" -function eye(::Type{T}, rows::Int) where {T} - return nda_eye(Int32(rows), T) -end -function eye(rows::Int) - return eye(DEFAULT_FLOAT, rows) -end - -@doc""" - cuNumeric.trace(arr::NDArray; offset=0, a1=0, a2=1) - -Compute the trace (sum of a diagonal) of the `NDArray`. -The accumulator type follows promotions of other reductions like 'sum'. -""" -function trace(arr::NDArray{T}; offset::Int=0, a1::Int=0, a2::Int=1) where {T} - T_OUT = Base.promote_op(Base.sum, Vector{T}) - return nda_trace(arr, Int32(offset), Int32(a1), Int32(a2), T_OUT) -end - -@doc""" - cuNumeric.diag(arr::NDArray; k=0) - -Extract the k-th diagonal from a 2D `NDArray`. -""" -function diag(arr::NDArray; k::Int=0) - return nda_diag(arr, Int32(k)) -end - @doc""" cuNumeric.ravel(arr::NDArray) @@ -111,7 +80,10 @@ copyto!(a, b); a[1,1] ``` """ -Base.copyto!(arr::NDArray{T,N}, other::NDArray{T,N}) where {T,N} = nda_assign(arr, other) +@inline function Base.copyto!(arr::NDArray{T,N}, other::NDArray{T,N}) where {T,N} + nda_assign(arr, other) + return arr +end @doc""" as_type(arr::NDArray, t::Type{T}) where {T} @@ -143,16 +115,24 @@ end # get_ptr is a blocking call that grabs the physical store # we have not tested across multiple processes or devices yet -function (::Type{<:Array{A}})(arr::NDArray{B,0}) where {A,B} - out = Array{A}(undef) +# NDArray-specific overrides of Core's AbstractArray constructors (NDArray <: +# AbstractArray): exact `Array{T}` / `Array{T,N}` / `Array` signatures so we win +# over `Array{T,N}(::AbstractArray)` (which would scalar-index). Bulk path uses +# `_copy_to_julia_array`; 1-d dispatches same-type (zero-copy) vs convert. +function (::Type{Array{T}})(arr::NDArray{S,0}) where {T,S} + out = Array{T,0}(undef) allowscalar() do - return out[] = convert(A, arr[]) + return out[] = convert(T, arr[]) end return out end -function (::Type{<:Array{A}})(arr::NDArray{B,1}) where {A,B} - return make_array(A, Ptr{A}(get_ptr(arr)), size(arr)) +function (::Type{Array{T}})(arr::NDArray{T,1}) where {T} + return make_array(T, Ptr{T}(get_ptr(arr)), size(arr)) +end + +function (::Type{Array{T}})(arr::NDArray{S,1}) where {T,S} + return T.(make_array(S, Ptr{S}(get_ptr(arr)), size(arr))) end # Copy logically into Julia's column-major storage. @@ -168,14 +148,14 @@ function _copy_to_julia_array(arr::NDArray{T,N}) where {T,N} return out end -function (::Type{<:Array{A}})(arr::NDArray{B}) where {A,B} +function (::Type{Array{T}})(arr::NDArray{S,N}) where {T,S,N} out = _copy_to_julia_array(arr) - return A === B ? out : copyto!(Array{A}(undef, size(arr)), out) + return T === S ? out : copyto!(Array{T}(undef, size(arr)), out) end -function (::Type{<:Array})(arr::NDArray{B}) where {B} - return Array{B}(arr) -end +(::Type{Array{T,N}})(arr::NDArray{S,N}) where {T,S,N} = Array{T}(arr) + +(::Type{Array})(arr::NDArray{T,N}) where {T,N} = Array{T}(arr) # conversion from Base Julia array to NDArray # Julia Arrays are column-major; Legate stores are row-major. For N>=2 we @@ -243,7 +223,7 @@ dim(::NDArray{T,N}) where {T,N} = N::Int Base.ndims(::NDArray{T,N}) where {T,N} = N::Int @doc""" Base.size(arr::NDArray) - Base.size(arr::NDArray, dim::Int) + Base.size(arr::NDArray, dim::Integer) Return the size of the given `NDArray`. @@ -258,12 +238,13 @@ size(arr, 2) ``` """ Base.size(arr::NDArray{<:Any,N}) where {N} = cuNumeric.shape(arr) -Base.size(arr::NDArray, dim::Int) = Base.size(arr)[dim] +Base.size(arr::NDArray, dim::Integer) = dim <= ndims(arr) ? size(arr)[dim] : 1 Base.isempty(arr::NDArray) = any(==(0), size(arr)) +Base.length(arr::NDArray) = prod(size(arr)) @doc""" - Base.firstindex(arr::NDArray, dim::Int) - Base.lastindex(arr::NDArray, dim::Int) + Base.firstindex(arr::NDArray, dim::Integer) + Base.lastindex(arr::NDArray, dim::Integer) Base.lastindex(arr::NDArray) Provide the first and last valid indices along a given dimension `dim` for `NDArray`. @@ -276,44 +257,33 @@ lastindex(arr, 2) lastindex(arr) ``` """ -Base.firstindex(arr::NDArray, dim::Int) = 1 -Base.lastindex(arr::NDArray, dim::Int) = Base.size(arr, dim) -Base.lastindex(arr::NDArray) = Base.size(arr, 1) +Base.firstindex(arr::NDArray, dim::Integer) = 1 +Base.lastindex(arr::NDArray, dim::Integer) = size(arr, dim) +Base.lastindex(arr::NDArray) = length(arr) +Base.IndexStyle(::Type{<:NDArray}) = IndexCartesian() Base.axes(arr::NDArray) = Base.OneTo.(size(arr)) Base.view(arr::NDArray, inds...) = arr[inds...] # NDArray slices are views by default. -Base.IndexStyle(::NDArray) = IndexCartesian() - function Base.show(io::IO, arr::NDArray{T,0}) where {T} - allowscalar() do - return print(io, "NDArray{$(T),0}(", repr(arr[]), ")") - end + print(io, summary(arr), "(") + @allowscalar show(io, arr[]) + return print(io, ")") end -function Base.show(io::IO, ::MIME"text/plain", arr::NDArray{T,0}) where {T} - println(io, "0-dimensional NDArray{$(T),0}") - allowscalar() do - return print(io, arr[]) - end +# Used by print(arr), println(arr), and nested displays +function Base.show(io::IO, arr::NDArray) + return show(io, Array(arr)) end -function Base.show(io::IO, arr::NDArray{T,N}) where {T,N} - return print(io, "NDArray{$(T),$(N)} with size ", size(arr)) -end +# Used for full REPL display +function Base.show(io::IO, ::MIME"text/plain", arr::NDArray) + summary(io, arr) -function Base.show(io::IO, ::MIME"text/plain", arr::NDArray{T,N}) where {T,N} - println(io, "NDArray{$(T),$(N)} with size ", size(arr)) - return Base.print_array(io, Array(arr)) -end - -function Base.print(arr::NDArray{T}) where {T} - return Base.show(stdout, arr) -end + isempty(arr) && return nothing -function Base.println(arr::NDArray{T}) where {T} - Base.show(stdout, arr) - return print("\n") + println(io, ":") + return Base.print_array(io, Array(arr)) end #### ARRAY INDEXING AND SLICES #### @@ -332,7 +302,7 @@ end Overloads `Base.getindex` and `Base.setindex!` to support multidimensional indexing and slicing on `cuNumeric.NDArray`s. -Slicing supports combinations of `Int`, `UnitRange`, and `Colon()` for selecting ranges of rows and columns. +Slicing supports combinations of `Integer`, `UnitRange`, and `Colon()` for selecting ranges of rows and columns. The use of all colons (`arr[:]`, `arr[:, :]`, etc.) returns a new Julia `Array` containing a copy of the data. Assignment also supports: @@ -349,10 +319,13 @@ Array(A) ``` """ ##### REGULAR ARRAY INDEXING #### -function Base.getindex(arr::NDArray{T,N}, idxs::Vararg{Int,N}) where {T<:SUPPORTED_NUMERIC_TYPES,N} +@inline function Base.getindex( + arr::NDArray{T,N}, idxs::Vararg{Integer,N} +) where {T<:SUPPORTED_NUMERIC_TYPES,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("getindex") acc = NDArrayAccessor{T,N}() - return read(acc, arr.ptr, to_cpp_index(idxs)) + return read(acc, arr.ptr, to_cpp_index(Int.(idxs))) end function Base.getindex(arr::NDArray{T,0}) where {T<:SUPPORTED_NUMERIC_TYPES} @@ -362,10 +335,11 @@ function Base.getindex(arr::NDArray{T,0}) where {T<:SUPPORTED_NUMERIC_TYPES} return read(acc, arr.ptr, zero_index) end -function Base.getindex(arr::NDArray{Bool,N}, idxs::Vararg{Int,N}) where {N} +@inline function Base.getindex(arr::NDArray{Bool,N}, idxs::Vararg{Integer,N}) where {N} + @boundscheck checkbounds(arr, idxs...) assertscalar("getindex") acc = NDArrayAccessor{CxxWrap.CxxBool,N}() - return read(acc, arr.ptr, to_cpp_index(idxs)) + return read(acc, arr.ptr, to_cpp_index(Int.(idxs))) end function Base.getindex(arr::NDArray{Bool,0}) @@ -376,17 +350,26 @@ function Base.getindex(arr::NDArray{Bool,0}) end #! TODO SUPPORT CONVERSION OF VALUES -function Base.setindex!(arr::NDArray{T,N}, value::T, idxs::Vararg{Int,N}) where {T,N} +@inline function Base.setindex!( + arr::NDArray{T,N}, value::T, idxs::Vararg{Integer,N} +) where {T,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("setindex!") return _setindex!(Val{N}(), arr, value, idxs...) end -function Base.setindex!(arr::NDArray{Complex{T},N}, value::T, idxs::Vararg{Int,N}) where {T,N} +@inline function Base.setindex!( + arr::NDArray{Complex{T},N}, value::T, idxs::Vararg{Integer,N} +) where {T,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("setindex!") return _setindex!(Val{N}(), arr, Complex{T}(value), idxs...) end -function Base.setindex!(arr::NDArray{T,N}, value, idxs::Vararg{Int,N}) where {T,N} +@inline function Base.setindex!( + arr::NDArray{T,N}, value, idxs::Vararg{Integer,N} +) where {T,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("setindex!") return _setindex!(Val{N}(), arr, convert(T, value), idxs...) end @@ -402,15 +385,17 @@ function _setindex!(::Val{0}, arr::NDArray{Bool,0}, value::Bool) end function _setindex!( - ::Val{N}, arr::NDArray{T,N}, value::T, idxs::Vararg{Int,N} + ::Val{N}, arr::NDArray{T,N}, value::T, idxs::Vararg{Integer,N} ) where {T<:SUPPORTED_NUMERIC_TYPES,N} acc = NDArrayAccessor{T,N}() - return write(acc, arr.ptr, to_cpp_index(idxs), value) + return write(acc, arr.ptr, to_cpp_index(Int.(idxs)), value) end -function _setindex!(::Val{N}, arr::NDArray{Bool,N}, value::Bool, idxs::Vararg{Int,N}) where {N} +function _setindex!( + ::Val{N}, arr::NDArray{Bool,N}, value::Bool, idxs::Vararg{Integer,N} +) where {N} acc = NDArrayAccessor{CxxWrap.CxxBool,N}() - return write(acc, arr.ptr, to_cpp_index(idxs), value) + return write(acc, arr.ptr, to_cpp_index(Int.(idxs)), value) end #### START OF SLICING #### @@ -424,98 +409,183 @@ function _setindex_slice!(lhs::NDArray, rhs::NDArray, slices) return nothing end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Colon, j::Int64) - return _setindex_slice!(lhs, rhs, slice_array((0, Base.size(lhs, 1)), (j-1, j))) +@inline _zero_based_index(i::Integer) = (Int(i) - 1, Int(i)) +@inline _zero_based_range(i::AbstractUnitRange{<:Integer}) = (Int(first(i)) - 1, Int(last(i))) + +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, ::Colon, j::Integer +) where {T} + @boundscheck checkbounds(lhs, :, j) + return _setindex_slice!( + lhs, rhs, slice_array((0, size(lhs, 1)), _zero_based_index(j)) + ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Int64, j::Colon) - return _setindex_slice!(lhs, rhs, slice_array((i-1, i))) +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, i::Integer, ::Colon +) where {T} + @boundscheck checkbounds(lhs, i, :) + return _setindex_slice!(lhs, rhs, slice_array(_zero_based_index(i))) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::UnitRange, j::Colon) +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, i::AbstractUnitRange{<:Integer}, ::Colon +) where {T} + @boundscheck checkbounds(lhs, i, :) return _setindex_slice!( - lhs, rhs, slice_array((first(i) - 1, last(i)), (0, Base.size(lhs, 2))) + lhs, rhs, slice_array(_zero_based_range(i), (0, size(lhs, 2))) ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Colon, j::UnitRange) +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, ::Colon, j::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(lhs, :, j) return _setindex_slice!( - lhs, rhs, slice_array((0, Base.size(lhs, 1)), (first(j) - 1, last(j))) + lhs, rhs, slice_array((0, size(lhs, 1)), _zero_based_range(j)) ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::UnitRange, j::Int64) - return _setindex_slice!(lhs, rhs, slice_array((first(i) - 1, last(i)), (j-1, j))) +@inline function Base.setindex!( + lhs::NDArray{T,2}, + rhs::NDArray, + i::AbstractUnitRange{<:Integer}, + j::Integer, +) where {T} + @boundscheck checkbounds(lhs, i, j) + return _setindex_slice!( + lhs, rhs, slice_array(_zero_based_range(i), _zero_based_index(j)) + ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Int64, j::UnitRange) - return _setindex_slice!(lhs, rhs, slice_array((i-1, i), (first(j) - 1, last(j)))) +@inline function Base.setindex!( + lhs::NDArray{T,2}, + rhs::NDArray, + i::Integer, + j::AbstractUnitRange{<:Integer}, +) where {T} + @boundscheck checkbounds(lhs, i, j) + return _setindex_slice!( + lhs, rhs, slice_array(_zero_based_index(i), _zero_based_range(j)) + ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::UnitRange, j::UnitRange) +@inline function Base.setindex!( + lhs::NDArray{T,2}, + rhs::NDArray, + i::AbstractUnitRange{<:Integer}, + j::AbstractUnitRange{<:Integer}, +) where {T} + @boundscheck checkbounds(lhs, i, j) return _setindex_slice!( - lhs, rhs, slice_array((first(i) - 1, last(i)), (first(j) - 1, last(j))) + lhs, rhs, slice_array(_zero_based_range(i), _zero_based_range(j)) ) end -function Base.getindex(arr::NDArray, i::Colon, j::Int64) - return nda_get_slice(arr, slice_array((0, Base.size(arr, 1)), (j-1, j))) +@inline function Base.getindex(arr::NDArray{T,2}, ::Colon, j::Integer) where {T} + @boundscheck checkbounds(arr, :, j) + return nda_get_slice( + arr, slice_array((0, size(arr, 1)), _zero_based_index(j)) + ) end -function Base.getindex(arr::NDArray, i::Int64, j::Colon) - return nda_get_slice(arr, slice_array((i-1, i))) +@inline function Base.getindex(arr::NDArray{T,2}, i::Integer, ::Colon) where {T} + @boundscheck checkbounds(arr, i, :) + return nda_get_slice(arr, slice_array(_zero_based_index(i))) end -function Base.getindex(arr::NDArray, i::UnitRange, j::Colon) +@inline function Base.getindex( + arr::NDArray{T,2}, i::AbstractUnitRange{<:Integer}, ::Colon +) where {T} + @boundscheck checkbounds(arr, i, :) return nda_get_slice( - arr, slice_array((first(i) - 1, last(i)), (0, Base.size(arr, 2))) + arr, slice_array(_zero_based_range(i), (0, size(arr, 2))) ) end -function Base.getindex(arr::NDArray, i::Colon, j::UnitRange) +@inline function Base.getindex( + arr::NDArray{T,2}, ::Colon, j::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(arr, :, j) return nda_get_slice( - arr, slice_array((0, Base.size(arr, 1)), (first(j) - 1, last(j))) + arr, slice_array((0, size(arr, 1)), _zero_based_range(j)) ) end -function Base.getindex(arr::NDArray, i::UnitRange, j::Int64) - return nda_get_slice(arr, slice_array((first(i) - 1, last(i)), (j-1, j))) -end - -function Base.getindex(arr::NDArray, i::Int64, j::UnitRange) - return nda_get_slice(arr, slice_array((i-1, i), (first(j) - 1, last(j)))) +@inline function Base.getindex( + arr::NDArray{T,2}, i::AbstractUnitRange{<:Integer}, j::Integer +) where {T} + @boundscheck checkbounds(arr, i, j) + return nda_get_slice( + arr, slice_array(_zero_based_range(i), _zero_based_index(j)) + ) end -function Base.getindex(arr::NDArray, i::UnitRange, j::UnitRange) +@inline function Base.getindex( + arr::NDArray{T,2}, i::Integer, j::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(arr, i, j) return nda_get_slice( - arr, slice_array((first(i) - 1, last(i)), (first(j) - 1, last(j))) + arr, slice_array(_zero_based_index(i), _zero_based_range(j)) ) end -function Base.getindex(arr::NDArray, i::UnitRange) +@inline function Base.getindex( + arr::NDArray{T,2}, + i::AbstractUnitRange{<:Integer}, + j::AbstractUnitRange{<:Integer}, +) where {T} + @boundscheck checkbounds(arr, i, j) return nda_get_slice( - arr, slice_array((first(i) - 1, last(i))) + arr, slice_array(_zero_based_range(i), _zero_based_range(j)) ) end -Base.getindex(arr::NDArray{T}, c::Vararg{Colon,N}) where {T,N} = Base.copy(arr) -function Base.setindex!(arr::NDArray{T}, rhs::NDArray{T}, c::Vararg{Colon,N}) where {T,N} +@inline function Base.getindex( + arr::NDArray, i::AbstractUnitRange{<:Integer} +) + @boundscheck checkbounds(arr, i) + return nda_get_slice(arr, slice_array(_zero_based_range(i))) +end + +@inline function Base.getindex( + arr::NDArray{T}, c::Vararg{Colon,N} +) where {T,N} + @boundscheck checkbounds(arr, c...) + return Base.copy(arr) +end + +@inline function Base.setindex!( + arr::NDArray{T}, rhs::NDArray{T}, c::Vararg{Colon,N} +) where {T,N} + @boundscheck checkbounds(arr, c...) return Base.copyto!(arr, rhs) end -function Base.setindex!(arr::NDArray{T,2}, val::T, i::Colon, j::Int64) where {T} - s = nda_get_slice(arr, slice_array((0, Base.size(arr, 1)), (j-1, j))) +@inline function Base.setindex!( + arr::NDArray{T,2}, val::T, ::Colon, j::Integer +) where {T} + @boundscheck checkbounds(arr, :, j) + s = nda_get_slice( + arr, slice_array((0, size(arr, 1)), _zero_based_index(j)) + ) nda_fill_array(s, val) return destroy!(s) end -function Base.setindex!(arr::NDArray{T,2}, val::T, i::Int64, j::Colon) where {T} - s = nda_get_slice(arr, slice_array((i-1, i))) +@inline function Base.setindex!( + arr::NDArray{T,2}, val::T, i::Integer, ::Colon +) where {T} + @boundscheck checkbounds(arr, i, :) + s = nda_get_slice(arr, slice_array(_zero_based_index(i))) nda_fill_array(s, val) return destroy!(s) end -Base.fill!(arr::NDArray{T}, val::T) where {T} = nda_fill_array(arr, val) +@inline function Base.fill!(arr::NDArray{T}, val::T) where {T} + nda_fill_array(arr, val) + return arr +end #### INITIALIZATION OF NDARRAYS #### @doc""" @@ -692,8 +762,16 @@ function reshape(arr::NDArray, i::Int...; copy::Val{C}=Val(false)) where {C} end # Ignore the scalar indexing here... -unwrap(x::NDArray{<:Any,0}) = @allowscalar x[] -unwrap(x::NDArray{<:Any,1}) = @allowscalar x[][1] # assumes 1 element +Base.only(x::NDArray{T,0}) where {T} = @allowscalar x[] + +function Base.only(x::NDArray{T,N}) where {T,N} + length(x) == 1 || + throw(ArgumentError("collection must contain exactly 1 element")) + + return @allowscalar x[firstindex(x)] +end + +unwrap(x::NDArray) = only(x) @doc""" ==(arr1::NDArray, arr2::NDArray) diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 0c0df9cac..743ccbb64 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -349,6 +349,17 @@ function Base.any(input::NDArray{Bool}; dims=Colon()) return _bool_reduction_impl(cuNumeric.ANY, input, dims) end +# Compare on-device against `zero(T)` / `_eye(T, n)` (identity filled with `one(T)`). +# Returns a 0D `NDArray{Bool}` — not a Julia `Bool`. +function Base.iszero(A::NDArray{T}) where {T} + return all(A .== zero(T)) +end +function Base.isone(A::NDArray{T,2}) where {T} + m, n = size(A) + m != n && return NDArray(false) # LinearAlgebra.isone: only square matrices + return all(A .== _eye(T, m)) +end + # Boolean multiplication is logical conjunction. cuPyNumeric's PROD reduction # uses a numeric fill identity, which Legate rejects for a Boolean target. function Base.prod(input::NDArray{Bool}; dims=Colon()) diff --git a/src/warnings.jl b/src/warnings.jl index db0ae37f4..c2cfc99b7 100644 --- a/src/warnings.jl +++ b/src/warnings.jl @@ -13,7 +13,7 @@ function repl_frontend_task() if !isassigned(_repl_frontend_task) _repl_frontend_task[] = get_repl_frontend_task() end - _repl_frontend_task[] + return _repl_frontend_task[] end @noinline function get_repl_frontend_task() if isdefined(Base, :active_repl) @@ -69,7 +69,7 @@ function assertscalar(op::String) return nothing end - _assertscalar(op, behavior) + return _assertscalar(op, behavior) end """ @@ -94,7 +94,7 @@ function assertpromotion(op, ::Type{FROM}, ::Type{TO}) where {FROM,TO} return nothing end - _assertpromotion(op, behavior, FROM, TO) + return _assertpromotion(op, behavior, FROM, TO) end @noinline function _assertscalar(op, behavior) @@ -119,26 +119,193 @@ end return nothing end +const _CUNUMERIC_MODULE = @__MODULE__ + +# Sentinel for stack frames with no recoverable module. Prefer this over `nothing` +# so `_module_of_stackframe` is type-stable as `Module`. Must not match Base / +# LinearAlgebra / cuNumeric checks below. +const _UNKNOWN_STACK_MODULE = Module(:__cuNumeric_unknown_stack_module__, false, false) + +@inline function _is_cunumeric_module(m::Module) + m === _UNKNOWN_STACK_MODULE && return false + m === _CUNUMERIC_MODULE && return true + pm = parentmodule(m) + while pm !== m + pm === _CUNUMERIC_MODULE && return true + m = pm + pm = parentmodule(m) + end + return false +end + +@inline function _is_linalg_module(m::Module) + m === _UNKNOWN_STACK_MODULE && return false + m === LinearAlgebra && return true + nameof(m) === :LinearAlgebra && return true + pm = parentmodule(m) + while pm !== m + (pm === LinearAlgebra || nameof(pm) === :LinearAlgebra) && return true + m = pm + pm = parentmodule(m) + end + return false +end + +# Note: LinearAlgebra (and other stdlibs) often have parentmodule === Base, so callers +# must check `_is_linalg_module` before treating a frame as Base. +@inline function _is_base_module(m::Module) + m === _UNKNOWN_STACK_MODULE && return false + m === Base && return true + nameof(m) === :Base && return true + pm = parentmodule(m) + while pm !== m + (pm === Base || nameof(pm) === :Base) && return true + m = pm + pm = parentmodule(m) + end + return false +end + +_module_from_def(def::Method) = def.module +_module_from_def(def::Module) = def +_module_from_def(_) = _UNKNOWN_STACK_MODULE + +_module_of_linfo(linfo::Core.MethodInstance) = _module_from_def(linfo.def) +_module_of_linfo(linfo::Method) = linfo.module +# Julia 1.12+ often stores a CodeInstance on stack frames; unwrap to MethodInstance. +_module_of_linfo(linfo::Core.CodeInstance) = _module_of_linfo(linfo.def) +_module_of_linfo(_) = _UNKNOWN_STACK_MODULE + +# When `frame.linfo` is missing (common for inlined frames), recover module from +# the source path so LinearAlgebra callers are not skipped and the walk does not +# fall through to loader frames like `Base.include_string`. +function _module_from_file(file) + (file === nothing || file === :none) && return _UNKNOWN_STACK_MODULE + f = string(file) + # Match stdlib path segments; avoid false positives on user paths when possible. + if occursin(r"(?:^|[/\\])LinearAlgebra(?:[/\\]|$)", f) + return LinearAlgebra + end + if occursin(r"(?:^|[/\\])cuNumeric(?:\.jl)?(?:[/\\]|$)", f) + return _CUNUMERIC_MODULE + end + # Base frames commonly appear as `./abstractarray.jl`, `./set.jl`, etc. + if startswith(f, "./") || occursin(r"(?:^|[/\\])[Bb]ase(?:[/\\]|$)", f) + return Base + end + return _UNKNOWN_STACK_MODULE +end + +function _module_of_stackframe(frame::Base.StackTraces.StackFrame) + m = _module_of_linfo(frame.linfo) + m !== _UNKNOWN_STACK_MODULE && return m + return _module_from_file(frame.file) +end + +# Keyword bodies often look like `#cholesky!#272`; surface `cholesky!`. +function _clean_stack_func_name(fname) + fname_sym = ifelse(fname isa Symbol, fname, Symbol(string(fname))) + s = string(fname_sym) + m = match(r"^#([^#]+)#\d+$", s) + return m === nothing ? fname_sym : Symbol(m.captures[1]) +end + +# Frames that are never the user-facing "triggering" API for enrichment: +# - loaders / client entry (`include_string` / Julia 1.12 `IncludeInto` from +# `include`ing tests, etc.) +# - keyword-call wrappers (`kwcall`) that would otherwise outrank cholesky/svd +# - AbstractArray iteration/indexing plumbing between NDArray getindex and the +# real stdlib caller (e.g. LinearAlgebra.cholesky / Base.unique) +const _SKIP_STACK_FUNCS = Set{Symbol}(( + :include_string, + :include, + :include_relative, + :_include, + :IncludeInto, # Julia 1.12+ callable include wrapper (Base.IncludeInto) + :eval, + :exec_options, + :_start, + :invokelatest, + :error, + :stacktrace, + :kwcall, # Base keyword-call wrapper; do not steal blame from cholesky/svd + :iterate, + :getindex, + :setindex!, + :indexed_iterate, + Symbol("macro expansion"), + Symbol("top-level scope"), +)) + +""" +Best-effort: outermost Base or LinearAlgebra frame above cuNumeric scalar-index +frames. Walk innermost-first, skip cuNumeric/Core (and frames with unknown +module), indexing/iteration plumbing, keyword-call wrappers (`kwcall`), and +Base loader frames (`include_string`, `IncludeInto` on Julia 1.12+, etc.). +Keep updating the candidate while still in Base/LinearAlgebra (last one wins) +so attribution names the user-facing API (`LinearAlgebra.svd`) rather than an +inner helper (`Base.lt`) or a loader (`Base.include_string`). Stop at the first +user/other-package frame and return that candidate, or `nothing` for the plain +message (e.g. user `Main`). Check LinearAlgebra before Base — stdlibs often +parent to Base. +""" +@noinline function _scalar_indexing_stdlib_caller() + caller = nothing + for frame in stacktrace() + m = _module_of_stackframe(frame) + # Skip frames with unknown module (same as previous `nothing` skip). + m === _UNKNOWN_STACK_MODULE && continue + (m === Core || _is_cunumeric_module(m)) && continue + clean_name = _clean_stack_func_name(frame.func) + clean_name in _SKIP_STACK_FUNCS && continue + if _is_linalg_module(m) + caller = (:LinearAlgebra, clean_name) + elseif _is_base_module(m) + caller = (:Base, clean_name) + else + # User / other package code — keep the last stdlib candidate, if any. + break + end + end + return caller +end + +# Returns (enriched::Bool, desc::String). Enriched = Base or LinearAlgebra stdlib caller. function scalardesc(op) - desc = """Invocation of $op resulted in scalar indexing of an NDArray. + caller = _scalar_indexing_stdlib_caller() + if caller !== nothing + modname, fname = caller + # Base/LinearAlgebra AbstractArray fallback — name the outer API first. + # No "Scalar indexing is disallowed." header (not part of the enriched template). + return true, + "`$modname.$fname` fell back to an AbstractArray implementation, which scalar-indexed an `NDArray`. " * + "This $modname path is probably not implemented yet for `NDArray`. " * + "Using `allowscalar` or `@allowscalar` might allow this function to work slowly, but it has not been tested." + end + + # Plain user-level scalar indexing (unchanged). + return false, """Invocation of $op resulted in scalar indexing of an `NDArray`. This is typically caused by calling an iterating implementation of a method. - This is very slow and should be avoided. + This is very slow and should be avoided. This can also happen if an external + method (i.e., LinearAlgebra.kron) is not re-implemented in cuNumeric.jl. Because + `NDArray`s subtype `AbstractArray`, the method call will dispatch to the + `AbstractArray` implementation, which often iterates over the array. If you want to allow scalar iteration, use `allowscalar` or `@allowscalar` to enable scalar iteration globally or for the operations in question.""" end function promotiondesc(op, ::Type{FROM}, ::Type{TO}) where {FROM,TO} - desc = """Invocation of $op resulted in implicit promotion of an NDArray from $(FROM) to - wider type: $(TO). This is typically caused by mixing NDArrays or literals - with different precision. This can cause extra copies of data and is slow. + return desc = """Invocation of $op resulted in implicit promotion of an NDArray from $(FROM) to + wider type: $(TO). This is typically caused by mixing NDArrays or literals + with different precision. This can cause extra copies of data and is slow. - If you want to allow implicit promotion to wider types, use `allowpromotion` or `@allowpromotion` - to enable implicit promotion.""" + If you want to allow implicit promotion to wider types, use `allowpromotion` or `@allowpromotion` + to enable implicit promotion.""" end @noinline function warnscalar(op) - desc = scalardesc(op) + _, desc = scalardesc(op) @warn("""Performing scalar indexing on task $(current_task()). $desc""") end @@ -150,9 +317,14 @@ end end @noinline function errorscalar(op) - desc = scalardesc(op) - error("""Scalar indexing is disallowed. - $desc""") + enriched, desc = scalardesc(op) + if enriched + error(desc) + else + # Plain path keeps the historical disallow header. + error("""Scalar indexing is disallowed. + $desc""") + end end @noinline function errordouble(op, ::Type{FROM}, ::Type{TO}) where {FROM,TO} @@ -165,7 +337,7 @@ end # NOTE: This is deprecated and should not be used from user logic. A proper solution to # this problem will be introduced in https://github.com/JuliaLang/julia/pull/39217 macro __tryfinally(ex, fin) - Expr(:tryfinally, + return Expr(:tryfinally, :($(esc(ex))), :($(esc(fin))), ) @@ -185,7 +357,7 @@ See also: [`@allowscalar`](@ref). allowscalar function allowscalar(f::Base.Callable) - task_local_storage(f, :ScalarIndexing, ScalarAllowed) + return task_local_storage(f, :ScalarIndexing, ScalarAllowed) end function allowscalar(allow::Bool=true) diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index db527d199..add8e8a06 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -158,7 +158,7 @@ end M = cuNumeric.zeros(T, 4, 3) sq = cuNumeric.zeros(T, 5, 5) v = cuNumeric.zeros(T, 8) - @test @inferred(cuNumeric.eye(T, 5)) !== nothing + @test @inferred(NDArray{T}(I, 5, 5)) !== nothing @test @inferred(cuNumeric.transpose(M)) !== nothing @test @inferred(cuNumeric.trace(sq)) !== nothing @test @inferred(cuNumeric.diag(sq)) !== nothing diff --git a/test/array/diagonal.jl b/test/array/diagonal.jl new file mode 100644 index 000000000..a04e16291 --- /dev/null +++ b/test/array/diagonal.jl @@ -0,0 +1,811 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +# Coverage for src/ndarray/diagonal.jl: diag/_eye/trace, Diagonal, UniformScaling. + +const DIAGONAL_NUMERIC_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) +const DIAGONAL_ARRAY_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) +const DIAGONAL_FLOAT_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_FLOAT_TYPES) +const DIAGONAL_COMPLEX_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_COMPLEX_TYPES) + +_nonzero_diag(::Type{T}, n) where {T<:AbstractFloat} = abs.(my_rand(T, n)) .+ one(T) +function _nonzero_diag(::Type{T}, n) where {T<:Complex} + return Complex.(abs.(real(my_rand(T, n))) .+ one(real(T)), zero(real(T))) +end +_nonzero_diag(::Type{T}, n) where {T<:Integer} = T.(collect(2:(n + 1))) +_nonzero_diag(::Type{Bool}, n) = fill(true, n) + +function _host_diag_compare(ref, out, ::Type{T}) where {T} + allowscalar() do + @test cuNumeric.compare(ref, out, atol(T), rtol(T)) + end +end + +function _host_scalar_compare(ref, out::NDArray{<:Any,0}, ::Type{T}) where {T} + allowscalar() do + @test ref ≈ out[] atol=atol(T) rtol=rtol(T) + end +end + +function _host_bool_compare(ref::Bool, out::NDArray{Bool,0}) + allowscalar() do + @test out[] == ref + end +end + +function _host_matrix_compare(ref::AbstractMatrix, out::NDArray, ::Type{T}) where {T} + allowscalar() do + @test cuNumeric.compare(ref, out, atol(T), rtol(T)) + end +end + +###### diag / identity / trace ###### + +@testset "diag" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + A = my_rand(T, 6, 5) + nda = NDArray(A) + @testset "k=$k" for k in (-2, 0, 2) + ref = diag(A, k) + _host_diag_compare(ref, cuNumeric.diag(nda; k=k), T) + _host_diag_compare(ref, LinearAlgebra.diag(nda, k), T) + end + end +end + +@testset "identity via I / _eye" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + n = 4 + ref = Matrix{T}(I, n, n) + _host_matrix_compare(ref, NDArray{T}(I, n, n), T) + _host_matrix_compare(ref, cuNumeric._eye(T, n), T) + end + # Untyped NDArray(I, ...) uses Bool; typed / _eye default to Float32 densify + ref_bool = Matrix{Bool}(I, 3, 3) + _host_matrix_compare(ref_bool, NDArray(I, 3, 3), Bool) + ref = Matrix{Float32}(I, 3, 3) + _host_matrix_compare(ref, NDArray{Float32}(I, 3, 3), Float32) + _host_matrix_compare(ref, cuNumeric._eye(3), Float32) +end + +@testset "trace" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + A = my_rand(T, 5, 5) + nda = NDArray(A) + @testset "offset=$k" for k in (-2, -1, 0, 1, 2) + ref = sum(diag(A, k)) + out = cuNumeric.trace(nda; offset=k) + @test out isa NDArray{<:Any,0} + _host_scalar_compare(ref, out, eltype(ref)) + end + end +end + +###### Diagonal constructors / densify / show ###### + +@testset "Diagonal constructors" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 4) + v = NDArray(d) + D = Diagonal(v) + @test D isa Diagonal{T,<:NDArray{T,1}} + @test D.diag === v + allowscalar() do + @test cuNumeric.compare(d, D.diag, atol(T), rtol(T)) + @test Matrix(D) ≈ Matrix(Diagonal(d)) atol=atol(T) rtol=rtol(T) + @test Matrix{T}(D) ≈ Matrix(Diagonal(d)) atol=atol(T) rtol=rtol(T) + end + + A = my_rand(T, 4, 4) + D2 = Diagonal(NDArray(A)) + allowscalar() do + @test cuNumeric.compare(diag(A), D2.diag, atol(T), rtol(T)) + end + end +end + +@testset "Diagonal show" begin + D = Diagonal(NDArray(Float32[1, 2, 3])) + s = sprint(show, D) + @test occursin("1.0", s) && occursin("2.0", s) + plain = sprint(show, MIME"text/plain"(), D) + # Summary must reflect the real Diagonal{T,<:NDArray} type, not Vector. + @test occursin("NDArray", plain) + @test occursin(string(typeof(D)), plain) + @test occursin("1.0", plain) + # Match Base Diagonal formatting (⋅ off-diagonals), not a dense Matrix dump. + @test occursin("⋅", plain) + dense = sprint(show, MIME"text/plain"(), Matrix(D)) + @test plain != dense +end + +###### Diagonal operators ###### + +@testset "Diagonal * Diagonal / ± / scalar" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + allowscalar() do + @test Matrix(D * D) ≈ Matrix(Dh * Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D + D) ≈ Matrix(Dh + Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D - D) ≈ Matrix(Dh - Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(T(3) * D) ≈ Matrix(T(3) * Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * T(3)) ≈ Matrix(Dh * T(3)) atol=atol(T) rtol=rtol(T) + end + end +end + +@testset "Diagonal broadcast on diag" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 5) + c = T(5) + Dh = Diagonal(d) + D = Diagonal(NDArray(copy(d))) + + ref = collect((Dh .* c).diag) + + # Out-of-place: stays Diagonal with NDArray diag + D2 = D .* c + @test D2 isa Diagonal{T,<:NDArray{T,1}} + @test D2.diag isa NDArray{T,1} + _host_diag_compare(ref, D2.diag, T) + + # In-place fused assign + D3 = Diagonal(NDArray(copy(d))) + D3 .= D3 .* c + @test D3 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref, D3.diag, T) + + # In-place .*= + D4 = Diagonal(NDArray(copy(d))) + D4 .*= c + @test D4 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref, D4.diag, T) + + # Zero-preserving Diagonal .+ Diagonal (matches Base structure) + ref_add = collect((Dh .+ Dh).diag) + D_add = D .+ D + @test D_add isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref_add, D_add.diag, T) + + D_add2 = Diagonal(NDArray(copy(d))) + D_add2 .+= D_add2 + @test D_add2 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref_add, D_add2.diag, T) + + D_add3 = Diagonal(NDArray(copy(d))) + D_add3 .= D_add3 .+ D_add3 + @test D_add3 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref_add, D_add3.diag, T) + end + + # Exact repro from the bug report + a = Diagonal(cuNumeric.ones(Int32, 5)) + a .*= Int32(5) + @test a isa Diagonal{Int32,<:NDArray{Int32,1}} + @test Array(a.diag) == fill(Int32(5), 5) +end + +@testset "scalar indexing message: LinearAlgebra / Base vs plain" begin + # Unsupported LA/Base fallbacks enrich; intentional user scalar indexing stays plain. + allowscalar(false) + + function _assert_enriched_fallback(msg, modfunc) + @test occursin("`$modfunc` fell back to an AbstractArray implementation", msg) + @test occursin("which scalar-indexed an `NDArray`", msg) + @test occursin("path is probably not implemented yet for `NDArray`", msg) + @test occursin("allowscalar", msg) + @test occursin("@allowscalar", msg) + @test occursin("might allow this function to work slowly", msg) + @test occursin("it has not been tested", msg) + # Enriched path replaces the generic iterating-method lead and omits the plain header. + @test !occursin("typically caused by calling an iterating implementation", msg) + @test !occursin("Scalar indexing is disallowed", msg) + @test !occursin("If you want to allow scalar iteration", msg) + @test !occursin("triggered via", msg) + @test startswith(lstrip(msg), "`$modfunc`") + end + + err = @test_throws ErrorException cholesky(Diagonal(NDArray(Float32[2, 3, 4]))) + msg = sprint(showerror, err.value) + _assert_enriched_fallback(msg, "LinearAlgebra.cholesky") + + # sortperm-based LA path must blame svd, not the inner Base.lt helper. + err_svd = @test_throws ErrorException svd(Diagonal(NDArray(Float32[2, 3, 4]))) + msg_svd = sprint(showerror, err_svd.value) + _assert_enriched_fallback(msg_svd, "LinearAlgebra.svd") + @test !occursin("Base.lt", msg_svd) + + # Base AbstractArray fallback (unique) should enrich with Base.. + err_base = @test_throws ErrorException unique(NDArray(Float32[1, 2, 1])) + msg_base = sprint(showerror, err_base.value) + _assert_enriched_fallback(msg_base, "Base.unique") + + # Call through a Main function so the first non-cuNumeric/Core frame is user code, + # not Base.include_string / Base.IncludeInto (Julia 1.12+) / client frames from + # `include`ing this test file. + function _plain_scalar_index_probe() + a = NDArray(Float32[1, 2, 3]) + return a[1] + end + err_plain = try + _plain_scalar_index_probe() + nothing + catch e + e + end + @test err_plain isa ErrorException + msg_plain = sprint(showerror, err_plain) + @test occursin("Scalar indexing is disallowed", msg_plain) + @test occursin("typically caused by calling an iterating implementation", msg_plain) + @test occursin("If you want to allow scalar iteration", msg_plain) + @test !occursin("triggered via", msg_plain) + @test !occursin("fell back to an AbstractArray", msg_plain) + @test !occursin("probably not implemented yet", msg_plain) + @test !occursin("might allow this function to work slowly", msg_plain) + @test !occursin("it has not been tested", msg_plain) +end + +@testset "NDArray iszero / isone" begin + z = cuNumeric.zeros(Float32, 3) + o = cuNumeric.ones(Float32, 3) + _host_bool_compare(true, iszero(z)) + _host_bool_compare(false, iszero(o)) + I = one(cuNumeric.zeros(Float32, 3, 3)) + _host_bool_compare(true, isone(I)) + _host_bool_compare(false, isone(cuNumeric.ones(Float32, 3, 3))) + _host_bool_compare(false, isone(cuNumeric.zeros(Float32, 3, 3))) + _host_bool_compare(false, isone(cuNumeric.ones(Float32, 2, 3))) +end + +@testset "Diagonal densifying broadcast" begin + D = Diagonal(NDArray(Float32[1, 2, 3])) + expected = Float32[2 1 1; 1 3 1; 1 1 4] + + # Out-of-place densifies to dense NDArray (Base densifies to Matrix). + R = D .+ Float32(1) + @test R isa NDArray{Float32,2} + @test Array(R) == expected + + A = cuNumeric.ones(Float32, 3, 3) + R2 = D .+ A + @test R2 isa NDArray{Float32,2} + @test Array(R2) == expected + R3 = A .+ D + @test R3 isa NDArray{Float32,2} + @test Array(R3) == expected + + # In-place still rejects off-diagonal writes (Base-style ArgumentError). + err_inp = @test_throws ArgumentError D .+= Float32(1) + @test occursin("off-diagonal", sprint(showerror, err_inp.value)) + + D_inp = Diagonal(NDArray(Float32[1, 2, 3])) + err_mat_inp = @test_throws ArgumentError D_inp .+= A + @test occursin("off-diagonal", sprint(showerror, err_mat_inp.value)) + + # Structure-preserving scale still stays Diagonal without @allowscalar. + D2 = Diagonal(NDArray(Float32[1, 2, 3])) + D2 .*= Float32(4) + @test Array(D2.diag) == Float32[4, 8, 12] + D3 = D2 .* Float32(2) + @test D3 isa Diagonal{Float32,<:NDArray{Float32,1}} + @test Array(D3.diag) == Float32[8, 16, 24] + + # 1×1 has no off-diagonals: Base allows in-place densifying-classified ops. + D1 = Diagonal(NDArray(Float32[5])) + D1 .+= Float32(1) + _host_diag_compare(Float32[6], D1.diag, Float32) + D1 .*= Float32(2) + _host_diag_compare(Float32[12], D1.diag, Float32) + + # 1×1 out-of-place densifies to NDArray (Base returns Matrix). + R1 = Diagonal(NDArray(Float32[5])) .+ Float32(1) + @test R1 isa NDArray{Float32,2} + @test size(R1) == (1, 1) + _host_matrix_compare(Float32[6;;], R1, Float32) +end + +@testset "Diagonal * NDArray / mul! / lmul! / rmul!" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + n, m = 3, 4 + d = my_rand(T, n) + dm = my_rand(T, m) + Ah = my_rand(T, n, m) # n×m + As = my_rand(T, n, n) # n×n + vh = my_rand(T, n) + Dh = Diagonal(d) + Dm = Diagonal(dm) + D = Diagonal(NDArray(d)) + D_m = Diagonal(NDArray(dm)) + A = NDArray(Ah) + A_nd = NDArray(As) + v = NDArray(vh) + + _host_diag_compare(Dh * vh, D * v, T) + _host_matrix_compare(Dh * Ah, D * A, T) # D (n)× A (n×m) + _host_matrix_compare(Ah * Dm, A * D_m, T) # A (n×m) × D (m) + _host_matrix_compare(As * Dh, A_nd * D, T) # square A*D + + C = cuNumeric.zeros(T, n, m) + mul!(C, D, A) + _host_matrix_compare(Dh * Ah, C, T) + + Cs = cuNumeric.zeros(T, n, n) + mul!(Cs, A_nd, D) + _host_matrix_compare(As * Dh, Cs, T) + + B = copy(A_nd) + lmul!(D, B) + _host_matrix_compare(Dh * As, B, T) + + B = copy(A_nd) + rmul!(B, D) + _host_matrix_compare(As * Dh, B, T) + end +end + +@testset "Diagonal \\ / / inv / ldiv! / rdiv!" begin + # Floats: full path including singular CUDA-style behavior + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + + _host_diag_compare(Dh \ vh, D \ v, T) + _host_matrix_compare(Dh \ Ah, D \ A, T) + _host_matrix_compare(Ah / Dh, A / D, T) + allowscalar() do + @test Matrix(inv(D)) ≈ Matrix(inv(Dh)) atol=atol(T) rtol=rtol(T) + end + + B = copy(A) + ldiv!(D, B) + _host_matrix_compare(Dh \ Ah, B, T) + + B = copy(A) + rdiv!(B, D) + _host_matrix_compare(Ah / Dh, B, T) + + # Singular: zeros become Inf/NaN (no SingularException; that needs a host Bool) + d0 = copy(d) + d0[2] = zero(T) + D0 = Diagonal(NDArray(d0)) + r = D0 \ v + Di = inv(D0) + Q = A / D0 + allowscalar() do + @test any(isinf, Array(r)) || any(isnan, Array(r)) + @test any(isinf, Array(Di.diag)) || any(isnan, Array(Di.diag)) + @test any(isinf, Array(Q)) || any(isnan, Array(Q)) + end + end + + # Complex: \ works via ./ ; inv/A/D blocked by missing __recip_type + @testset verbose=true for T in DIAGONAL_COMPLEX_TYPES + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + _host_diag_compare(Dh \ vh, D \ v, T) + _host_matrix_compare(Dh \ Ah, D \ A, T) + end + + # Integers with explicit allowpromotion for inv / \ + # `\` uses ./ → typically float(T) (Float64 for Int32/Int64). + # `inv` / `A / D` use cuNumeric.__recip_type (Float32 for Int32, Float64 for Int64), + # which is intentional NDArray promotion — not Base's float(Int32)==Float64. + @testset verbose=true for T in (Int32, Int64) + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + FT = float(T) + RT = cuNumeric.__recip_type(T) + Dh_div = Diagonal(FT.(d)) + Dh_inv = Diagonal(RT.(d)) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + allowpromotion() do + _host_diag_compare(Dh_div \ FT.(vh), D \ v, FT) + _host_matrix_compare(Dh_div \ FT.(Ah), D \ A, FT) + allowscalar() do + @test Matrix(inv(D)) ≈ Matrix(inv(Dh_inv)) atol=atol(RT) rtol=rtol(RT) + end + _host_matrix_compare(RT.(Ah) / Dh_inv, A / D, RT) + end + end +end + +@testset "Diagonal det" begin + # Floats: det is prod of the diagonal (0D NDArray, not a Julia scalar) + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + @test det(D) isa NDArray{<:Any,0} + _host_scalar_compare(det(Dh), det(D), T) + _host_scalar_compare(det(Matrix(D)), det(D), T) + + # 1×1 + d1 = T[T(5)] + D1 = Diagonal(NDArray(d1)) + _host_scalar_compare(det(Diagonal(d1)), det(D1), T) + end + + # Integers: prod may widen (e.g. Int32 → Int64); needs allowpromotion + @testset verbose=true for T in (Int32, Int64) + d = _nonzero_diag(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + allowpromotion() do + _host_scalar_compare(det(Dh), det(D), T) + _host_scalar_compare(det(Matrix(D)), det(D), T) + end + # 1×1 + D1 = Diagonal(NDArray(T[7])) + allowpromotion() do + _host_scalar_compare(T(7), det(D1), T) + end + end +end + +@testset "NDArray ± Diagonal" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + n = 3 + d = my_rand(T, n) + Ah = my_rand(T, n, n) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + _host_matrix_compare(Ah + Dh, A + D, T) + _host_matrix_compare(Dh + Ah, D + A, T) + _host_matrix_compare(Ah - Dh, A - D, T) + _host_matrix_compare(Dh - Ah, D - A, T) + end + + # Bool promotes under + + @testset "Bool with allowpromotion" begin + Ah = Bool[1 0; 0 1] + d = Bool[true, true] + A = NDArray(Ah) + D = Diagonal(NDArray(d)) + allowpromotion() do + allowscalar() do + @test Array(A + D) == Ah + Diagonal(d) + @test Array(A - D) == Ah - Diagonal(d) + end + end + end +end + +###### UniformScaling ###### + +# Prefer typed UniformScaling (one(T)*I) so narrow integers do not promote against +# Int64 λ from 2I / -I. Plain I (Bool λ) is fine for +/* on numeric arrays. + +@testset "UniformScaling constructors / copyto!" begin + @testset verbose=true for T in DIAGONAL_ARRAY_TYPES + n = 3 + J1 = one(T) * I + ref = Matrix{T}(J1, n, n) + E = NDArray{T}(J1, n, n) + _host_matrix_compare(ref, E, T) + E2 = NDArray{T}(I, (n, n)) # Bool λ → ones on diagonal still + _host_matrix_compare(Matrix{T}(I, n, n), E2, T) + + C = cuNumeric.zeros(T, n, n) + copyto!(C, J1) + _host_matrix_compare(ref, C, T) + + C = cuNumeric.ones(T, n, n) + copyto!(C, zero(T) * I) + _host_matrix_compare(zeros(T, n, n), C, T) + + # Rectangular scaled identity (skip Bool: Bool(2) is inexact) + if T != Bool + J2 = T(2) * I + R = NDArray{T}(J2, 2, 3) + allowscalar() do + @test Array(R) == Matrix{T}(J2, 2, 3) + end + C = cuNumeric.zeros(T, 2, 3) + copyto!(C, J2) + allowscalar() do + @test Array(C) == Matrix{T}(J2, 2, 3) + end + else + R = NDArray{Bool}(I, 2, 3) + allowscalar() do + @test Array(R) == Matrix{Bool}(I, 2, 3) + end + end + end + + # Untyped NDArray(I, ...) uses λ's type (Bool for I) + E = NDArray(I, 2, 2) + @test eltype(E) == Bool + allowscalar() do + @test Array(E) == Bool[1 0; 0 1] + end + E = NDArray(I, (2, 2)) + allowscalar() do + @test Array(E) == Bool[1 0; 0 1] + end +end + +@testset "NDArray ± / * UniformScaling / one / oneunit" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + Ah = my_rand(T, 3, 3) + A = NDArray(Ah) + J1 = one(T) * I + J2 = T(2) * I + + _host_matrix_compare(Ah + J1, A + J1, T) + _host_matrix_compare(J1 + Ah, J1 + A, T) + _host_matrix_compare(Ah * I, A * I, T) + _host_matrix_compare(I * Ah, I * A, T) + _host_matrix_compare(Ah * J2, A * J2, T) + _host_matrix_compare(J2 * Ah, J2 * A, T) + _host_matrix_compare(Matrix{T}(I, 3, 3), one(A), T) + _host_matrix_compare(Matrix{T}(I, 3, 3), oneunit(A), T) + + # Subtraction / A+2I: signed & float/complex only (unsigned -one wraps; skip) + if T <: Union{AbstractFloat,Complex} || (T <: Signed) + _host_matrix_compare(Ah - J1, A - J1, T) + _host_matrix_compare(J1 - Ah, J1 - A, T) + _host_matrix_compare(Ah + J2, A + J2, T) + end + end + + # Plain I / 2I (Bool/Int64 λ) — natural API for floats & complex + @testset verbose=true for T in (DIAGONAL_FLOAT_TYPES..., DIAGONAL_COMPLEX_TYPES...) + Ah = my_rand(T, 3, 3) + A = NDArray(Ah) + _host_matrix_compare(Ah + I, A + I, T) + _host_matrix_compare(I + Ah, I + A, T) + _host_matrix_compare(Ah - I, A - I, T) + _host_matrix_compare(I - Ah, I - A, T) + _host_matrix_compare(Ah + 2I, A + 2I, T) + _host_matrix_compare(Ah * (2I), A * (2I), T) + _host_matrix_compare((2I) * Ah, (2I) * A, T) + end + + # Bool: * I keeps Bool; ± I needs promotion + @testset "Bool UniformScaling" begin + Ah = Bool[1 0; 0 1] + A = NDArray(Ah) + allowscalar() do + @test Array(A * I) == Ah + @test eltype(A * I) == Bool + @test Array(one(A)) == Ah + @test Array(oneunit(A)) == Ah + end + allowpromotion() do + allowscalar() do + @test Array(A + I) == Ah + I + @test Array(I - A) == I - Ah + end + end + end +end + +###### Diagonal ↔ UniformScaling ###### + +@testset "Diagonal ± / * UniformScaling / copyto!" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + J1 = one(T) * I + J2 = T(2) * I + + Di = D + J1 + @test Di isa Diagonal + allowscalar() do + @test Matrix(Di) ≈ Matrix(Dh + J1) atol=atol(eltype(Di)) rtol=rtol(eltype(Di)) + @test Matrix(J1 + D) ≈ Matrix(J1 + Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * I) ≈ Matrix(Dh * I) atol=atol(T) rtol=rtol(T) + @test Matrix(I * D) ≈ Matrix(I * Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * J2) ≈ Matrix(Dh * J2) atol=atol(T) rtol=rtol(T) + end + + if T <: Union{AbstractFloat,Complex} || (T <: Signed) + allowscalar() do + @test Matrix(D - J1) ≈ Matrix(Dh - J1) atol=atol(T) rtol=rtol(T) + @test Matrix(J1 - D) ≈ Matrix(J1 - Dh) atol=atol(eltype((J1 - D).diag)) rtol=rtol( + eltype((J1 - D).diag) + ) + end + end + + copyto!(D, J1) + allowscalar() do + @test Array(D.diag) == ones(T, 3) + end + copyto!(D, zero(T) * I) + allowscalar() do + @test Array(D.diag) == zeros(T, 3) + end + end + + # Plain I on floats + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + allowscalar() do + @test Matrix(D + I) ≈ Matrix(Dh + I) atol=atol(T) rtol=rtol(T) + @test Matrix(D - I) ≈ Matrix(Dh - I) atol=atol(T) rtol=rtol(T) + @test Matrix(I - D) ≈ Matrix(I - Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * (2I)) ≈ Matrix(Dh * (2I)) atol=atol(T) rtol=rtol(T) + end + end + + @testset "Bool Diagonal + I with allowpromotion" begin + D = Diagonal(NDArray(Bool[true, false])) + allowpromotion() do + Di = D + I + @test Di isa Diagonal + allowscalar() do + @test Array(Di.diag) == [2, 1] + end + end + end +end + +###### Structured Diagonal vs densified Matrix path ###### + +# Compare Diagonal{NDArray} ops to the same math on densified host Matrices, +# and check that D + I stays Diagonal while A + I densifies to NDArray. + +@testset "Diagonal vs dense" begin + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + Dh = Diagonal(d) + Md = Matrix(Dh) # densified host counterpart + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + + _host_matrix_compare(Md * Ah, D * A, T) + _host_matrix_compare(Ah * Md, A * D, T) + _host_diag_compare(Md \ vh, D \ v, T) + _host_matrix_compare(Ah / Md, A / D, T) + allowscalar() do + @test Matrix(inv(D)) ≈ inv(Md) atol=atol(T) rtol=rtol(T) + end + _host_matrix_compare(Ah + Md, A + D, T) + + Di = D + I + @test Di isa Diagonal + allowscalar() do + @test Matrix(Di) ≈ Md + I atol=atol(T) rtol=rtol(T) + end + + Ai = A + I + @test Ai isa NDArray + @test !(Ai isa Diagonal) + _host_matrix_compare(Ah + I, Ai, T) + _host_matrix_compare(Ah * I, A * I, T) + end + + # Mul / add / structure for remaining numeric types (skip inv / div) + @testset verbose=true for T in (DIAGONAL_COMPLEX_TYPES..., Int32, Int64) + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + Dh = Diagonal(d) + Md = Matrix(Dh) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + J1 = one(T) * I + + _host_matrix_compare(Md * Ah, D * A, T) + _host_matrix_compare(Ah * Md, A * D, T) + _host_matrix_compare(Ah + Md, A + D, T) + + Di = D + J1 + @test Di isa Diagonal + allowscalar() do + @test Matrix(Di) ≈ Md + Matrix{T}(J1, n, n) atol=atol(T) rtol=rtol(T) + end + + Ai = A + J1 + @test Ai isa NDArray + @test !(Ai isa Diagonal) + _host_matrix_compare(Ah + J1, Ai, T) + end +end + +###### Native LinearAlgebra API (supported Diagonal paths) ###### + +@testset "Diagonal eigen / eigvals native" begin + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = T[T(3), T(1), T(2)] + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + + # eigvals is a copy of the diagonal (NDArray) + λ = eigvals(D) + @test λ isa NDArray{T,1} + _host_diag_compare(eigvals(Dh), λ, T) + @test λ !== D.diag + + # unsorted eigen: values == diag copy, vectors == I (NDArray) + F = eigen(D) + @test F.values isa NDArray{T,1} + _host_diag_compare(d, F.values, T) + @test F.vectors isa NDArray{T,2} + _host_matrix_compare(Matrix{T}(I, 3, 3), F.vectors, T) + + @test eigvecs(D) isa NDArray{T,2} + _host_matrix_compare(Matrix{T}(I, 3, 3), eigvecs(D), T) + end +end + +@testset "Diagonal native reductions / predicates / norms" begin + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = abs.(my_rand(T, 3)) .+ one(T) # positive → isposdef + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + + _host_scalar_compare(tr(Dh), tr(D), T) + _host_scalar_compare(sum(Dh), sum(D), T) + _host_scalar_compare(zero(T), prod(D), T) # n>1 off-diagonals + _host_scalar_compare(maximum(Dh), maximum(D), T) + _host_scalar_compare(minimum(Dh), minimum(D), T) + _host_bool_compare(isposdef(Dh), isposdef(D)) + _host_bool_compare(true, issymmetric(D)) + _host_bool_compare(true, ishermitian(D)) + @test isdiag(D) + _host_bool_compare(false, iszero(D)) + _host_bool_compare(false, isone(D)) + _host_bool_compare(true, iszero(Diagonal(cuNumeric.zeros(T, 3)))) + _host_bool_compare(true, isone(Diagonal(cuNumeric.ones(T, 3)))) + _host_bool_compare(true, istriu(D)) + _host_bool_compare(true, istril(D)) + _host_bool_compare(false, istriu(D, 1)) + _host_bool_compare(false, istril(D, -1)) + + _host_scalar_compare(opnorm(Dh), opnorm(D), T) + _host_scalar_compare(norm(Dh), norm(D), T) + _host_scalar_compare(cond(Dh), cond(D), T) + _host_scalar_compare(logdet(Dh), logdet(D), T) + + # matrix functions via f.(diag) broadcast (Base Diagonal methods) + allowscalar() do + @test Matrix(sqrt(D)) ≈ Matrix(sqrt(Dh)) atol=atol(T) rtol=rtol(T) + @test Matrix(exp(D)) ≈ Matrix(exp(Dh)) atol=atol(T) rtol=rtol(T) + end + end +end diff --git a/test/array/linalg.jl b/test/array/linalg.jl index 836d47249..19a25d002 100644 --- a/test/array/linalg.jl +++ b/test/array/linalg.jl @@ -111,7 +111,7 @@ end @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) n = 5 ref = Matrix{T}(I, n, n) - out = cuNumeric.eye(T, n) + out = NDArray{T}(I, n, n) allowscalar() do @test safe_compare(ref, out, atol(T), rtol(T)) end @@ -126,7 +126,7 @@ end ref = sum(diag(A)) # widens ints like trace's accumulator out = cuNumeric.trace(nda) allowscalar() do - @test ref ≈ out[1] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) + @test ref ≈ out[] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) end end end @@ -140,7 +140,7 @@ end ref = sum(diag(A, k)) out = cuNumeric.trace(nda; offset=k) allowscalar() do - @test ref ≈ out[1] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) + @test ref ≈ out[] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) end end end From 8f44dec5b9ccd6a53ce676c8c5095c7f0e73329a Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sat, 15 Aug 2026 13:32:13 -0400 Subject: [PATCH 07/49] Support more unary/binary ops (#177) * support more unary/binary ops --------- Co-authored-by: krasow --- benchmark/__plot_results.jl | 301 ++++++++++++++++++++++++++++++++++++ docs/src/api_binary.md | 8 +- docs/src/api_unary.md | 5 +- src/ndarray/binary.jl | 34 ++-- src/ndarray/unary.jl | 90 +++++++---- test/array/binary/tests.jl | 83 +++++++++- test/array/unary/tests.jl | 20 ++- 7 files changed, 486 insertions(+), 55 deletions(-) create mode 100644 benchmark/__plot_results.jl diff --git a/benchmark/__plot_results.jl b/benchmark/__plot_results.jl new file mode 100644 index 000000000..fd1ed646a --- /dev/null +++ b/benchmark/__plot_results.jl @@ -0,0 +1,301 @@ +#!/usr/bin/env julia +# Weak-scaling plots (1/2/4/8 GPUs) for the benchmark result CSVs. +# One figure per benchmark, three panels: throughput, time/step, parallel efficiency. +# +# CSV schema (see src/core.jl save_result): +# implementation,gpus,N,M,trial,time_ms,throughput,correctness +# `throughput` is the benchmark's `total_flops` divided by elapsed time. For +# Gray-Scott that unit is Gpoint-updates/s; other benchmarks report GFLOP/s. +# +# save_result appends and does NOT encode the code-path variant, so a cuNumeric CSV +# holds alternating runs: baseline, @accelerate, baseline, @accelerate, ... +# (a run boundary = the GPU count resetting downward). +# cuPyNumeric / CUDA.jl have no accelerated path -> a single block. +# +# Encoding: color = implementation; line style = code path +# solid = baseline, dashed = @accelerate. + +using Plots +using Statistics + +gr() + +function parse_args(args) + results_dir = "results" + out_dir = nothing + single_cunumeric_run = :baseline + hide_baseline = false + output_suffix = "" + + for arg in args + if startswith(arg, "--single-cu=") + value = Symbol(lowercase(last(split(arg, "="; limit=2)))) + value in (:baseline, :accelerated) || + error("--single-cu must be baseline or accelerated") + single_cunumeric_run = value + elseif arg == "--hide-baseline" + hide_baseline = true + elseif startswith(arg, "--out=") + out_dir = last(split(arg, "="; limit=2)) + elseif startswith(arg, "--suffix=") + output_suffix = last(split(arg, "="; limit=2)) + else + results_dir = arg + end + end + isempty(output_suffix) && hide_baseline && (output_suffix = "_no_baseline") + + results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) + if out_dir === nothing + out_dir = if basename(normpath(results_dir)) == "results" + joinpath(@__DIR__, "plots") + else + joinpath(@__DIR__, "plots", basename(normpath(results_dir))) + end + else + out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) + end + return (; results_dir, out_dir, single_cunumeric_run, hide_baseline, output_suffix) +end + +const CONFIG = parse_args(ARGS) +const RESULTS_DIR = CONFIG.results_dir +const OUT_DIR = CONFIG.out_dir +const SINGLE_CUNUMERIC_RUN = CONFIG.single_cunumeric_run +const HIDE_BASELINE = CONFIG.hide_baseline +const OUTPUT_SUFFIX = CONFIG.output_suffix + +# filekey, family label, color, marker, can_contain_accelerated_blocks +const FAMILIES = [ + ("cunumeric", "cuNumeric.jl (fused)", "#2a78d6", :circle, true), + ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond, true), + ("cupynumeric", "cuPyNumeric", "#eb6834", :rect, false), + ("CUDA.jl", "CUDA.jl", "#008300", :utriangle, false), +] + +const INK = "#0b0b0b" +const MUTED = "#898781" +const GRIDCOL = "#e1e0d9" +const IDEALCOL = "#c3c2b7" + +struct Row + gpus::Int + time_ms::Float64 + thr::Float64 +end + +# Parse a CSV into runs, split wherever the GPU count resets to a smaller value. +function load_runs(path) + rows = Row[] + for line in eachline(path) + isempty(strip(line)) && continue + f = split(line, ',') + push!(rows, Row(parse(Int, f[2]), parse(Float64, f[6]), parse(Float64, f[7]))) + end + isempty(rows) && return Vector{Row}[] + runs = [Row[]] + for (i, r) in enumerate(rows) + i > 1 && r.gpus < rows[i - 1].gpus && push!(runs, Row[]) + push!(runs[end], r) + end + return runs +end + +# Aggregate trials per GPU count -> sorted vector of (gpus, t, tsd, h, hsd). +function aggregate(rows) + by = Dict{Int,Vector{Row}}() + for r in rows + push!(get!(by, r.gpus, Row[]), r) + end + sd(x) = length(x) > 1 ? std(x) : 0.0 + return [ + (gpus=g, t=mean(getfield.(by[g], :time_ms)), tsd=sd(getfield.(by[g], :time_ms)), + h=mean(getfield.(by[g], :thr)), hsd=sd(getfield.(by[g], :thr))) + for g in sort(collect(keys(by))) + ] +end + +# Build the series (color+marker+linestyle+agg) present for one benchmark. +function series_for(bench) + series = [] # NamedTuple(label,color,marker,ls,agg) + for (key, fam, color, marker, splits) in FAMILIES + path = joinpath(RESULTS_DIR, "$(bench)_$(key).csv") + isfile(path) || continue + runs = load_runs(path) + isempty(runs) && continue + if splits && bench == "grayscott" + if length(runs) == 1 + label = + SINGLE_CUNUMERIC_RUN === :accelerated ? "$fam · accelerated" : + "$fam · baseline" + ls = SINGLE_CUNUMERIC_RUN === :accelerated ? :dash : :solid + push!( + series, (label=label, color=color, marker=marker, ls=ls, + agg=aggregate(runs[1])) + ) + else + # Repeated harness runs append alternating baseline/accelerated blocks. + baseline_rows = reduce(vcat, runs[1:2:end]) + accelerated_rows = reduce(vcat, runs[2:2:end]) + push!( + series, + (label="$fam · baseline", color=color, marker=marker, + ls=:solid, agg=aggregate(baseline_rows)), + ) + push!(series, + (label="$fam · accelerated", color=color, marker=marker, + ls=:dash, agg=aggregate(accelerated_rows))) + end + else + push!( + series, + (label=fam, color=color, marker=marker, ls=:solid, + agg=aggregate(reduce(vcat, runs))), + ) + end + end + return series +end + +function throughput_label(bench) + return bench == "grayscott" ? + "Throughput (Gpoint-updates/s)" : "Throughput (GFLOP/s)" +end + +function addline!(p, s, y; kw...) + return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, + lw=2.2, ls=s.ls, marker=s.marker, ms=6, msc=s.color, markerstrokewidth=0.8, + label=s.label, kw...) +end + +# One legend key: a short line sample (+ optional marker) with a text label. +function swatch!(p, x, y, color, ls, marker, label) + plot!(p, [x, x + 0.032], [y, y]; color=color, lw=2.6, ls=ls, label="") + marker !== nothing && scatter!(p, [x + 0.016], [y]; color=color, marker=marker, + ms=6, msc=color, markerstrokewidth=0.8, label="") + return annotate!(p, x + 0.045, y, text(label, 9, INK, :left)) +end + +# Grouped legend: color/marker = implementation, line style = code path. +function build_legend(series) + pl = plot(; framestyle=:none, legend=false, xlims=(0, 1), ylims=(0, 1), + grid=false, ticks=false) + # implementations present, in FAMILIES order, matched by color + present = [ + (fam, color, marker) for (key, fam, color, marker, _) in FAMILIES + if any(s.color == color for s in series) + ] + annotate!(pl, 0.015, 0.74, text("Implementation", 10, INK, :left)) + xs = range(0.18, 0.80; length=max(length(present), 1)) + for ((fam, color, marker), x) in zip(present, xs) + swatch!(pl, x, 0.74, color, :solid, marker, fam) + end + annotate!(pl, 0.015, 0.26, text("Line style", 10, INK, :left)) + has_baseline = any(endswith(s.label, "· baseline") for s in series) + has_accelerated = any(endswith(s.label, "· accelerated") for s in series) + if has_baseline && has_accelerated + swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") + swatch!(pl, 0.40, 0.26, MUTED, :dash, nothing, "@accelerate") + swatch!(pl, 0.70, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + elseif has_accelerated + swatch!(pl, 0.18, 0.26, MUTED, :dash, nothing, "@accelerate") + swatch!(pl, 0.52, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + elseif has_baseline + swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") + swatch!(pl, 0.48, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + else + swatch!(pl, 0.18, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + end + return pl +end + +function positive_ylim(vals; pad=0.12) + isempty(vals) && return (0, 1) + hi = maximum(vals) + hi > 0 || return (0, 1) + return (0, hi * (1 + pad)) +end + +function main() + mkpath(OUT_DIR) + files = filter(f -> endswith(f, ".csv"), readdir(RESULTS_DIR)) + benches = unique( + String[ + m.captures[1] for f in files for (key, _, _, _, _) in FAMILIES + for m in (match(Regex("^(.*)_" * replace(key, "." => "\\.") * "\\.csv\$"), f),) + if m !== nothing + ], + ) + + for bench in benches + series = series_for(bench) + if HIDE_BASELINE + series = filter(s -> !endswith(s.label, "· baseline"), series) + end + isempty(series) && continue + + common = (xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", + framestyle=:box, grid=true, gridcolor=GRIDCOL, gridalpha=1.0, + foreground_color_text=INK, tickfontcolor=MUTED, legend=false, + xlims=(0.85, 9.4)) + + # Panel 1: throughput (higher better) + throughput = [x.h for s in series for x in s.agg] + p1 = plot(; ylabel=throughput_label(bench), title="Throughput", + ylims=positive_ylim(throughput), common...) + for s in series + addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) + end + + # Panel 2: time per step (lower better; ideal = flat) + p2 = plot(; ylabel="Time / step (ms)", title="Time per step", common...) + for s in series + addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) + end + + # Panel 3: parallel efficiency = thr(p)/(p*thr(1)); ideal = 1.0 + efficiencies = Float64[] + for s in series + i1 = findfirst(x -> x.gpus == 1, s.agg) + i1 === nothing && continue + base = s.agg[i1].h + append!(efficiencies, [x.h/(x.gpus*base) for x in s.agg]) + end + p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", + ylims=positive_ylim(vcat(efficiencies, [1.0])), common...) + hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.4, label="") + for s in series + i1 = findfirst(x -> x.gpus == 1, s.agg) + i1 === nothing && continue + base = s.agg[i1].h + addline!(p3, s, [x.h/(x.gpus*base) for x in s.agg]) + end + + # grouped legend panel: color/marker = implementation, style = code path + pl = build_legend(series) + has_baseline = any(endswith(s.label, "· baseline") for s in series) + has_accelerated = any(endswith(s.label, "· accelerated") for s in series) + style_title = if has_baseline && has_accelerated + "solid = baseline · dashed = @accelerate" + elseif has_accelerated + "dashed = @accelerate" + elseif has_baseline + "solid = baseline" + else + "implementation comparison" + end + + fig = plot(p1, p2, p3, pl; layout=@layout([grid(1, 3); leg{0.16h}]), + size=(1400, 600), dpi=200, + plot_title=titlecase(bench) * " — weak scaling ($style_title)", + plot_titlefontsize=12, left_margin=6Plots.mm, + bottom_margin=6Plots.mm, top_margin=4Plots.mm, + background_color="#fcfcfb") + + out = joinpath(OUT_DIR, "$(bench)_weak_scaling$(OUTPUT_SUFFIX).png") + savefig(fig, out) + println("wrote $out") + end +end + +main() diff --git a/docs/src/api_binary.md b/docs/src/api_binary.md index 7285ae15e..ae6d72e82 100644 --- a/docs/src/api_binary.md +++ b/docs/src/api_binary.md @@ -6,7 +6,13 @@ The following binary operations are supported and can be applied elementwise to pairs of `NDArray` values: -- `+`, `-`, `*`, `/`, `^`, `<`, `<=`, `>`, `>=`, `==`, `!=`, `atan`, `hypot`, `max`, `min`, `lcm`, `gcd` +- `+`, `-`, `*`, `/`, `^`, `%`, `<`, `<=`, `>`, `>=`, `==`, `!=`, `&`, `|`, `⊻`, `<<`, `>>`, `atan`, `hypot`, `max`, `min`, `lcm`, `gcd`, `fld`, `mod`, `rem`, `copysign` + +## Differences from Base Julia + +- `div` / `÷` are not provided. cuPyNumeric floor-divide matches Julia `fld` (toward `-Inf`), not truncated `div`. For example `-7 ÷ 2` is `-3` in Julia and `-4` for `fld`. +- `&`, `|`, `⊻` are integer and `Bool` bitwise ops. `<<` and `>>` are integers excluding `Bool`. +- `copysign` is float-only. ```@autodocs Modules = [cuNumeric] diff --git a/docs/src/api_unary.md b/docs/src/api_unary.md index e2099a911..da783bccf 100644 --- a/docs/src/api_unary.md +++ b/docs/src/api_unary.md @@ -6,11 +6,14 @@ The following unary operations are supported and can be broadcast over `NDArray`: -- `-`, `!`, `abs`, `acos`, `acosh`, `asin`, `asinh`, `atan`, `atanh`, `cbrt`, `conj`, `cos`, `cosh`, `deg2rad`, `exp`, `exp2`, `expm1`, `floor`, `imag`, `isfinite`, `log`, `log10`, `log1p`, `log2`, `rad2deg`, `real`, `sign`, `signbit`, `sin`, `sinh`, `sqrt`, `tan`, `tanh`, `^2`, `^-1` or `inv` +- `-`, `!`, `~`, `abs`, `acos`, `acosh`, `asin`, `asinh`, `atan`, `atanh`, `cbrt`, `ceil`, `conj`, `cos`, `cosh`, `deg2rad`, `exp`, `exp2`, `expm1`, `floor`, `imag`, `isfinite`, `isinf`, `isnan`, `log`, `log10`, `log1p`, `log2`, `rad2deg`, `real`, `round`, `sign`, `signbit`, `sin`, `sinh`, `sqrt`, `tan`, `tanh`, `trunc`, `^2`, `^-1` or `inv` ## Differences from Base Julia - The `acosh` function in Julia will error on inputs outside of the domain (`x >= 1`), but cuNumeric.jl will return `NaN`. +- `round.(A)` uses IEEE round-to-nearest-even (the `RINT` kernel), matching Julia `round(x)` / `RoundNearest`. Only that 1-arg path is wired. `digits`, `sigdigits`, and `RoundingMode` are not supported (`round.(A; digits=n)` errors). +- `floor`, `ceil`, `trunc`, and `signbit` are float-only kernels. `round` also supports complex values. Bool and integer inputs are not accepted (Julia's `floor`/`ceil`/`trunc`/`round` on integers are identity). +- `~` is bitwise not on integers. On `Bool` it matches `!` (the invert kernel rejects `Bool`, so we use logical not). ```@autodocs Modules = [cuNumeric] diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 5631cfa5f..5428f701a 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -1,11 +1,11 @@ # Still missing: -# # Base.copysign => cuNumeric.COPYSIGN, #* ANNOYING TO TEST -# #missing => cuNumeric.fmod, #same as mod in Julia? -# # Base.isapprox => cuNumeric.ISCLOSE, #* HANDLE rtol, atol kwargs!!! -# # Base.ldexp => cuNumeric.LDEXP, #* LHS FLOATS, RHS INTS -# #missing => cuNumeric.LOGADDEXP, -# #missing => cuNumeric.LOGADDEXP2, -# #missing => cuNumeric.NEXTAFTER, +# # Base.isapprox => cuNumeric.ISCLOSE, # rtol, atol kwargs +# # Base.ldexp => cuNumeric.LDEXP, # LHS floats, RHS ints +# # missing => cuNumeric.LOGADDEXP, +# # missing => cuNumeric.LOGADDEXP2, +# # missing => cuNumeric.NEXTAFTER, +# # Base.div / ÷ — FLOOR_DIVIDE matches Julia `fld` (toward -Inf), not +# # truncated `div`. Do not ship as `div`: `-7 ÷ 2` is -3 in Julia, -4 for fld. # Binary ops which are equivalent to Julia's broadcast syntax global const binary_op_map = Dict{Function,BinaryOpCode}( @@ -24,14 +24,18 @@ global const binary_op_map = Dict{Function,BinaryOpCode}( Base.:(==) => cuNumeric.EQUAL, #* BE SURE TO DEFINE NON-BROADCASTED VERSION (BINARY_REDUCTION), Base.lcm => cuNumeric.LCM, Base.gcd => cuNumeric.GCD, - # Base.xor => cuNumeric.LOGICAL_XOR, #! DO LATER - # Base.:⊻ => cuNumeric.LOGICAL_XOR, #! DO LATER - # Base.div => cuNumeric.FLOOR_DIVIDE, #! THESE ARE IN-EXACT FOR INTS? - # Base.:(÷) => cuNumeric.FLOOR_DIVIDE, #! THESE ARE IN-EXACT FOR INTS? - # Base.:(>>) => cuNumeric.RIGHT_SHIFT, #! DO LATER - # Base.:(<<) => cuNumeric.LEFT_SHIFT, #! DO LATER - # Base.:(&&) => (cuNumeric.LOGICAL_AND, Bool, :same_as_input), #! CANNOT OVERLOAD WTF? (see Base.andand) - # Base.:(||) => (cuNumeric.LOGICAL_OR, Bool, :same_as_input), #! CANNOT OVERLOAD WTF? + Base.:(&) => cuNumeric.BITWISE_AND, # integers and Bool + Base.:(|) => cuNumeric.BITWISE_OR, + Base.:(⊻) => cuNumeric.BITWISE_XOR, + Base.:(<<) => cuNumeric.LEFT_SHIFT, # integers, not Bool + Base.:(>>) => cuNumeric.RIGHT_SHIFT, # integers, not Bool + Base.fld => cuNumeric.FLOOR_DIVIDE, # matches Julia fld, not div/÷ + Base.mod => cuNumeric.MOD, + Base.rem => cuNumeric.FMOD, + Base.:(%) => cuNumeric.FMOD, # Julia `%` is rem + Base.copysign => cuNumeric.COPYSIGN, # floats only + # Base.:(&&) => (cuNumeric.LOGICAL_AND, Bool, :same_as_input), # cannot overload (see Base.andand) + # Base.:(||) => (cuNumeric.LOGICAL_OR, Bool, :same_as_input), # cannot overload ) global const floaty_binary_op_map = Dict{Function,BinaryOpCode}( diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 743ccbb64..d67586ee4 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -26,19 +26,24 @@ global const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( global const unary_op_map_no_args = Dict{Function,UnaryOpCode}( Base.abs => cuNumeric.ABSOLUTE, - # Base.conj => cuNumeric.CONJ, #! NEED TO SUPPORT COMPLEX TYPES FIRST + # Base.conj => cuNumeric.CONJ, # handled as a special case below Base.:(-) => cuNumeric.NEGATIVE, - # Base.frexp => cuNumeric.FREXP, #* annoying returns tuple - # missing => cuNumeric.GETARG, #not in numpy? - # Base.imag => cuNumeric.IMAG, #! NEED TO SUPPORT COMPLEX TYPES FIRST - # missing => cuNumerit.INVERT, # no bitwise not in julia? - # Base.isfinite => cuNumeric.ISFINITE, #* dont feel like looking into Inf rn - # Base.isinf => cuNumeric.ISINF, #* dont feel like looking into Inf rn - # Base.isnan => cuNumeric.ISNAN, #* dont feel like looking into Inf rn - # Base.modf => cuNumeric.MODF, #* annoying returns tuple - #missing => cuNumeric.POSITIVE, #What is this even for + # Base.frexp => cuNumeric.FREXP, # returns a tuple + # missing => cuNumeric.GETARG, + # Base.imag => cuNumeric.IMAG, # handled as a special case below + Base.:(~) => cuNumeric.INVERT, # integers only; kernel rejects Bool + Base.isfinite => cuNumeric.ISFINITE, + Base.isinf => cuNumeric.ISINF, + Base.isnan => cuNumeric.ISNAN, + # Base.modf => cuNumeric.MODF, # returns a tuple + # missing => cuNumeric.POSITIVE, Base.sign => cuNumeric.SIGN, - # Base.signbit => cuNumeric.SIGNBIT, #! Doesnt support Bool, I do not feel like dealing with this right now... + Base.signbit => cuNumeric.SIGNBIT, # floats only; kernel rejects Bool/int + Base.ceil => cuNumeric.CEIL, # floats only + Base.floor => cuNumeric.FLOOR, # floats only + Base.trunc => cuNumeric.TRUNC, # floats only + # 1-arg `round.(A)` only (see __broadcast below). ROUND needs extra_args. + Base.round => cuNumeric.RINT, ) ### SPECIAL CASES ### @@ -105,6 +110,17 @@ function Base.:(-)(input::NDArray{Bool}) return out end +# Broadcast `.-` on Bool: Julia `-true === -1`, so promote then NEGATIVE. +@inline function __broadcast( + ::typeof(Base.:(-)), out::NDArray{O}, input::NDArray{Bool} +) where {O<:Integer} + assertpromotion(".-", Bool, O) + promoted = unchecked_promote_arr(input, O) + result = nda_unary_op!(out, cuNumeric.NEGATIVE, promoted) + destroy!(promoted) + return result +end + function Base.sqrt(input::NDArray{T,2}) where {T} return error("cuNumeric.jl does not support matrix square root.") end @@ -169,6 +185,32 @@ for (julia_fn, op_code) in unary_op_map_no_args end end +# INVERT rejects Bool; on Bool, Julia `~` is the same as `!`. +@inline function __broadcast(::typeof(Base.:(~)), out::NDArray{Bool}, input::NDArray{Bool}) + return nda_unary_op!(out, cuNumeric.LOGICAL_NOT, input) +end + +@noinline function _unsupported_round_broadcast() + return throw( + ArgumentError( + "cuNumeric.jl only supports round.(A) (default RoundNearest / IEEE rint). " * + "digits, sigdigits, and RoundingMode are not supported.", + ), + ) +end + +# Reject keyword forms before fusion captures Julia's keyword wrapper as a GPU callable. +@inline function Base.Broadcast.broadcasted_kwsyntax( + ::typeof(Base.round), ::NDArray; kwargs... +) + return _unsupported_round_broadcast() +end + +# Only the 1-arg `round.(A)` method above is supported (IEEE rint / RoundNearest). +@inline function __broadcast(::typeof(Base.round), ::NDArray, ::NDArray, extra...) + return _unsupported_round_broadcast() +end + # Some functions always return floats even when given integers # in the case where the output is determined to be float, but # the input is integer, we first promote the input to float. @@ -192,29 +234,9 @@ for (julia_fn, op_code) in floaty_unary_ops_no_args end end -# global const unary_op_map_with_args = Dict{Function, Int}( -# Base.angle => Int(cuNumeric.ANGLE), -# Base.ceil => Int(cuNumeric.CEIL), #* HAS EXTRA ARGS -# Base.clamp => Int(cuNumeric.CLIP), #* HAS EXTRA ARGS -# Base.floor => cuNumeric.FLOOR, #! Doesnt support Bool, I do not feel like dealing with this right now... -# Base.trunc => Int(cuNumeric.TRUNC) #* HAS EXTRA ARGS -# missing => Int(cuNumeric.RINT), #figure out which version of round -# missing => Int(cuNumeric.ROUND), #figure out which version of round -# ) - -# for (base_func, op_code) in unary_op_map_with_args -# @eval begin -# @doc """ -# $($(Symbol(base_func))) : A unary operation acting on an NDArray -# """ -# function $(Symbol(base_func))(input::NDArray, args...) -# out = cuNumeric.zeros(eltype(input), size(input)) # not sure this is ok for performance -# extra_args = cuNumeric.StdVector{cuNumeric.LegateScalar}([LegateScalar(a) for a in args]) -# unary_op(out, $(op_code), input, extra_args) -# return out -# end -# end -# end +# CLIP / clamp needs extra_args (lo, hi). nda_unary_op! currently ccall's +# without extra scalars, so clamp is not wired. Do not revive the old +# StdVector{LegateScalar} path unless that C API exists. @doc""" Supported Unary Reduction Operations diff --git a/test/array/binary/tests.jl b/test/array/binary/tests.jl index 794c2e19a..de1d2f13a 100644 --- a/test/array/binary/tests.jl +++ b/test/array/binary/tests.jl @@ -36,11 +36,33 @@ function test_binary_operation(func, julia_arr1, julia_arr2, cunumeric_arr1, cun end function test_binary_function_set(func_dict, T, N) - skip = (Base.lcm, Base.gcd) + # Random data includes zeros / huge shift amounts; these are tested separately. + skip = ( + Base.lcm, Base.gcd, Base.fld, Base.mod, Base.rem, Base.:(%), Base.:(<<), Base.:(>>) + ) # not defined for complex. skip_on_complex = ( - Base.:(<), Base.:(<=), Base.:(>), Base.:(>=), Base.max, Base.min, Base.atan, Base.hypot + Base.:(<), + Base.:(<=), + Base.:(>), + Base.:(>=), + Base.max, + Base.min, + Base.atan, + Base.hypot, + Base.:(&), + Base.:(|), + Base.:(⊻), + Base.copysign, + Base.fld, + Base.mod, + Base.rem, + Base.:(%), + Base.:(<<), + Base.:(>>), ) + skip_on_float = (Base.:(&), Base.:(|), Base.:(⊻), Base.:(<<), Base.:(>>)) + skip_on_integer = (Base.copysign,) # kernel is float-only; Bool <: Integer @testset "$func" for func in keys(func_dict) @@ -51,6 +73,14 @@ function test_binary_function_set(func_dict, T, N) continue end + if T <: AbstractFloat && (func in skip_on_float) + continue + end + + if T <: Integer && (func in skip_on_integer) + continue + end + (func in skip) && continue arrs_jl = make_julia_arrays(T, N, :uniform; count=2) @@ -98,6 +128,55 @@ function run_binary_ops_tests(types) end end + # fld/mod/rem/% need a non-zero divisor. FLOOR_DIVIDE is fld, not div. + if (T <: cuNumeric.SUPPORTED_INT_TYPES && T != Bool) || T <: AbstractFloat + if T <: Integer + arr_jl_div = my_rand(T, N; L=(T <: Unsigned ? 1 : -20), R=20) + arr_jl_den = my_rand(T, N; L=1, R=20) + else + arr_jl_div = my_rand(T, N) + arr_jl_den = my_rand(T, N) + arr_jl_den = map( + x -> abs(x) < T(1) ? copysign(one(T), iszero(x) ? one(T) : x) : x, + arr_jl_den, + ) + end + arr_cn_div = @allowscalar NDArray(arr_jl_div) + arr_cn_den = @allowscalar NDArray(arr_jl_den) + + allowscalar() do + for func in (fld, mod, rem, Base.:%) + @test safe_compare( + func.(arr_jl_div, arr_jl_den), func.(arr_cn_div, arr_cn_den), + atol(T), rtol(T), + ) + end + end + end + + # Shifts: small non-negative amounts. Left shift uses a small lhs to + # avoid C++ undefined behavior on signed overflow. + if T <: cuNumeric.SUPPORTED_INT_TYPES && T != Bool + max_shift = T(min(7, 8 * sizeof(T) - 1)) + arr_jl_sh = my_rand(T, N; L=0, R=max_shift) + arr_jl_lshift_lhs = my_rand(T, N; L=0, R=7) + arr_jl_rshift_lhs = my_rand(T, N) + arr_cn_sh = @allowscalar NDArray(arr_jl_sh) + arr_cn_lshift_lhs = @allowscalar NDArray(arr_jl_lshift_lhs) + arr_cn_rshift_lhs = @allowscalar NDArray(arr_jl_rshift_lhs) + + allowscalar() do + @test safe_compare( + arr_jl_lshift_lhs .<< arr_jl_sh, arr_cn_lshift_lhs .<< arr_cn_sh, + atol(T), rtol(T), + ) + @test safe_compare( + arr_jl_rshift_lhs .>> arr_jl_sh, arr_cn_rshift_lhs .>> arr_cn_sh, + atol(T), rtol(T), + ) + end + end + allowscalar() do @test unwrap(arr_cn == arr_cn) @test !unwrap(arr_cn == arr_cn2) diff --git a/test/array/unary/tests.jl b/test/array/unary/tests.jl index 558a2c877..3257c0ab0 100644 --- a/test/array/unary/tests.jl +++ b/test/array/unary/tests.jl @@ -47,14 +47,19 @@ function test_unary_operation(func, julia_arr, cunumeric_arr, T) end end -skip_on_integer = (Base.acosh, Base.atanh, Base.atan, Base.acos, Base.asin) -skip_on_bool = (Base.:(-), skip_on_integer...) +skip_on_integer = ( + Base.acosh, Base.atanh, Base.atan, Base.acos, Base.asin, + Base.ceil, Base.floor, Base.trunc, Base.round, Base.signbit, +) +skip_on_bool = skip_on_integer skip_on_complex = ( Base.tanh, Base.deg2rad, Base.rad2deg, Base.sign, Base.cbrt, Base.exp2, Base.expm1, Base.log10, Base.log1p, Base.log2, Base.acos, Base.asin, Base.atan, Base.acosh, Base.asinh, Base.atanh, + Base.ceil, Base.floor, Base.trunc, Base.signbit, Base.:(~), ) +skip_on_float = (Base.:(~),) function test_unary_function_set(func_dict, T, N) default_generator = (T == Bool) ? :uniform : :unit_interval @@ -73,6 +78,10 @@ function test_unary_function_set(func_dict, T, N) continue end + if func in skip_on_float && (T <: AbstractFloat) + continue + end + domain_type = get(SPECIAL_DOMAINS, func, default_generator) # :uniform is the only generator capable of generating bits @@ -126,6 +135,13 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) allowpromotion(T == Bool) do return test_unary_function_set(cuNumeric.unary_op_map_no_args, T, N) end + + if T <: AbstractFloat + @testset "round is 1-arg only" begin + a = cuNumeric.ones(T, 4) + @test_throws ArgumentError round.(a; digits=1) + end + end # Special cases for unary ops that dont use . syntax @testset "- (Negation)" begin arr = my_rand(T, N) From 1b18c5533c75d35f39bd85e1a1f064a9fb64efe9 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sat, 15 Aug 2026 13:32:57 -0400 Subject: [PATCH 08/49] Support argmin, argmax, var, std, mean (#178) * support var, std, argmin, argmax --------- Co-authored-by: David Krasowska --- .buildkite/run_developer_ci.sh | 2 +- docs/src/api_unary.md | 3 + .../include/ndarray_c_api.h | 3 + lib/cunumeric_jl_wrapper/src/ndarray.cpp | 154 ++++++++++++++- src/cuNumeric.jl | 4 +- src/ndarray/unary.jl | 187 +++++++++++++++--- test/array/unary/tests.jl | 143 +++++++++++++- 7 files changed, 461 insertions(+), 35 deletions(-) diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 402db7115..29d7db738 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -20,7 +20,7 @@ sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" export PATH="$CMAKE_ROOT/bin:$PATH" cmake --version -# Clean slate so cached state doesn't leak across Julia versions. +# Clean slate so cached state doesn't leak across Julia versions. rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml diff --git a/docs/src/api_unary.md b/docs/src/api_unary.md index da783bccf..3ea0442b3 100644 --- a/docs/src/api_unary.md +++ b/docs/src/api_unary.md @@ -14,6 +14,9 @@ The following unary operations are supported and can be broadcast over `NDArray` - `round.(A)` uses IEEE round-to-nearest-even (the `RINT` kernel), matching Julia `round(x)` / `RoundNearest`. Only that 1-arg path is wired. `digits`, `sigdigits`, and `RoundingMode` are not supported (`round.(A; digits=n)` errors). - `floor`, `ceil`, `trunc`, and `signbit` are float-only kernels. `round` also supports complex values. Bool and integer inputs are not accepted (Julia's `floor`/`ceil`/`trunc`/`round` on integers are identity). - `~` is bitwise not on integers. On `Bool` it matches `!` (the invert kernel rejects `Bool`, so we use logical not). +- Full reductions (`sum`, `mean`, `var`, `std`, `argmax`, …) return a 0-d `NDArray`, not a Julia scalar. Use `unwrap` or `A[]` (with `allowscalar`) to read a host value. +- `var` / `std` match Julia / StatsBase sample statistics (`corrected=true`, divisor `n-1`). Complex is not supported. +- `argmax` / `argmin` are 1-d only (matching Base's `Int` return, not `CartesianIndex`). The result is a 0-d `NDArray{Int64}` of the 1-based index. Complex is not supported. ```@autodocs Modules = [cuNumeric] diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index b13ddd1ac..dd612d188 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -91,6 +91,9 @@ void nda_unary_op(CN_NDArray* out, CuPyNumericUnaryOpCode op_code, CN_NDArray* input); void nda_unary_reduction(CN_NDArray* out, CuPyNumericUnaryRedCode op_code, CN_NDArray* input); +CN_NDArray* nda_unary_reduction_axes(CuPyNumericUnaryRedCode op_code, + CN_NDArray* input, const int32_t* axes, + int32_t num_axes, bool keepdims); CN_NDArray* nda_get_slice(CN_NDArray* arr, const CN_Slice* slices, int32_t ndim); CN_NDArray* nda_attach_external(const void* ptr, size_t size, int dim, diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index 9a925abc0..5fb5190bf 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -30,8 +30,11 @@ #include #include #include +#include #include +#include #include +#include #include #include "ndarray_c_api.h" @@ -237,18 +240,157 @@ void nda_unary_reduction(CN_NDArray* out, CuPyNumericUnaryRedCode op_code, out->obj.unary_reduction(op_code, input->obj); } +// COUNT_NONZERO's result is an integer count, not the input dtype. +// Passing res_dtype (not dtype) keeps the source type. dtype=int64 would +// cast the input before counting. +static std::optional unary_red_res_dtype( + CuPyNumericUnaryRedCode op_code) { + switch (op_code) { + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_COUNT_NONZERO: + return legate::int64(); + default: + return std::nullopt; + } +} + +static bool is_arg_reduction(CuPyNumericUnaryRedCode op_code) { + switch (op_code) { + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX: + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMIN: + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX: + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMIN: + return true; + default: + return false; + } +} + +} // extern "C" + +// Matches cupynumeric Argval { int64_t arg; T arg_value; }. +// Templates cannot live in the extern "C" block. +template +struct ArgvalCompat { + int64_t arg; + T arg_value; +}; + +template +static void fill_arg_identity(NDArray& acc, const legate::Type& argred_type, + bool is_argmax) { + ArgvalCompat id; + id.arg = std::numeric_limits::min(); + id.arg_value = is_argmax ? std::numeric_limits::lowest() + : std::numeric_limits::max(); + if (argred_type.size() != sizeof(id)) { + throw std::runtime_error("argred identity layout mismatch"); + } + acc.fill(Scalar(argred_type, &id, true)); +} + +static void fill_arg_identity(NDArray& acc, const legate::Type& src_type, + const legate::Type& argred_type, bool is_argmax) { + switch (src_type.code()) { + case legate::Type::Code::BOOL: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT8: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT16: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT32: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT64: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT8: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT16: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT32: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT64: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::FLOAT32: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::FLOAT64: + fill_arg_identity(acc, argred_type, is_argmax); + break; + default: + throw std::runtime_error( + "argmax/argmin are not supported for this element type"); + } +} + +// Public unary_reduction fills identity via type_dispatch on the *output* +// type. For ARGMAX that output is a struct, which Legate rejects. Fill from +// the source dtype (like Python), then launch SCALAR_UNARY_RED and GETARG. +static CN_NDArray* nda_arg_reduction(CuPyNumericUnaryRedCode op_code, + CN_NDArray* input) { + const bool is_argmax = + op_code == CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX || + op_code == CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX; + auto* runtime = cupynumeric::CuPyNumericRuntime::get_runtime(); + const std::vector scalar_shape{}; + auto argred_type = runtime->get_argred_type(input->obj.type()); + // C++ get_argred_type does not attach redops; Python does this on first + // use. record_reduction_operator throws if called twice for the same type. + static std::unordered_set registered_argred; + if (registered_argred.insert(input->obj.type().code()).second) { + auto ids = cupynumeric_register_reduction_ops( + static_cast(input->obj.type().code())); + argred_type.record_reduction_operator( + legate::ReductionOpKind::MAX, + legate::GlobalRedopID{ids.argmax_redop_id}); + argred_type.record_reduction_operator( + legate::ReductionOpKind::MIN, + legate::GlobalRedopID{ids.argmin_redop_id}); + } + NDArray acc = runtime->create_array(scalar_shape, argred_type); + fill_arg_identity(acc, input->obj.type(), argred_type, is_argmax); + + auto task = + runtime->create_task(CuPyNumericOpCode::CUPYNUMERIC_SCALAR_UNARY_RED); + task.add_reduction(acc.get_store(), is_argmax ? legate::ReductionOpKind::MAX + : legate::ReductionOpKind::MIN); + task.add_input(input->obj.get_store()); + task.add_scalar_arg(Scalar(static_cast(op_code))); + task.add_scalar_arg(Scalar(input->obj.shape())); + task.add_scalar_arg(Scalar(false)); + runtime->submit(std::move(task)); + + NDArray idx = runtime->create_array(scalar_shape, legate::int64()); + idx.unary_op( + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_GETARG), + acc); + return new CN_NDArray{NDArray(std::move(idx))}; +} + +extern "C" { + CN_NDArray* nda_unary_reduction_axes(CuPyNumericUnaryRedCode op_code, CN_NDArray* input, const int32_t* axes, int32_t num_axes, bool keepdims) { + if (is_arg_reduction(op_code)) { + return nda_arg_reduction(op_code, input); + } std::vector axis_vec(axes, axes + num_axes); NDArray result = input->obj._perform_unary_reduction( static_cast(op_code), input->obj, axis_vec, - std::nullopt, // dtype - std::nullopt, // res_dtype - std::nullopt, // out - keepdims, {}, // args - std::nullopt, // initial - std::nullopt // where + std::nullopt, // dtype + unary_red_res_dtype(op_code), // res_dtype + std::nullopt, // out + keepdims, {}, // args + std::nullopt, // initial + std::nullopt // where ); return new CN_NDArray{NDArray(std::move(result))}; } diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index e05a6e191..0204b3b46 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -38,7 +38,7 @@ using cunumeric_jl_wrapper_jll import Base: axes, convert, copy, copyto!, inv, isfinite, sqrt, -, +, *, ==, !=, isapprox, read, view, maximum, minimum, prod, sum, getindex, setindex!, - sum, prod + sum, prod, argmax, argmin using LinearAlgebra import LinearAlgebra: mul! @@ -49,7 +49,7 @@ import Random: rand!, randn!, randexp! using StaticArrays: SVector using StatsBase -import StatsBase: var, mean +import StatsBase: var, mean, std include(joinpath(@__DIR__, "../deps/version.jl")) include("utilities/preference.jl") diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index d67586ee4..ef234d694 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -250,11 +250,18 @@ The following unary reduction operations are supported and can be applied direct • `minimum` • `prod` • `sum` + • `mean` + • `var` / `std` (sample / `corrected=true`; real types only) + • `argmax` / `argmin` (1-d only) -These operations follow standard Julia semantics. +Full reductions return a **0-d `NDArray`**, not a Julia scalar. Use `unwrap` or +`A[]` (with `allowscalar`) when you need a host value. Reduction over specific dimensions is supported via the `dims` keyword argument, -following the same semantics as Julia's base reduction functions. +following the same keepdims semantics as Julia's base reduction functions. +Multi-axis `dims=(1,2)` is implemented as sequential single-axis reductions +(the C++ kernel accepts only one axis at a time). `argmax`/`argmin` are 1-d +only (Base's N-d / `dims=` path returns `CartesianIndex`). Examples -------- @@ -264,6 +271,7 @@ A = cuNumeric.ones(5) maximum(A) sum(A) +mean(A) # Reduce over a specific dimension B = cuNumeric.ones(3, 4) @@ -275,10 +283,8 @@ sum(B, dims=(1,2)) # 1×1 result ``` """ global const unary_reduction_map = Dict{Function,UnaryRedCode}( - # Base.argmax => cuNumeric.ARGMAX, #* WILL BE OFF BY 1 - # Base.argmin => cuNumeric.ARGMIN, #* WILL BE OFF BY 1 + # ARGMAX/ARGMIN: 1-d Base.argmax/argmin below, not this map. #missing => cuNumeric.CONTAINS, # strings or also integral types - #missing => cuNumeric.COUNT_NONZERO, # Base.count(!Base.iszero, arr) Base.maximum => cuNumeric.MAX, Base.minimum => cuNumeric.MIN, #missing => cuNumeric.NANARGMAX, @@ -288,12 +294,10 @@ global const unary_reduction_map = Dict{Function,UnaryRedCode}( #missing => cuNumeric.NANPROD, Base.prod => cuNumeric.PROD, Base.sum => cuNumeric.SUM, - #missing => cuNumeric.SUM_SQUARES, - # StatsBase.var => cuNumeric.VARIANCE #! dies horribly?? wth + # VARIANCE opcode is unused: compose sample var from mean / sum instead. ) -#! IT WOULD BE NICE IF THESE JUST RETURNED SCALARS WHEN APPROPRIATE -# #*TODO HOW TO GET THESE ACTING ON CERTAIN DIMS +# Full reductions return 0-d NDArrays (not Julia scalars). That is intentional. function _unary_reduction_apply(out, op_code, input::NDArray{T}, ::Type{T}) where {T} return nda_unary_reduction(out, op_code, input) @@ -331,17 +335,21 @@ function _unary_reduction_impl(base_func, op_code, input::NDArray{T,N}, dims::In return _unary_reduction_axes_apply(op_code, input, T_OUT, axes) end +# cupynumeric throws if axes.size() > 1. Compose keepdims single-axis reductions +# so `sum(A; dims=(1,2))` matches Julia's 1×1 (etc.) shape. function _unary_reduction_impl(base_func, op_code, input::NDArray{T,N}, dims::Tuple) where {T,N} - if length(dims) > 1 - error( - "$(base_func): reducing over multiple dimensions is not yet supported. Got dims=$dims" - ) + n = length(dims) + n == 0 && return copy(input) + n == 1 && return _unary_reduction_impl(base_func, op_code, input, dims[1]) + result = input + owned = false + for d in dims + next = _unary_reduction_impl(base_func, op_code, result, d) + owned && destroy!(result) + result = next + owned = true end - # single element tuple - T_OUT = Base.promote_op(base_func, Vector{T}) - is_wider_type(T_OUT, T) && assertpromotion(base_func, T, T_OUT) - axes = Int32[dims[1] - 1] - return _unary_reduction_axes_apply(op_code, input, T_OUT, axes) + return result end # Generate code for all unary reductions. @@ -358,9 +366,27 @@ function _bool_reduction_impl(op_code, input::NDArray{Bool}, ::Colon) return nda_unary_reduction(out, op_code, input) end +function _bool_reduction_impl(op_code, input::NDArray{Bool}, dim::Integer) + return nda_unary_reduction_axes(op_code, input, Int32[dim - 1], true) +end + +function _bool_reduction_impl(op_code, input::NDArray{Bool}, dims::Tuple) + n = length(dims) + n == 0 && return copy(input) + n == 1 && return _bool_reduction_impl(op_code, input, dims[1]) + result = input + owned = false + for d in dims + next = _bool_reduction_impl(op_code, result, d) + owned && destroy!(result) + result = next + owned = true + end + return result +end + function _bool_reduction_impl(op_code, input::NDArray{Bool}, dims) - axes = collect(Int32, (d - 1 for d in (dims isa Integer ? (dims,) : dims))) - return nda_unary_reduction_axes(op_code, input, axes, true) + return _bool_reduction_impl(op_code, input, Tuple(dims)) end function Base.all(input::NDArray{Bool}; dims=Colon()) @@ -388,10 +414,125 @@ function Base.prod(input::NDArray{Bool}; dims=Colon()) return _unary_reduction_impl(Base.prod, cuNumeric.ALL, input, dims) end -#! ONLY ADD ONCE REDUCTIONS RETURN A SCALAR -# function StatsBase.mean(arr::NDArray{T}) where T -# return sum(arr) ./ prod(size(arr)) +# Number of elements a reduction with `dims` collapses. Used by mean/var/std. +_reduction_nelem(arr::NDArray, ::Colon) = Int(prod(size(arr))) +_reduction_nelem(arr::NDArray, dim::Integer) = Int(size(arr, dim)) +function _reduction_nelem(arr::NDArray, dims::Tuple) + n = 1 + for d in dims + n *= Int(size(arr, d)) + end + return n +end + +# Divide an NDArray by a count without going through 0-d broadcast, which +# unwraps to a Julia scalar in Broadcast.copy. +function _div_nelem(arr::NDArray{T}, n::Integer) where {T} + FT = float(T) + return (FT(1) / FT(n)) * arr +end + +""" + mean(A::NDArray; dims=:) + +Arithmetic mean of `A`. Full reduction returns a 0-d `NDArray`, not a Julia +scalar. With `dims`, the reduced axes are kept as size 1, matching Base. +""" +function mean(arr::NDArray; dims=Colon()) + s = sum(arr; dims=dims) + result = _div_nelem(s, _reduction_nelem(arr, dims)) + destroy!(s) + return result +end + +""" + var(A::NDArray; corrected=true, mean=nothing, dims=:) + std(A::NDArray; corrected=true, mean=nothing, dims=:) + +Sample variance and standard deviation (`corrected=true`, divisor `n-1`), +matching Julia / StatsBase. Real types only. Returns a 0-d or reduced +`NDArray`, not a Julia scalar. +""" +function var(arr::NDArray{T}; corrected::Bool=true, mean=nothing, dims=Colon()) where {T<:Real} + μ = isnothing(mean) ? cuNumeric.mean(arr; dims=dims) : mean + centered = arr .- μ + isnothing(mean) && μ isa NDArray && destroy!(μ) + sq = centered .^ 2 + destroy!(centered) + s = sum(sq; dims=dims) + destroy!(sq) + n = _reduction_nelem(arr, dims) + denom = corrected ? n - 1 : n + result = _div_nelem(s, denom) + destroy!(s) + return result +end + +function std(arr::NDArray{T}; corrected::Bool=true, mean=nothing, dims=Colon()) where {T<:Real} + return _sqrt_ndarray(var(arr; corrected=corrected, mean=mean, dims=dims)) +end + +# SQRT kernel rejects 0-d (shape [] vs [1]). Wrap the host sqrt back into a 0-d array. +function _sqrt_ndarray(v::NDArray{T,0}) where {T} + s = T(sqrt(unwrap(v))) + destroy!(v) + return NDArray(s) +end + +function _sqrt_ndarray(v::NDArray{T}) where {T} + out = similar(v) + nda_unary_op!(out, cuNumeric.SQRT, v) + destroy!(v) + return out +end + +# function _count_nonzero(input::NDArray, dims) +# nz = input .!= zero(eltype(input)) +# result = sum(nz; dims=dims) +# destroy!(nz) +# return result +# end +# +# """ +# count(A::NDArray{Bool}; dims=:) +# count(!iszero, A::NDArray; dims=:) +# +# Count `true` values in a `Bool` array, or nonzeros in a numeric array. +# Returns a 0-d or reduced `NDArray` of integers, not a Julia `Int`. +# """ +# function count(arr::NDArray{Bool}; dims=Colon()) +# return _count_nonzero(arr, dims) # end +# +# function count(::ComposedFunction{typeof(!),typeof(iszero)}, arr::NDArray; dims=Colon()) +# return _count_nonzero(arr, dims) +# end + +# Kernel ARGMAX/ARGMIN are 0-based. Julia indices are 1-based. +function _indices_to_one_based(raw::NDArray{Int64}) + result = nda_add_scalar(raw, Int64(1)) + destroy!(raw) + return result +end + +""" + argmax(A::NDArray{<:Any,1}) + argmin(A::NDArray{<:Any,1}) + +1-based index of the first extremum, as a 0-d `NDArray{Int64}` (not a Julia +`Int`). 1-d only. Complex arrays are not supported. +""" +function argmax(arr::NDArray{T,1}) where {T} + T <: Complex && throw(ArgumentError("argmax/argmin are not supported for complex arrays")) + raw = nda_unary_reduction_axes(cuNumeric.ARGMAX, arr, Int32[], false) + return _indices_to_one_based(raw) +end + +function argmin(arr::NDArray{T,1}) where {T} + T <: Complex && throw(ArgumentError("argmax/argmin are not supported for complex arrays")) + raw = nda_unary_reduction_axes(cuNumeric.ARGMIN, arr, Int32[], false) + return _indices_to_one_based(raw) +end # function Base.reduce(f::Function, arr::NDArray) # return f(arr) diff --git a/test/array/unary/tests.jl b/test/array/unary/tests.jl index 3257c0ab0..509ee9eb3 100644 --- a/test/array/unary/tests.jl +++ b/test/array/unary/tests.jl @@ -115,10 +115,132 @@ function test_unary_reduction_dims( end end - # we are testing a multi axis reduction. This will throw a runtime error. - # https://github.com/nv-legate/cupynumeric/blob/main/src/cupynumeric/ndarray.cc#L1132 + # Multi-axis is sequential single-axis keepdims reductions (C++ allows one axis). + # Integer `prod` of 10×10 random values overflows Int64; association then + # disagrees with Julia, so skip that case (1D/single-axis prod is still tested). + if N >= 2 && !(func === Base.prod && T <: Base.BitInteger) + julia_res = func(julia_arr; dims=(1, 2)) + cunumeric_res = func(cunumeric_arr; dims=(1, 2)) + n = size(julia_arr, 1) * size(julia_arr, 2) + scale = maximum(abs, julia_arr) + allowscalar() do + @test safe_compare( + julia_res, + cunumeric_res, + reduction_atol(T, n, scale), + reduction_rtol(T, n), + ) + end + end + end +end + +function test_mean_var_std( + julia_arr::AbstractArray{T,N}, cunumeric_arr::NDArray{T,N} +) where {T,N} + n = length(julia_arr) + scale = maximum(abs, julia_arr) + atolv = reduction_atol(T, n, scale) + rtolv = reduction_rtol(T, n) + allowpromotion(true) do + allowscalar() do + @test isapprox(mean(julia_arr), unwrap(mean(cunumeric_arr)); atol=atolv, rtol=rtolv) + if T <: Real + @test isapprox(var(julia_arr), unwrap(var(cunumeric_arr)); atol=atolv, rtol=rtolv) + @test isapprox(std(julia_arr), unwrap(std(cunumeric_arr)); atol=atolv, rtol=rtolv) + end + end + for d in 1:N + nd = size(julia_arr, d) + atol_d = reduction_atol(T, nd, scale) + rtol_d = reduction_rtol(T, nd) + allowscalar() do + @test safe_compare( + mean(julia_arr; dims=d), mean(cunumeric_arr; dims=d), atol_d, rtol_d + ) + if T <: Real + @test safe_compare( + var(julia_arr; dims=d), var(cunumeric_arr; dims=d), atol_d, rtol_d + ) + @test safe_compare( + std(julia_arr; dims=d), std(cunumeric_arr; dims=d), atol_d, rtol_d + ) + end + end + end if N >= 2 - @test_throws Exception func(cunumeric_arr, dims=(1, 2)) + allowscalar() do + @test safe_compare( + mean(julia_arr; dims=(1, 2)), + mean(cunumeric_arr; dims=(1, 2)), + atolv, + rtolv, + ) + if T <: Real + @test safe_compare( + var(julia_arr; dims=(1, 2)), + var(cunumeric_arr; dims=(1, 2)), + atolv, + rtolv, + ) + @test safe_compare( + std(julia_arr; dims=(1, 2)), + std(cunumeric_arr; dims=(1, 2)), + atolv, + rtolv, + ) + end + end + end + end +end + +# count API is commented out for now (see src/ndarray/unary.jl). +# function test_count(julia_arr::AbstractArray{T,N}, cunumeric_arr::NDArray{T,N}) where {T,N} +# allowpromotion(true) do +# allowscalar() do +# @test count(!iszero, julia_arr) == unwrap(count(!iszero, cunumeric_arr)) +# if T == Bool +# @test count(julia_arr) == unwrap(count(cunumeric_arr)) +# end +# end +# for d in 1:N +# allowscalar() do +# @test safe_compare( +# count(!iszero, julia_arr; dims=d), +# count(!iszero, cunumeric_arr; dims=d), +# 0, +# 0, +# ) +# end +# end +# if N >= 2 +# allowscalar() do +# @test safe_compare( +# count(!iszero, julia_arr; dims=(1, 2)), +# count(!iszero, cunumeric_arr; dims=(1, 2)), +# 0, +# 0, +# ) +# end +# end +# end +# end + +function test_argmax_argmin() + # 1-d only. Fixtures, not random: ties must pick the first extremum. + v = Int32[1, 5, 3, 5, 2] + allowpromotion(true) do + allowscalar() do + vn = NDArray(v) + @test unwrap(argmax(vn)) == 2 + @test unwrap(argmin(vn)) == 1 + @test unwrap(argmax(vn)) == argmax(v) + @test unwrap(argmin(vn)) == argmin(v) + + c = NDArray(ComplexF32[1, 2]) + @test_throws ArgumentError argmax(c) + @test_throws ArgumentError argmin(c) end end end @@ -214,11 +336,17 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) if include_bool_reductions # Test things that only work on Booleans julia_bools = rand(Bool, N) + julia_bools_2D = rand(Bool, isqrt(N), isqrt(N)) allowscalar() do cunumeric_bools = NDArray(julia_bools) @test any(julia_bools) == any(cunumeric_bools)[] @test all(julia_bools) == all(cunumeric_bools)[] end + allowpromotion(true) do + cn2 = @allowscalar NDArray(julia_bools_2D) + test_unary_reduction_dims(any, julia_bools_2D, cn2) + test_unary_reduction_dims(all, julia_bools_2D, cn2) + end end end @@ -250,6 +378,15 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) test_unary_reduction_dims(func, julia_arr_1D, cunumeric_arr_1D) test_unary_reduction_dims(func, julia_arr_2D, cunumeric_arr_2D) end + + @testset "mean/var/std" begin + test_mean_var_std(julia_arr_1D, cunumeric_arr_1D) + test_mean_var_std(julia_arr_2D, cunumeric_arr_2D) + end + end + + @testset "argmax/argmin" begin + test_argmax_argmin() end end end From ceaff8776eaa1740c2b822fdb1798ec91fb2724d Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sun, 16 Aug 2026 14:54:31 -0500 Subject: [PATCH 09/49] use concurrency queues to limit jobs of same type clobbering each other --- .buildkite/developer.pipeline.yml | 5 +++++ .buildkite/run_developer_ci.sh | 30 +++++++++--------------------- 2 files changed, 14 insertions(+), 21 deletions(-) diff --git a/.buildkite/developer.pipeline.yml b/.buildkite/developer.pipeline.yml index 340cf6644..b5e1ac0a4 100644 --- a/.buildkite/developer.pipeline.yml +++ b/.buildkite/developer.pipeline.yml @@ -10,6 +10,11 @@ steps: # concurrent jobs contend on Pkg locks. Separate from the JLL cache. cache_dir: "${HOME}/.cache/julia-buildkite-plugin-developer-{{matrix.julia}}-{{matrix.fusion}}" command: ".buildkite/run_developer_ci.sh" + # Serialize jobs of the same (Julia, fusion) cell cluster-wide so concurrent PRs can't + # clobber that cell's shared developer depot (wrapper overrides + .ji). Different cells + # use different caches and still run in parallel. + concurrency: 1 + concurrency_group: "cunumeric/developer/{{matrix.julia}}/{{matrix.fusion}}" artifact_paths: - "deps/build.log" - "deps/*.log" diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 29d7db738..5be27506e 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -24,27 +24,15 @@ cmake --version rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml -# Isolate each run in a disposable hardlink-clone of the plugin cache (CACHE_DEPOT) so -# concurrent PR jobs on the shared cuda-queue cache can't clobber each other's wrapper -# overrides or .ji (a .ji baked against a stale wrapper .so segfaults). The clone (WORK_DEPOT) -# is near-free on one filesystem and shares the read-only artifact store; every build write -# lands in WORK_DEPOT and is discarded on exit, and CACHE_DEPOT is only ever read. -CACHE_DEPOT="${JULIA_DEPOT_PATH:-}"; CACHE_DEPOT="${CACHE_DEPOT%%:*}" -[[ -n "$CACHE_DEPOT" ]] || CACHE_DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" -WORK_DEPOT="$(mktemp -d "${HOME}/.cache/cn-ci-depot.XXXXXX")" -trap 'rm -rf "$WORK_DEPOT"' EXIT -cp -al "$CACHE_DEPOT/." "$WORK_DEPOT/" -export JULIA_DEPOT_PATH="$WORK_DEPOT" -echo "Isolated build depot: $WORK_DEPOT (hardlink clone of $CACHE_DEPOT)" - -# Reset the cloned wrapper + libcxxwrap wiring so each rebuilds fresh in this depot. Both -# wrapper CMakeLists and build_jlcxxwrap derive libcxxwrap's path from DEPOT_PATH[1] and bake -# absolute paths into JlCxx's cmake config + rpath, so a libcxxwrap built in another run's -# depot can't be reused here — it must be rebuilt in place. Dropping the dev override lets -# `using Legate` fall back to the stock JLL until build_jlcxxwrap rebuilds the custom one. -rm -rf "$WORK_DEPOT"/dev/libcxxwrap_julia_jll \ - "$WORK_DEPOT"/packages/*/*/override \ - "$WORK_DEPOT"/compiled/v*/{CxxWrap,Legate,cuNumeric,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} +# Build directly in the plugin's persistent cache so artifacts, precompile, and libcxxwrap all +# stay warm across runs. Safe because the pipeline serializes same-cell jobs (concurrency_group +# in developer.pipeline.yml) — only one job per (Julia, fusion) cell writes this depot at a time. +# Dev mode rebuilds the wrapper .so each run, so drop stale overrides and the .ji that bake +# @wrapmodule bindings (cuNumeric/Legate + wrapper JLLs) — a cached .ji would mismatch the fresh +# .so and segfault. libcxxwrap's dev build is kept and reused. +DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" +rm -rf "$DEPOT"/packages/*/*/override \ + "$DEPOT"/compiled/v*/{cuNumeric,Legate,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} LEGATE_BRANCH_INPUT="${BUILDKITE_MESSAGE:-}" if [[ "${BUILDKITE_PULL_REQUEST:-false}" =~ ^[0-9]+$ ]]; then From ae4713d065ab1583c081fb808771ea023fb4b3fe Mon Sep 17 00:00:00 2001 From: krasow Date: Sun, 16 Aug 2026 15:28:17 -0500 Subject: [PATCH 10/49] lock/unlock --- .buildkite/developer.pipeline.yml | 24 +++++++++++++++++++----- .buildkite/run_developer_ci.sh | 6 +++--- 2 files changed, 22 insertions(+), 8 deletions(-) diff --git a/.buildkite/developer.pipeline.yml b/.buildkite/developer.pipeline.yml index b5e1ac0a4..5696d00f9 100644 --- a/.buildkite/developer.pipeline.yml +++ b/.buildkite/developer.pipeline.yml @@ -1,4 +1,13 @@ steps: + - label: ":lock: Enter developer CI" + command: "true" + concurrency: 1 + concurrency_group: "cunumeric/developer" + agents: + queue: "cuda" + + - wait: ~ + - group: ":hammer: Developer" key: "developer" steps: @@ -10,11 +19,6 @@ steps: # concurrent jobs contend on Pkg locks. Separate from the JLL cache. cache_dir: "${HOME}/.cache/julia-buildkite-plugin-developer-{{matrix.julia}}-{{matrix.fusion}}" command: ".buildkite/run_developer_ci.sh" - # Serialize jobs of the same (Julia, fusion) cell cluster-wide so concurrent PRs can't - # clobber that cell's shared developer depot (wrapper overrides + .ji). Different cells - # use different caches and still run in parallel. - concurrency: 1 - concurrency_group: "cunumeric/developer/{{matrix.julia}}/{{matrix.fusion}}" artifact_paths: - "deps/build.log" - "deps/*.log" @@ -40,3 +44,13 @@ steps: fusion: - "on" - "off" + + - wait: ~ + continue_on_failure: true + + - label: ":unlock: Leave developer CI" + command: "true" + concurrency: 1 + concurrency_group: "cunumeric/developer" + agents: + queue: "cuda" diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 5be27506e..e4004be1f 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -20,13 +20,13 @@ sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" export PATH="$CMAKE_ROOT/bin:$PATH" cmake --version -# Clean slate so cached state doesn't leak across Julia versions. +# Clean slate so cached state doesn't leak across Julia versions. rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml # Build directly in the plugin's persistent cache so artifacts, precompile, and libcxxwrap all -# stay warm across runs. Safe because the pipeline serializes same-cell jobs (concurrency_group -# in developer.pipeline.yml) — only one job per (Julia, fusion) cell writes this depot at a time. +# stay warm across runs. Developer builds are serialized across PRs, while each matrix job in a +# build uses a separate (Julia, fusion) cache. # Dev mode rebuilds the wrapper .so each run, so drop stale overrides and the .ji that bake # @wrapmodule bindings (cuNumeric/Legate + wrapper JLLs) — a cached .ji would mismatch the fresh # .so and segfault. libcxxwrap's dev build is kept and reused. From b2cd2d248baa51c3b8f9fe442523f87f7e1c3737 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sun, 16 Aug 2026 22:08:36 -0400 Subject: [PATCH 11/49] Add more linear algebra ops (i.e., eigen) (#183) * Add cholesky and eigen, split out batched linalg --- TODO.md | 27 ++ docs/src/api.md | 2 +- docs/src/linalg.md | 166 +++++++++-- lib/cunumeric_jl_wrapper/src/types.cpp | 2 + lib/cunumeric_jl_wrapper/src/wrapper.cpp | 42 +++ src/cuNumeric.jl | 3 + src/ndarray/batched_linalg.jl | 93 ++++++ src/ndarray/detail/linalg.jl | 254 +++++++++++++--- src/ndarray/linalg.jl | 200 +++++++++++-- src/ndarray/promotion.jl | 10 +- test/analysis/type_stability.jl | 52 +++- test/array/batched_linalg.jl | 190 ++++++++++++ test/array/linalg.jl | 350 +++++++++++++++++++---- test/util.jl | 4 + 14 files changed, 1242 insertions(+), 153 deletions(-) create mode 100644 src/ndarray/batched_linalg.jl create mode 100644 test/array/batched_linalg.jl diff --git a/TODO.md b/TODO.md index b1bbe0b09..34663324d 100644 --- a/TODO.md +++ b/TODO.md @@ -60,3 +60,30 @@ points so they do not fall through to scalar-indexing Base paths. - Still fallthrough / unsupported on `Diagonal` (e.g. `svd`, `pinv`, `cholesky`, host `AbstractArray` RHS): leave alone unless fixing is cheap; densify intentionally when needed + +**Decompositions** +- Done: `cholesky`, `eigen` / `eigvals` / `eigvecs`, `svd`, `qr`, and `\` are + wired to `LinearAlgebra` for 2D `NDArray`; `batched_solve` / `batched_cholesky` + / `batched_eigen` / `batched_eigvals` cover one batch dimension +- `eigh` / Hermitian eigen — the `SYEV` task is already wrapped, but there is no + entry point until `Hermitian` / `Symmetric` work on `NDArray` +- `PosDefException` from `cholesky` — a non-positive-definite input now raises a + catchable `ErrorException` (since the launchers call `task_throws_exception`), + but mapping it onto `LinearAlgebra.PosDefException` needs the failing pivot + index, which the task message does not carry +- The decomposition launchers must keep calling `task_throws_exception`, as + cupynumeric's own launchers do. Besides propagating task exceptions it grows + each leaf allocation pool by `--max-exception-size` (4096 bytes), which the + batched GPU kernels depend on: `CuPyNumericMapper::allocation_pool_size` only + declares one 16-byte-aligned `int32` of ZCMEM for `SOLVE` / `GEEV`, while + `solve.cu` allocates `batchsize` pointer arrays plus `batchsize` infos and + `geev.cu` allocates an info per matrix. Without it, `b > 1` aborts on GPU +- `adjoint(::NDArray)` (already on the P1 list) would unblock `Cholesky.U`, + `SVD.V`, destructuring a `Cholesky` as `L, U`, and `ldiv!` / least-squares `\` + on `NDArrayQR` (which needs `Q' * b`) +- More than one batch dimension — needs `domain_from_shape` in Legate.jl to + handle more than three dimensions, plus a cupynumeric `POTRF` instantiated for + `DIM >= 4` +- Multi-GPU cuSolverMp paths (`MP_POTRF`, `MP_SOLVE`) for large single matrices; + `cupynumeric_has_cusolvermp()` gates them and they need an NCCL communicator +- Batched `svd` / `qr` diff --git a/docs/src/api.md b/docs/src/api.md index 2533dc73b..a187a79e1 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -4,6 +4,6 @@ Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, a ```@autodocs Modules = [cuNumeric] -Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl"] +Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl"] Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!, :randn, :randn!, :randexp, :randexp!, :default_rng, :random, :random!)) ``` diff --git a/docs/src/linalg.md b/docs/src/linalg.md index 1052ca4a9..bfa798b6d 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -1,11 +1,28 @@ # Linear Algebra -cuNumeric.jl provides matrix multiplication, batched solves, SVD, QR, and related -helpers for `NDArray`. - -`solve`, `svd`, and `qr` accept `Float32`, `Float64`, `ComplexF32`, and -`ComplexF64`. Integer and `Bool` inputs require `@allowpromotion` or -`allowpromotion` and produce `Float64` outputs. +cuNumeric.jl provides matrix multiplication, solves, Cholesky, eigen, SVD, QR, +and related helpers for `NDArray`. + +All of the decompositions accept `Float32`, `Float64`, `ComplexF32`, and +`ComplexF64`. Integer and `Bool` inputs are converted to `Float64`. As +everywhere else in the package, that conversion needs `@allowpromotion` (or +`allowpromotion`) only when it widens the element type, so `Int64` and `UInt64` +pass through silently while `Int32`, smaller integers, and `Bool` do not. + +## Which entry point to use + +Single matrices go through the standard `LinearAlgebra` functions and return the +standard factorization objects. Stacks of matrices have no `LinearAlgebra` +equivalent, so they get their own `batched_*` names. + +| Single matrix (2D) | Returns | Stack of matrices (3D) | +| --- | --- | --- | +| `LinearAlgebra.cholesky(A)` | `Cholesky` | `cuNumeric.batched_cholesky(A)` | +| `LinearAlgebra.eigen(A)` | `Eigen` | `cuNumeric.batched_eigen(A)` | +| `LinearAlgebra.eigvals(A)` | `NDArray` | `cuNumeric.batched_eigvals(A)` | +| `LinearAlgebra.svd(A)` | `SVD` | not supported | +| `LinearAlgebra.qr(A)` | `NDArrayQR` | not supported | +| `cuNumeric.solve(A, b)`, `A \ b` | `NDArray` | `cuNumeric.batched_solve(A, B)` | ## Matrix multiply @@ -29,52 +46,139 @@ Pages = ["ndarray/binary.jl"] Filter = t -> t isa Function && nameof(t) === :mul! ``` -## Solve (batched) +## Solve -`cuNumeric.solve(A, b)` solves linear systems and returns an array with the same -shape as `b`. `A` has shape `(..., m, m)` and `b` has shape `(..., m)` or -`(..., m, n)`. +`cuNumeric.solve(A, b)` solves a linear system and returns an array with the same +shape as `b`. `A` is a square `m × m` matrix and `b` has shape `(m,)` or +`(m, n)`. `A \ b` is equivalent. ```julia A = cuNumeric.rand(Float32, 64, 64) b = cuNumeric.rand(Float32, 64) -x = cuNumeric.solve(A, b) +x = cuNumeric.solve(A, b) # or: A \ b B = cuNumeric.rand(Float32, 64, 4) -X = cuNumeric.solve(A, B) +X = A \ B +``` -As = cuNumeric.rand(Float32, 8, 32, 32) -Bs = cuNumeric.rand(Float32, 8, 32, 2) -Xs = cuNumeric.solve(As, Bs) +## Cholesky + +`LinearAlgebra.cholesky(A)` returns a `Cholesky` object holding the lower factor +`L`, with `A ≈ L * L'`. + +```julia +using LinearAlgebra + +A = cuNumeric.rand(Float32, 64, 64) +A = A * cuNumeric.transpose(A) + 64 * NDArray{Float32}(I, 64, 64) # make it SPD + +F = cholesky(A) +L = F.L # LowerTriangular view of F.factors ``` -## Singular value decomposition +Three things differ from Base: + +- The input is **not** checked for being Hermitian; only its lower triangle is + read. There are no `Hermitian` / `Symmetric` overloads. +- A non-positive-definite input raises an `ErrorException` carrying the task's + `"Matrix is not positive definite"` message, not a `PosDefException`. The + `check` keyword is therefore not supported. +- `F.U` (and hence destructuring as `L, U = F`) needs `copy(F.factors')`, which + falls back to scalar indexing until `adjoint(::NDArray)` is implemented. Use + `F.L` or `F.factors`. + +## Eigendecomposition + +`LinearAlgebra.eigen(A)` returns an `Eigen` object; `eigvals` and `eigvecs` +return the pieces individually. Column `j` of the eigenvector matrix is the +eigenvector for eigenvalue `j`. + +```julia +A = cuNumeric.rand(Float64, 64, 64) + +F = eigen(A) +values, vectors = F # or: eigvals(A), eigvecs(A) +``` -`cuNumeric.svd(A, full_matrices=true)` returns `(U, S, Vh)` for a 2D `m × n` -array. With `k = min(m, n)`, the output shapes are: +Eigenvalues and eigenvectors are **always complex**, even when `A` is real with +real eigenvalues: `ComplexF32` for `Float32` / `ComplexF32` input and +`ComplexF64` otherwise. This follows the underlying LAPACK `geev` path and +differs from Base, which returns real factors for such input. CUDA.jl's +non-symmetric branch behaves the same way. -- Full: `U` is `m × m`, `S` has length `k`, and `Vh` is `n × n`. -- Thin: `U` is `m × k`, `S` has length `k`, and `Vh` is `k × n`. +On a GPU this requires `cusolverDnXgeev`, added in CUDA 12.6.2. When it is +missing, the eigen entry points throw an error rather than silently falling back +to host execution. + +The symmetric/Hermitian path (`eigh`, backed by the `SYEV` task) is not wired up +yet; it needs `Hermitian` / `Symmetric` support on `NDArray` first. + +## Singular value decomposition + +`LinearAlgebra.svd(A; full=false)` returns an `SVD` object with +`A ≈ F.U * Diagonal(F.S) * F.Vt`. ```julia A = cuNumeric.rand(Float32, 128, 64) -U, S, Vh = cuNumeric.svd(A, false) +F = svd(A) +U, S, Vt = F.U, F.S, F.Vt ``` +With `k = min(m, n)`, the output shapes are: + +- Thin (`full=false`, the default): `U` is `m × k`, `S` has length `k`, and `Vt` + is `k × n`. +- Full (`full=true`): `U` is `m × m`, `S` has length `k`, and `Vt` is `n × n`. + +Note that `full=false` is the Julia default, whereas numpy's +`full_matrices=True` is not. Destructuring as `U, S, V = F` also gives you the +*adjoint* of `F.Vt`, and `F.V` is a lazy `Adjoint` wrapper, so operating on it +falls back to scalar indexing until `adjoint(::NDArray)` is implemented. Prefer +`F.Vt`. + `S` is real-valued for both real and complex inputs. ## QR decomposition -`cuNumeric.qr(A)` returns the economy-size factors `(Q, R)` for a 2D `m × n` -array. With `k = min(m, n)`, `Q` is `m × k` and `R` is `k × n`. +`LinearAlgebra.qr(A)` returns an `NDArrayQR`, holding the economy-size factors +with `A ≈ F.Q * F.R`. For a 2D `m × n` input and `k = min(m, n)`, `Q` is `m × k` +and `R` is `k × n`. ```julia A = cuNumeric.rand(Float32, 128, 64) -Q, R = cuNumeric.qr(A) +F = qr(A) +Q, R = F # or: F.Q, F.R ``` -SVD and QR currently accept only 2D arrays; batched decompositions are not -supported. +`NDArrayQR` is a `LinearAlgebra.Factorization` but not one of Base's QR types. +Base's default `QRCompactWY` stores a blocked Householder representation and +`QR` stores `factors` plus `τ`; the backend runs `geqrf` followed by `orgqr` and +discards `τ`, so neither is constructible. The practical difference is that +`F.Q` is a materialized `NDArray` rather than a lazy `QRCompactWYQ`. + +SVD and QR accept only 2D arrays; there are no batched versions. + +## Batched decompositions + +The `batched_*` functions apply an operation to every matrix in a stack. They +take exactly one batch dimension, i.e. shape `(b, m, m)`, and return plain +tuples or arrays rather than factorization objects. + +```julia +As = cuNumeric.rand(Float32, 8, 32, 32) +Bs = cuNumeric.rand(Float32, 8, 32, 2) + +Xs = cuNumeric.batched_solve(As, Bs) # (8, 32, 2) +Ls = cuNumeric.batched_cholesky(As) # (8, 32, 32), needs SPD blocks +values, vectors = cuNumeric.batched_eigen(As) # (8, 32) and (8, 32, 32) +``` + +Each matrix is factored on a single processor, so this is a good fit for many +small matrices and a poor one for a few large ones. Two or more batch dimensions +are rejected: the `POTRF` task body is only instantiated up to three dimensions, +and Legate.jl builds launch domains for at most three dimensions. + +Every caveat listed under Cholesky and Eigendecomposition applies here too. ## Helpers @@ -188,6 +292,10 @@ scale or shift; prefer `Diagonal` and `I` instead. There is no public `eye`. ## Not available yet -There is no public dense-matrix `cholesky`, `eig`, `lu`, matrix `inv`, or -`ldiv!` yet (beyond the `Diagonal` / `NDArray` paths listed above). -Elementwise `inv` / `^-1` are unary operations, not matrix inverse. +There is no public dense-matrix `lu`, matrix `inv`, or `ldiv!` yet (beyond the +`Diagonal` / `NDArray` paths listed above). Elementwise `inv` / `^-1` are unary +operations, not matrix inverse. + +Also missing: `eigh` / Hermitian eigen (needs `Hermitian` and `Symmetric` +support on `NDArray`), batched SVD and QR, and the multi-GPU cuSolverMp paths +for Cholesky and solve. diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index 7ea80f70d..8bf7706fb 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -171,6 +171,8 @@ void wrap_linalg_ops(jlcxx::Module& mod) { legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_SOLVE}); mod.set_const("SVD", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SVD}); mod.set_const("CQR", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_QR}); + mod.set_const("POTRF", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_POTRF}); mod.set_const("SYEV", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SYEV}); mod.set_const("GEEV", diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 907d3c425..fd1785c0c 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -55,6 +55,19 @@ void* nda_store_to_ndarray(legate::LogicalStore st) { return static_cast(new CN_NDArray{cupynumeric::as_array(st)}); } +// Legate.jl wraps ManualTask::add_input/add_output for a whole partition, but +// not the overloads taking a projection. GEEV needs one: its eigenvalue store +// has one fewer dimension than the launch domain. Julia picks the source +// dimensions; this only translates them into a SymbolicPoint. +static legate::SymbolicPoint make_projection(const std::vector& dims) { + std::vector exprs; + exprs.reserve(dims.size()); + for (auto d : dims) { + exprs.push_back(legate::dimension(static_cast(d))); + } + return legate::SymbolicPoint{std::move(exprs)}; +} + #if LEGATE_DEFINED(LEGATE_USE_CUDA) void register_tasks() { auto library = get_lib(); @@ -99,6 +112,35 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { mod.method("get_lib", &get_lib); mod.method("nda_store_to_ndarray", &nda_store_to_ndarray); + // True when the loaded cuSolver provides cusolverDnXgeev. Without it + // cupynumeric has no GPU eigenvalue kernel for general matrices. + mod.method("cusolver_has_geev", &cupynumeric_cusolver_has_geev); + + mod.method("add_input_proj", + [](legate::ManualTask& task, + std::shared_ptr part, + const std::vector& dims) { + task.add_input(*part, make_projection(dims)); + }); + mod.method("add_output_proj", + [](legate::ManualTask& task, + std::shared_ptr part, + const std::vector& dims) { + task.add_output(*part, make_projection(dims)); + }); + + // Marking a task as throwing also grows its leaf allocation pools by + // --max-exception-size (4096 bytes by default), which the cupynumeric + // decomposition tasks rely on: their mapper only declares enough zero-copy + // memory for a single status flag, while the batched GPU kernels allocate one + // per matrix. cupynumeric's own Python launchers always set this. + mod.method("task_throws_exception", [](legate::ManualTask& task, bool value) { + task.throws_exception(value); + }); + mod.method("task_throws_exception", [](legate::AutoTask& task, bool value) { + task.throws_exception(value); + }); + // Legate.jl Scalar has no vector constructors. BITGENERATOR (and similar) // tasks take fixed-array scalars; these helpers pack them from Julia // pointers. diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 0204b3b46..468904acc 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -74,6 +74,8 @@ const SUPPORTED_NUMERIC_TYPES = Union{ const SUPPORTED_SOLVE_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} const SUPPORTED_SVD_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} const SUPPORTED_QR_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const SUPPORTED_CHOLESKY_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const SUPPORTED_EIG_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} const SUPPORTED_ARRAY_TYPES = Union{Bool,SUPPORTED_NUMERIC_TYPES} const SUPPORTED_TYPES = Union{SUPPORTED_ARRAY_TYPES,String} @@ -180,6 +182,7 @@ include("ndarray/random/random.jl") include("ndarray/unary.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") +include("ndarray/batched_linalg.jl") include("scoping/scoping.jl") # From https://github.com/JuliaGraphics/QML.jl/blob/dca239404135d85fe5d4afe34ed3dc5f61736c63/src/QML.jl#L147 diff --git a/src/ndarray/batched_linalg.jl b/src/ndarray/batched_linalg.jl new file mode 100644 index 000000000..90c929c8c --- /dev/null +++ b/src/ndarray/batched_linalg.jl @@ -0,0 +1,93 @@ +# Batched linear algebra: operations over a stack of matrices, dispatching on 3D +# arrays. These have no Base.LinearAlgebra counterpart, so they keep the +# `batched_` prefix and return plain tuples rather than `Factorization` objects. +# +# All of them take exactly one batch dimension, i.e. shape (b, m, m). See +# `MAX_BATCHED_DIM` for why. + +""" + cuNumeric.batched_solve(A, B) + +Solve a stack of linear systems `A * X = B`. + +`A` must have shape `(b, m, m)` and `B` shape `(b, m, n)`. The result has the +same shape as `B`. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a single system use [`cuNumeric.solve`](@ref). +""" +function batched_solve( + a::NDArray{A}, b::NDArray{B} +) where {A<:_SOLVE_ACCEPTED,B<:_SOLVE_ACCEPTED} + O = promote_type(_solve_eltype(A), _solve_eltype(B)) + return _solve_check_a_dims_batched( + checked_promote_arr(batched_solve, a, O), checked_promote_arr(batched_solve, b, O) + ) +end + +function batched_solve(a::NDArray, b::NDArray) + bad = eltype(a) <: _SOLVE_ACCEPTED ? eltype(b) : eltype(a) + return throw(ArgumentError("array type $bad is unsupported in batched_solve")) +end + +""" + cuNumeric.batched_cholesky(A) + +Cholesky factor of every matrix in the stack `A`, returned as a single array of +the same shape. Each `A[i, :, :]` is factored independently into a lower +triangular `L` with `A[i, :, :] ≈ L * L'`; the upper triangle is zeroed. + +`A` must have shape `(b, m, m)`. As in `LinearAlgebra.cholesky`, only the lower +triangle is read and Hermitian-ness is not checked. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. +""" +function batched_cholesky(a::NDArray{T,N}) where {T<:_CHOLESKY_ACCEPTED,N} + _assert_batched_dims(:batched_cholesky, N) + return _cholesky(checked_promote_arr(batched_cholesky, a, _cholesky_eltype(T))) +end + +function batched_cholesky(a::NDArray) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in batched_cholesky")) +end + +""" + cuNumeric.batched_eigen(A) -> (values, vectors) + +Eigenvalues and right eigenvectors of every matrix in the stack `A`. + +`A` must have shape `(b, m, m)`. `values` has shape `(b, m)` and `vectors` has +the shape of `A`, with `vectors[i, :, j]` the eigenvector for `values[i, j]`. + +Both results are always complex, even for real input. See `LinearAlgebra.eigen`. +""" +function batched_eigen(a::NDArray{T,N}) where {T<:_EIG_ACCEPTED,N} + _assert_batched_dims(:batched_eigen, N) + return _eig(checked_promote_arr(batched_eigen, a, _eig_eltype(T))) +end + +function batched_eigen(a::NDArray) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in batched_eigen")) +end + +""" + cuNumeric.batched_eigvals(A) + +Eigenvalues of every matrix in the stack `A`, as a complex array of shape +`(b, m)`. See [`cuNumeric.batched_eigen`](@ref). +""" +function batched_eigvals(a::NDArray{T,N}) where {T<:_EIG_ACCEPTED,N} + _assert_batched_dims(:batched_eigvals, N) + return _eigvals(checked_promote_arr(batched_eigvals, a, _eig_eltype(T))) +end + +function batched_eigvals(a::NDArray) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in batched_eigvals")) +end diff --git a/src/ndarray/detail/linalg.jl b/src/ndarray/detail/linalg.jl index 0362a84f5..e9e8a11f1 100644 --- a/src/ndarray/detail/linalg.jl +++ b/src/ndarray/detail/linalg.jl @@ -17,6 +17,12 @@ function choose_nd_color_shape(shape::NTuple{N,Int}) where {N} return Tuple(color_shape) end +# One batch dimension is the ceiling for every batched op: +# - the POTRF task body is only instantiated for 2 <= DIM < 4 +# - Legate.jl's `domain_from_shape` builds launch domains for at most three +# dimensions and yields an empty domain past that +const MAX_BATCHED_DIM = 3 + function prepare_manual_task_for_batched_matrices(full_shape::NTuple{N,Int}) where {N} initial_color_shape = choose_nd_color_shape(full_shape) tilesize = Tuple( @@ -45,6 +51,7 @@ function solve_batched(a::NDArray{T,N}, b::NDArray, x::NDArray) where {T,N} domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) lib = cuNumeric.get_lib() task = Legate.create_manual_task(rt, lib, cuNumeric.SOLVE, domain) + cuNumeric.task_throws_exception(task, true) Legate.add_input(task, tiled_a) Legate.add_input(task, tiled_b) @@ -61,17 +68,47 @@ const _SOLVE_ACCEPTED = Union{SUPPORTED_SOLVE_TYPES,_SOLVE_PROMOTABLE} _solve_eltype(::Type{T}) where {T<:_SOLVE_PROMOTABLE} = Float64 _solve_eltype(::Type{T}) where {T<:SUPPORTED_SOLVE_TYPES} = T -# `a` must be at least 2D, `b` at least 1D. -function _solve_check_a_dims(a::NDArray{<:Any,0}, b::NDArray) - throw(ArgumentError("0-dimensional array given. Array must be at least two-dimensional")) +# `a` must be at least 2D, `b` at least 1D. `solve` takes only the 2D case; +# stacked systems go through `batched_solve`. +function _solve_check_a_dims_2d(a::NDArray{<:Any,0}, b::NDArray) + return throw(ArgumentError("0-dimensional array given. Array must be two-dimensional")) +end +function _solve_check_a_dims_2d(a::NDArray{<:Any,1}, b::NDArray) + return throw(ArgumentError("1-dimensional array given. Array must be two-dimensional")) +end +_solve_check_a_dims_2d(a::NDArray{<:Any,2}, b::NDArray) = _solve_check_b_dims(a, b) +function _solve_check_a_dims_2d(a::NDArray{<:Any,N}, b::NDArray) where {N} + return throw( + ArgumentError( + "$N-dimensional array given. Use `cuNumeric.batched_solve` for " * + "stacked systems of shape (...,m,m)", + ), + ) end -function _solve_check_a_dims(a::NDArray{<:Any,1}, b::NDArray) - throw(ArgumentError("1-dimensional array given. Array must be at least two-dimensional")) + +function _solve_check_a_dims_batched(a::NDArray{<:Any,N}, b::NDArray) where {N} + _assert_batched_dims(:batched_solve, N) + return _solve_check_b_dims(a, b) +end + +function _assert_batched_dims(f, N::Integer) + N < 3 && throw( + ArgumentError( + "$N-dimensional array given. `$f` requires shape (b,m,m); use the " * + "matching `LinearAlgebra` or `cuNumeric` function for a single matrix", + ), + ) + N > MAX_BATCHED_DIM && throw( + ArgumentError( + "$N-dimensional array given. `$f` supports at most one batch " * + "dimension, i.e. shape (b,m,m)", + ), + ) + return nothing end -_solve_check_a_dims(a::NDArray, b::NDArray) = _solve_check_b_dims(a, b) function _solve_check_b_dims(a::NDArray, b::NDArray{<:Any,0}) - throw(ArgumentError("0-dimensional array given. Array must be at least one-dimensional")) + return throw(ArgumentError("0-dimensional array given. Array must be at least one-dimensional")) end _solve_check_b_dims(a::NDArray, b::NDArray) = _solve(a, b) @@ -94,15 +131,101 @@ function _solve(a::NDArray{T,N}, b::NDArray{S,N}) where {T,S,N} " (size $(size(b)[end-1]) is different from $(size(a)[end]))", ), ) - prod(size(a)) == 0 || prod(size(b)) == 0 && return zeros(T, size(b)...) - x = zeros(T, size(b)...) + prod(size(a)) == 0 || prod(size(b)) == 0 && return cuNumeric.zeros(T, size(b)...) + x = cuNumeric.zeros(T, size(b)...) solve_batched(a, b, x) return x end # Mismatched batch dimensions function _solve(a::NDArray{T,N}, b::NDArray{S,M}) where {T,N,S,M} - throw(ArgumentError("Batched matrices require signature (...,m,m),(...,m,n)->(...,m,n)")) + return throw(ArgumentError("Batched matrices require signature (...,m,m),(...,m,n)->(...,m,n)")) +end + +# cholesky + +""" + potrf!(out, a; lower, zeroout) + +Run the cupynumeric `POTRF` task, writing the Cholesky factor of `a` into `out`. + +The task is batched over the leading dimension: for a `(b, m, m)` input it +factors each `m × m` block independently. `lower` selects which triangle holds +the factor, `zeroout` zeros the opposite triangle in-task. +""" +function potrf!(out::NDArray{T,N}, a::NDArray{T,N}; lower::Bool, zeroout::Bool) where {T,N} + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + + @task_scope "potrf" begin + task = Legate.create_auto_task(rt, lib, cuNumeric.POTRF) + cuNumeric.task_throws_exception(task, true) + + l_a = nda_to_logical_array(a) + l_out = nda_to_logical_array(out) + + in_var = Legate.add_input(task, l_a) + out_var = Legate.add_output(task, l_out) + + Legate.add_scalar(task, Legate.Scalar(lower)) + Legate.add_scalar(task, Legate.Scalar(zeroout)) + + # Each matrix must live on one processor; only the batch axes may split. + Legate.add_broadcast(task, l_a, CxxWrap.StdVector(UInt32[N - 2, N - 1])) + Legate.add_constraint(task, Legate.align(out_var, in_var)) + + Legate.submit_auto_task(rt, task) + end + return out +end + +# eigen + +""" + geev!(a, ew, ev) + +Run the cupynumeric `GEEV` task, writing eigenvalues into `ew` and, unless `ev` +is `nothing`, right eigenvectors into `ev` as columns. + +`ew` has one fewer dimension than `a`, so the launch domain maps onto it through +a projection that drops the trailing axis. The task keys eigenvector computation +off the number of registered outputs, so `ev` must not be added when only +eigenvalues are wanted. +""" +function geev!(a::NDArray{T,N}, ew::NDArray, ev::Union{NDArray,Nothing}) where {T,N} + full_shape = size(a) + tilesize, color_shape = prepare_manual_task_for_batched_matrices(full_shape) + + tiled_a = Legate.partition_by_tiling(nda_to_logical_store(a), collect(tilesize)) + tiled_ew = Legate.partition_by_tiling( + nda_to_logical_store(ew), collect(tilesize[1:(end - 1)]) + ) + tiled_ev = if ev === nothing + nothing + else + Legate.partition_by_tiling(nda_to_logical_store(ev), collect(tilesize)) + end + + # 0-based source dimensions of the launch domain. + proj = CxxWrap.StdVector(Int32.(0:(N - 1))) + proj_ew = CxxWrap.StdVector(Int32.(0:(N - 2))) + + @task_scope "geev" begin + rt = Legate.get_runtime() + domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) + lib = cuNumeric.get_lib() + task = Legate.create_manual_task(rt, lib, cuNumeric.GEEV, domain) + cuNumeric.task_throws_exception(task, true) + + cuNumeric.add_input_proj(task, tiled_a.handle, proj) + cuNumeric.add_output_proj(task, tiled_ew.handle, proj_ew) + if tiled_ev !== nothing + cuNumeric.add_output_proj(task, tiled_ev.handle, proj) + end + + Legate.submit_manual_task(rt, task) + end + return nothing end function svd_single(a::NDArray{T,N}, u::NDArray, s::NDArray, vh::NDArray) where {T,N} @@ -133,9 +256,9 @@ function _svd(a::NDArray{T,2}, full_matrices::Bool) where {T} k = min(m, n) S = real(T) # cuSolver requires full square buffers regardless of full_matrices - u_buf = zeros(T, m, m) - s = zeros(S, k) - vh_buf = zeros(T, n, n) + u_buf = cuNumeric.zeros(T, m, m) + s = cuNumeric.zeros(S, k) + vh_buf = cuNumeric.zeros(T, n, n) svd_single(a, u_buf, s, vh_buf) # Backend factors are logically ordered; only thin strided views need materialization. u = full_matrices ? u_buf : copy(u_buf[:, 1:k]) @@ -149,22 +272,6 @@ const _SVD_ACCEPTED = Union{SUPPORTED_SVD_TYPES,_SVD_PROMOTABLE} _svd_eltype(::Type{T}) where {T<:_SVD_PROMOTABLE} = Float64 _svd_eltype(::Type{T}) where {T<:SUPPORTED_SVD_TYPES} = T -function _svd_check_dims(a::NDArray{<:Any,0}, full_matrices::Bool) - throw(ArgumentError("0-dimensional array given. Array must be at least two-dimensional")) -end - -function _svd_check_dims(a::NDArray{<:Any,1}, full_matrices::Bool) - throw(ArgumentError("1-dimensional array given. Array must be at least two-dimensional")) -end - -function _svd_check_dims(a::NDArray{<:Any,2}, full_matrices::Bool) - return _svd(a, full_matrices) -end - -function _svd_check_dims(a::NDArray, full_matrices::Bool) - throw(ArgumentError("cuNumeric does not yet support stacked 2d arrays")) -end - # qr function qr_single(a::NDArray{T,N}, q::NDArray, r::NDArray) where {T,N} @@ -191,8 +298,8 @@ function _qr(a::NDArray{T,2}) where {T} m, n = size(a) k = min(m, n) # cuSolver requires full square buffers regardless of output shape - q_buf = zeros(T, m, m) - r_buf = zeros(T, n, n) + q_buf = cuNumeric.zeros(T, m, m) + r_buf = cuNumeric.zeros(T, n, n) qr_single(a, q_buf, r_buf) # Host conversion assumes contiguous storage, so materialize the economy slices. q = copy(q_buf[:, 1:k]) @@ -205,18 +312,87 @@ const _QR_ACCEPTED = Union{SUPPORTED_QR_TYPES,_QR_PROMOTABLE} _qr_eltype(::Type{T}) where {T<:_QR_PROMOTABLE} = Float64 _qr_eltype(::Type{T}) where {T<:SUPPORTED_QR_TYPES} = T -function _qr_check_dims(a::NDArray{<:Any,0}) - throw(ArgumentError("0-dimensional array given. Array must be at least two-dimensional")) +# cholesky/eigen guards. Both run in floating point only, so int/bool inputs +# promote to Float64 the same way solve/svd/qr do. + +const _CHOLESKY_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool} +const _CHOLESKY_ACCEPTED = Union{SUPPORTED_CHOLESKY_TYPES,_CHOLESKY_PROMOTABLE} +_cholesky_eltype(::Type{T}) where {T<:_CHOLESKY_PROMOTABLE} = Float64 +_cholesky_eltype(::Type{T}) where {T<:SUPPORTED_CHOLESKY_TYPES} = T + +const _EIG_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool} +const _EIG_ACCEPTED = Union{SUPPORTED_EIG_TYPES,_EIG_PROMOTABLE} +_eig_eltype(::Type{T}) where {T<:_EIG_PROMOTABLE} = Float64 +_eig_eltype(::Type{T}) where {T<:SUPPORTED_EIG_TYPES} = T + +# GEEV always produces complex eigenvalues and eigenvectors, even for real input. +_eig_complex_eltype(::Type{Float32}) = ComplexF32 +_eig_complex_eltype(::Type{ComplexF32}) = ComplexF32 +_eig_complex_eltype(::Type{Float64}) = ComplexF64 +_eig_complex_eltype(::Type{ComplexF64}) = ComplexF64 + +function _check_square_matrices(f, a::NDArray{<:Any,N}) where {N} + N < 2 && throw( + ArgumentError( + "$N-dimensional array given. Array must be at least two-dimensional" + ), + ) + sz = size(a) + sz[end - 1] != sz[end] && + throw(ArgumentError("Last 2 dimensions of the array must be square in $f")) + sz[end] == 0 && throw(ArgumentError("Input shape dimension 0 not allowed in $f")) + return nothing +end + +""" + _cholesky(a) + +Lower Cholesky factor of each trailing square block of `a`, with the upper +triangle zeroed. Only the lower triangle of the input is read; the input is +assumed Hermitian without being checked, matching cupynumeric. +""" +function _cholesky(a::NDArray{T,N}) where {T,N} + _check_square_matrices(:cholesky, a) + out = cuNumeric.zeros(T, size(a)...) + potrf!(out, a; lower=true, zeroout=true) + return out end -function _qr_check_dims(a::NDArray{<:Any,1}) - throw(ArgumentError("1-dimensional array given. Array must be at least two-dimensional")) +""" + _eig(a) + +Eigenvalues and right eigenvectors of each trailing square block of `a`. Both +are always complex. +""" +function _eig(a::NDArray{T,N}) where {T,N} + ew = _alloc_eigenvalues(a) + ev = cuNumeric.zeros(_eig_complex_eltype(T), size(a)...) + geev!(a, ew, ev) + return ew, ev end -function _qr_check_dims(a::NDArray{<:Any,2}) - return _qr(a) +""" + _eigvals(a) + +Eigenvalues only. Registering just the one output is what tells the backend task +to skip the eigenvector computation. +""" +function _eigvals(a::NDArray) + ew = _alloc_eigenvalues(a) + geev!(a, ew, nothing) + return ew end -function _qr_check_dims(a::NDArray) - throw(ArgumentError("cuNumeric does not yet support stacked 2d arrays")) +function _alloc_eigenvalues(a::NDArray{T,N}) where {T,N} + _check_square_matrices(:eigen, a) + _assert_geev_available() + return cuNumeric.zeros(_eig_complex_eltype(T), size(a)[1:(end - 1)]...) +end + +function _assert_geev_available() + (Legate.num_gpus() > 0 && !cuNumeric.cusolver_has_geev()) && error( + "eigen requires cusolverDnXgeev, which the installed cuSolver does not " * + "provide. Upgrade CUDA (12.6.2 or newer) or run without GPUs.", + ) + return nothing end diff --git a/src/ndarray/linalg.jl b/src/ndarray/linalg.jl index 075820f85..ae73a9f0f 100644 --- a/src/ndarray/linalg.jl +++ b/src/ndarray/linalg.jl @@ -1,48 +1,196 @@ +export NDArrayQR + # Type/dim guards dispatch on one argument at a time, then forward to `_solve`. """ cuNumeric.solve(A, b) -Solve linear system(s) `A * x = b`. +Solve the linear system `A * x = b`. -`A` must have shape `(..., m, m)`. `b` must have shape `(..., m)` or `(..., m, n)`. -The result has the same shape as `b`. Batch dimensions are supported; the -implementation always uses the batched Legate `SOLVE` path. +`A` must be a square `(m, m)` matrix. `b` must have shape `(m,)` or `(m, n)`. +The result has the same shape as `b`. Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. -Integer or `Bool` inputs promote to `Float64` only when promotion is allowed. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a stack of systems of shape `(b, m, m)` use +[`cuNumeric.batched_solve`](@ref). + +See also `\\`. """ -function solve(a::NDArray{<:_SOLVE_ACCEPTED}, b::NDArray{<:_SOLVE_ACCEPTED}) - A, B = eltype(a), eltype(b) +function solve(a::NDArray{A}, b::NDArray{B}) where {A<:_SOLVE_ACCEPTED,B<:_SOLVE_ACCEPTED} O = promote_type(_solve_eltype(A), _solve_eltype(B)) - # int/bool -> float is an implicit promotion, disallowed unless `allowpromotion` - A <: _SOLVE_PROMOTABLE && assertpromotion(solve, A, O) - B <: _SOLVE_PROMOTABLE && assertpromotion(solve, B, O) - return _solve_check_a_dims(unchecked_promote_arr(a, O), unchecked_promote_arr(b, O)) + return _solve_check_a_dims_2d( + checked_promote_arr(solve, a, O), checked_promote_arr(solve, b, O) + ) end function solve(a::NDArray, b::NDArray) bad = eltype(a) <: _SOLVE_ACCEPTED ? eltype(b) : eltype(a) - throw(ArgumentError("array type $bad is unsupported in solve")) + return throw(ArgumentError("array type $bad is unsupported in solve")) +end + +""" + A \\ b + +Solve `A * x = b` for a square 2D `NDArray` `A`. Equivalent to +[`cuNumeric.solve`](@ref). +""" +Base.:\(a::NDArray{<:Any,2}, b::NDArray{<:Any,1}) = solve(a, b) +Base.:\(a::NDArray{<:Any,2}, b::NDArray{<:Any,2}) = solve(a, b) + +""" + LinearAlgebra.cholesky(A::NDArray{T,2}) -> Cholesky + +Cholesky factorization of the Hermitian positive-definite matrix `A`, returned as +a `LinearAlgebra.Cholesky` object holding the lower factor `L` with `A ≈ L * L'`. + +Only the lower triangle of `A` is read and, unlike Base, it is *not* checked for +being Hermitian. A non-positive-definite input raises an `ErrorException` from +the task rather than `LinearAlgebra.PosDefException`, so the `check` keyword is +not supported. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a stack of matrices of shape `(b, m, m)` use +[`cuNumeric.batched_cholesky`](@ref). +""" +function LinearAlgebra.cholesky(a::NDArray{T,2}) where {T<:_CHOLESKY_ACCEPTED} + factors = _cholesky(checked_promote_arr(cholesky, a, _cholesky_eltype(T))) + return LinearAlgebra.Cholesky(factors, 'L', 0) +end + +function LinearAlgebra.cholesky(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in cholesky")) +end + +""" + LinearAlgebra.eigen(A::NDArray{T,2}) -> Eigen + +Eigenvalues and right eigenvectors of the square matrix `A`, returned as a +`LinearAlgebra.Eigen` object. Column `j` of `F.vectors` is the eigenvector for +`F.values[j]`. + +Values and vectors are **always complex**, even when `A` is real with real +eigenvalues. This follows the underlying LAPACK `geev` path and differs from +Base, which returns real factors for such input. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a stack of matrices of shape `(b, m, m)` use +[`cuNumeric.batched_eigen`](@ref). +""" +function LinearAlgebra.eigen(a::NDArray{T,2}) where {T<:_EIG_ACCEPTED} + return LinearAlgebra.Eigen(_eig(checked_promote_arr(eigen, a, _eig_eltype(T)))...) +end + +function LinearAlgebra.eigen(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in eigen")) +end + +""" + LinearAlgebra.eigvals(A::NDArray{T,2}) + +Eigenvalues of the square matrix `A` as a complex `NDArray`. See +`LinearAlgebra.eigen` for the supported element types and for why the +result is always complex. +""" +function LinearAlgebra.eigvals(a::NDArray{T,2}) where {T<:_EIG_ACCEPTED} + return _eigvals(checked_promote_arr(eigvals, a, _eig_eltype(T))) +end + +function LinearAlgebra.eigvals(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in eigvals")) end -function svd(a::NDArray{<:_SVD_ACCEPTED}, full_matrices::Bool=true) - A = eltype(a) - O = _svd_eltype(A) - A <: _SVD_PROMOTABLE && assertpromotion(svd, A, O) - return _svd_check_dims(unchecked_promote_arr(a, O), full_matrices) +""" + LinearAlgebra.eigvecs(A::NDArray{T,2}) + +Right eigenvectors of the square matrix `A`, as columns of a complex `NDArray`. +See `LinearAlgebra.eigen` for the supported element types. +""" +LinearAlgebra.eigvecs(a::NDArray{<:Any,2}) = LinearAlgebra.eigen(a).vectors + +""" + LinearAlgebra.svd(A::NDArray{T,2}; full=false) -> SVD + +Singular value decomposition of `A`, returned as a `LinearAlgebra.SVD` object +with `A ≈ F.U * Diagonal(F.S) * F.Vt`. + +With `m, n = size(A)` and `k = min(m, n)`, `full=false` gives an `m × k` `F.U` +and a `k × n` `F.Vt`, and `full=true` gives `m × m` and `n × n`. `F.S` has length +`k` and is real-valued for both real and complex input. + +Destructuring an `SVD` yields `(U, S, V)` — the adjoint of `F.Vt`, not `F.Vt` +itself. `F.V` is a lazy `Adjoint` wrapper, so operating on it falls back to +scalar indexing until `adjoint(::NDArray)` is implemented; prefer `F.Vt`. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. +""" +function LinearAlgebra.svd(a::NDArray{T,2}; full::Bool=false) where {T<:_SVD_ACCEPTED} + return LinearAlgebra.SVD(_svd(checked_promote_arr(svd, a, _svd_eltype(T)), full)...) end -function svd(a::NDArray, full_matrices::Bool=true) - throw(ArgumentError("array type $(eltype(a)) is unsupported in svd")) +function LinearAlgebra.svd(a::NDArray{<:Any,2}; full::Bool=false) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in svd")) end -function qr(a::NDArray{<:_QR_ACCEPTED}) - A = eltype(a) - O = _qr_eltype(A) - A <: _QR_PROMOTABLE && assertpromotion(qr, A, O) - return _qr_check_dims(unchecked_promote_arr(a, O)) +""" + NDArrayQR <: LinearAlgebra.Factorization + +QR factorization of an `NDArray`, holding the explicit factors `Q` and `R` with +`A ≈ Q * R`. + +The backend returns materialized factors rather than the packed Householder +representation Base uses, so this is a distinct type from `LinearAlgebra.QR` and +`QRCompactWY`. It destructures as `(Q, R)` and exposes `F.Q` and `F.R`. +""" +struct NDArrayQR{T,M<:NDArray{T,2}} <: LinearAlgebra.Factorization{T} + Q::M + R::M +end + +Base.iterate(F::NDArrayQR) = (F.Q, Val(:R)) +Base.iterate(F::NDArrayQR, ::Val{:R}) = (F.R, Val(:done)) +Base.iterate(::NDArrayQR, ::Val{:done}) = nothing +Base.size(F::NDArrayQR) = (size(F.Q, 1), size(F.R, 2)) +Base.size(F::NDArrayQR, d::Integer) = d == 1 ? size(F.Q, 1) : size(F.R, d) + +function Base.show(io::IO, ::MIME"text/plain", F::NDArrayQR) + summary(io, F) + println(io) + println(io, "Q factor: ", summary(F.Q)) + print(io, "R factor: ", summary(F.R)) + return nothing +end + +""" + LinearAlgebra.qr(A::NDArray{T,2}) -> NDArrayQR + +Reduced QR factorization of `A`, with `A ≈ F.Q * F.R`. For an `m × n` input and +`k = min(m, n)`, `F.Q` is `m × k` and `F.R` is `k × n`. + +See [`NDArrayQR`](@ref) for why this is not a `LinearAlgebra.QRCompactWY`. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. +""" +function LinearAlgebra.qr(a::NDArray{T,2}) where {T<:_QR_ACCEPTED} + return NDArrayQR(_qr(checked_promote_arr(qr, a, _qr_eltype(T)))...) end -function qr(a::NDArray) - throw(ArgumentError("array type $(eltype(a)) is unsupported in qr")) +function LinearAlgebra.qr(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in qr")) end diff --git a/src/ndarray/promotion.jl b/src/ndarray/promotion.jl index b13437fbb..20e5ab607 100644 --- a/src/ndarray/promotion.jl +++ b/src/ndarray/promotion.jl @@ -5,7 +5,15 @@ is_wider_type(::Type{A}, ::Type{B}) where {A,B} = sizeof(A) > sizeof(B) checked_promote_arr(arr::NDArray{T}, ::Type{T}) where {T} = arr function checked_promote_arr(arr::NDArray{T}, ::Type{S}) where {T,S} - is_wider_type(S, T) && assertpromotion(promote_type, T, S) + return checked_promote_arr(promote_type, arr, S) +end + +# `op` only names the caller in the error message, so that e.g. cholesky reports +# itself rather than `promote_type`. +checked_promote_arr(op, arr::NDArray{T}, ::Type{T}) where {T} = arr + +function checked_promote_arr(op, arr::NDArray{T}, ::Type{S}) where {T,S} + is_wider_type(S, T) && assertpromotion(op, T, S) return as_type(arr, S) end diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index add8e8a06..9d0240749 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -127,14 +127,14 @@ end @testset verbose = true "svd" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A = cuNumeric.NDArray(T[1 0; 0 1]) - @test @inferred(cuNumeric.svd(A)) !== nothing - @test @inferred(cuNumeric.svd(A, false)) !== nothing + @test @inferred(LinearAlgebra.svd(A)) !== nothing + @test @inferred(LinearAlgebra.svd(A; full=true)) !== nothing end @testset "promote $(T)" for T in (Int32, Int64, Bool) A = cuNumeric.NDArray(T[1 0; 0 1]) allowpromotion() do - @test @inferred(cuNumeric.svd(A)) !== nothing + @test @inferred(LinearAlgebra.svd(A)) !== nothing end end end @@ -142,17 +142,59 @@ end @testset verbose = true "qr" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) A = cuNumeric.NDArray(T[1 0; 0 1]) - @test @inferred(cuNumeric.qr(A)) !== nothing + @test @inferred(LinearAlgebra.qr(A)) !== nothing end @testset "promote $(T)" for T in (Int32, Int64, Bool) A = cuNumeric.NDArray(T[1 0; 0 1]) allowpromotion() do - @test @inferred(cuNumeric.qr(A)) !== nothing + @test @inferred(LinearAlgebra.qr(A)) !== nothing end end end +@testset verbose = true "cholesky" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + A = cuNumeric.NDArray(Matrix{T}(I, 2, 2)) + @test @inferred(LinearAlgebra.cholesky(A)) !== nothing + B = cuNumeric.NDArray(reshape(Matrix{T}(I, 2, 2), 1, 2, 2)) + @test @inferred(cuNumeric.batched_cholesky(B)) !== nothing + end + + @testset "promote $(T)" for T in (Int32, Int64, Bool) + A = cuNumeric.NDArray(T[1 0; 0 1]) + allowpromotion() do + @test @inferred(LinearAlgebra.cholesky(A)) !== nothing + end + end +end + +@testset verbose = true "eigen" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + A = cuNumeric.NDArray(Matrix{T}(I, 2, 2)) + @test @inferred(LinearAlgebra.eigen(A)) !== nothing + @test @inferred(LinearAlgebra.eigvals(A)) !== nothing + B = cuNumeric.NDArray(reshape(Matrix{T}(I, 2, 2), 1, 2, 2)) + @test @inferred(cuNumeric.batched_eigen(B)) !== nothing + @test @inferred(cuNumeric.batched_eigvals(B)) !== nothing + end + + @testset "promote $(T)" for T in (Int32, Int64, Bool) + A = cuNumeric.NDArray(T[1 0; 0 1]) + allowpromotion() do + @test @inferred(LinearAlgebra.eigen(A)) !== nothing + end + end +end + +@testset verbose = true "batched_solve" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) + A = cuNumeric.NDArray(reshape(T[2 1; 5 7], 1, 2, 2)) + b = cuNumeric.NDArray(reshape(T[11, 13], 1, 2, 1)) + @test @inferred(cuNumeric.batched_solve(A, b)) !== nothing + end +end + @testset verbose = true "linalg ops" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) M = cuNumeric.zeros(T, 4, 3) diff --git a/test/array/batched_linalg.jl b/test/array/batched_linalg.jl new file mode 100644 index 000000000..6621fdd76 --- /dev/null +++ b/test/array/batched_linalg.jl @@ -0,0 +1,190 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): Ethan Meitz +=# + +# Batched (3D+) linear algebra. These entry points are deliberately not +# LinearAlgebra methods: Base has no batched equivalent. + +const N_BATCH = 3 + +function batched_spd(::Type{T}, b, n) where {T} + RT = real(T) + out = zeros(T, b, n, n) + for i in 1:b + B = my_rand(T, n, n; L=RT(-1), R=RT(1)) + out[i, :, :] = B * B' + T(n) * Matrix{T}(I, n, n) + end + return out +end + +function batched_random(::Type{T}, dims...) where {T} + RT = real(T) + return my_rand(T, dims...; L=RT(-1), R=RT(1)) +end + +@testset "batched_solve" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) + n, nrhs = 4, 2 + A_ref = batched_spd(T, N_BATCH, n) + b_ref = batched_random(T, N_BATCH, n, nrhs) + + x = cuNumeric.batched_solve(cuNumeric.NDArray(A_ref), cuNumeric.NDArray(b_ref)) + + allowscalar() do + X = Array(x) + @test size(X) == (N_BATCH, n, nrhs) + for i in 1:N_BATCH + @test isapprox( + A_ref[i, :, :] \ b_ref[i, :, :], X[i, :, :]; atol=atol(T), rtol=rtol(T) + ) + end + end + end +end + +@testset "batched_solve promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + A = cuNumeric.NDArray(reshape(T[1, 0, 0, 1], 1, 2, 2)) + b = cuNumeric.NDArray(reshape(T[1, 1], 1, 2, 1)) + + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" cuNumeric.batched_solve(A, b) + else + @test cuNumeric.batched_solve(A, b) isa NDArray{Float64} + end + + allowpromotion() do + x = cuNumeric.batched_solve(A, b) + allowscalar() do + @test safe_compare( + reshape(Float64[1, 1], 1, 2, 1), x, atol(Float64), rtol(Float64) + ) + end + end + end +end + +@testset "batched_solve rejects 2D input" begin + A = cuNumeric.zeros(Float64, 3, 3) + b = cuNumeric.zeros(Float64, 3, 1) + @test_throws "requires shape (b,m,m)" cuNumeric.batched_solve(A, b) +end + +@testset "batched_cholesky" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + n = 4 + A_ref = batched_spd(T, N_BATCH, n) + out = cuNumeric.batched_cholesky(cuNumeric.NDArray(A_ref)) + + allowscalar() do + L = Array(out) + @test size(L) == (N_BATCH, n, n) + for i in 1:N_BATCH + Li = L[i, :, :] + @test istril(Li) + @test isapprox(A_ref[i, :, :], Li * Li'; atol=atol(T), rtol=rtol(T)) + end + end + end +end + +@testset "batched_cholesky promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + vals = reshape(T[1, 0, 0, 1], 1, 2, 2) + A = cuNumeric.NDArray(vals) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" cuNumeric.batched_cholesky(A) + else + @test cuNumeric.batched_cholesky(A) isa NDArray{Float64} + end + allowpromotion() do + out = cuNumeric.batched_cholesky(A) + allowscalar() do + @test safe_compare(Float64.(vals), out, atol(Float64), rtol(Float64)) + end + end + end +end + +function batched_eigen_residual(A_ref, values, vectors, i) + C = eltype(values) + Ai = C.(A_ref[i, :, :]) + Vi = vectors[i, :, :] + return maximum(abs.(Ai * Vi .- Vi * Diagonal(values[i, :]))) +end + +@testset "batched_eigen" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 4 + A_ref = batched_random(T, N_BATCH, n, n) + values, vectors = cuNumeric.batched_eigen(cuNumeric.NDArray(A_ref)) + + allowscalar() do + vals, vecs = Array(values), Array(vectors) + @test size(vals) == (N_BATCH, n) + @test size(vecs) == (N_BATCH, n, n) + @test eltype(vals) == complex(T) + for i in 1:N_BATCH + @test batched_eigen_residual(A_ref, vals, vecs, i) <= + max(atol(T), rtol(T) * n) + @test isapprox( + sort(vals[i, :]; by=x -> (real(x), imag(x))), + sort(LinearAlgebra.eigvals(A_ref[i, :, :]); by=x -> (real(x), imag(x))), + atol=atol(T), + rtol=rtol(T), + ) + end + end + end +end + +@testset "batched_eigvals" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 4 + A_ref = batched_random(T, N_BATCH, n, n) + values = cuNumeric.batched_eigvals(cuNumeric.NDArray(A_ref)) + + allowscalar() do + vals = Array(values) + @test size(vals) == (N_BATCH, n) + for i in 1:N_BATCH + @test isapprox( + sort(vals[i, :]; by=x -> (real(x), imag(x))), + sort(LinearAlgebra.eigvals(A_ref[i, :, :]); by=x -> (real(x), imag(x))), + atol=atol(T), + rtol=rtol(T), + ) + end + end + end +end + +@testset "batched dimension limits" begin + # Exactly one batch dimension: 2D belongs to the LinearAlgebra entry points, + # and 4D exceeds both the POTRF task and Legate's launch-domain construction. + @testset "$f" for f in + (cuNumeric.batched_cholesky, cuNumeric.batched_eigen, + cuNumeric.batched_eigvals) + @test_throws ArgumentError f(cuNumeric.zeros(Float64, 3, 3)) + @test_throws ArgumentError f(cuNumeric.zeros(Float64, 2, 2, 3, 3)) + @test_throws ArgumentError f(cuNumeric.zeros(Float64, 2, 3, 4)) + end + + @test_throws ArgumentError cuNumeric.batched_solve( + cuNumeric.zeros(Float64, 2, 2, 3, 3), cuNumeric.zeros(Float64, 2, 2, 3, 1) + ) +end diff --git a/test/array/linalg.jl b/test/array/linalg.jl index 19a25d002..c5014032f 100644 --- a/test/array/linalg.jl +++ b/test/array/linalg.jl @@ -247,8 +247,12 @@ end A = cuNumeric.NDArray(T[1 0; 0 1]) b = cuNumeric.NDArray(reshape(T[1, 1], 2, 1)) - # int/bool requires promotion to float. Will throw without allowpromtion() - @test_throws "Implicit promotion" cuNumeric.solve(A, b) + # int/bool converts to float; only a widening conversion needs allowpromotion() + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" cuNumeric.solve(A, b) + else + @test cuNumeric.solve(A, b) isa NDArray{Float64} + end # ...allowed under @allowpromotion, result is Float64 allowpromotion() do @@ -261,32 +265,51 @@ end end end -function check_svd_reconstruction(ref_A::AbstractMatrix, u, s, vh, tol_a, tol_r) - U = Array(u) - S = Array(s) - Vh = Array(vh) - A_rec = U * Diagonal(S) * Vh +@testset "solve rejects batched input" begin + A = cuNumeric.zeros(Float64, 2, 3, 3) + b = cuNumeric.zeros(Float64, 2, 3, 1) + @test_throws "batched_solve" cuNumeric.solve(A, b) +end + +@testset "backslash" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) + A_ref = T[2 1; 5 7] + b_ref = T[11, 13] + A = cuNumeric.NDArray(A_ref) + + x_vec = A \ cuNumeric.NDArray(b_ref) + x_mat = A \ cuNumeric.NDArray(reshape(b_ref, 2, 1)) + + allowscalar() do + @test safe_compare(A_ref \ b_ref, x_vec, atol(T), rtol(T)) + @test safe_compare(A_ref \ reshape(b_ref, 2, 1), x_mat, atol(T), rtol(T)) + end + end +end + +function check_svd_reconstruction(ref_A::AbstractMatrix, F, tol_a, tol_r) + A_rec = Array(F.U) * Diagonal(Array(F.S)) * Array(F.Vt) return isapprox(ref_A, A_rec; atol=tol_a, rtol=tol_r) end -function check_svd_orthonormality(u, vh, tol_a, tol_r) - U = Array(u) - Vh = Array(vh) +function check_svd_orthonormality(F, tol_a, tol_r) + U = Array(F.U) + Vt = Array(F.Vt) ku = size(U, 2) - kv = size(Vh, 1) + kv = size(Vt, 1) ok_u = isapprox(U' * U, Matrix{eltype(U)}(I, ku, ku); atol=tol_a, rtol=tol_r) - ok_vh = isapprox(Vh * Vh', Matrix{eltype(Vh)}(I, kv, kv); atol=tol_a, rtol=tol_r) - return ok_u && ok_vh + ok_vt = isapprox(Vt * Vt', Matrix{eltype(Vt)}(I, kv, kv); atol=tol_a, rtol=tol_r) + return ok_u && ok_vt end @testset "svd square matrix" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A_ref = my_rand(T, 5, 5) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) + @test F isa LinearAlgebra.SVD allowscalar() do - @test check_svd_reconstruction(A_ref, u, s, vh, atol(T), rtol(T)) - @test check_svd_orthonormality(u, vh, atol(T), rtol(T)) + @test check_svd_reconstruction(A_ref, F, atol(T), rtol(T)) + @test check_svd_orthonormality(F, atol(T), rtol(T)) end end end @@ -294,41 +317,38 @@ end @testset "svd tall matrix (m > n)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A_ref = my_rand(T, 6, 4) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda, false) # thin SVD for reconstruction test + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) # thin, for reconstruction allowscalar() do - @test check_svd_reconstruction(A_ref, u, s, vh, atol(T), rtol(T)) - @test check_svd_orthonormality(u, vh, atol(T), rtol(T)) + @test check_svd_reconstruction(A_ref, F, atol(T), rtol(T)) + @test check_svd_orthonormality(F, atol(T), rtol(T)) end end end -@testset "svd thin output shapes (full_matrices=false)" begin +@testset "svd thin output shapes (full=false)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) m, n = 6, 4 k = min(m, n) A_ref = my_rand(T, m, n) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda, false) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - @test size(Array(u)) == (m, k) - @test size(Array(s)) == (k,) - @test size(Array(vh)) == (k, n) - @test check_svd_reconstruction(A_ref, u, s, vh, atol(T), rtol(T)) + @test size(Array(F.U)) == (m, k) + @test size(Array(F.S)) == (k,) + @test size(Array(F.Vt)) == (k, n) + @test check_svd_reconstruction(A_ref, F, atol(T), rtol(T)) end end end -@testset "svd full output shapes (full_matrices=true)" begin +@testset "svd full output shapes (full=true)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) m, n = 6, 4 A_ref = my_rand(T, m, n) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda, true) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref); full=true) allowscalar() do - @test size(Array(u)) == (m, m) - @test size(Array(s)) == (min(m, n),) - @test size(Array(vh)) == (n, n) + @test size(Array(F.U)) == (m, m) + @test size(Array(F.S)) == (min(m, n),) + @test size(Array(F.Vt)) == (n, n) end end end @@ -336,10 +356,9 @@ end @testset "svd singular values non-negative and sorted" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A_ref = my_rand(T, 5, 5) - nda = cuNumeric.NDArray(A_ref) - _, s, _ = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - sv = Array(s) + sv = Array(F.S) @test all(sv .>= 0) @test issorted(sv; rev=true) end @@ -350,10 +369,9 @@ end @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) n = 4 A_ref = Matrix{T}(I, n, n) - nda = cuNumeric.NDArray(A_ref) - _, s, _ = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - @test safe_compare(ones(T, n), s, atol(T), rtol(T)) + @test safe_compare(ones(T, n), F.S, atol(T), rtol(T)) end end end @@ -365,10 +383,9 @@ end v1 = T.(collect(1:5)) v2 = T.(collect(1:4)) A_ref = v1 * v2' - nda = cuNumeric.NDArray(A_ref) - _, s, _ = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - sv = Array(s) + sv = Array(F.S) @test sv[1] > atol(T) @test all(sv[2:end] .< sqrt(atol(T)) * 100) end @@ -379,10 +396,14 @@ end @testset verbose=true for T in (Int32, Int64, Bool) vals = T == Bool ? T[1 0; 0 1] : reshape(T.(collect(1:4)), 2, 2) A = cuNumeric.NDArray(vals) - @test_throws "Implicit promotion" cuNumeric.svd(A) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.svd(A) + else + @test LinearAlgebra.svd(A) isa LinearAlgebra.SVD + end allowpromotion() do - u, s, vh = cuNumeric.svd(A) - @test eltype(Array(u)) == Float64 + F = LinearAlgebra.svd(A) + @test eltype(Array(F.U)) == Float64 end end end @@ -390,8 +411,7 @@ end @testset "qr reconstruction" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) A_ref = my_rand(T, 6, 4) - nda = cuNumeric.NDArray(A_ref) - q, r = cuNumeric.qr(nda) + q, r = LinearAlgebra.qr(cuNumeric.NDArray(A_ref)) allowscalar() do Q = Array(q) R = Array(r) @@ -406,8 +426,7 @@ end @testset "qr square matrix" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) A_ref = my_rand(T, 5, 5) - nda = cuNumeric.NDArray(A_ref) - q, r = cuNumeric.qr(nda) + q, r = LinearAlgebra.qr(cuNumeric.NDArray(A_ref)) allowscalar() do Q = Array(q) R = Array(r) @@ -422,9 +441,13 @@ end @testset verbose=true for T in (Int32, Int64, Bool) vals = T == Bool ? T[1 0; 0 1] : reshape(T.(collect(1:4)), 2, 2) A = cuNumeric.NDArray(vals) - @test_throws "Implicit promotion" cuNumeric.qr(A) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.qr(A) + else + @test LinearAlgebra.qr(A) isa cuNumeric.NDArrayQR + end allowpromotion() do - q, r = cuNumeric.qr(A) + q, r = LinearAlgebra.qr(A) allowscalar() do @test eltype(Array(q)) == Float64 @test isapprox( @@ -434,3 +457,226 @@ end end end end + +@testset "qr returns an NDArrayQR factorization" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) + A_ref = my_rand(T, 6, 4; L=real(T)(-1), R=real(T)(1)) + + F = LinearAlgebra.qr(cuNumeric.NDArray(A_ref)) + @test F isa cuNumeric.NDArrayQR{T} + @test F isa LinearAlgebra.Factorization{T} + @test size(F) == (6, 4) + @test size(F, 1) == 6 + @test size(F, 2) == 4 + @test size(F, 3) == 1 + @test sprint(show, MIME("text/plain"), F) isa String + + allowscalar() do + @test isapprox(A_ref, Array(F.Q) * Array(F.R); atol=atol(T), rtol=rtol(T)) + end + end +end + +# Hermitian positive-definite, with a diagonal shift to keep it well conditioned. +function spd_matrix(::Type{T}, n) where {T} + RT = real(T) + B = my_rand(T, n, n; L=RT(-1), R=RT(1)) + return B * B' + T(n) * Matrix{T}(I, n, n) +end + +@testset "cholesky reconstruction" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + n = 6 + A_ref = spd_matrix(T, n) + F = LinearAlgebra.cholesky(cuNumeric.NDArray(A_ref)) + + @test F isa LinearAlgebra.Cholesky + @test F.uplo == 'L' + + allowscalar() do + L = Array(F.factors) + @test size(L) == (n, n) + @test istril(L) # `zeroout` clears the upper triangle in-task + @test isapprox(A_ref, L * L'; atol=atol(T), rtol=rtol(T)) + end + end +end + +@testset "cholesky of the identity" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + n = 4 + A_ref = Matrix{T}(I, n, n) + F = LinearAlgebra.cholesky(cuNumeric.NDArray(A_ref)) + allowscalar() do + @test safe_compare(A_ref, F.factors, atol(T), rtol(T)) + end + end +end + +@testset "cholesky factorization object" begin + n = 5 + T = Float64 + A_ref = spd_matrix(T, n) + F = LinearAlgebra.cholesky(cuNumeric.NDArray(A_ref)) + + L = F.L + @test L isa LowerTriangular + @test parent(L) === F.factors + @test size(F) == (n, n) + + allowscalar() do + @test isapprox(A_ref, Array(parent(L)) * Array(parent(L))'; atol=atol(T), rtol=rtol(T)) + end + + # `F.U` (and hence destructuring as `L, U = F`) needs `copy(F.factors')`, + # which falls back to scalar indexing until `adjoint(::NDArray)` lands. + @test_throws "scalar-indexed" F.U +end + +@testset "cholesky rejects bad shapes and types" begin + @test_throws ArgumentError LinearAlgebra.cholesky(cuNumeric.zeros(Float64, 3, 4)) + @test_throws ArgumentError LinearAlgebra.cholesky(cuNumeric.zeros(ComplexF64, 0, 0)) +end + +# The POTRF task raises this itself. It only surfaces as a catchable error +# because the launcher marks the task as throwing; without that Legate aborts +# the process. Not a PosDefException: the pivot index is not reported. +@testset "cholesky of a non-positive-definite matrix throws" begin + A = cuNumeric.NDArray(Float64[1.0 2.0; 2.0 1.0]) + @test_throws "Matrix is not positive definite" begin + F = LinearAlgebra.cholesky(A) + allowscalar() do + sum(abs.(Array(F.factors))) + end + end +end + +@testset "cholesky promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + vals = T == Bool ? T[1 0; 0 1] : T[2 0; 0 2] + A = cuNumeric.NDArray(vals) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.cholesky(A) + else + @test LinearAlgebra.cholesky(A) isa LinearAlgebra.Cholesky + end + allowpromotion() do + F = LinearAlgebra.cholesky(A) + allowscalar() do + L = Array(F.factors) + @test eltype(L) == Float64 + @test isapprox(Float64.(vals), L * L'; atol=atol(Float64), rtol=rtol(Float64)) + end + end + end +end + +# Eigenvectors are only unique up to sign/phase, so compare the residual +# `A*v - λ*v` rather than the vectors themselves. +function eigen_residual(A_ref, values, vectors) + C = eltype(values) + return maximum(abs.(C.(A_ref) * vectors .- vectors * Diagonal(values))) +end + +sort_spectrum(v) = sort(v; by=x -> (real(x), imag(x))) + +@testset "eigen residual" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 5 + A_ref = my_rand(T, n, n; L=real(T)(-1), R=real(T)(1)) + F = LinearAlgebra.eigen(cuNumeric.NDArray(A_ref)) + + @test F isa LinearAlgebra.Eigen + + allowscalar() do + values, vectors = Array(F.values), Array(F.vectors) + # geev always produces complex output, even for a real input matrix + @test eltype(values) == complex(T) + @test eltype(vectors) == complex(T) + @test size(values) == (n,) + @test size(vectors) == (n, n) + @test eigen_residual(A_ref, values, vectors) <= max(atol(T), rtol(T) * n) + @test isapprox( + sort_spectrum(values), + sort_spectrum(LinearAlgebra.eigvals(A_ref)), + atol=atol(T), + rtol=rtol(T), + ) + end + end +end + +@testset "eigen of a diagonal matrix" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 4 + A_ref = Matrix{T}(Diagonal(T.(1:n))) + values = LinearAlgebra.eigvals(cuNumeric.NDArray(A_ref)) + allowscalar() do + @test isapprox( + sort_spectrum(Array(values)), + complex(T).(1:n), + atol=atol(T), + rtol=rtol(T), + ) + end + end +end + +@testset "eigen factorization object" begin + n = 5 + T = Float64 + A_ref = my_rand(T, n, n; L=-1.0, R=1.0) + F = LinearAlgebra.eigen(cuNumeric.NDArray(A_ref)) + + # an Eigen destructures as (values, vectors) + values, vectors = F + @test values === F.values + @test vectors === F.vectors + + allowscalar() do + @test eigen_residual(A_ref, Array(values), Array(vectors)) <= rtol(T) * n + end +end + +@testset "eigvals and eigvecs agree with eigen" begin + n = 4 + T = Float64 + A_ref = my_rand(T, n, n; L=-1.0, R=1.0) + nda = cuNumeric.NDArray(A_ref) + + values = LinearAlgebra.eigvals(nda) + vectors = LinearAlgebra.eigvecs(nda) + + allowscalar() do + @test eigen_residual(A_ref, Array(values), Array(vectors)) <= rtol(T) * n + end +end + +@testset "eigen rejects bad shapes and types" begin + @test_throws ArgumentError LinearAlgebra.eigen(cuNumeric.zeros(Float64, 3, 4)) + @test_throws ArgumentError LinearAlgebra.eigvals(cuNumeric.zeros(Float64, 0, 0)) +end + +@testset "eigen promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + vals = T == Bool ? T[1 0; 0 1] : T[2 0; 0 3] + A = cuNumeric.NDArray(vals) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.eigen(A) + else + @test LinearAlgebra.eigen(A) isa LinearAlgebra.Eigen + end + allowpromotion() do + F = LinearAlgebra.eigen(A) + allowscalar() do + @test eltype(Array(F.values)) == ComplexF64 + @test isapprox( + sort_spectrum(Array(F.values)), + sort_spectrum(ComplexF64.(LinearAlgebra.eigvals(Float64.(vals)))), + atol=atol(Float64), + rtol=rtol(Float64), + ) + end + end + end +end diff --git a/test/util.jl b/test/util.jl index b9c3e596c..6654765ad 100644 --- a/test/util.jl +++ b/test/util.jl @@ -27,6 +27,10 @@ const DOMAIN_GENERATORS = Dict{Symbol,Function}( :positive => (T, N) -> (x=rand(T, N); T <: Signed ? abs.(max.(x, -typemax(T))) : x), ) +# The package only gates promotion when the target type is wider in bytes, so +# Int64/UInt64 -> Float64 needs no `@allowpromotion` while Int32/Bool do. +promotion_is_gated(::Type{FROM}, ::Type{TO}) where {FROM,TO} = sizeof(TO) > sizeof(FROM) + rtol(::Type{Float16}) = 1e-2 rtol(::Type{Float32}) = 1e-5 rtol(::Type{Float64}) = 1e-12 From bd99a72c7989185377a8d9976b801c778220b4d0 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sun, 16 Aug 2026 23:39:52 -0400 Subject: [PATCH 12/49] Dynamic Mode Decomposition Example + Benchmark (#184) --- .gitignore | 8 ++ benchmark/benchmarks.toml | 33 +++++++ benchmark/src/benchmarks/dmd.jl | 122 +++++++++++++++++++++++++ benchmark/src_py/benchmarks/dmd.py | 51 +++++++++++ docs/make.jl | 1 + docs/src/api_hdf5.md | 7 ++ docs/src/benchmarks/howto.md | 4 +- docs/src/examples/dmd.md | 138 +++++++++++++++++++++++++++++ docs/src/examples/grayscott.md | 19 ++++ docs/src/linalg.md | 3 + examples/Project.toml | 2 + examples/data/gray-scott.h5 | Bin 0 -> 493568 bytes examples/dmd.jl | 80 +++++++++++++++++ examples/gray-scott.jl | 23 ++++- src/ndarray/detail/linalg.jl | 8 ++ src/ndarray/linalg.jl | 3 + src/ndarray/ndarray.jl | 47 ++++++++++ test/array/linalg.jl | 5 ++ test/io/hdf5.jl | 34 +++++++ 19 files changed, 584 insertions(+), 4 deletions(-) create mode 100644 benchmark/src/benchmarks/dmd.jl create mode 100644 benchmark/src_py/benchmarks/dmd.py create mode 100644 docs/src/examples/dmd.md create mode 100644 examples/data/gray-scott.h5 create mode 100644 examples/dmd.jl diff --git a/.gitignore b/.gitignore index 0311f9e68..d431bc365 100644 --- a/.gitignore +++ b/.gitignore @@ -17,6 +17,14 @@ logging/* debug debug/* +# example outputs (examples/data and the docs copy of gray-scott.gif are tracked) +examples/*.h5 +examples/*.gif +/gray-scott.h5 +/dmd-mode.h5 +# h5write emits a virtual dataset backed by this sidecar directory +*_legate_vds/ + # benchmark outputs benchmark/results** benchmark/results/* diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml index 70d5a47a6..1d8b3194f 100644 --- a/benchmark/benchmarks.toml +++ b/benchmark/benchmarks.toml @@ -47,6 +47,39 @@ fusion = false N = [2000, 2832, 4000, 5656] M = [2000, 2832, 4000, 5656] +################################# +# DMD # +# Snapshot matrix N × M # +# (space × time), r=min(20,M-1).# +# FLOPs (m=N, n=M-1): # +# 2mn² + 11n³ thin SVD # +# 2mnr + mr B = X2 V Σ⁻¹ # +# 2mr² à = U' B # +# 25r³ eigen(Ã) # +# 2mr² Φ = B W # +# M fixed ⇒ n,r fixed ⇒ W ~ N. # +# Weak scaling: N / P constant. # +# N = 50000 * P. # +# M is the intensity knob: AI # +# ~ n²/M ~ M, independent of P. # +################################# + +[[dmd_baseline]] +T = "Float32" +gpus = [1, 2, 4, 8] +cpus = 16 +fusion = [true, false] +N = [50000, 100000, 200000, 400000] +M = 512 + +[[dmd_lifetimes]] +T = "Float32" +gpus = [1, 2, 4, 8] +cpus = 16 +fusion = false +N = [50000, 100000, 200000, 400000] +M = 512 + ################################# # Monte-Carlo Integration # # Work ~ N. Scale N linearly # diff --git a/benchmark/src/benchmarks/dmd.jl b/benchmark/src/benchmarks/dmd.jl new file mode 100644 index 000000000..8099f34c5 --- /dev/null +++ b/benchmark/src/benchmarks/dmd.jl @@ -0,0 +1,122 @@ +# Exact DMD of an N×M snapshot matrix (space × time). Rank is min(20, M-1), +# matching examples/dmd.jl. +# +# N is spatial degrees of freedom (rows of X), not a grid side length. +# The SVD is tall-skinny: X1 is N × (M-1). With M (and rank r) fixed, every +# term below is Θ(N) or O(1), so weak scaling is N ∝ P — not the N³ of a +# square SVD. + +abstract type AbstractDMD{T} <: AbstractBenchmark{T} end + +Base.@kwdef struct DMDBaseline{T} <: AbstractDMD{T} + N::Int + M::Int +end + +Base.@kwdef struct DMDLifetimes{T} <: AbstractDMD{T} + N::Int + M::Int +end + +name(::DMDBaseline) = "dmd_baseline" +name(::DMDLifetimes) = "dmd_lifetimes" +dims(b::AbstractDMD) = (b.N, b.M) +data(b::AbstractDMD{T}) where {T} = "DMD with T=$(T), N=$(b.N), M=$(b.M)" +allowed_types(::Type{<:AbstractDMD}) = cuNumeric.SUPPORTED_FLOAT_TYPES + +function build_benchmark(::Type{A}, ::Type{T}, N, M) where {A<:AbstractDMD,T} + return A{T}(; N=N, M=M) +end + +_dmd_rank(b::AbstractDMD) = min(20, b.M - 1) + +# X1 is m×n, m=N spatial points, n=M-1 snapshots, r = min(20, n). +# 2mn² + 11n³ thin SVD (Golub & Van Loan, Matrix Computations, 4ed, §8.6) +# 2mnr + mr B = X2 V_r Σ_r^{-1} (GEMM + column scale) +# 2mr² à = U_r' B +# 25r³ eigen(Ã) (LAPACK xGEEV) +# 2mr² Φ = B W +function total_flops(b::AbstractDMD) + m = b.N + n = b.M - 1 + r = _dmd_rank(b) + return ( + 2 * m * n^2 + 11 * n^3 + + 2 * m * n * r + m * r + + 2 * m * r^2 + + 25 * r^3 + + 2 * m * r^2 + ) +end + +function initialize(b::AbstractDMD{T}; mod=cuNumeric) where {T} + # X1 is N×(M-1); the SVD backend requires m >= n. + b.N >= b.M - 1 || throw( + ArgumentError("DMD snapshot matrix is N×M with N ≥ M-1 (got N=$(b.N), M=$(b.M))") + ) + X = mod.rand(T, b.N, b.M) + GC.gc() + return (X,) +end + +_dmd_T(A) = A isa NDArray ? cuNumeric.transpose(A) : transpose(A) +_dmd_row(v) = v isa NDArray ? cuNumeric.reshape(v, (1, length(v))) : reshape(v, 1, length(v)) + +# svd / eigen return factorizations whose stores the lifetime rewriter cannot +# see, so those stay outside the macro. The GEMM lift is wrapped. +# +# Do not form Diagonal(1 ./ S) inside @analyze_lifetimes: the rewriter treats +# `1 ./ S` as a last-used temp and destroy!s it, while Diagonal still holds +# that same vector. Scale columns with a broadcast instead (same math). +function _dmd_factors(X, r) + n = size(X, 2) + X1 = X[:, 1:(n - 1)] + X2 = X[:, 2:n] + F = svd(X1) + rk = min(r, length(F.S)) + return X2, F.U[:, 1:rk], F.Vt[1:rk, :], F.S[1:rk] +end + +let body = quote + Sinv = eltype(X)(1) ./ _dmd_row(S) + B = (X2 * _dmd_T(Vt)) .* Sinv + à = _dmd_T(U) * B + (B, Ã) + end + @eval _dmd_project(::DMDBaseline, X, X2, U, Vt, S) = $body + @eval _dmd_project(::DMDLifetimes, X, X2, U, Vt, S) = @analyze_lifetimes $body +end + +function _dmd_compute!(b::AbstractDMD, X, r) + X2, U, Vt, S = _dmd_factors(X, r) + B, à = _dmd_project(b, X, X2, U, Vt, S) + E = eigen(Ã) + CT = Complex{eltype(X)} + Bc = X isa NDArray ? cuNumeric.as_type(B, CT) : CT.(B) + return E.values, Bc * E.vectors +end + +run!(b::AbstractDMD, X) = _dmd_compute!(b, X, _dmd_rank(b)) + +correctness_supported(::AbstractDMD) = true + +function check_benchmark_correctness( + b::AbstractDMD{T}, gs::GlobalSettings; mod=cuNumeric, atol=1e-3, rtol=1e-3 +) where {T} + mod === cuNumeric || return "skipped" + + Xh = rand(T, b.N, b.M) + X = NDArray(Xh) + r = _dmd_rank(b) + # Values, not lifetimes: compare against the baseline body on host and device. + ref = DMDBaseline{T}(; N=b.N, M=b.M) + λ, _ = _dmd_compute!(ref, X, r) + λh, _ = _dmd_compute!(ref, Xh, r) + + mag = sort(abs.(Array(λ)); rev=true) + magh = sort(abs.(λh); rev=true) + return isapprox(mag, magh; atol=atol, rtol=rtol) ? "pass" : "fail" +end + +register_benchmark("dmd_baseline", DMDBaseline) +register_benchmark("dmd_lifetimes", DMDLifetimes) diff --git a/benchmark/src_py/benchmarks/dmd.py b/benchmark/src_py/benchmarks/dmd.py new file mode 100644 index 000000000..e0b74180c --- /dev/null +++ b/benchmark/src_py/benchmarks/dmd.py @@ -0,0 +1,51 @@ +import cupynumeric as np + +from core import register_benchmark + + +class DMD: + name = "dmd_baseline" + + def __init__(self, T, N, M): + self.T, self.N, self.M = T, N, M + self.r = min(20, M - 1) + + def dims(self): + return self.N, self.M + + def total_flops(self): + # Same breakdown as benchmark/src/benchmarks/dmd.jl + m = self.N + n = self.M - 1 + r = self.r + return ( + 2 * m * n * n + + 11 * n * n * n + + 2 * m * n * r + + m * r + + 2 * m * r * r + + 25 * r * r * r + + 2 * m * r * r + ) + + def initialize(self): + X = np.random.rand(self.N, self.M).astype(self.T) + return (X,) + + def run(self, state): + (X,) = state + n = self.M - 1 + X1 = X[:, :n] + X2 = X[:, 1:] + U, S, Vt = np.linalg.svd(X1, full_matrices=False) + r = self.r + U = U[:, :r] + Vt = Vt[:r, :] + B = (X2 @ Vt.T) * (self.T(1) / S[:r]) + At = U.T @ B + _, W = np.linalg.eig(At) + ctype = np.complex64 if self.T == np.float32 else np.complex128 + _ = B.astype(ctype) @ W + + +register_benchmark("dmd_baseline", DMD) diff --git a/docs/make.jl b/docs/make.jl index d95c9f4eb..628a272bc 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -38,6 +38,7 @@ makedocs(; "Initialization" => "examples/initialization.md", "Monte-Carlo" => "examples/montecarlo.md", "Gray-Scott" => "examples/grayscott.md", + "Dynamic Mode Decomposition" => "examples/dmd.md", ], "Performance Tips" => [ "Kernel Fusion" => "perf/kernel_fusion.md", diff --git a/docs/src/api_hdf5.md b/docs/src/api_hdf5.md index c60787935..5ad2aa34b 100644 --- a/docs/src/api_hdf5.md +++ b/docs/src/api_hdf5.md @@ -22,6 +22,13 @@ restored = cuNumeric.h5read("checkpoint.h5", "field"; layout=:row) Synchronize before accessing the file outside the runtime, moving or deleting it, or exiting immediately after the write. +`h5write` removes a leftover empty or truncated `.h5` before launching the +write, which is what otherwise aborts `HDF5CombineVDS`. A valid HDF5 file is +left for Legate to overwrite. The `*_legate_vds` sidecar is left alone: after +a Legate write that directory is the data, and the `.h5` is only an index. +`h5read` rejects a missing path, a directory, or a file that does not start +with the HDF5 signature, instead of opening it in a Legate task. + ## Dataset layout ```julia diff --git a/docs/src/benchmarks/howto.md b/docs/src/benchmarks/howto.md index 6e6e8e6e4..851f3bdd3 100644 --- a/docs/src/benchmarks/howto.md +++ b/docs/src/benchmarks/howto.md @@ -47,7 +47,9 @@ n_correctness_iter = 5 - `cupynumeric` / `cuda`: optional comparison backends - `check_correctness`: one CPU-reference check per config (not per timed iter), recorded in the CSV -Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, `grayscott_baseline`, `grayscott_lifetimes`, …). Names must match what `src/benchmarks/*.jl` registers. +Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, `dmd_baseline`, `dmd_lifetimes`, `grayscott_baseline`, `grayscott_lifetimes`, …). Names must match what `src/benchmarks/*.jl` registers. + +DMD's `N` is the number of spatial degrees of freedom (rows of the snapshot matrix), not a grid side length. The SVD is of the tall-skinny `N × (M-1)` matrix `X1`. Thin SVD plus the rank-`r` lift is `Θ(N)` when `M` and `r` are fixed, so weak scaling is `N ∝ P` (same idea as Monte Carlo, not GEMM's `N ∝ P^{1/3}`). The flop count is in `src/benchmarks/dmd.jl`. ```toml [[gemm]] diff --git a/docs/src/examples/dmd.md b/docs/src/examples/dmd.md new file mode 100644 index 000000000..9677cc9b2 --- /dev/null +++ b/docs/src/examples/dmd.md @@ -0,0 +1,138 @@ +# Dynamic Mode Decomposition + +Dynamic mode decomposition takes a sequence of states from a simulation or an +experiment and finds the best linear operator that advances one state to the +next: + +```math +x_{k+1} \approx A x_k +``` + +Its eigenvectors are spatial patterns and its eigenvalues say what each pattern +does per step: ``|\lambda|`` is growth or decay and ``\arg(\lambda)`` is +rotation. For a field on an ``N \times N`` grid, ``A`` is ``N^2 \times N^2`` and +far too large to form. DMD never does. Stack the states as columns of +``X = [x_1 \; \dots \; x_n]``, split it into + +```math +X_1 = [x_1 \; \dots \; x_{n-1}], \qquad X_2 = [x_2 \; \dots \; x_n] +``` + +and take the rank-``r`` SVD ``X_1 = U \Sigma V^*``. Projecting ``A`` onto the +columns of ``U`` gives an ``r \times r`` matrix whose eigendecomposition carries +the dynamics: + +```math +\tilde{A} = U^* X_2 V \Sigma^{-1}, \qquad +\Phi = X_2 V \Sigma^{-1} W +``` + +where ``W`` holds the eigenvectors of ``\tilde{A}``. The columns of ``\Phi`` are +the exact DMD modes, back in the full ``N^2``-dimensional space. + +## Snapshots from Gray-Scott + +The [Gray-Scott](./grayscott.md) example writes its `u` field to +`gray-scott.h5` every 20 steps, one flattened frame per column: + +```julia +snapshots = cuNumeric.zeros(Float32, N * N, n_steps ÷ snapshot_interval) + +# inside the time loop +if n%snapshot_interval == 0 + snapshots[:, n ÷ snapshot_interval] = cuNumeric.reshape(u, (N * N, 1)) +end + +cuNumeric.h5write(SNAPSHOT_FILE, "u", snapshots) +# h5write is asynchronous, so flush before another process opens the file. +cuNumeric.Legate.runtime_sync() +``` + +The snapshots never touch the host: they are assembled into a device array and +handed straight to [HDF5](../api_hdf5.md). + +## Decomposition + +Run `examples/gray-scott.jl` first to produce the file, then: + +```julia +# found in examples/dmd.jl +using cuNumeric +using LinearAlgebra +using Printf + +const SNAPSHOT_FILE = "gray-scott.h5" + +function dmd(X::NDArray{Float32,2}, r::Int) + n = size(X, 2) + X1 = X[:, 1:(n - 1)] # states x_1 … x_{n-1} + X2 = X[:, 2:n] # the same states advanced one snapshot + + F = svd(X1) + r = min(r, length(F.S)) + U = F.U[:, 1:r] + Vt = F.Vt[1:r, :] + + # Σ⁻¹ stays a Diagonal instead of a dense r×r matrix. + Sinv = Diagonal(1.0f0 ./ F.S[1:r]) + + # X2 V Σ⁻¹ appears in both the projected operator and the exact modes. + B = X2 * cuNumeric.transpose(Vt) * Sinv + à = cuNumeric.transpose(U) * B + + E = eigen(Ã) # always complex, even for a real à + Φ = cuNumeric.as_type(B, ComplexF32) * E.vectors + + return E.values, Φ +end + +X = cuNumeric.h5read(SNAPSHOT_FILE, "u") +n_points, n_snapshots = size(X) +N = isqrt(n_points) + +λ, Φ = dmd(X, 20) + +# Only r eigenvalues, so ranking them on the host costs nothing. +vals = Array(λ) +order = sortperm(abs.(vals); rev=true) + +println("mode |λ| cycles/snapshot") +for i in order[1:min(5, end)] + @printf("%4d %6.4f %+8.4f\n", i, abs(vals[i]), angle(vals[i]) / 2π) +end + +# The slowest-decaying mode is the pattern the simulation settles into. +lead = order[1] +mode = cuNumeric.reshape(abs.(Φ[:, lead:lead]), (N, N)) +cuNumeric.h5write("dmd-mode.h5", "leading", mode) +cuNumeric.Legate.runtime_sync() +``` + +On 100 snapshots of a ``100 \times 100`` grid this prints something like: + +``` +100 snapshots of 10000 points + +mode |λ| cycles/snapshot + 17 0.9945 +0.0091 + 18 0.9945 -0.0091 + 14 0.9928 +0.0000 + 19 0.9825 +0.0202 + 20 0.9825 -0.0202 +``` + +Every ``|\lambda|`` sits just below 1, which is the pattern settling rather than +growing, and the oscillatory modes come in the conjugate pairs a real operator +must produce. + +## Notes + +- `svd`, `eigen`, and matrix multiply all run on device; see + [Linear Algebra](../linalg.md). +- `Diagonal(1.0f0 ./ F.S[1:r])` scales by ``\Sigma^{-1}`` without building a + dense ``r \times r`` matrix. Note the `1.0f0` — an `Int` literal would try to + widen the `Float32` singular values. +- `eigen` is always complex, so `B` is converted with `cuNumeric.as_type` before + multiplying by the eigenvectors. +- The eigenvalues are only `r` numbers, so sorting and printing them on the host + is cheap. The modes stay on device. diff --git a/docs/src/examples/grayscott.md b/docs/src/examples/grayscott.md index dc0fca426..59ef00463 100644 --- a/docs/src/examples/grayscott.md +++ b/docs/src/examples/grayscott.md @@ -5,6 +5,9 @@ using cuNumeric using Plots +# Flattened u snapshots land here for examples/dmd.jl to analyze. +const SNAPSHOT_FILE = "gray-scott.h5" + struct Params{T} dx::T dt::T @@ -63,12 +66,16 @@ function gray_scott() n_steps = 2000 # number of steps to take frame_interval = 200 # steps to take between making plots + snapshot_interval = 20 # steps to take between saved snapshots u = cuNumeric.ones(dims) v = cuNumeric.zeros(dims) u_new = cuNumeric.zeros(dims) v_new = cuNumeric.zeros(dims) + # One flattened frame per column, the layout DMD wants. + snapshots = cuNumeric.zeros(Float32, N * N, n_steps ÷ snapshot_interval) + u[1:15,1:15] = cuNumeric.rand(15,15) v[1:15,1:15] = cuNumeric.rand(15,15) @@ -79,6 +86,10 @@ function gray_scott() u, u_new = u_new, u v, v_new = v_new, v + if n%snapshot_interval == 0 + snapshots[:, n ÷ snapshot_interval] = cuNumeric.reshape(u, (N * N, 1)) + end + if n%frame_interval == 0 u_cpu = u[:, :] heatmap(u_cpu, clims=(0, 1)) @@ -86,6 +97,11 @@ function gray_scott() end end gif(anim, "gray-scott.gif", fps=10) + + cuNumeric.h5write(SNAPSHOT_FILE, "u", snapshots) + # h5write is asynchronous, so flush before another process opens the file. + cuNumeric.Legate.runtime_sync() + return u, v end @@ -93,3 +109,6 @@ end u, v = gray_scott() ``` ![Simulation Output](../gray-scott.gif) + +The snapshots written to `gray-scott.h5` are the input to +[Dynamic Mode Decomposition](./dmd.md). diff --git a/docs/src/linalg.md b/docs/src/linalg.md index bfa798b6d..a47daf389 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -138,6 +138,9 @@ falls back to scalar indexing until `adjoint(::NDArray)` is implemented. Prefer `S` is real-valued for both real and complex inputs. +The backend only factors tall or square matrices (`m >= n`), matching +cupynumeric. A wide input throws `ArgumentError`. + ## QR decomposition `LinearAlgebra.qr(A)` returns an `NDArrayQR`, holding the economy-size factors diff --git a/examples/Project.toml b/examples/Project.toml index 9153064ac..44251c5ef 100644 --- a/examples/Project.toml +++ b/examples/Project.toml @@ -1,7 +1,9 @@ [deps] CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" Legate = "1238f2cf-6593-4d60-9aca-2f5364e49909" +LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" +Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" [extras] diff --git a/examples/data/gray-scott.h5 b/examples/data/gray-scott.h5 new file mode 100644 index 0000000000000000000000000000000000000000..09bf0a657511c905f1b89064f3a8168b1f3cc005 GIT binary patch literal 493568 zcmeFYWmJ{n+wDs!5-NgpOCu==O3X>8bc@~Hpppg(7$~S92AHTQf{KFOieh&cDhgtv zsHp7wefR#4amM+2KAipGWsEhR1&g)rXWe;S^Ea>D03W}u@>=pzUAjp8=cnWU{7B@; z{`a5k{?7{m{a4$C1^7ft8C(fiFri*N{&SrprTkz2^Pe~V=PEBHAt@ozxvPYPq>`j$ z(%Jv>ITHT=dGi1H8m*Cd^uKS(@V{>5|M`%Rlr)x*kdl&=lu-KL&FKB7wg1=e(mf?4 zdl=|T$o%)~1`_IBB-As-^G@RKrRBSbZze4%@t@wA_rJSDV$**4E?0f&b}&|LKAMUwgnV<_atxT0zm-6kSw>M*auoIjNu*!<3w243EjGYaI_oDdEj6V5-oBY0>+1XX{8@!^FKnmYuuEh2y` zG<~@&&y$vq+?ng@!ljK))NFO&0gXb~FH^&$*QWTiE(T|pW+P_84y-f0gt>p4(0xNE z?u_b5qe+Gw-(bb#^-i?xPe!@;@$cPWYRiSQ_@!6a1P998L=EiAe{W(+LiS0c{VY-$Adiz-7$=jir`6>^69jj2V_ZB8?Xho!r zJU7nQAT3Pzv5y_6)b{7a8=g#F8bHHOA=LO8!O=z0Y&;snu`^;QBNNSck0bahAdI6& z44|!HAgAT}vg4E|JEXyV*)FUZ?abUu{js8|6OQ}az<0+uysKZ0O*%FB+w~C|TH5it zq$_PVYqRQ$8AtEz!z~T2R6XX+wVpv-loG}TBcizNeGD)Bjb&S<;{9dr=Cyg$P?2-Lu3Jx@`#l43|F+}M(HcXM= z#g^_IUaL!;YZkN(apbyKceYFWGGcTv56lavLqs&YU5KT_l{ikE633P=G2D7Iiq};m z*r686^R+=-+Tq7QC2vkJCEw3>R2r{_o6jYorQv`>5;I`ewhs5*Phi%%ml#$h z$#9(>tgF{!{5ea;nf0ZOIw-N%m+#wy`LuHc7hQ^GQlB_BSjW@lLL8GkVtG&^nr`6{ zRPhbvr3XQ5=Hl>quFd1M{ARK?yHWYhe<5AJ&vM; zVFcgHhjMR05I-ODqim%Y4{sCiz2m|W1D*N9T@{W75{Q(tLrmjjJS$v_(AuN8uJ;@b zOC)HnsYII(y4){g$-xU8IV9Vi>NdXoyf~P@io&@yF`5rg#!{y$jx(L(xN>6*qXMIt zpB2tIqe9s5GLVImVtn6w^2c3w2K?>Mv8hf}&ojohy%HGsus3QH#zHM%H8#9HgzLQ? zVWHn2TxjUZ&)>A!E6|+9fBSHWy&D_;c~dVrh$$Pxm@zeq?GIx(xh|F+ez9D+DVjNq zWbdhAwDB3hxd<`9q#hHM&R&GlZ+?7V^8xxkM#e}cJPCY(32BPlr}iu!jVIZY;l z#&bj2JU^K0?E}aRABIiwpsKYShqgQONOwoNEU=^F^keYyG{$iAUT_Hvz`PS_*nDgQ zYR?@<>e1(T7B9i|YrFHOhc5rQSa8M)2mTIqV~=can!5*5Wnu^u{)X}O^KhOu3+EWe zP%*Z_Tr?qoUTr=ctM5sZW;dSB?Z{S`me}` z{_`rND&Ins?Kh049N*7X=cR7OwEbp73!8pyuOsy?`tq59*cVO@;DJ>k+;Ct3E0+Z^ zW~V=c6MgvLfCtsKyV2IAA3b+E@T;>eFB$Y^Tap>KeC$P8*<55AIU_8~1yS7xp{wI! zT-4o)*I&+{-|+^dU6Z7Cp%PVZ=~8N-1(zB-@I@b2#$-^YPE671L#Y63Ufd?;<^*FN+G)Te>EPF$N74Co9P+f?5q2>Qr9YA} zxBnWf|FIwXmu^Av^jBdeN2~V`DmJwUMN&L9Yjq?DhL6Ox$xImlBMLGXf#G zeH7ZHa{`AGe-$es-EjqbRyJeG0vX=<*@Kg+bXhpmjBWRA*#6Ohb7nfxG}MW? zcO2OG$d+U5d-IOHDVy&ZaF&KPTP;;7liZy;3G!@h@5Ha6Qf!(yku%gC`LyyK;@9M( zv+-ce&WS{q31jg3&LWt7--xWWhcV;dJ**tqiu`t27WMDR`bm1s^fG0O_)ZyTt$9q@ znr9za@U@l+AD`2wlARWVB74%?svA8!cj2sPDPDT^8@FzM#So7VI5&I-8=Cz2bEXn^ zT|5G#l6i1wiN}T5MBKAU!pyupTvXYChT&(Bc_nWDilNy=22ph#u+&V2W^0|B4DVNm>g7+1f*jFyL3TyO_Iy0f|ENi4_b zS#ZpY7OekLf{4je;TSRyU6+kUPQ`rON-Tu!sl7OEdI_6OHo&m-4@~}bVTg4Pwz{iP zt5TiXlT~RvP>Fx#<(a%wn%&xdqu|CzJoxhpHys{eq5E~H53R#<|FhUK_9Q9-Q@Fc* zI1jYD({fY~?%aO~HhY(&PIWADyAMTU$TU3g%0Z~gMwHf6BVqhi4D0g-e>?w1?mrm@ z?~~)pnW8N(S<1#qa+2$Jyj=ATDbi0xtaJ<4CF-!;`y|5CYSC78099kE&{uLVGDoB_ zuI~iSdKbbqnikw(@&~T^RT%ERP>jVWG`=2<&FZN*W}k09t>NiEDqUPINBmk2%n z8Ai=N;c)LK#)#*y-Z!A`)IGRQs>hk*rx7N17*qZCLGgDvCM_t#z@uBSRBa1>{Mv-W zC)2sedm8=p22x|K8?W1`Gj#M*^j}r5Q{)FpD9RMsl;*=ljx{N6KV5x#m2IlvhDEEEyqn6XCIFHq1Nc;M~?CoOap? z$8j}Sdh`sUj$VZMy>qzka}+@pm9QOBhHK`Vu<+J;Vbvtbn+C060tkX0zO zT#c=FSL0&uOz!(Mn;97sSobxKe&Ptx@z$r}uU6E}sfK;pGQfT|<}}X4N7wlX3tEm$ z!}aKIx)UqU>_f`@gD{ld57{p}p>I?Q<1uUT;q_`bsINlp(|in5$j8WUE1`HXA18xX zVMg0(WVsY#>w-+$2c$6c@l>9-8&0FV5Kf!yz_CHPTM2BRyx5Hh(82d0C#MHPD%R6y=^L@b#VOXN3%&_Mgj5QZs0$I*x982eNfv07nPg)5A)Yzgs@T zAoVOV)pudk#8S*E+=NF;JJ2>_F9zuxK-Rzmn50#ONlF!1<4}QL`Bmszeh@44k7Cr| zQ`k4Dj;&%-kesT!AHvSY-^I_V1*7$KO(_#*3xWH??lsUotWO*iOnZE^U1|7 z?0!vwO%C1pDOH)O7FQThUB-ydO-xwRWXdI?-Neh8v_76e|DrS+CCug0=d-x> z<`g>DjG>(FAfCJy%ITiuJr!H7`Dwrl!_}G9rcBSp%1l!1$x}yF*|$`U{hHJ`d6_z2 z*=zE`MQs*b(qrLaLz)aR=D4G#RDEwwG+8pT#EPdyyYlxldGAvOr%I-Cx{H0U}(izR;Ae6mEFn^kn! zxKfw3?fT+Idr@nv2}hZl({iLGZ_Kdb+(9-hkg?-!(QeF>Oy+mWWTV(W%HmS^XG}6b z_L$BEE#o;oZx}reN7Jd+kBjP^xxBw6rH2^M%}tx)Xp;E*FL74S!~W#$1EvRYekJUHr(Z6$8DnBKesZu@p%R}KS|@z`nkl4 zB-Y)V%F*`YIK0mgzTY0fs*hgmJJ*3mH70x}qr<^Y>dc5zqh_i)_ldq%NY~`BnVO70 zs>zE=+VmW$OFeS~sy*$+?*S&fGt-PcQY_eSWN+?Lv1W*9SLSRcw`|H_Y+RaHOLG|N zG>gBUPNwVF(OfiqAgi|y;Nx01&P=r89ZGBqP-6Sh9^BGfg~gjycx=21 zOE#&{Lq?S?)6{7iqQ$o=x-3|!&&$^hsr1o^3ZG1f{bno`>-WXZOulo^U`uW)w-wB0 zPxBdUd@_N4=EJ#jbu>K`eR+AP1E*Xzr1D%9vRs}Aqh#54zbx0~$g$|3JPTYEXnRqf zt6SumJzkOO?cF%%M-RH(?a8+B>QuO>Nxd#Q++eB4=ba50{Wp`}=ZU`Fo6b37Qn+Jp z5`uMZU>SUNb*!a83u*P@`dMr-HRgKCU&RdogO@+)05kJsQ8d2b$+l}&9JN@DZ@x=$;)Z72+V&jV$327T?pNqv)`-o2-s9)J zclfaCEy|a_#iLj6@j2-ehE{yXyZmL4X(3n7uu`IGxtX(oj+$#@j)uv zE+%vTnW?-SI)<;eCh%HY0PDXw@YYssMy`GadM0IX z$M_V6b)UtLLniW4^-vn?g!9WuSAICui?7mT>Hh2)Y&M_3yd{+|P^p0Ts(qNLUW?-u z$FS?^G3-`6jGJxMnB`UtYnvLxsvbi^>}lu(TtH@*di-jH0^CsL9uVwezfe{J2aVfGAlA#BXyRim0`ya!w!>8~x=Nvv> z$`qQ17~9q)?roSzmC-|3K4bu8);V&yl{zgdKVjv05u<)ykM|4mMBK3)DWR)yY|DCN zzu$uO_GQSwwiP$DH{#pjLe#sjhU4Hh@YYxhJ1Oy=`i;=Pwh8~@x4|NMC$xh1pr*1C zid!@JaCR!a3TM#c{b<_C#jy7V@NbPFhl{hiV%9m-R;`D&at=)M7vi*P4kjw+W2W?a z@TG|Bq{{I1=@zJbU58mBc08`T0!OR!(Ea5KXb0xQ?nnWqY+Q{36yaj>I;^?05l1RF zq1VVv&MuzIjV4og#BKH8E6>N^kZKDk(>y$0Hk zHeuzBU6`3%ft7vBp*wmrG(Q!={8v7VR;|G2$9WjNI1hcM1D{-&9 z1nNuoV4Fn^TI7zwX6j)K53ItLfbBTiMQG+OYjMzL4Mx8$z|xMDxFsjz-msOpQ=E^x zgH|C#c?~AVXRtACCa;$dV|l7S{k6?_{-p$`7#>B<_~mF=GefjJ35%wrLch~e{9C*p zn`A2B=6@WTA1@&??*`(+uj6&;1^ilf1QSiFaP7oS6#8w&uEG)wU%3wIHAQ$gyb!Mj z6~aKh2_MkNNHyNQZpMbP`6gn^1<(e21|96Y~B=sZPG zlG}~Cu#>pgbQ3F!USOKmd${g-kEi~xFy8qdY+NtnPVOm4ZK_3G?mi5#-HmMHGQ51U z1#g@;!>Mc&r1Cal!mU&mnT?~x!9bqrZo~_hK0&YBPCR;$i431nIR0TAhPlrJ>+_NO zrVQ;Tj$`?(J23zG4ntjkq4OR|dNfFJq@e^oMeghM>K*pfKSd7_C+~Q785nW~y_JvR z_|}8CqrD#s{#0UE*4Tiy8GUY_9-WOyR`FSNdWMf9e3n4j|u z3kKiCq7PT$ol*ziDv>+3okplyGBvhD@$e5Lp8Nd@%i@c1uXYlS|4qcGt>e+2orT{I zi}C8>e(c_I6`!}h!}s43?D0#1H@Ed|nDh|ZTFGFDIlmbgJ z2RkndJ$*UASVBG;^Q)!?yW9@o&aA7+)LD<7aI-b;NTdchAGk+`;gC8iSGlCgAV9 z99;OY33=y^A{Hf(Q$=5{%jw5sw&qygupm@~AW4RsUkIsUReZ64Zk$#^SvP3@EtWr3Wu8F~UihWRpW?jsJJgvo<_zV4foeSBU4h356VPix7%Vysf3Zb78t2&&xWo&AmUP6dRh} z`%vwG4U4NRIW5AJAF2&G)K-@g8k#ICR^@=*%1r(f&b`CGqJ8XQ{P`JzF*ieypD-Q< zu}d(?WV^_l&tgaDD{PDJpo^w->14;Fkc;Rpq%@HGYmI-_ODT*a)U0dCs^==(CPn9cIE6APxcY4M6EuMH44FOC=TM8mjU#j z>Bp#YZ~7H^@O*E=>E6_rk>Sp%QIP-M8yix*@xyy6Jevxj zUB3_O3vMAS^c!B9%dsL@osaJuG3vAp*LLm4nDgYM7GD-D2;#5g5UL#u<=wHNT=!rA zhvx_Jp3ojY|MX#frHGSML9+xG<{ouo@N@@WSZ2r0)t8XtYlWaBQ+!<#jK?OKm^^1A zddQwYMBa0Ft4i>RQ+L*inB4rbIhR;CaQ$Cb%0+q8)-I5q1tHuuF`RZ45ey#}!G2%D zxa(91t*nA+6%@cHO1`vOz~1bN>1eRvioCutomN{N>GS^kDKDH}19V&y%K3JSf`T^+pY6gG_NfHySq27Guzk zGI+<=VZK`v5_~&xju`j`8$%}Ux8mFFPJF3LmRk7n%Z*^VD+~7fB9a|e(HtTfO@*D2 zT(&Blp<1Dw92?9m2Jqy4A7*%avhcV&J?FYGV!1Pa`$wUzT^4O7mZ#p9}hJe?X!^%ap!@s8n{oLCl*jpeCc zG4wQ!;9c0%;aMQl`T&1BTjJaraawMthNK&ie6zObB~+i=_9{80L%W$#MtTnv}y5X@&*r%T<6XK*Iv4R%>L&-^i_X&|42vA zk}df3>;h62HDOFgCoWT0<&P1)xcR3wI|MI&x6Oln{`xaaCzOBWqImCYEHm5^m@#4? zf4U5$Nk=@}72>#kPc*atL~vr880V5;&gdRM^GqN1G4SN#ZV*b(ZTE1$f9`!0-QZU} z4;#;J!w}y~IN9+5ll#c>q=OpuMjFxYi4DJ8?#CA~Vy`F`dPkXHaamEc66_b75@>RK zAkQuzNSWXSN<|1wJjiKVRv`%s5Z(6zw){3uVo`D5_b;v7#V> ziaQ6=W9mR!_e$Uf-8g3Lj%M)N2xeXiWuBPVk9YesUvSj3^F8<`LhQ|zx7=l4|8{?S ztRFI7rD1^f7F?Nn0jnLG(5JBzk1MLudU!9^wOjK{=YCwd(}N>KKOYecUG7gLdkCiA z(IbKX3?4;N;+6Hjg*9dx6 z#IQmmo-+*-xb9j!WroFZalaT|-x|qXN5W_|djMBI4&c`=zLawI&0e~Eon*lk@{T;&%bk4aLok>h6T_L% zJDLd_V>zfKj+r8_GTs}*wG*P)|42CJRfjM#GKh7BeoWc!#hhAjt&A(fwmDNb-W%)c zWN^IM42_3|Agm$}|Ato~%=I?zCx3_AL3y6-smbKyCcJUdj*pyOXaz4CHw18|btngE zL^8V|nhn`8ESHI4YfKa)o5QKyGnB911W_U0pZZh1IWv-6N>{2)66f+42ae9oN64F= zxVFd`7N(JcXDmYBxb0YGa}kY!?{WK`3}2j6q0DaszFKI-A+wx#;5j(zjxXEef|)QX zjGkite6ACmQ8$v4UBbD(ID|@q$CerUQ!>F@urqSuJr}OXa}t_OANqZ^p_c!B+}hO_ zev?em*x`pKMsqMauvo;OhcTkZW9>a>8GnrMPD=SId4zycHwLZFXq>Y&lP@( z_47ly)-Q}E!$Wz`X#iga2Qu@UFR!?HakDnW+1sD%K00#2P&=BLTCp+CoH;^s&`q6& zgIDa)x}LZZKSnUKW$4jRj^q97ap!L{LiTrJzR)H5e=%USnia1v>B~>y?)*B|o6cwb zsU8r-G0%fIXjdQ=^!&N_x;IBCd+^?8F_wd!`F30%W=2|b^8|AaX*OckTYb8X>BCM_ z%dmXO0H6nb{?{?-4E$~2z z7gg1Lxp9Io%`SLT@`wjp-nr3gr|_iwap1dqHoOyML7XyTK(Zb+t28-rt|}FUXK350 zSNJ2Jfmh1|Vfru>53Wzf;p5BEJ8l;)pS%dG?u}S=a7CsYbHv_K5#z{!X(Be(vSe(aF{{n=cve+|FO!woX?|B~^p)f2rbH(F zR^n8{GVI?v3hSFf@h^7>YUStSvP3bAK3605&nBXbKRF;{!8Uj^flh3|e^x zcMFp^>2)LzJ~R@Zlm>LzuEUTGQ!wTFV4)XHMBSGxq&zRi(TM#>EVzP$*WY4~Q3qbW zk)z#K!EN<~)_p;ZYqL}s9^aMnzdCV){a-l!{D}4HFQIw!Hb%8wgk8`Hlr$Xv=A4dvS9{rP&jBA<1f!Kk98u+JHXB|0PFy&y^W9F}7B&r+18)gb4I$XWCn zVeiBeyzu^;YIj8 zZ8c&R6`(^t9}z-lD7`*K==O0ebnDNwA{9OtqwIKM1CB|jVgAGMcq05@+4r*G=ClSi z4|ZVqmLquKUylag`&gjy1k)5A;Bwz9D4%>B4%t<>dA1B`+NC(Nb}h!3u7ZN$3i$6@ zhBKp=qUP`tIO;Ef@z4xbl+Wa@mBYDzL?Gp@EGh9+hF5lIQ1WTnGwv8Rvcj@B5St#_kMxeg!eeL<(CTYRxXVH{tU0nA&ljwKvP3?z_T-~cXzkAHQ@P7*CS%OUvblF0Pkr#@&l#rNsEZ)wH%s0*|@%B3FJ!)AX&K{UbD79$zdBV*O#I~@b{K8EAg{a9wu+e!~61;81$k5 zkK0#cjB^q8))wQI-v<0lC_%}fQdnNgWcQXdcAAjPOHGs5WHpSvEyDP-rXTC8bm-Up z4{TmtK;FV_uxnn8@$(AMx}XR}ex-Q(V>^y4+KI@O+u@^8iYw=fv0&O-yiQq%51mV( zbz(EdIc-Noq3{F9?t$&#N~}=Xk4{+!(D`*H`#()*x2JPh96Ox}W@G7=mB1Ur{2AZb zjyi%Ns-F0ao3G9xchPS24BLYIysfCcx(iztRbh4SY8W)_$2iNq&^0W_?F+kb?cpBW zE!l^8Wry&0;8EybJPDtk=djhJ4i@w3;Wz6lia!fq!Qu?D=2JMYb{2mxnZ!9_Zf|cC z@xxj2ma;hm?{sDQv1WV-yNu4C4r5@|A=EV-f$gBvSl4kLT}AxqJmf42rk%pCW2f*$ zXt>*YUWSLx4Q#n|7ZYS3BeeZFrWU_O{M>i&{wz3fTPFSgX3(vB8ixf7&&=Lw{B&U~ zA3RRvC7S^pF|;46-3|HYnH)R1e8j!g_p!9#IyxELgy+S32>SjQ`!7C)g3A-+cs+vG zw?}CH@C<)iUt{soW|-H1!RX93Y`gde$#*0u+ayK1cp0v5&lEg1lXH~PY3nwZ9d~B( z@Zw1{I68tBmt(jMzO)@|&k!jco^S3#<>EF}4{3(n_xH%|`w^R?zX{&_11BoKgwD&S(DGwwHe=2m-hv?YTJ{^g0Kw1T~pc6 zIGfiG%%FeuM8=IB!S@4WS(4<>C$pW{l4ioAtJL}LcQ@voRG^fdBHyTY@D$2|})|20Es`JcEEuPQP<=IbyRhsmoyn(UM%}h96aJuGmnOwaygMO3K z*rGa@H!^4ONaPf{YL21im_&vvh0Td+Al+gfs%&~qw8yJrNK)O?n~dymuj z^y*v&uTA2o=BbQI97l=tAq+DReoUbYAzS3#hGvwxB(yg_P2MWfVC^GK-uKt0?Hg_8 zJ=bQjr4BEp>r(rwK2NUd#i8LQ9Dd%63au72`_`KyE?To^v@OqzcIQ11ysR~YrSj<< zpq(P}>16ILnnv@`@l?G$lt~YwsCCPigS$9Ww$6h66Aidnuq^dcT3qlYf;fqOZYCdIjlg3RMF+` zMjd8%(d7guJ^meIzzIUzQ#xqEr9I5~!_tyQT2@^7#+pSlY?&+C?JD-lMfDl1+m}WK zp@SQkB(cP6D(6lb%X9q)(?$5Y!>c`cD83Ijyfdc$9&O5mtJCh78V^fouq8~B!Fxpx zm7vK-BQ=?uuO+NR^ z+m^Fwvtl|!x=i4bZo_!CPZSkIKDVjLo-@J?czCWd#fil95wct+{E4p`s2mUsuKc%9Xi`Nzs1&5aO5x9O>yycWd(#*^ zWez!a1_#U(-m}`F?6f|Dj}DN1vn*LUq$kG@k)+nQ_jnX6>Ns9L!`U$nDCzzV2eRHn z#i9|)*Bj7PwE;sU-eRM7Gi7OcvkHpz@Mb znz$xY>-`kks*mQ=KXHtF;K!4b?b#%$$)_=ryqEtBGtw`kX}~$`IC%l1ldj;+=NlN| zbrVLjucF%e5?*~P zi#p~~A!a5+yeIHwuOZZt4COej{_J0?Pv>$OI$wJPo7KnQtXT=a4Hf7ye?NXds>M3N zf0tz+g#{Wej4`$v{4UJ%^1^;bqY`M<>-_qw+dT8Igw%iFvTtx*A=-i+EM| zFgLbr$Go?jF*dar5v}>KYRSXFLn|;(Mex_W)i@zy%%Q`JVI95!x${eLGkFURm2Jc6 z6`54~o5Hpw)0yZvifYRuDd*wJ>)UmCS>hL(_ng4Rhimb9W;P-QWZ`XYE*@VgK*Gii z*!^lNN^frm1>?3hDZ){~feL5m;@RFDG>U7>j3o$7UW)r~mtpcB!DUkgqkSg4jN8`0 z-&=T9gjdAz$Ry5gPUO^kzT9SO!L+u{d~={4<92R9-jOT}X-LJ+gY)rtz*1O;3FaYK z^CH047vo%`7+T9%7$5layu zatH0Y490g&roxCZbaaVe^{KuLSg67)!Y6S$W;?#;F2dEES?Km>HV%agALIT4tVJ&0*5h3fi3bXLm8*tyHmY55Xd?~{WUj~C-w&0;jo%Eqkp z9C%0OLcUW5&+E>hm*Nn5XLxhsLqop3_5*s+2Qd9fF49&^ht9yMa8sLyJBoQ&`l$qu zbE^Y$ybF) zIkXyI7p=nRngXOtrSi+<(OlK!Lz^re#tUZo`%fvt+UKI?>nP| zc=_TCO4dF^(UlgA9RC*+-K43uTbexD0o&1E@u;Z*%|Gv(+VJE#o*gn#BTG=J5N&VAK6w?v1XcIq+DP?wIC8VtRtOy^CCJh`kh6~{{O;K)`a z7QchSh-aAUcn>nqucQ6pWtgi^p~ogS2Ix!j*Sqb|$(w=wpAw)`G72M4%oqD{5%MGU z;cL=WOtfgiuX|DyQ624%Xwvb8ArA~Pqv10P?sv5y8%^lY$&fnRw7J+ojW>rW@xmZ^ zE*>Vs9rb^)bz!UU6MsPRwYTUIJX)+%6E-B&qeo;GT6-lxYkvX~%Vt7%dLB+n?1VAvg}G}w5Hh?g+lFXTZL|^doq98LlO6LWII#aL2OjBa&n3FnRLL_JT9py|ZPBBz zsV3LV6P_)dZaf?-$5mmSSkWR$x$<}#tdQgG7wb{Ia0ql{!=dOg5_=YAp)PzQhP^zD z*supMuKR%zkL76kQ;jPJ^`i2f-ZZ8IwdeKYlxeQC{_e^~e;1B2aOUD52YQ&~s?@A1-QLE{;Q1r9HG?c)`kcB1(?0z_2G3Xmq}Y ztVbUaDlN-1A*$Tmr5BgzTQhu>6U`=o<9GPb$}f;+LUVpJI)t9h16a~Kn0NOF@Wc~e z#vJwHHG#pu&vND29A_>#<-p}(cAVvF&0P_9QR3w&Y8K2fcTNx%&Q8Odb?ae&UwEe7 zp2G3vUs#!Rqx3^u`_L7#$en_S)ABqfhPsS5Vc8YF8%gF_Q`D+ zJ^GIJdcie>58=GRZOr+TZMie7KND|y(zkmc_i2Z5gP2DpN8;GkOK{M|i5#AjNbl5v zl&Fp8t);QNs2a_Gkr5p070QgiLEK&KPpP#&tU2S!C13Zs@4V9Cem%_#LyH$-@Ue2l zU%w8cb>Gk&EzeU*nmq7LFm_{mQ8($r(EJQL_18^?3cqxn-il2_%zsCFipIkN(29O26|Grjnsc=QqChiDljHn6K6P$VM)I?Tn9mH-M z2XWQXM7Etv;E`2v9IY0^y+M(zunnX0vtYI_3Sh4|U%s8}#f6K@-1YiP;BB=JM)b`_ zS@|BUZMg~KbKg-uR6*1Q885d5YHE}iHuk{ zh?i#!;so)0htT_u`p0qE@o3I(h~SUfP>vBy&*+msYpZ;iR_jT#%yRb|uf_N-@)UYw z7Pfcnf=l=7nB4Rga-vqO^p*zePns~f&5j>;x$tg>7fsZH`2JrQ$DWDiw;}Pu*D;Xc z9}@ZcY9eQE8ps#N;(20nEER7=aiQ45K9z*f&@_k_c8L1p7;kC?dC=hC5%H-K8jJ{;HElOwx;*~1##1H;KB+af}t?-YjZzUwsQ>=8*v$V;RV6BYk;rqbKJmfB|(b)ElacGfk36 zZnj2P|M4hD6dL`f8tCdgMv&28)OvPfg_qDxmYTD)uLCy^cjHfUAK_&UV$10;-d_{N zL9(&jAr;5pD`VOCGn&V@MY3B%813r^aBEB;Uo7(Fsd=8*ZNZmy4m>*9jsAJwl&}t@ zn}_iKR0*%dSg~i$4I=_VxG^e-FFyFu$w>T9GdM8Jg(tH_&BYx%-k)v7!o}vy*A*JX z$H5re!yaSC`r`MDsVEehdadF<{9AhyHP&B6J|+BsWojI3Zp6KkHdJ&JzS6Vc)CwP_ z2;M#?K=^~s1~YYrVBGcolpXBDF>^fl&(w|OiO!U>=))m5toUe~8RZ&8T|vqJqUb#1 zxqRC&Ub0%Y$d)9jP@=+dC?uj}Cp5H6!zeO~gp8(A(k|`2R9f2p?WLu(_tci>d|r6d zNAB{w@9VyvC;H-}^)U2)HwRb8tidY%6KJCT7`Mka!Y)#c z0~X4=-_(j zP-IfI78j_S(k{=M15LV7eQ!_ZS-LV$vK4nffpy{q4<6Bz;q$t4isT#Hj_ODy<#yca zZ^&Ldba*3Ijf%$2xof;4?J`5zRMv;@hs(qh7>VF50a)`y&M0+RnAdY9lm{NhBKJFZ z>r;=8FH~7_L6_a$n6a&DCqDXON5}Xc{Ne4y%)yS_@VpzFiw-#7xFdCH%~@w=NR^k` z9GlUKZKo%LSOfa3dy< zK7|<>wOC~F7e!Gm`M$XxBi5Vo`dusbmghS6a2FnbVZ#@ZRvg*eoEIh-@ENtadbj9j zHKLLG{6x)wH*lHq2yc$xMDc@a)EG?W({5nOJ4N18*^YyKlQG>h3Mm&y#}u#F;}iNXS4C`=rG%q$)Zu-8LGvI)h!rVt-zIg z>+$#fb8P5$2MxzB;^L7L`0@S_KAhfzWrO1x@!uduiC#Q#!b1c*EyI>hQ}JNW2;99o z1v*pb;ze*NUai`PV@och^O#q-qt}F+^jmP$6ipUK>&kPZ%Tb*)`9P@!6)T(4qRm&V z9r6@E+TX}aAIc+yf)0jO_PPFuHS&iID+J_SCDl531&_IfFo&NVbS>`GJDqHg?Kw& z-%sGy{JroEF30lhGT3x0#)&gU$d>HD>)aeHUX_iP>T{sdER}8>V|eO_FeRQl(rvsZ z-M-wy>T6Oa*qH`{J~6mAI04T*vN6iH81tr=qwmTJ3>$S38eMN8G59V@I$lRm_bRCT z*^4I=H%dpK6mNp|h}SaR$ETrK7Q98h77{Wo`T695mO9 znvNaWZ)!6()}O@o4U6D?EE$=pv3Sxt4QH<9;oZg(^wHRiiUR3PjyjGr4^ALI?=Y$_ zZN-tTVt$NSghBW7(ZokG%qy~R{qQWr-$+Nt&@_DNlm_2+X;6x4kfd!a6eKuZ=%7jN<9zxp+L*Yglie-o)!v#(hwDw+e|v}{bhG8YH!bCGVCC!O^?l$EBlYG5)w zi(>i3XAJEOLikRMmOYvXJWq5fuoel?6 zIPs_CIHV(aFmM>NoCebKffExXEB|WlU$`7Sk4ci5Eht@zg9?%@*ir)h=d1CqY9oB4 zZykMZ9R^vIA~B#CDxVf(n$l9-kzTT^K^ew|tij*m>!B9D31+Rf;NqvPsGhV9|J8~o zc`%jJa}qi4##GKp9>bJlVbotKd$3encJbw;}}qQ3V90W5qIf7{MY^lhW)yOW)mMEe(YoEo;}4} zo#zOZp4ZPeY3%bjna&N;weCBWrG3V7x%W{1D<8mFT0OZ_ysy*STCnWSCtNVOk6w{C zpmg~r7AoDtvx$$O->DX-Qy!pVcMaZtsKJ)a578;*DSlsj4fm5Du}4?D*>}I-?)eXi zFPd;;eN#S=UfaRG>2wy4-=;B<#%XcFu9(EHE+bj(8_durvOn^-XQ!D)T+kqyoZY{X z_u~Wd*1m_uiqG==e8=gRKkzT_JL>blLF2BW+<`VQI+#|wqV=W zt=RFL8pk%*V1W3)+tZ{Ux+|3nQfBhd>jZu{GKJyEW0|xff)%?4vR2NGtE_AopV*eG z^;&VuSw$ZIPl2-}qy9-pnO>eM?4+&2&@sw1DOF~@rz#gKx8!JHVEA^^q=USJ;{)2z zQMgj)M;h?$DMPOQm(HUbT#{EUf)Y~?LqoZS~F?b@&e~h61dtp6Uc=66yN1ENV zq|GpWt`e{Jhf!;;UDTRcq3Ue9UV~pwY4GwO4c5)l;NeD1E-uvJ$bm=&bTXSqM4-7Xk`*@0u!jaU?653a}O(~^)unG<_7Zlv}M*Y11d;QwZ$kS4l|MG-N}fd6O9S+OglxkP-^O&nF1WWGZjp9!EcyVVrEH_U zYKm`7d zR@}MTniiQhOmXeP>b>IE%XP0yhj2|r8UqVcc+p$>-B;pSRXv%*u8*N*as)Lt3}X3p zH@X$u({gfqdKXG|G)s>k&Gh(Wl0NlI+wyc}TY6cyY^`m;o6bhmd}_jWazA9< zwBXcNmVAHBiboe(^K?raevTt?+4XS_CVE;rda)mY{Q?+@1y%sY^YVkpeXs$Ao}c9TowhWc6$;`J>LfG#)^z4+O=C^Ocsg|*#$l4RJ!s#T z<&*6g)<^CY(e!hgw&1)Z70&fgWsc+;;|*KVe`O0|Vsk!h(VTCbT8Q7;inim`h0CPH zne%j5Al|sHv7pXm%WPv_Ir6t>e%{fyP4qO(DWi^JG+l zEz6qf@%Gq8Xp}rapOY6*5?zHccIQxeLHyqxS8? z+vyt4=-kG|%zHR`@)6oxtAj%53&fv#g)uMF*c)&%cDp0nx61SpHK~Y!Um(CY3t?wno-b`n9XetjT%;3qRlR2?SPgs8JiVQ^kN-{g+xg%9C;SLVg@ z3%hb?vWDnoPhkH<85meh1WAx!=@Oil$wX!EVC-1;li)QlL=BfOCek5OI__J78 zWYMwmzWVS56X#aIrM?(zyUxXhUs*8Sm4^XZi;=QpB`hP>qq^4ye7;&H_p20!uiX@K%d#z;UY|@vZH#j=K!wgV&Mv>Oa)i zoj^s)-KhAw4yX1kN9X;EP(9bitp?xIUSQ6hcUYwN z24iC$;hDi@eB51$j=_77+oc>A99O|}%2GIwT!5Pc3b5qAJUn&E#f@)s(ZOdXUE4%c z^|>2sdg&25HHaCt9QwJ_F_mMGmn?qw;%unpmf*^wow)t{ELP2Xh^v!6L#bC2b{n9` zXRQ_4=GZ@UO#g^(zO~r$=qhT>s<0;N5cGC!$M%csF|Bs35VGj*jy9Mo!o`#D5Lp+TapQVEmnaL6PkWPo7GB;kauB* z?5W;AwxEleHOFmnj#Tupp=&yqC~PO zYY+d1S3v+{!~S9V<01(vhGJ980PJit0pAmIaU?|Y$BRzjlxHpWlr7W6?KNDf8s^FlW1Tsz-id84cjx&*wtN-WiBVRv1{Ildwd_Cpf7Rih znHuz2)`|%`TW zioPfu$I(4_+}eeon)Tu7Tkd3*2M6?Ur%$pQUn)xn*s&MqsrBHunRZNAXu~OG9T+PN z%fKqx1L_)Zkcv6$d(Fc|RbttjzEEsE3f*qyLR<0+b%)Ntv)K#S?^IxS7fovVnlSf} zH8W>)XH||fwchsOlEwYFT3Cz|NBYn*&zqZuN`Gyd2mP)4@Jpd9?=R@Z7wEynt9C3~ z&{?{Rk}vkQVag02NdQa;4QE_Lh#F!b4vU3HM>*3@^HRdC>i+c!HwO>fY%MyM&UPREkY3`a(}VfAF@Wn222jCSx_hFxmyP%3s5#zT;wwCr2rP44&oKH5Egw1=75bsTq_>1pNl_R=lgNSJRe3o zi#{dHH1v1lVl`J9S@mX)x;efFHpTYBuDF7UxG8}uphH~T7FixEi#^s*U;~NvqC^KQmF7W3c>C`pk z`fz?1FJ8Oh&X?QWXmrb!(I4Dl)?XQ#A)VpmH5$ch7Gg#DKAfC=2OW}sA#ippUMM$U zv~Z3re%iAt(3uB)+?n{no4>roQx6HGl6nZ6NN@0yUntkK31Ov05RESlW^$)N9O~># zzlQ#NT7ZZz3}&_ze`1s(belE0zEX5R{QP))mVQX#pcAk<`xH6Su`)8&Voil9 zHP_kj*Uugt@fd8AApBeLR=d3%%yE~4sh$u<*Gofq|N0OHoD8Fqba7{%4PyKtIad`7 zVrsfCC(xTO|9LX=eP8;wccapz^R9Os8(gdNT=2WsEc6n+X0Gx@+&b|dqYG5nIJXTS z?zWJezpdmtd$GytK6HL5e4D=m8T}@ZTZPZ-(=VLgR}AIv-Qr0e4ChppAyobp!Zizn zxa^%|$^H!F;X4C3VrPF2+wI9`Kl)PDaf_=-?(zP+5;wNmowp;|`1cIW97oPnr*5`uw4Es*9lizX--x% zC%5LvIzxISc4U$8ku0jEPx{h>mRASx?$W`G?;pa2bwj8X8Ns^pNS>^UWR>i_-?F-fcgXbGj^#br^GLHogCW?16KNC&)R%!RoaVT`|e_T z;$L{qRudMd5zmH5H`Ky`CW)f2qJf#(QWr0{0h5^ z*`|L{GDMA`hvZ!EXw7tG2Np)UvU0R1b6k8G{cSKO6bt{jT{!O*MzHWwB#+#QWcZ^9 z!hR^X8Vuo&iV(ii2;w}0!BqO-M-O3Nzu4p@85Hr_S1fn+eAg7C-}HxelRP}A-H8){ zw=l;17utVq#YWLbmg#onM&<52G1G-}S9q|uu%Fxc4VEsMK1Le;DaUmnFwFwVs9QWaOalJK5Qnvef=9j%$yuXzq{ca-Z+#4 zo)6`iqv5o97{>PtLm2ockWSizsd3MbT_^bPFz`5 z?Gw%g8nW*{9>%);A-u&P)+G++zzE69yb;ECUoYu;xbwNG8!vuwp@Ls)TrN>S`=ajf z&zpv>Lzm;j*kef8_88v1nlf;k278P#;g0d07Oi(%=R@;n(z_Qv+VBQ2+QXUK9yzhD*AR@F)*2Tw zx=5B_Bo0{4$Cv2ccz5FlYIoIRp)ket-|5pl!;)GFcAWaG7n7a|i#W4C|D5q-PhqZA zwhm(8n;@R@3gW7GDHuyuBfq_Q$JXx|8r@T5~ zd4mm}YXrk4Z4SKb*WNk}vq+@6zdrTwIKnBlbi4-7Tzg`;I-sTQI;?p9?eFvq?u=7Icx$>r!(2NKbwf z{bSC>0o0Ptkoq|vE^6MNJ2O1kSN1!do4PPx$%$u6?RmXLXCB?qfn&CrbBTlWIUcv; zUDHJ{KIDbpU;5!`ztO0YJza*w29$p}jj+n6I6Ay3gRiO6_CF(jv+T&E3-+ws(ThDV zkTH$!yw$OvbZz_bxA^}3Wq+ZN=EAZgPVAJ>jcLLn92DMx&P~l^U2e;ybRAZ1Qs?_AtI626dI@f9+6}7~SK;#bgRI-i-11mkn6IXcbne9X4c$0j zs}~mxb!CcV`PLr;r+$}?L%9?0Wp!h>JDoX3*7cpIOzAaWpIs6(`CYQ({n{$gP5i~? z@j-n5Rf%bmZQFG-8tNB9FkX8C4sM@|g=(wOM(wCDQfpA0`U498HRrpOHas=SoZBKg z@twXshseHC?Uy4*?C3%FXZHM&+L_J0I`DO_Da%~+*=d3X?Wd@+&x0l$?fVTLuU}!* z@LIIGGm@H_7F3XI{y5bf6ulmf(&*vvxtoaS?(?z#K{-;AtDx@v1g`lFFszjhZism0 z$IMuM*ovD!+OWsp&g^-x6TfV=WaB>*MwRI?{LfVRe1ZZyPTvauF2bbH7>U>>6Oh$56HoG%!bN8n3_o4Og5fW)A)pBl zYqg@&IbC`88L_&?jE)`6sk_;jFNBA7x|?J&uPd`=|6lwpehcjj_wZ@_C3H`%M1=1? zwZjgmP~E& zE11VkFLA2;YY@s1P}n$ljE1M4?VaQh0RICyu6)wOW48Ji`<~5AwK;ho|Ei|Kz%13lOwiWv%%f2;oGIal&iq2r0B-6CDybQ0NE=KK{d>q-C zjndV#@m?nbGk2t5;_)QBIhjJoo0B*pGl)(lJ>=RN^fJ4R<`c>gJzy5PADW8BA2U#D zHWzLiis96|96fg)hS~D-*yeQ|T7Pc{`?wk|E03aSnJ~yctwg`(i?M%vKDOy+BU57* zjJhP_lkE&Vw@g4#RU9r`#^P3EDkr+ea;(EJE{yHRdm&bgb8E)lIE~APmf+g?bVM9U zK!*+)m@;-ABG#9{E4&z*C=QK-`>&G1@T6=%9N*2y z#pf9qa55F=dd|W7??rHHQHuR5Hey-jHrUH&^5DQ)$7scK0 zF}+w_guV;2uulA3#|QIpf65X(7*&S+(sg(dy%r&AG8@CQ5I0-TLGV4%$V@WOc6$cy zjhux)=Ck2#m5B>8w_7&0DIG~&@I&nI4+awE5JeJFjji%M?U_J|SV|KD7l|;wz zzIPoNAIkCSz(VNmnTH|zg?M7JMA#sun0ICkVzt&_-QN<7{8ohWl{xTo&PK5NTr|td zg}P`@+s+lB^lTAAuP%T?$s+Vxwiw|97Xv4Si&&H*GXfGAer+P}+=>#GhMc!{IdR+) z*-Nhbhv9cFA#BtZRFp1<&fKN&T)zTuUDu-b;7#z`z6oQU)?&A?*SbGggbN!N3Xfnh z<}6%>|GJdo)QD9G4p@s{`WsN3vBs274FF1ccc7EE8rVd+S?I@Yg-5W6J zS~>i4w&Qf*9vH7VfJZYcu($D;_|Yd(-s&`##Y(ULT{`oH6=>0KCZpBk`EZByLYEHb z@`k}|a3t-AbR>qV)79rQ_DCKiD)=OhoH+&$$CL0%I15Gp3&NT^2iF+M@bs@lvUw$n z_nbuE_UG{DZZ&dSUB~Jzw^4ch9%fH|gbV6VggaFS&-!$lJWQi|WiodQ=kr3t6jps7 z!@13e^7Z@y96qooUq}bgQMDCsCWyzXSBqT~JuK)htbRWb?f)^FO?Zrsbr0~opa$(0 z)nKyoLo8~j!(zwR*!uee4z&4(XB&S)yZkTiS2be9-X`=E|F>l?>GIu3;nACk%=e5F z_S-}nHja?FGQm`mbN%%yd+H__bIZ(@eB}EN6-U3o$ni6}9{DD@il2Cu|4TTKKTv$< zo6G?ChCLI0VusN_r1VmtRS)4a-&0|-LkrILZbjd}t+}a{I-7`p^m0ZzZv0MeLKW{@w4{~-=EJFq9G4;7XFnA>cTnZq|Wsdh%rSHEMw9-)%ZAX(AHfYoGr!E~^=+paRTUsd^GG6?$w`nJMhyn4X)#j#-xF!VYYsKwr^iNAH8tfUb8|kEebsOK_H+<`u=;2^zvZRTxO)mW z98Kg?<#=I)#n5Dm^mJtZ?^NK&If>w_F1Ac^7OzpfKm*-2y!BR>p?&mtFifANqxBi_ zK#%UVdMq?)OZ%yYoYvcf%HM?-J;s7=g|Z%`Tk*4;wW1bvVv$^TK((-3tE8v7QM{{} zNldj!V1DovPTMq=mk&kq=-9!$XyeX`WC#A-V@2)nCS3T~Kr+`dYh#Xhv8%-UKX1fD ze_^kA8FAJaV}4#N{l7l#Sp2L#6MIp*_r$McjXSb?t1ARcY2V80P5QF7f$XQhj~J&k&CDXcY4 zVvqUp^q(=ABS(+nqMyQ>(H+Em(Vkqo*t1mHFm^I`YQcFuwx6lTeMj`!^lV!O?rTfq z_xc>LL7&~$w`I_2L%NSNVaIwiE_buw5zz+yJ6JL3c1KR@)rr+|KO65!XY$B2&KE{- zZ=*yetHkk*^CV%JMbqS9C|?wcW;C}KXNgB26)$`Do;sW*?3*}iP3EuDV#adm{<>-N zm7Nx^U(#fYTUwmaQkO@L>al0I0V}r~@!AO!`jwmUW>h=AI%dJ`^8EgtpU#B$^8SB2 zlf4&D=goanS^aPv|I3SFZs=fU3tKF0sVyHaF=5z5O$NVg&O3!FyfRvq!KN+P=Uofx zb!frH{i<}aSLH0f=KQp_C1;FR<8Dn&>FrA=AYYf78G2mh+?KbC47gBu1xZoDj`xt6 z1S^v08xt>UQw*&;kK&VtP(I4-&w+Z5H1(2fLZAlyikmTL?r)^p{)AodADDJ%gqC^} zJ{sGAg0f%u9{Lj-kN?KswK8vQodVnWD)Ig%6@I?doDK;a&Z~0sBUL+1Z5WKczEs zQyRaE$3Agl0x@FZsUjb-+xaV&W@lsT;j&>*D;m&$#aZP=9AziuHipaRPd?m(5p4mi8- zMUO`5|BgI@pGJqUZ^bTbR^5Ub9m`>rv;}9k?m$@6y)dsm0G06-ICu9LY_w0J|Ju`N z{8)uD$*MdOUi|kqGq~PlGJ`Eg@~h04yOit7(qap)p07*`#k=VEN;s~%tFhB|1saOV z5aYT5N`JPXyXbrOu?f0nWjL{>7{%_3G55z3v~w@P?t)U3$FD+oueETJ9Nx4uoA6Ei zqZ7xsKyO<*OTQ;`dGmNW9~#Hg$A>UewIA2Yx%K-vHQq0LjDt~okZipSb8=+n){-K) z+85)zeHjLc7ij6f9!^(Q;;q>te4mmBO9SbiR?dap)m&jB&x6W#@m!Y`iO#wJ$zNo~ z$l@ii?<(EZY3bbSHj{sTOFm`K2;Q$B$h>SvmbKHT-N6s2JAV+1s+VA`(Ofvso`adm z^HB0g=61NqesYBPzme-vE*Vdg()sW}HtzMCh3G*UD7Ym1?Gd80{+@+|Z?j=EDGS9d zvvJ*XE|!$fMTUl~RgY${V)+CHFAd}AE`2yB#Da6)G-2+lQ`oGx44*e=V!_W;6h6(w zxxWQisUTjj{N8>!Y{Q?0+t45^(seDCq98R76{eZcx{!_&TT)QdZzf*lB_gnQBFYnz zU>28*8!b~YeM~CC64JOu)bWx^#}dhc$Oc3w-MFMP?IbKdpMdJ) z@mQ*xfbzlwjPpsro%B?;Xf~C47U8_S%az`l23%G23MVzU;E`=ME^eEKCFkQ1P>_lF zt_yJ_aUBwy?}z^)>C(qufy3kLsIRWZ_3%p6jNXo8Ygfq`X^D(#DMUz64y-$6qOMsw zOs>y_;(v+oE1QAOKGR{H>;=AETj8yw)B~8S++3vKwtwwlkB}!^{qf^mF z?C7-;b{@ru91p`T*7LdQ|L3W0>A$4g-ieOyc1~8hwok@+P@f;Z^mIrWdwXDPsXzQ zIap-A8eM-?;K{N(`0V@*&MQ^8s#Kd(RvJ^cuO-iQ@5GbZotd_^6OT^rz+Z)C>|$cT z7IE4<9NUW4=1NpQ`3LQ1e!%>8PvtyYgS-#IbNC%e|5I%k<+le#p79vw6biR3BanJB z1B=6#Vynj_C$d(CuBZBr?2o^?I--pk9cR3gDtK>sEY|g%<(DE z9$J8ePn(3feg?tgpGn5NDgO*tXP9`-?{5o-`LZol&N(cVoTdDop^kXa69Iiut}Ld!_2gqAz53;f6ZAbdCbGofw|Ib2=WX1;ZoBG)alGd zo8K!?y6!Nh8rHxe;U{u5TXM+)$*oOq&(mwWGOW21N30gFIk+#wl|A`xi6Jxa4*k4x9Ey*kn&kkDnlg zs6yOcy%ql@$+@@YEn*Vg+8J>pebHJq z6>3vgz$WqtZWlg4YMTZmA8t+aNs=L)ZAI(9_EdZ!omQW|OzYC0hIziM^BW}B^5?Ib zL3}H0-ioXN+#{W*){)}*R7nR~e8}CEy}0RQ58l=6##tx2vZ8x223*nyVyv-CEez(8 z8%kSQj_x5BG2Z4KR(De7M#-a=)tU3QpWHJxPMlOE?B`tZxy^hzFTkH)H3F&YE%Qrd zMysRb1vZEKGo;Fo-Jgp7z1@okEIp_-&5f%s$_&I&y*MVoktidKF`%Ez3ObKJ= zg&`bta|qj?2@`!bgwv{nXc7`Y<0XT*daf^Tig%>0BH8u&zSOdDV_1bPw%RG8y}B)G z<0fFQLNOHP9fI}B`$)}hz_l@ItQuy-2^%_6KcG9`&2gdcQ1J+#`EZx;rT5wdQM)#T z=evb*fp}G~8$x(V`k5Oq1W>n!zho`^nA^@ry50SFd$>Dic6XDWwkzi}{pi{>;h$^R zMknM&%tX6znez}*1=sg4k#b6rsz0^qBRTUoCSBzG<;3AF+@xdaMOOGR|4aaD*MzXf zQ@r0b;j}#v&Ncgn@ci#k{<#;-4$}jUulVuHdLNoe)}U#E2WK7Z!_=n-U2o|& zx_-Fs2IsIWM0eYa6qC#Fef}ByvYYcnh&~I1B^A9?=5K}ztNNs{W8Zl5?8`xns0t!y z33s{8P&SE=V0B3Z9rq69lQ!WTs}n|T@uWZZ3FOa7GRJh3A4i4wuy(S{csTFDJ%7qv zfB$F-v-kb6$|Mf~qH%vKyoDBFzp%1JYktm=IgQpGsb!1f?ByT@kys)?S zA?$X02q#8Hu@Da6Me`_LG74})y~A#b2MBbFHR!5?ehIN-ome_f?7zFa2$ zR6v_h+H4M|hh-$krbO{S$(W{y)|eU;!3E>Od8%g^UsXvLS2)qm2L2pt=EvjxeK>80 z7kij2a7|TI#2RlhMNJn&Pe^);q z6P9?l>7jh{FP!INBV`^^6kB|V;u!rXVOd7Vx*bl3m@q~PYjIh3;Wa}%J^6cWhxzc@ zF)um=EpVOHOc~na1|zY&2npj4ppVM~ta;mr2}T;+=qK5jm7Qqd-h*%A!I_c$INCsT z#L58PlrHRU@j_j^BDuFHN*M7`TyQy(b-g2KB!0A&oa;8m2eW3a%&A%--kq#%YgTv* zyS5+O_bYaNv_%oV@BI*DUw~-gNT#%@!O7~sc)LN3+kP2wUW~Bu9UK^%CG&gb`F{{9 z`_7gDbQJcpheJ49td3yEN0E$rAIU>Mg!?maDCf8gVduIKjzAEb4;oC>P6PS#j}JA! zda;#_C!hRV@4CNQ0hJfMaP)W%4z}L`pTRd!r1}H*-nZo4bOR1?wBkk=VSQCOOMcXy zQJ;MH;FLdqOb_ORmP5FD`cO9C6v4$SBiJE#C@097BWqqLeOm={sC0mr$+~ss+5q9? z_GeEYPo5ptmjn8rcOCPmDRzAlEbG?U7%R;9Z-8St0!Gg#AZTZQj z7tJ>IVUzp)`QQD4GP5<1?}J0>aAFA8JPc>Z?r?fe8^Uf$vS#WAGkin6$+m}|}gLy}o&p|PvJS1zSf_4~d#Pe@DEQrGniS~44Al0J>u+46H zH%K2n;wc%`!-Z>WdeZ4cAP$?UOLoE@6(`5y%-4l*4%sKUmD?x~o$>wSmaG&%F(JBx z%-ytS{@mVFo+>Pgzx}CXFN}2M0Q!#);th{rrf!h=;dcV~)Z3r0rH|X;wKp5h`_brk zAI>gyrJIlJ1)OCjcCJ06`)yg zX~rtyhCEr+gF5xDRIHUb#sNMIcOS@!CH}N_m32gz4Bp~#_fr`_w*oKvl)H17AK5Cm zH;*pp!PR^0xYnvO?=jJ73x#@nM5><$Xdu=}j5A zLo3MvwjO07co%(ci`Qb@)Yc5xl!XywmU(#B zWS#BbhL<0xF?$`zR zaAdQvo6ev*;R$|@Xh50hVRk>-aD5Uu}eOiZDABb@}SOA=|Du<&bsZN=FOnlX7& zJq9d&ib`C^{lzjnc-ejgrEJCVs&#l%wGt2S#xnA_?7c2Ze<)GrQchnEzs)mX+G;%F zn@xw}om_m8KIGVx{g}A;GJYk!#L_|kaC3v?{70&D@jNa56zy>RAL&8Myat6S-w^Fu zhm+s0!l3X3q89H)^FQk`)pG@cx-G`g=0!-TXXV!J!mm(a@QoA5RVc!l zw&Fc6jlsSpY52Im5Dm#|;aqVDDrHx&&HA}`vOnN(uPHUU3s?SHQ!am84@Kjr*x-H@ zKV~0?qOgNkj$Mb6ze~|psR&i8a^NCdwy~bGU|*AplqEB%b$b+V*2(*Plo_2PK4AuS z;IMFn>pR9mpK+)ZjbqoNg=n*G1NNObj8z@}L*wj+NcDS#9rbUp?ATK*`Evuh`X_O{ z&n{@5S_|*BOHpT0h$ADjvG8C9986}S`K|=4+Z&4`=2J1FB!&NmPvSty4jB*bA(q7>T)8?U|RSB|GHzDuF9@&RU?>GJsdQaaWYg`!` z%I9O2=v9i)L1uxE;fbI? zUJ2<<>v-w^I)8;pv%`p~mn^zPCJy>$VE6Mmz^5Xl`AT0vdjp<(Z$S&Ua*PWo!@aoq zSlC8%H5q_DVrn9;H;VQ&G#;C$$Dv1c90DP8$mD&Oduay7x10e(mvk1@BywdB@qZUb zvgHhKZn;Du3QedISxnS$NWa4#J;g;c|XDW53O0;QMJ* zpFNspBZC>$aZu-4+F3L@1 zQKv+i@iUd4Uq-VyER08m!LmQvj^}2y;p1;VkuUdcPmcrGY_$zb)wUsN<1W;>?uU{2 z0jxYHdO?}YO<1`B`v-1-q5SOLx!ZAS@*Z@(ctG++6}Z;E5~VsParnn6gxj9RKjB`! z9+1X=HZ$q_K2GLAOyrJv!#S{B0PpUnFj}nXB;MlgoO-N3a|cyH#7F<))nq3e7L_pM2XtuwPIh*dw zy;Y0koH`s?{RE@(AEMc>8rc1*LEHTg5uNoEroxh+JmVv-rhh|}{LUs_`-_Js8(~-? zGnIY`@48z$pPZJiSs)_s|{%>qRFta9jvh_CM7S_spdthrev24nsCEsyk z^*6ko|3kds-;#;{gTf&{aVV`GEi>xT`1dD#cKt&_x&p_>DoO5Gm6;(ecy355{`=8d zxCClE`!}7@aq=0@Po<96OggPg;Qg2>bafib?dcJ$ebNF6%3dqX0E_cGl#g|FpSXvki}F>$B}C z>3F<0V6gaQo6FKU!dzxck4%>@0!14S}JBhHD1rB+;iK^VNx zwD|myI_K(YP=AspXW0wO_M-g!&%)1rtsz}cEqa{M;h!DSnMgCB-t_y=h>ra+^|fv=+Q}h-dud7Uh*3)8cV0`k?cETFm=bd^ZpkHjvm#K zeJ`8x-AO}M*ckGPtr7nk8Z%i>_I#<50~>BcheTrzTx-hy-t8FryggU;v|3Gupsv{@( zb)v#*b8eq(%rg~6^x9+0RclO`y-kB1=8+(t$`o3C9Ls*@QM@Br4!6JVw42w1 zcJ&>ldt%0?4l>hXsUhuFi1wKyobV)L7N0hvd6p4%QjIyk$dseowPWno_H=nK?CCch zseHH-uY`BzvOU7ylk0vzEg6pW!lf84S*)f>^cxq?_3o4Z$I*F*)%^c&JSCOc(x4%f zN+pzxx^LP_lqOMj`DiGjv?x&_5-GAOd#{qc_Y5J)-g}eqdwzd^d@i4>>*Acwd7t-r zzMl7eKW^SPo4_WrukaT3(>tyH{3rgwUxA%@qprR9jU}6YLRT1ldK`CE^43fAS+Grb z=YirO^wsCP==Qv%X2eFxMz^@vk+<%3=FRPreU@GA+zrBwlQm);&)}!X zX>6E2nd28ka-mrmpFj3xqIi${ia)j7!-yL^HRvV!Z@~%`9+;rYGqTgQUD2G!OV#+L zjVhm}s<1_|DpTJ#=a-$Wc*|Fvd)7+MWREtDm+3H2CZ~Ae062*LdnH)1$F`}oou0%Q z(sdkcF8N={I)6wHmQGv`PDzs;!x8Bm{@R9C6Oevx}(%^&QlY{FRqO@(9Fj9>PsvZ|;#%k5k8#*kK27yZ|6S~kM)!5@@}mZ z;U&D9FPYBO$8e!hc)~@;kra3a+d?kl!mX=Nl@1lR%-hI%T#N5>?+fF%9>tI4TK<{J z6}G~Q{~Sl79+6aZ6CR!LPR`G>`~Ul5x~BqlKHNl5T@~&u+m41VlK*Y92U}zJc0=j;YMj9eq^w?r`{Ey)I(ai5 znruPo^mQ%l+K_P^*A$V7fNp|gVm}$m{p7Kb5tR&O8&8Q zMj5_3mf>8R5~Panuy>bi9DFkyd)8z@Y0F&bug<{?;UDXZR{2zN!$H#NVdt|LyJyKR zBSQKJn$6&e*|7|27)6b5(ql(^Uarugp5t5CyHr4_&r*2mE7bi zv{os{wQ*&r=`0-|ae0{dGZT5%X^0<|f&l|kQRzc6d1`_>zLtGsGUu~^|#Y- z{OB~;g{Cpsa4N$#j^SJJwK-ebu#4#In$tdL-daRIS7;Yt>5-_ zuqfV*zP}G3``tn0ZQcWy>E-xtx)gm^;r_V~}?N)N)1J zR;<92p{Jp9R^x#9)F);h!dj&*7~N2UF?t24nKuu99kOs_XF5KJ-mNPdLZ6Ls zxMDpSi!x%cU;KP!4(lYR=|cRwwiaCi zDzG!_0>*c&!{xm%p||TDCMv!{a_f5t-FXftE>_~>;LR|%S&exSOJUe*0i4R`;In52 z+Bc@*kw!9-_D#o?K}lTwFPult^y7A6J>*(n!H$zlpxYp4nhTMzdNU0jg#Y8RcO~Ac z?}cOhMf9r|236m0sNdIwYpj(St=E)}6^-~S*^;jP@1lRN^H}}w0JJ`C!#>AS99~-l zs~N)IoiraQ2ePr^SQff3OW@-7{=8Fa#+}yhu=8XYI{!+>n1J!{=p2L2*JeW@x(GFA zb|Ay>JO-COhMnYVXAV=L$FtVd*4E+)ODzV!Yt4%ZDs-v%gSJiINq6CWBtE>1Un7rV zf9Y=2{kIX8Th}1&{BqQ47NS9AG9PbnVeaFW9R2D5PR-52+O%=dJ2PJL^Qk!AQO+K9 zo6srn6wEF?K+W_9v<_6`1@V@vcW%!^M~(UTOb0rN*X)M)1=r@QQ}L22J1aL~_dcKD zJN7Yr7Tmz9bJFdSbO<-S_M&R7XwVlUIK8zEOLQNh?)+k`_m6DHXx!9Fk z3jJk=(C0!O;@$^?q?oKyj=-3W)-l0!aXmam} z7TmpAi4|HuaY_F*9+}jmPmi0(dvp<@RiU(BsL5dGt?=nG8J7xzad5&oRExJf_E|B` zwBLtIdDR#@=_AtSNI%L_VL5*?;ysxga&p9zCBFEEtX`aZ$DZzMd+_~5YksBliTQVA z%g^n(K)o%8e{03t#wyamqCktFA8@jHk9%_eZ@T;$qOQpfWA{)j8ykeJ8>b-W!$RzS zzgZX(=dgdn3rxSO$X%D!xo&{u1-^Ho+eKSWU+2IT;~d%U4j6P7^eJ&<(>uMH>~F_y z5#1Qv+>*uSof-O8_&HU&(($H2y{j!5D}4dZX-T4b3vQ&&%++zwe&vAo^F5IACHIk?71xGr_$h`!Zh`c^#$qP+MlTBJr{;x6{UpCS6fDZ3Kqh5N4jo zR`mD1f;-k9@m*7uF-vtgP|g}#cG&Rk*xvjr-KF^%9_-!Jhl}L#qU-%xDSE<>o08cZ z=P#KiKc16*QG3ZHy%7C*z0CX@M-O04l@n73_hr2BD|=tI#-Pbf;kmpUO509?*RVpI zc)A}wzTZQ5?H`<)-p4dN)f5Z>z& z!t=j^I7A~*xR^uP&c>G>-ac&6&Wooi-8nAAMfB7D{N(Ax#FURtC-VL{RVg_@acdHG z{9O(0hLcbbzg}Q*QwHDAVn=rq@uXPuqpbtieUq$FvM>a9ucqAy)NjtMG2~-Y+t}s4;L9X|aYx)nF=56@BwISUHSg`J%PJJIycc~L)J`mzk9clmpN9?%2NWj%A!yYbg4XJ8*x zok~H(<+Zr_{xo(ie1(DYl^MHRo1Iz<=W(w!M_qQ{#q|DkdFjEM`h&U8$e%VprI&W) zFh>0wF3k8)ZtpUj$}>XvY(fyPw+WzMoS%3pgnJ$B&A&dL%m{H~sC0>T>bBd-ubTqQ zw>sl+aTanmZAP!%S5cV!1-hQisiLaS{U1#EyK{FwyWN*A-CcNOuow5Z__DNJAmeU@ zFfl5Wao0!CQ*k5@n~$L8^5N82FFT>wAddg$&-9*tJc7Y2b@t}*NuJW1;>O1jE1Ys) zDkAK;7uJlE{=d1h6Vkbj(z*teA8*Aya=!IGYR>vKJtfoW#QPoG7_K5Z>{UN%&ko`@ zv*EH^8^Jr;VN@9s#)7PoR8Ski-|}-0r4UY98OQ~nh6?*gvfvgzEE(&?gLggXIeM|v zlXc42e9IU5Aq&x7w*pIF--G$8e>n214gV9aRsT7b!bq|wCPIAU?wl)n=mOE$mB$9N z<+tJ7ICUglZ-y~UJ)F-6g>l*K5nS9joJSvoaPTbAChz;RNi#oQ=`@(3W4t-2&Xe1Z zEOdJ4rGh7C{jsb^J`P&$N3~`>YVIj8^s_oG+Zc0+S69xwZ%?a#(vw>2PG#|(ceM29 zmxI9!a1EvAp^;2$8!jIEa2}WwMkUd_%_NJvSA7`&Z42T!z&d`2C@rTRDt>VuOVP62K3tLxlx0nA*?1xkx-D zrOv{*E!yA|klhqxYX;(SO%7gT2a?C!#LA{W(Q0H%8b4^qzv2geHnRs${Ow0w;oMF) z@aBcrzLLKTWUC&-IBQKPx851S6SX7Q;$|qn>ka3@ufe=HCy>o93)B3}5Oxx+`;mA@ zylf-`EcdPU>6e`@D@umHxik9h&&0Yd8=x#&($d)Hc(>T2=cOA8t z>;V{{UkavczaG=X~*_esBQ2>;u_vg|HfK4&~0lzU&?7!>GoAtncS88FOb2 zKIKHs8GX2Tig>p!bZ5_-o1y306;SVQ zl8ao_m2JdNVRq4pR(o8jWa`D`+JoseQThj3`tgV4f6q$(cg_RJI7fO=dA4ZW=KtFz z_hqmD>}fa8mg|~WankcHe3Bpx&($jF-mR@UQ zwreK(=rBX}YpTl^UK)&dYRScSoOpBeC8QroL^J(BuWgH z?E}u3H)D?iE#62nrq_B4&i&S%lihnuR>+ZuyMi~yIdMXGUq)8jbNobG4v(;8;_*%l zoo2vW7iE_veO`*e!t-^LE*Hb^5Yq`4ELEnp;%Z#&9);OW!_a(FES7J|!GB)T>$Bt# zcCNpRkw?EH*hY=Fwzg&KX&rdT#e&MU-KhGtCmo#a*;~Fh7ESNQ+Fcfm(iWc2!}i># zrp@y8Ex6B2iPnMNh3obbQ#$k#cycrX-?zs!Zeb7@(t+=D0^hVzPXlMf36E7`*h^ImhIV3&P$(e zHRtJ8iX5l=85gcR6c*=o9Qbt#C%zm&(pT|<%1%EoQ)aDBO7zs&i4lpJ7`}ZXd~{Nfe2eSt@XZn6EF3A!vi;nE@?izB&9(`WR*QB0GbB2i) z(0j8Wo*UjUTY9jTZjQhi*Rm$WFHrahB$sYS1caRYFj(%cU5DPthln!o`pU-vt*D?fpuyks%Fh#hqL3DYUR#;5hik< z{esEZEt$kz)HEib?Z2s#`^&_vPK%+?yBvqNRbpM-1uV0v#XbM081MBQ7b5F%$ny&B z&pwFI4qM=AvP)y6Ht65pVgQuE){+ z;7)XFvIh4&3lK0e7sdI)R8vZUXYv%Nw1~mciIbp}{2xYtnJAgUNPM`LO1)cg^u7_s zr+LydkZ(?VwPv*4aRFy%m%#7RY}{X)j5gzD!Af=>bH6OZM~8I?8MqU%?iF}FZ7(i0 z*^F*Q%i${A=XHNFp-`TR=Q~7mWgHYEqoogP64DOHf19Ha<`E6`Yf*T)D~*ruC(z{2 z1o{^G^Qm?}PWm9)M8OYuEjo-r_MU~ zUt5l|9T#9)cm|3VCy8e|0nvF=;5|DY%H88JOM40;g}apLI2~$p64A~w5fu*Uth*?^ zX+5I(_P|JHejCU&4{TX*T9XTdUto5#eb^pch!DLT>|H+>zeHD6?64GfHWTGf83!4JxU|z>LXvmR$ z`>PrJ^*oMkmyKuf-axuYC*I&SUFd1moZUCxgPOxG@h=qPo3HTeRPxbdX(3!+m!K|g z4H8ER=YC)zo|(@_y6i^Uw3>~cH?nYOO*RJk&cnm11<=%4gxtxC5uCULm&%qxBXKFV zAC%uWTCS~x1g_dNi3eWDTsCqrQ*HXtZM-qVqLmn?aSK7mcH@?-^xsV_K{K`0*ykj< zlrI|)(XbwK)7PN+gJrlPpTE|R^YQP>66n1tLddfc+?coqeb$ztv~)cpY&K!^Vfo%Y zz8RKxB`=vFO!H2Od=n)4uhs-k+#SM#F&;d#qdN!3=`r@@AJj(Ogw5Ims4m_iOycdB zmAD(;<7H=BS%JZSccJZt%?M~FJwh$k20#TiI4Snu%zXE{8#o6 zza5@P4*DrfMgR4Yi2-0%4vyLLAq zWW`?`xvW6t#Z6cy`tQ|lvI{U8Cekvb1OGj&Y0&P-6^A3w(LG=h~We(w(Lh`JQ4cF~A zq*puXf*1WLV7?0fX-P+?WCcA+)c9?qDwQ%+_$pI{eT~$lo4ExwpSNb~%Ni^W)#i*9 zI&86AkK5hbF*{bWl%kiZtrV84VH(?nB~w32^s)}|{9-(bQyWHeeMk`N;K^5sy?OtO zIlBxo;OYt5;z!irFDv2p4bYShT}}4sugOnAk})1EIpBSo++p9Ai;VTiH|=<8tmvf+ zjk!RWyWe<8UX{+bX=zk3mY&w~L>7LW!u5Ni8JsnqCapqwsh@cCH%hhW zJ6je`>devY47kc)ymGa=^gE}=vkCeRFX>N~d1HE=5zikrV!oykYn~di*;^wPG&5nK~YzektW@^SRQxz<#ZOEOXK+MDcn3KiR1btP))9_9nviw)Fq57^@dWZ zmn#ik+w=Jp;V<6nK<+ajf3@eoTL#=-V#uA7g^9D(K+c>793;GJlT2fd((TB$t2%Sc zJ;@&5kdEaNOWx=s`M~+s^pMBRIFZhltA%eGoWg*wiJTTPm1BFv&{cN=M{N#e&h5c; z?a`kef!%ppbmaB@4EQBdkKKRj(&V=u{SQdb*jRlYo2$z^LXoS9Ld{vQ5hgZ{i{&+n3WD>paj^Qfd(H-q9 z*#oCOoS!ai!j%R*o1@K9qVs+T)Zm6ZO`17qv2m#;O-D$_#yfSI{u3{+leTd5ba*jS zpFS7FBY)YDE6SwP+SP=svO6+d^xrk|oiw8+l@`Zm@V|g*Y(H``Yoj8Wr5MKgqrTjF zR2V?ge{{#sh=WFH(6CXBElX7RRQ95_AY0hEo)%aXKzmFT5(M0?mKAFPK@@yq% zp*HN)Lz4sKd0f=jq1S$0UVEWOMbUp7{pEKyO{JE28KMoQvPH9KzAqWeXkY2ZI_tp) zf%crRURXWv+6Wg}iS@w^*nRg0Jo6f$CVGP1+dty%Y(PoIcLce8!?kNa;C-$U7kWsC z<3%M>vl(X^sIugT8jobR;KU^@xl{Q3SDU7DT{r1M$eqq*hH?Cr6Grw}EuSXbh z{5fp*y@64SkEr|g1s3z<`G_X6w0$Ffy%%OczT7h&%-|UH1U~5!P0Mv-sC+tr&F_oP z`Lz{iEz#zm{lD=btrkn}iudT7_?WWJAVGM_X@@Q&b>KzpxpfMHf@ZctL*%;g+Ex* z-A=Ls?YO0-0;6`_#IGY&xEU|p$UEC`Drq+^&8mdo@BQd=ZZAqJwqlCOdfd(|$JwSE zaHaPa*iMx0-4=VWcTfd}kE=pWw?o+4^9aIfkHFzVI#rcYxpCh#PS1$u7LU<_{u|)JB`vAiMc)47|G<>7spd*eX2BE0@<~YF}~EU5S@J$rq-=NN3$~2?J9?}c+Y;9rc0k#GOy2>!p_3QcUc?4vO-t3 zcxTCTRV}$xN%*^xKl^h*e39jO7$U5cy%v&V+`bZr_Ljn|V<}YkFURGAg*c=zS2C-a zm~>_~a<0um*NSY^2jt?X@_aeZF2KQxg)pvP1iP1u@O*SS|GrG3gLf=l){SEPY#*-O zZ^x@vI-Eb@4c^|~hg}_)Vo$HR=<+@b0WU>IZnYGnKCMJ>SQ$P|D#gf8#i(+W`Au@Z z*JX|jxF>nXBgqIXOM%0zRPS z(}eHTxaSX4A05S!?}a$CIumW0q{CBsSZX)rp=I21l&h_SYw8x{zT1q;O-iABxd2b( zx#cvUiNhC@5I1icydO@%n|pC+aEZg;PVpGCdI~HTOoeC1sTk#xCLZJ|{O&x47K+}S zKEs-)9EAbj_A1OimEfxAzxQ8G!C)4ZyQc!D&R@U|nL?JYeu;>%cUTR-1*$)%!z^+tOyqNG85z#2 z^?mv7taRxvzl3Y;mtfnEDNvg_5pSMP$I2tQ7~&<~iM4w%=F3Hxx;@72)Ne>U-h_XQ zl-cr~lCVe`p}6feM&#YWh1hejY%8<&?X9T)z6PFl%P{QqLaaKNi>lXiaJo7Z3)~a< zN_2vN6jMG`e1rGPOQDsTj5qhkqp@2oeovE5=g!4Al(7@1s?Xzs+Y=Q3XplU!3VoGj zzBJNex}g^BF0^Jh7ZpbK{R5Bx-b&V^4&&}$LQ{hyh%nm)g@fyH{qia_JywKPj|-5u zAeNIBI5R%01x-||a92FDf2zh|_xTCvGg^4y(zla%WHWMmpGHafLwHFSK=}qWHan(8 z1h%K?17qI0(SfhN3Txz-E?Xt2Gh>Y^ho4ZO^VCmh6)2n@yK0nMoW_~r1F*H-jd6a$ z0n?NYjY2E>xYpxlQa(=poPf|jVF>&gk0;ghaAaN?`jj2Pf#R^h4`IXwPo@> zBi=P2|tid1kVY?!bZ#`pjFQNk21TFZ(HRiPjJFzVH&CXV>ApcQrOt z2;*VhaE9(zXIcrN4;Vifh1#uISQscB08@@2LA@Sr++=4ps1*+`ZpVgxW_)35 zOU20ER2IECJ=%r4l-=27mb+wc-RP$4%x$8(SP$*Xrq}H`rCK`OV!Cq9M^mQn>cIFH z?YXj8mj|_4@#xGb#1C@7@C}|&emoh4S&MNadncCcyC!~%Pw4qYg^F#2Ut7|V{c^3j z^mZ@y$dSy{DpxLTHISi)yxBoChq|@i>=QPSpN9!UR5+K9Ov&00eYrbCGC9ko2UNkD z=Z~55#EZ_13AuoQzCDC5V1=|WKPZgJfTP12MC>|-W*U#N+o1_}^i&uBsCW~TEx50$ zCvV1zM!3m^>!Sv;a<>mx%lPangBWfykTK&zhFPLeILg+a^gdvf^9GWX#yzc%ya>$ou&wLmrrO(qR zdazN;ReWpxd0Yc*IQ!hG;L#taG3C8crkR4Nackkd_%u?4ndLlInQsqP`{Iv8_7PxYpo;R&V`0}A-=Th{8MK1`U^UM&|MFjKFl0X)W@aO0IlEKy-%m=@` zSh?MU9mcuxyXOG@oC3CQ`qJsT-8K0%!O^qE_>k$Q} zG`?xe!2G^k_C>m5^dv*9?aP}Vg=@V_`eA)S=~*>`-9L?>vhfJAb_2P*dA`%fXU#AocPLH_ScD@=mDq8k4!@lg7!a<` z>$ygZxn;>^)9m@DT)gJ<+&Roc`lR0urGs?A-qs40OyWpx`xM5mw&C0}IgIlaMzXkT zC|9)|CQPy*wieEKS%9B(Gm9pA%3J>aK<4~j?6lKJ1s&~&qJRBD=uEFbO1C;}mmJw8 z*;#dT6fgIDOQ!U+XRipb$xwGDY51^U|4{jk3#MYd^eP9BD`;KM_7Y*keb+t z#A&UmdftFuu;64>JKhL&;+xTKtaXq~>PJ5&EeYZ)Q{i39<6M35yWYq`{L2$l5)N%x)nu7&PY2;1&d{HrO3JoCh;C)ud^x(ySbRpZL^ z?}#sH!S7w$(f+9!FTU^2ktzLnd#?-U?e*el=_Ggb3}nY=lG`61%5%jd7*aNZ#pTj3 z{dpL}?nvLX@C^fE0T|! zIcXSmbA!2EI?SK08!G*gLs&C&5S7J4w6N5TCS#p>E1GO`P!E0As$qs&U-+#MwoS*C zX!BUSNar8pu$CfY{54qf(1;qx#m}Z?&#WI#^!i`cS~`q2Nk-z(lM$(r#^IK@CNBd5I$|2C><=Svcrh4kw`nW z7*BKepvA0P2z&A!??ekVl|9Mu1*UuxY0J)M`>@5K0c;=O$!m*+!93ZI1MB?x>39I6 zZ3Ea_^7Si!4dL0AKJ-35kmp*v^LgU{-mY|_(X2kKU1(3WKeAumEu9kYyTd%ePBf}e z{L{!rKi&1hjz5E6<6fXsjuM@uqw2SfF^6xq9XgEPAmzs;U<3vF5VAnrMdFA+;hvWdvTMw zH;bzV()X7;vt{=A9!C09_vNO8_Wa({mfvSv@`0);um3aV@wE0-?$=png%votH2~i` z`(jPh@H4X2ug~w#*&bn~ViNEp?{(23MA3 zyYhLAGcDeNPKABhxXYgH``L0=7fVho?Zla_45@olhc&%5xK4PGN4}DcSvPQ`N_t`1 z4a3C&qj5-dkuH-8age*PbA@!mu6&DL)lI1^yzSENMl^~smwlftm4rRwKdc|;8p*Xb z(vc2h`Y`#K9s4eleuz@hV>LRli)1X`dTCSpVN2n_H={!r1-|d}9lwMPa->R?u4Uzt z>x{?MoulA>B^Ey}W}{^1YTS%CfTilUapUzD4BD^45q-4Tp_MVSUv!athz;*-?7>gJ zdeV4A4_<=}Kh8I&n`%eeyl+RJ9$IvG+nk4_A1gHW8@8Ny0sDYDWD7%T_OLN@P&egH zk6RefDj&Ix(YPBBiLX;qpwWF1V!bwC+=Ann()&JEmHb4!z8a%P$aiwFA-j(6%>4o8 z?A<6Do0Azk9qK^C&HB9bN_NmyYQn>4!Y(tUV2c?Oq0J;Z012KdiWrQdrET371v z!vF)G|J$Ah_jEb-syY*+Re3D=ADaGthh-DQd->l*40k?+j}LaBVzlJs`>n?Fs$vBH zj;F~B@xMJb=84@;5nNpgMgLjYR2&0U_vzq?Yy>2&fY-iV=vj4M{L+$Pd-(&IHp-m1 zr3J?6@O*nT7Gpe9rwFRe#tVThIbYuzJHg@bhRDPTb z5An8_xQEjGn(VYMv|vW)1vo7$#93d_GB?Lz-10P7sV{)igw;~SE#1Si3-jMxk8{tZ z>!{#2S|$F2)vsUJbpH)HO}>i>v1f5*eI?WmY{Eqk$t&+zjG{TYD3o(%tivoIBNZ7Z zljJdG@J_$+Oc?Az?c`3(-1G|_iz?A0KMzlCCZW5|G&H`-kgknIxTIAo-SKRJMA&BeGmH3wTVG7#1y8Q%va;79X#%r~5j zuG3=hp-^_+#ZmNr5x}%82Ub4Rrdph20xdS9)8QOcWzRssm1Oa&&cVt%1$Y!thU}`{ zD9t;HY~gU(m7GD(pNHV1zD<1mE8#XWA0Lc!@IZD&9fFf#Fk=cPuZzLeWl{LsI|>1a z!l28OM zScZt?ZAkmG8>z2%LaBH?s;ig5?q@EJKhD6sf)oTznTDC0;xKt!3})Iy!~AA6CJm28 zc)?`Ip%cINrU~;XflG%)@briP>C1E!u9PwBLjIt^RP?r^!b;N2hUr;3r$)}fv+6u_ zJ6VLTSc{}V>#%TQDLjIU&`a)d%calw=EFp6$x1+S`c#~3Iu+d=rXsIA0bxB8QKFIz z^QNgdHYOEa`=s-5eIl<+mK=_I7<)B($#-}U-t%tD0JnFz7k&VFy`>i{aUR-T&W3I` z>4Ga*f-adW5PxwMBA%{*$G*iln=%(0cco*{q8XUEDjBXt(vK`TYjx)gST~uC7AkX~ z{B15YBm-feoFn?R>p#cSMdt1JdCp8Mw~+U|CATko1mB_JF??Hsh%(8% zxGYAS--U4BR)XOtR>Lf1C9H%wxb??81b@kb?=RVp7)rmIeJ+}tFTl`SlABLij8y5e zt}9*!o!-S*EB=kwqW@M;P2rXN1paJNY7jk zOY!EXU)c)9Tap(VwF7KsoNP3g(UPsH8rX6cU$a*cX%ZR4Xebx65-laZh=pvWRUM{fI;;Z6x!{AaZUxQd=6q# z(NVNra1t&}&mueXJerFSBT)3;Rs+*m)O`k%TTkV$7n5k?KAQWJ{dryHgjo-)_|167X5eTb8hvfQDdviZd zwd=yD`O?ohvW5|!f!rPjz$#-U&%=xOxeZE?}y+atem!#9QMmp++uXyfW zThU4Mm{ioBY3fF_cI?0vHXXU`Q73M)F{O^^gJ1Wj^Fn$WS9D5Y=*~pG6V_;#$I?3Pe+d)949(%mni9Bi_z!99eQjzLZ5?b_34tN&-w#;EUeJu3&}Y= z9brH%Q)6NOoA76k&KxTp6qn4*`SO|tV@$g;O&(WC^xrX}|DITy!e)_4Y|u;Kc*#Qf z-;ZFXcwYOD7)qN%(%Dhbiwh=O^3{D4@dn7QWvKxptPFWfc6f(pN*?y60cV~t;GMV9 z#oM_9HLrB`O))|@)ihHK?5 zxv94a$4U>bO}H6-R!Lu7p(S&|g`fDxnwEjxxJVw?P4wT^qW_xgNa6Y1B<^=fVEH29 zUALADoR3=FN zw`;Enthg`luh|e@UDKb1_1!tn$&^)+bDnoZk1zH0*!8U*tF!g_tg}A1^wFcCXu|)s z(Px$DzacLTS)R~=`FlIEdV6Q8r&z+8GJEVxpxXJ65O%Z8-!tmwa$Z>F=u`FO7T zJc*&-L}Qs6%$P|7g%R9`rJpP~zonu0JG5CbS+c%v8tm??$%(a^d@KIl!@@P_zgwL; zbsCK6s?G3cI^2<}&yZ8?x%RBQJ}ZsM0VdSX?Z_a}f8S;cTl`ike;k{^_d(N`Ic_q8 zQX;v)D2zwn`EsRn?!`^(CfRSvr^_z)bEoE=DWBnWYRp+8Yyd6k@!r}@ zJg6#cJ5D-+d$wZxTW$EoM3axKh56G`hqonLzwoIZKR2XPNA%w%ol~hJ`tJhcsVr(8 zO~379IY9i{4L?0ty~3Wa8cZ1Fq|TQ`O7sqDK)(sXr{D1lEwdV7w6qaJ{eB@f;~Uy6 z`HBT$Kd>sb5leKN@Z(-3er{BzlA7=Zj;GC&4wR(c$lTZ0mao-X~?QUU(7{ zIv#;@i^DKgIR>wfr!c$WJf{731>SzuX!8CR#yALD^z=RCE8a(M(SJ9-NoAaF5|3Sp zqe-(!nrnyhh?W=gqwJV*rX4qqRiI|?Td4S2g(>dak?y_?k0oC*o)yTLR|%WQJ(#t5 z3$A9a6MeE=SheE8a@Yd5=pATmwFk?6DzGD>3da2op?=U2Y*RTZ89KS1n+X3|^xwqn zXub{_%_(kvbPskUL*?@~PnmI-Yj7Y|G{85j@$bQM+&j7wZP%BfDpE3mqZtst!W81q^-i_e{0}9xD0KdmBX*^IwXnydqqA!r?aQ9 z_qPast_%_On4D$1b>$w>0CW%4!&6v8b*e?^vn)?IybIATe<|#5m*C~Lwb-G)7ACba zGj&>s=1t~emSk)nS!Q9H_$}>pbKti*7mxhrV|0%^#Qa-`71sI4bjgQ<=)e13B=Oad zSVrs{#U1N>XeS(oW3zO)UJlVCk}5D@=2Dy+JQr1Gv!Ihb504U-;Ewr9jGtPHf3w$O zqE|73S}wq~UYW2cNX1OU6tq&5exGZp!s$&zMzfiy?kl~xb23moJQIh6!TapH+^0nU z-B3A^O9uzDZZ_$eU_!N|zgW2J7)oQ7AzIFbPu`~?Hf%Pmo$}E8b}{yQmSgC;%`oY> z8A)ktfpZ1uDxFD(L|^vKO2#>lL^x+AV5xNc>{uO-*5jw(^VO;7SCIg9?*zmRO5>Pk zQ+R&p7}j+0rpaMzK3gH*ujO7`I~=ew?U6 zUFrciyY0s4nPrH*E%ejEL8xP{$$;~KqT#0AWlM<&rAG!;(P+~k2z6pt#cWDYz`p4npv{+nDk48>O zGH0hnQsK7Qj;jKK!7by0t#iF!xa5UPFy0@Eg?EY$;YElR#;rTQZ{`Iwx4D>&khFS45pm}H- zj_4(*ZmDPt-Om%FV4gH>Oo}f-i{A7r8w2B7|YfP`+3BCgvV#& z?ErZly%V_G&7a;irrh-Q4IcQHVX8+89vDu*(Ppvu=8%OwqS?0nw*#H*&dXW*G4lH~ zV560ac=`V4+tcKgCz?Edur(LRnSZ#^Urav!4rO}vP(OVI&pI4~qOJ5Y@85_H6>Bhc z;d1m`wp936v8=h|%yEq^*!^{tFpjbiYC0Z<$0p!cV;b^TE=Hx>W?T}lX6ru>(08tM z`VLa#l?lQaZqlA-mm2fn$`0HjUaN9pavZ&=&db(nbWdx-d-h*MCw~G(_nX+~d=`0I z4q<8cy-;!5COq5-rv9|zNB;*1@3t69DKcX7U9G{Sce8}f85$B9d+5Gpb{ipFZmp_eR3H4;_hR3X(QVF zX~|h0?WptJlocPP55&4RcM3(!k% zWCk6SB)&pV6F&Z*&rzO(c10^@&xpaAyAJr7>WOKdlcA#_=j$Ik;Hq^Ezb1Z!iS({* zI^LG)fgNcu(3&|Vz4*%?jEr?<{8LZa5`>0{s^5O$}aX% z8ybd6PqBjq%~N`^xVIx?6J1!-e;`lH7cI(5xRc}kxI{8%J$nc@^^j!A9?IEfw>vL) zc42~irDm8(uWTm=rcCHbGX-0w)n%f|EG@X5=!y-MBQdjZK1P~s!%MrX_@Mp?zf)B? z;kyo1&voMR1{)4v*PAEm`}0SI2ix@W;TX~2PKoZRH7t;s=7Ah=Q8H({{W#Qq2#-e% zV$zU-+*9wyLr#)EUI)q6_G1S-;a9%wf}P<_Q8m{VWh0_+vF$RL?5@Jhi*;DP^)J?c zXie9922{|p;G*3-^EnBNpam?S;89n6Hy z`(r4lW%|%!&re*M#_$?1QhdD}zt-87hJu0r!8$Fa=$ z8A?->Xm?eUd0`!B>0m{(v|gMVLY{ssc}H6x77rRqi$22GejmcQVZ-@fj&!!J9>#RZ zJ8$_N$n<^wqC5CeW9VRB8R*Sb;-j7~tnm|vow@kK5vQr||2mzT2sHD_K)a&#=;?S7 ztN*^oXA3#g8|ZSFXb_6)B(p!K4~tcu+1FFN0>g%Iq=)S4wL^IOz;Ieyji7DBi2oz$ zyyLNM|MyQwq@rb1M$6Waj5rP>BP5&9-dnjVWF@0WN+~LZifF5(y`{bP($vu2`}g{N zzJK)S?(wMmx?I=$eZJ50bsWd@;QHl}yrC5#ncHx#RSczvWI)r+UdDcW`%#!W zGSky~UL z9>`7>KFr_e#lKyaJN;;`g7EHs(6pA>qwK*q{kjFOs^2)ZswLIreEVv$DSO)Y;q@FR zs^`0Nr`(tQO#+#IC6sNX+c^7k6rYES{#!7Vvu_FudaU^I^Ms|D7|z2cq4dfPB2W19 z=KF!{>g>bGXT7*{Q-PCNlnNeQ9*DYA3t@Y{3eIk~VDs)5HuY=C&jrHt(lHf(ytQys zop>qGojZh=-Pjnw#ud^(+be=ccSbQA^MpOZtxD6&mz!-+}e}jQMp( zZ&q14@?24Q1=LnK}L522fU#~xWxH2*2rw#s1Hi-a;_P%vLD z2w-iVcp8@b@X-ax;zID3tSOo{i0qsm=&;6_vB`LMAJn#~l(nFA)+Ljfk zj952BW)hQZStPtM%jupxxojZAGXi<2zkJ6h!x{5^FpoZwj+=ww93h-*v$erg)C}af zo`dM3=*vmPq8YomGpn5|clROt)|+A$G%!roNw>!{&_T{R?G+ASpZJ?2l@*vJtX7Q< z@~k&C<=%2@uG#6xJn^06p7!Rc7osU23Sw`BiU%!>ifttCT@}LhD}!jK9l+bpe(a>> z!~5Bu9D3c2vyGi;wA+z8mfKTZo_ozdbiiDbJ~-HO4DMdc#k$+uP&wu@T5tP^rpjvE z+fSQQZgpX4y%k%E-%Cxnm=kP0S+>ZRZBzUi^D~gujY0Gq7sL@Y;>VHs^^6(5eC{xS z-PX9v4BbW6a3_Y3vi~2)`D>Tn%r812x??}Yzpz1D4#thVbj;aRiqA>Mu*KpDT1PAL zjgib32Ft8U(Uiyatf{@dpJatx=`q}kho}0=v(%4&mk*+9jURp1`ZD0;04mJ%;M0As zOkM~+{ccY$dtqf4_GF)63oc2L^}wVVjW5JtQmhN!6a`?pQ!I7~=f|?R3Q4~&!!_#z z#$Qoo%hGl%|7^q#J1rP>(}r2!DiKbd+n{<0Shxnurfe+SfJ@0y=&?Ki4>$p7AB7FNwG_SY9LA*7 z2jXY^jjdJE3AZ><+<0rM8v|opnAQ#)@yebTZDe-* zsRs@9%&2|63!kKSqD#IuGhJIVwx9*C?ssG2?gwzWoPmquhG74X;b_&6fag`B6YQ6s zn?+~fFI?Qp@+R~(k*>Xd!m1XZy|zIQPG4=qLC@`_H?1Ely2w7p-;QmX_Tkf4R(#=P z%Hf|n3sf_L-Z;ZKGNBRfd$M`ma5RlBoAodsv9^b$6DZ(9Bp?Y_MwACw!j`E=$ZSqAQ#AT%oq24Ictkyn`LssnfXN2zc1njU zCu?!lMm6!QDsp6-k4XRY2o-f#v9d=UW-1=Wx!rp($aV_LK7a*w>Ktft5TOa#NGpjJ zKIk-j2uell)1?^uy8;tWA44C>Dp&9Mg6WP;*-Wxp?m6ulV68{$=&`F!dybzkUBjzX z8Gh|A{`$Seg^K%#$~X_V_L6n;*@+_0O6X;*gIeesY%Gjn#C>53gmvM*>y4;Tt3chR zG@Lyc4U^upFiI^`W>y8b;TR&i#dA z|5vzj;x^j0IU^b2{cy>a=ecx^mR1*H`qrg5Few+a92a8ko7v1Rj$*focI=SaO8Or! zBc)~)Jocnu-TfKzvnF76k8F&QT;{4L+tFXE7DoND^oQ;3>i+ zSfRjC&WEt^Y94OaCgN<-EZl0Gi-D&Xp@rXC)J)oeQ(DK7_~|kd7u-c^=zW-fypH!a zr_e5`8i{5bVS9fyIwvi~x{29XaX(FT_7q&#p97tkSmbKY#s=kCaNn1}^A1yZ+9i~2 zuiCSZrDz7VFHzijJMQ`Cz^5b;Do^I1W@b7DKUjjLYl?Ar;x=r*ydN#PN*~;%W9WQk zuk@g9!tPk8Ssvsflu9HzzH#^|C&gr88i4L zV7TOm1~5?Hl0khn7%=rF=6Fjtfy}*Z#-`v;++4(YOBdSwW$;ca#*BHD7<73XMmBCi zhe7Mny<$10ywAd>-Scp6jN}z`<6-$`Celn}U@6}G-eoa3czY(sJeZBMPh-*YXc8~E z&*Fw=<5{*Ugq6a0dMo*?ZM~YXPt|GM7Y*{_u`H~_y3Z&~6TE$IoDJ$yol)ksb%{{T#Q>bOR>{tIYL4TF~)Z_zUHkF zzw=tC9g;b4Rs!=c3jcZhWWIME&f}}aSG>)UMI*a1D_4!1Y_7!Z*sXZ_YX{PH z?m|?nYGFPl(P@FL!WTKviTW*7*hi}#n_8W~`siv* zbdX$c)NaXU)SyDC28#=K;=FXF#c!&R&hJWSe%vZKpIx}`vJVl#hoSWSI6jBhVMy3H z6ee9nr0r$AdzH-Sz$CWykLP*6|2QLJGHVn^^8A`$x>dUIK#v~WoF~~lT_qX~e1?QB zSD^Rm3_dBHMX&htfWrki_?*Qhds$CZj$@L%w}+eSkoDmLRt>z49R4PH9W6rZ6w z@FiwlZbY}nx0ojSZ~o0B#vGl)ZtrHX-Hd4Vl{s`yV-z=B^T}j}J*Rhfjjp6S3lX*036#e9U^WdD9oM&x$LOe@D z`id5FR%UWP6u5uUAFQ#HbNPmUc=-FToU4B0>Fpn=&KKQc=U;4`+=SVCRoGD5j8A4a z=lO>%_(*)9@16I zwdcR%I+97%qkTn3@n`9Cz~5x9UnAON_ax@doLxi*_SgL{rn;>2Gg`R|^1Q73p) zHMk#Jg;?;*1p_X~6<*1KHoT=I&-IKpG?!W0E$uc8_iN2b@vV8jRa-`HZBL_(x>Sho z$QPfbKewxNMd){7czst!^zY6Nq7VMsn@rvGB>kNZ-nKYdS|7v)5&Po)Gp%nP|X)mvyPzRgW3_bh)vmE+-o6a>7FKn11ib8+8V( zjql7ex4ZJz+wOdI*@RET-`D4*aGvCKcZ#Mm=3o+U6(`6nTJn6(GZ|YvjeVTQbN1Px zG&Bw3vpRQ1f3;=q5$Tp_*@ZEQ`drjQ_G+U0?s?XU#zmdjOhNv;Stnlf(dVFL(nssr zh0i3!(8a=pPen%^tzf|&qJ{2i-s6AyUj^Ym8(vG|{&NYudn``6 z!+5G!u=vqD*vrA5{}e6R-mxnkWDa}4L7&ZX^!fF#KBpbiXJuD?E;P{Rw?PK{mtn+c z;Ts*T>CRr?Oc?#vj43r1jES_O*}?zI{~j0p*Zgb}o7N=o&FVM?kDAGu$EUGL)Ohy3 zE@$c1LA;UXE^{JV>Dn{rtcHSm z`fRyf_|G3YQ|(PxnjGp*#jz&5apM2-zg@RTH*HW7H_O^lp%6=x#u?N#oy>aoQ8HJR zZZ1Xf9S1vb+fFn3R(E2Sc$GRV)gtw^xch?^kB@52p0!$ZIHJk00!^|(ayya(c%iM$ zyL%*W-lQWptLyV)g8|cJ7N+pNGe?O2+bb)XAMPZw=z{o?qGz!@Cz=ZT$8&t#P~Mvu zz#utS9qrMR|AZ5vV=I1);d1ZBs&dRaRl4tK#)YbCbZ#a5_M=VZIU@Zo|5WLcqt3BY zTJXD$=+IL{|4k8A^+f43>Lkq>6LhI2`QJ7n$&9r~vVj0Mss@^slZ1iuxX zW27P{H&*IxkK7{RC z4`Tj3NB-Jj%uMyxRIU1l?v5{D^Zh(K5-%b==_; zk<1Y8+CT^C|E=l3&I5)2lW_y(sr#^2XDhD9ZHA}xmE7LG3qSwu#vIowjFr6a!($t; z{o{I^eqN3TL7Nf2b~}OsC37hJXX~f?Q2+8E+PWOU<=~@GJ1frxjYL-7n8oI`Q|P;Q z1jpSFK7oM~OQoN-PriINiMOCJdM7R#7USkD$)w(1jca2{@%3*R(xd)A+eLDeC56~8 zezWc0m%vndVcM=dw4}+&dNgUMf_Yv67AHub+nq&-T)hlx8&;uwN->;U7GuZELY$QR z@4AlCjai)z#kvg4mTZ92@hmJ2S%~?Hi=g!>7rlgm(z^droR@wL^)boJYZ5Oy%QSW} z9l=qde(Wro%;hR=I;Opafq4xMg)T?f^h|6PO>0$M7RF9r3e7*OVB=qc`@4#fZdHgU zH)J0tI_S)GDcI9I1(_KsFnO5*ZO+B)m# zH-<5Fnlon$U%)Fzf!5oO;rWUJytGI|=j0^3JUkDFf^uNHw*ZTzl;Kf&1xAmpfboJ7 zEWacgY{~-kRiB4AE74%>5)nT>9%Y8Ha29`W-QU@Wh>JtV?0B4(bGAyj^pqR@$Gz7_ z^7daJPI+s^L#i5lJNydDTC7F<&w03bXBG^;Cm_I2xY9-W(7w119iz76Q*I63SMSA< zK~-p5vR-@?%kgn!jxeAzkvM%Gw&W+ndwv{p%4cG9%m0vjEd~en$KYzi4Eza9T&EIdmiPn zm!KPc2H$rcfXnO6=xtJr{gRO$xj7g6`z=6M$z5*EO~%kY@u<^?MUdSrc)0zC_=9u! zTX6z6o$_VjNK;-GX4a!_`*3#kLVOC7J{{4gPaaM}oWdf^9Z?LEdAkrkstz5`+{W&} z=O|h%8SwQDIISoByUAw|)?vTowzi_;^Ezw@T8W7ICFs{L8;%7T7(ag=GDGI#<-;Tx z`^T{>hchw3ky(GWSSwk@+@;HrdUhuEb)A4OMzgTae*v~nT8(exs_@_J)5tD=2Q6|c?5q84tg})JWp%KFd--F)#^U%~gf<@DJ!s~iDPV20}xgEJ1-bU&2-N2tA2{Z8flq*oxwjW%%V# zj3kFuh%$+$=`iUzlfKwt%MapIw@iGxItEK>#t7p+0p|>t;6&_3Bucln&7J$0BtFG| z{hM)8Ol!6@)8qc9vghb#$e#N}{~fBsB{^C&7_Uw4} z;hpn7OpV%!?=8l2o_P<(dOXFV@5`VWGy$K|hoWY8G}JV*(Am8NBmeHluIO8czWW(J zE~s)yRU2MdCAwXYZj9D8qscFGe%WG1mCEicZEM8-FZ6hGQd_P|Zb4TQ={m{(gY|FU zp;?EgfcYJCJ#rbn`bKeZ_qMF`-G&2ZF*rFp97`KVVP3mb%-k;?l~p_8aq$vnX1*0R zf(j3mv}V2Rf1kulf9i27KH6qY9fiKESY$)}tupU)vEnNuQ|@0ae#MpY-dbt%oTDby z)(UfHtrE>@rB89=Cj@oyVOhpY@tS17b3!mQ14Hq9;tZ7DUxec+75Hdy3P&$K!|Db_ zPO5Lk)x$dSa2J`4smYAA$hmwdGQ}Z2p%gA1Ixjrli>&Yjc=KQ&}8;|WW z;Q76}Y-`w-fx4}jSfWnLhrPME<7VvZ7=|U~o@g^-C@h!H!^!nU2#nkh+uA#r^!o=i z4>adCLtXxS-IX(%^k99sEjv7yp5Fpje)f0gm(T93ujtRTl4Sh*(f+-4>K!I3>Pm+tCzW$Gj5Bp+Rme>;0>0F z2YS2A-6z@@vDDK-bQ>F5X9{byM&_~3Ui4A-;Z|Aebo%;G(O>`vd3&&=l^ea|NY`2Y zxTv))nJF1;eJd(8H=}w)cRv4q9qW$UV~vs}zU25p?^7~18y2D5>L5yGcKH3~Ul?Sx zM=*@cvaj$4%Wpn(w<%S>k&-CTg?E@HJ=)q}S z+*mC#&X>@B(vN9Ni__K|5#N)mZZ1T@-}V@I%nX_TMc|J^7QB5bu}%M+%oT)1prlH# z)v|xg>ntoVOKMgLBeUcP$JbiU|UlC{;%2;u=@KTq2p%x1%bI3z^Qat(eA7rn>F zXaGALb?1OyuB@ztu*CY&Ag;mv4Us#c+4NS45tdJ(KYinX2t%(#px}$ zF|`Bl)*I7)eQ$ny-A^*xZfrTxn-32RVx#EHr*cAR`bnN;qDPiAhVs2(2+M`ddpSY6 zO7{=sQR!hmG}%-9xc&Lq(1kjKLHm@SPM)Gms`a-)JLg%rxTgrtBag!6rQB;lO6*nH znp!aAQ#}jr9MhM{Zjyz&EIs2!1F7s1$aOtK+2dt66QUxRkQG6( zF|S7eHw^aUCSiZOc?gqqv^%?=aAj6}FitmBmtK2wMCgvdzlFJY@MSyJW?#kk+%I^z zMV(s}b(vq*jhUN!a(WL3Zt`;FQuP7ckt{6cia@qV3E_y(p&Zyej0p;%+~OF_p!R_@ z5RED4lP@)d)zYQLgQe+i{E+6%k7t}{en1mSX^PNY)CA5$MzbyTk8$OUJ(t&y0$CFd0_vg~v!l_$z$4R&FuhZzic9L&MkdBfPVWOPG&fhOE z%U6Y6lG^g{QzLdo$w(K^=nHhIHc~#R-@|Qu3mwsc%?c)FM9?6HpA{iu~^+koj zJlHObTV6@e^PE7AI5~);9|&*u^8ilj?aAov{Tbb}#>wT~Kd1FaiE$?LV3#LMI`4DX z*;;z}wlxzkvNnGSi>af!_*3d7{}=Ac{M-RFER@dd6k+Umhtuw@uoi}kCb}kye%B&7 zKP-Yngd?I870QO+LHrvQz`RsH+HdsXIN<{qCV6ngt&L7&rzt|Isz1JL$be?GaBq%Z zLd2raSaDv?LFu}jHL^ReFYLw6HgYYe$^BU1%_Hglyb&70FD(c2R(T{RwH?AzVG_G! zOD9(ANV?h&<|@N5o~#XK6VYs+h7Y1Z59oG%0MC!|q~(WoPRhAWFx1FX<}>q=leh&# zimxDaLY#?UyEaamfRrb(uJO zV>3?PxQuTLzQD3yb6Pyo<;^YKdGd5G9v#z<6J?gxZnro875KAmcnE9c=e@ToQhd8n zykItj7ek`BsI$qg=Xicxq2hadwanVbSMmg@JHTGMRb7yyp>4-K!BS zH;rO5Iiv3%9m&h*2h;0X7?lb_sMJ1)LB9UH6F86wVbZ6X;mL_Vgr}~4)hTDA654hk zDm~}o#eC_KnIn9Z$FC8#xG7C@+R^Q3XUWf7aV+g%Ilo`o` z+4O7#gKkE!+0DWH*EO8BtwLF_Q~H%#22e-%#pQp!xwhVuCLjA#!^f3Z7pq~LiYm5G zwTHpx*_a(x1RI+ph_!x#F8)o}VVM@ox)|`5vN_j=+3>m}_)dHfMIC(E+A4sB(!=|3 zSs3qJlAiLz;;ASMUt<`$m}g^j*GkywoW)j`Myz|I!kZ&R^L97l%4HTj7}b{|DbQl6JKJdb@TC4A zPTv;5jynT++%b@oi=-pGzaJe_d?X7le1b>PPu|ppD;1m=qGivMQ*3Bs-kblQ*Yfku z0MW^8vFvv+`j^f{Kw1&3QV-yt@Mnh={f4_$3r;W73(3p;6w&%wi)-kqH|-`<`j7yGcHz6U>zGNVsgS3W7$=Oleyp0aJrCeqI} zvfhj1C7)#0HW!|&Mk4#fFobTOjnO-DVYg=^jGvsq=a9!3HtZiF16y!FfiA!G>MGse z7Hn8)O~*ENoY2{Ud3)^H%+yv`ma>nTZpF9erfeJBg~wlZh@Qq{snOetu9IHDc*Pnl8xx05wi7TZY!=MdW?{lb$+DKqnSa7n*i3kfdrg&T z8P|$;e#%U+ogrsO2p4ivH?DH(!u%9{&WqBa<9g{F%T}jaGbP?w@)b9>G+^q|JIJ|z z5z8i>Lh}A2a4nlk4cT|FwiOpYs73cTi}5+=Kba{^#fHw(50$(at1C7jp!Oh68s9{n z&U@(mr^Gzv7IbUbmIvQ!%Q;7f15dQ2_P&-Jo~g?2+6t`i{s9wTJ;D=}s~EZJB)S#t z!?u1^u=%$M-%yUTssHiRvOq>XFy>jSZy0l8JLavLk2WV}z|#3YoT-qR#fW?)C<{~T z;t^Cey@eeY-{5s;1vWj?lr5Xcv(iku;-)EcnlM)fe0qh?|L)@H!Sm3N97Ds{ooI7u zqiAf!7}aMLd;(YC)SP8_o*XN=o}p}izn^%r+A`|*9rS)GY?l@DP;oQ{HEHpfDV;^1 zoC+~=&t_Z;I*PH~uH$;k=a_i=12o2b#cjWjsL6Z|l@>QK@$V_zKDiG+K5T_&=k;=~ zUx}tOmf}*}BAlL-jUTxSuyDY9JhzbZeu8jY5BczPgc(iys!+wfR(gU~;AfM$XnJ%O zj%BCF*@0jNPah6y%<9mt-ucb z#RyT%lI&|bmU+&DBnvU*^<1p2m0uqc_^|{9KmlZeCRo*Cv(CzdAr~tA_r~8 zjOL3FwqY)Od(FkihD;0!%fpO)MKGLRj#-lTyq~-YjqTQe<$Ee+3OGq7dFd<3=5!YZF+$s5OU)!M0CG=4ZojQ0~?pq$?{ zI3WO zC43&H$E0I}WP(RzNS8!oHYSKa;M%z*xHNScs{1ZShx}yz959EjFK5tj!8oq^7f!cm z4@OI8&)w_7@BH=-4z|ajrn&)oo}zQRFT>;RE8x6u6-Lfj1DE})@kZu|6jHh<%;3c!w(em#{+HQn}@C+WAl)`fNDnz>$ zphU7NtwaYuIiwU*oyzfZ*Cwn{mOj9!U1+(q2F2zFFed+y@EMc&cC%=M>tfk`$aKzl zFpi5?M$-S54_g`8vF^1YC+5kFKI=IOTb;(s(|fVE&u;jQu7=^>8mPz`y5i0*tV`J@ zox+=O-h3<6-&A4A*BT6XeF(e5#p65rG~7Ke;LNKlaDF10qSd$2x-pp^Ba>*79M46Y z{$rP~ley#5NLDI`GG3nD$`^XEWKRdajBHB9f3MJCqHr^hoyVJl=P)nx0zx`pgv+XP zm^i2oiyxoFgVa-q$vTInov*@3W_qpeK7`7UXR=>>g^t7D;o8Dal7IdR^)JcPdz{1; zPv)@1dNwT^qB(lYc;5Iogqw%>vv?U9?O?$IcO8Cs-IVqBzCdByOKkLij;XyHP}lhd zE^m2`%8^e{>Gud%&OJuljs`p*`W82YaXM_pFN|8Fz}GfP+`U?b2Y0DTZ;=|ce?CtLA?mh4${+w#Q5c6=!M z@4{HoR4NlWZecv{^_a!f3DInMXFM0T9VXe@K)UPo=by)-g&KBc^9XGQ*|*}f<;`i{ zpw4_Y=M}T&>=LNXFZpV$(^O}*x(4t5Yso*NHI24y$MwQZ*0<4P^bld*ie@?Dh5?8D zP3E=|$%k4easIM7{CRygm2S@9a`(wpIyj2U--NU5x)1-;b>bO+D}Fm{#AepI%zWFH z8&9_3x36ueqt=!UF>UznnslD@Y{R~<+Ok8jwy@Y_|C=oA;J*g!C0Wr#`>u3t*`3jg zr5{&x*cJzpxnxn2?7tG2^i_JTMIUUwdMXQEkLA!#Ls)4Lz#ZbB8#&5`-CCK_bbuj0 z7Ixq`T|KJEj4~=nk2;OIR2iYmrt5_JVcLO4n>z79h9OO5KQObXo1iR==~`sUw^8Oi zbHn0){l9%hQ~7)%iHCP4(7rT|C39x7xoGWir^ZY7h-8E;#BZwUNsnpvyy_%*!tkzq zJlTLsG7H_cuM;nP>O{|touo^w6YWAfvC}@`F>4#~miYOqV!HG5K@-*-mK<)r1zUJp z@%p|VER&zt<+8BU?ItSiHQb!LBm>!a z%Zeuk2sc|^S5Ml;~jc$&u!V>9uBHd*RH->r6Z_K@y9 zk1kY|yz$}_9r@&OM;6<3;+T&eSt;4yEYW~{FLh$7p&?zabY_=P-NaXB%y}zJxqgf} z16o_sZJree$?Mvjkn5^KdU#_JxYAsPdwc%l_PVJ&@MA2!-bb-^spJrZr}XrKH49o9 z)3Ad++en_fwd|`l#p}@JrVjrl>G0ewZC1S1riQ;RyY7}Ax>@?XsAa^K;-NDM>&7KL zj5+S93B$w8xI?bz{@aq-d`J=vZqA{%aV!s+#!xwZG7r~{V#11W);#m(uwjxjIcd(K z)%tY1(~h6^XmOzEqaL%gIOLiZJp#1om8L15ttOA^w5EDVTRu$C=4cZ=Cd}!`1=;%I zT{UE2-_G1r*+uf!$qdL5KSM*Jd{6Q0v33?MZ$?wac!GFEhH*)sK#pAN$~v!J42m}7 zq@S%hy;_}tk*eHeqRPK2&6pe5jCJ!>nVa5}&sR4k-I`J7i#oj;TQI&{li|(U2*a%% zFLW1w?`IuO8ly*-U()Y9IGNf`iTqg@$Hw^oSh9MGXa=LHz9gK#XAR(P@ja(CbYX`l zTFhLk%u_3VqDcKKO!U6Ne9#Z5rv5-LgKs#M`x%dqf5xGqlCM(viwe~yynaGibU0O3 zhp2Jau;%RCy9EvNg+=-;nQvMobBn)pSzn&b|AdJg5hkAt}NhZApnHxX~9 zumFxIaIVrz%-?b!&kx^4R^JE6E~v+%X7xCbe-BS5+=h`nYx<8BEm->rw0}un%IPic z-~Wgn#^2DV&o8t&`d73?1-=vg*P%E`_TY2qF?}YtoSDMDYeuokB9xD%4{k(|WSnfY zIYRc!Z+qQGEl*>V)iE6Vb_^=1wb)rwhyEXG5$k*miBk{3IO7nsaSWv zjOGp3v8ds;Ft+Z)?_WJUjGw^si7*3{gjEn3M?L=;93fiwclRh^0r*gVMPCN^bYj`L zCRC`rg@V}qi2Ak#O_Mg^+P1A|yjKO&!MjlBy8}nOH(}F>G6XFuL-L&pv`pEG*59kJ z?s_$BhU`Px-Gk^NJK0NGClG$@Bnrj5(Xm@1|F)aWS$0#YH)$k$-VpzGtW>Q0>&ngk z&G?_kU99}R3uX(7(eXc;;#7x zdm$Saj^#jWTrS?4FM;cTrP!M+eIOnya8thkjg!TXWE;|=o^9U;c`leiFRLGzJ%Dh(};%*C|yWbAyHD9o}%I5i|9X?i05#3s_QeY7yWBbc`xoMqgR zQH6~dBlFQ$GAG^DB?kHy(v3Sh6)OT3Bll)8Y-)GF>iQ8lXPkvb$VE)ZmyVpAL(m){ z%%a2VVdA2 zR%Z3CLvp_@!k-Z_(0Vo*O05!bzRN-=bXkL|idATJ@+6|2Z(_5-Q%ul$h5myZP@8xU zA(f)P?mLV=F1uhKC7n4oCAjpb0Lf$X5PmQRg&VW*JZ?VTuaPsM3Y!7tYB zh}0X{(@+Slud{J({CMariNWo$nK-Yu67^fQ!E=5srf#_h9nJR$*zgxapv22cN?bJH z4-);}K{u%$YR#@n9zbRXH}>Jrk!|=epbQ89uEEo)Li{?KkG5HP*yTBk#=)Vy_S}r$ z6~E!ko=R*nnu}Pgv6wM&64WeHVSRQP9K$z3tNjUlb-s_9AD>~lN0}#*H0Y?K#otfG zn^f1573Z6A_6bF9FZ+U}rZ3R1{0=lnorgo%QB(|<40Y)?@ikW_>8zEO3#Y89yy;>`0u+G&0BQjpG#eM z?}G{3hzIRjZ!<33BwUYm;;&Ng$TMBq^XkP`+}1)(X8uh$t@<0>OQj>;_X%D`+{MFP z!pIz?OOFM65WYVSy_E-}L&p*5=a7tXsWLN_jLh3Z=dgH`oKbR=Sa(#Dl|daT#%=La zS#p->jnlMjX!5ELw~g)1`J=7qQZ3yd5KRw79xFq7*y8;e z(PF&~cdU2dw=qsMz2L-G+Kznp!H&J!*)Zv9PgWU8=CH9lJ;RMS_gDu$oZp`N?`W~z z&KA5@EVIzh|d_jZE{P#(g(7Zzza+D?O zp36LUwIkCdlcE*Sp9l5bxj?x;4Rc($EE0@9>A>b^`|@IVZ^ousQf0XbSJewwak&A1 zALzifetMkXmX5pEols-#j(yL@;Z*O1Q2e?Pw}prPX5$OAf1$({0j=rgV8EBDX0-0w zhj$Ew2Qtf*O$XN;z1Zo7JKxp1^88SVurBOJuP9p@G_~fhD^~pX z*-Wyd#vJ?eC7y(M;ADF%r0wy>54$+jo-KflQw<{f+(z$%Kj7S1gXcc$a;;@|Za&?U zs^a^P4|d_jAWz;ChON#nKQ8z(h<58F@6_Ip=Pvq4pXmVhu=3!xo34CeBDE)e{ph;D zmebYxaMgpJ{PJooxKa-Fz&ZQIGc(}GF z-?%$)gZP7c=nde%Y(H+P3gp7{5c+%#7XMYh3=l}y``^@IMWhyQ&xfVZ}J zaDAd1S0*|$=!6qDyw$=(D@DZr>WO7JQ!!E41TIJRBc#Xs1{o=KBZz?^sqQ9;e*7sk@)@_&b>-t50TvFx3 zfO0RsUE|K9a-Gk#c4pB5bu1aKi2Q}Uae7}gZfUH`TjW0aP0kB3__yb}b&vlFNg6UNXoC597BHAzWo1#OQ7Q%zO2J zT;Q`!yx7jsUHXt*se9w8(|0`uxEtHS)hHf||E$G=C&y8o+JKw_Wv;o?hNpv!_^z!b z`?=dPR^6Fb=6cfcu<))g1oGn!(c?x9X5Dq+dpC<@oJ9oNtO;lP?V&s+9n4W$fjlf4 ztjcs>$wdv|%QGI#@$1hPVdtGD4Of73s3RPlQ*g6>9X^`Zp=bST^xM{yu_xMbpIR3x z7Fh95p&d*6x-cZ$i|;o1@!~q+iq8y_H7tVtGa`AqTJouPBY4Noh3tjg6X?YUrE z7p|4LVX3z0PwQN${mP4@|B9#Jbr3h~3Zs5t1ZT->CW;PxQZbSjrVnP{)G#Upg|J<{ z^yYT=r!ogJKGd6vjwu6p6Y^0Bg>-@NUV?$89IclN`MABo6(BKebTC*j`prJO%hICENC zwmK`?;0#MT`*fKOKja?zzgt|$&=JMnsslzemJ#t44cA=j3AIL9i5a`RsC zlt`YvV2~fbdHD$6QTXlRg(!dO%9mFav0170>V=i7K2R*eR8f4DQ$$Cpb-1<=ebgprTJxS{7@(Fg{!m038Ari8M)9V_yzk(oUiJ)23#-=#fBA9)9A|9;_E zu4Ed;CsTUHm_HMG(_J+Faf{q|texa=g9p*vCx{yFdi-j$4V z&>-d}`0{nw0Nzyfpu=%j-rP&7C_3?{_Dm$6)P~vFo*24r1V&1}N%z$j^v<}5Pe0$| zh*vZADV6VdSQk!DvtnvDJH8KbmRzRvkLOF)R&we!fwEqu2Xm8uFjapGYxeU2szWJVL2Y|4VQ*7R_4q`^ox-qRYuwZb-cmS(Uk-)pUAUTc z4X-0UquO1KKee=(9?^wMTUzlyOIxnmE4>dp`tzN}0O@4%;fV@gmjCl%grPST76^MH zy+1c~c42<76O}E5G5AFE(UU#7u9yKWA=cd!;6f6j#U<6y+c3`fr96qrd5 z?+lflu-|(Lc5mL|kxEl)#I|A9bOX8^GGzmM)90ZbSI%{kPFQDJk8!5XeQ>?9Bi~2c zaopHG-1Vpjz3-TDOI$ae>R`xB!w&3lsJ&!ATJwndAYsBOunAXU%CQLu4;_lN?WSYr zI`K>kqjk~ogIIm-4z4|w+079(+IqF;NDCtx^f2S^&b@?HXT!Nxb{wr`$Fheu+!-KV z#^zSc_c7&T=^GAq)aMtOubuL0O^aU|G??CuzH60fD~!5U+dDG&)=5mgC1=OFamZ~o z9y{J7W2VbeOqKJAQeZ7E&Ug$Ty}$5&+niemYBS-OA*22nvyEg zv%9lugc0}L>cCFt+tFrVE55F7#oA025^;H}RKaY{)qiHqHlLhnI)Ar{T zD4$vZ(}%IBd?P(cH{;QE_(D92F9u{S=~;aVp>tm1wY36oyc4c;KP^s*YsW-u`OfCF z;Zon0jA-7B=a(qbtI1b<%zJ^cUUzY<$3>_}k7nlfgZN-tg9qxum0CBGlN&;K$={Np z_Y}F(dmqkB$$>^kVS$~Tju~}x@!Wd}bQLz>ZpZ!5e|;J8%NsDyPWm-}E3smQbi)l+ zl~P0{F4*`J={9em;#!Zs7q3ENqi~Zw4j@xFugbogK$kM)3lG67uNdF^#_`J&$ttQi zv*4O8d#FE$Z)64X6f-fa@;?O6nvFge(=pp4A9^3kk@Ryvf+PdzKCvDO%Np_S;|H8} ze~0M_&rsuh3%Qc%>+?X?_l(`>@^}+MLf1)m>S|=@6~I?|Qnza@!vLS9n6`F~crGR| zSp1d^jaCfXsLtc3FQWC9BAkwyhlz)0VaL%#WX)cH%l!)Qr%45Fo!pCu!F7lWzJjX@ zu4C`zOPF+9W&qlIQPXiNPL10D6{R&eF+#d?FE4@33(1sTmhNH8Y}^gWM#Ir;_}eA& zoc}aV93{`qP-k}gX-GTq?AjU~#{E{yv9w1j%Kyc~J2Vwvh0oDib0tRJDaVM*J5iFn z9|y`0;6g$*0%mN&_EE*~buK`ZFs?&G7h&0wEQFV4VxTaIc`qF%hcfV8*vd7-7Qn$R ziOSDngttG6pFI4yUbrG&y6yP$=L^)>3wz}55{&zv3Xj?pm=DOnt0%eWb95z|d#^{t zhYCzxTp=1`DGU}Cp!2a@G-gZZ@QX}bcFn-S#x!&@OvA{)G+`m6<8pZ>Ce|-N%GGSF zeJii)F`El*C-R`@VD3}tr`OnR@*AKPzZdFdupU0IA@@6wT(n~ss)vypBgy!{$zt9*z-QCi~8N^$xTU?|7~7-hX_&fF2$j5hj1P%ld2>2tTP!)e@{9 zy8=3uD=}g~Av*S2ig%^iD0`NH4?Qzr^&=C z2iD?4Q8I%U&EbcIG0ZC;&+vT_Gjk*z z9R$Bu!^dzB+NkWs1@palZ&!_Yl`2d&-;RVxSyPwq!5fuBup4p$X1&fxPU)g}s;|Q# z{thgn9^!D{Cn$N7%*xqGym%mPnK`%!{81EynRWHqZ;31 z-L5-GnRE#+jW6IFFABr@BA(TsM@ilp80elx>xtr1`+pprcR-K*|NRve3X!Jv4lSuv zI!C2NN|UUJkc^~KX}Il?5h;YEWo3_yvUg->XJ+sHv3}3*?+<_7+*j{+*Y&zyuje`E zab91>!p99b|KI`AeV*Z^@oPLh^Z~u4hd`mtFJyiBi^pGcnED}`ne8%#%QltY(?Hcpx9l9+9M7V6Z3gw{ zq;l(rLfEzVgKsx!@lTZG0BSlh?SG)6wEPIZONu=*Yz$9a&bR z!Of8xw0NaKlX^|jEOq4U)QQYDWTt4NAz@~G7;C|uLE=~cAgo}~2iqQzPV1sm#JUc$vp zs$Wg#hqEcHPfTR<#CRS{9?nbRznw3cfpG>N%zSRg{k6g^8fDCQ**irx8qn^g0iPEe z@Rg4NV>TF2)z(O6N+$H0V9vvBtvF<$4U?n0aD6vh?vy;&&lCr4mFsrCmcxlpvpMZ+ zCNdY7$o7{VsH?vSo0rJ;R?laGhNDx8#mz-_7Rg zhnXDoBAwr)C*D6Rkvqr6b7kOgJ{GUK-XnibGj!*xa$EixEP8iyBaZ*7&wU5=IaD-Y z=L&t!_0i|XW%?YbZb;)SV>-c%$u}&xs-rbOt97R3L;1`rZCPbt&%tut)v{Ka?aAiM zMVZ3oOXmy46b8CY;-*z&xaHhXKG_n=TRVL?!pWJ7!>qYju6gXa%y7+hxWq_@K7(|q zrK!W~<=SldKWz?Oq$~bLecpX;$n;zjzPV{e{d;mx)mZV%3>yg|bm3XKpO@?s&i1%$ zMt_^ek;5eaJADeIs(&D!4G^stGKGdM0rWR*))?wRDdK_`66AMKDeK9JMaZFqk&2BY)=qoejXVWd%!CQkD6O`!u>wie$SGc8r#kYv> za69@Pzcjz$4g+-8#IwYNuwAXF@P|W!@!mOjrXR!JkYkuV z_#_f?Pa|THO{>|066}=d-(jBwx zSRMuju0({N?CrJI!SlH6?edqw*=7-T-IM)Y<|0t?zG|J8!Q}mNym+-j_R}k|^WSPT zj9iOq(atQz19fm_2G<-;j#7~^P?Lx*XYad_6Lre9>J=*RnmJZJvp|s&?;;m z$~~9j;-Zz98MO|6eSa zo{O0fXD6?Vg1(Zx$3%~8NVcskYgBSb< zBijY|(0T=YB8y=iUjel@m56BAfK{WV%XPz2#H?BfW9jR4kDiCK_H)t4Vvc0)XXA0m z9DJQH7c=taVVig!uZsVvhjR)&I>t$lS_sE|apI*v?YX|ieY7kphxhC`upga@X5Z5= z!fZBoZBdfLKdW+^D^vE1ERb}zozeFa*4dLNwUVKw;MEzD@V6tNumaJNUMQ5gn&wVmB z<%wP`v#oXQ)?w16ZMgTL1{*ccq3MzA>FUm7*N3Cfy0!;tyEfyuTPb>IuEoNEg|Kp1 zhIFsR_<2LlV@1Lt{5&6*BIjddSSHJQC$Lg2lt!iYEDLGLn3PjkIDQ$9cTd5~OsbwfN$U`r$44$1nlCOfjd9_x0<&_(u^=}=L#7o<>Z47RA zNyO1h;=Mj!fSzBspz~buZ?3zAFWui@M&K_z{-?m5DT-{R{}*kdKVj**XE1zs8*SUw zWBeHLPTC#Bd+(i?-Dfib+{%$EEb!CQ)?tI?R0fNmKrg|bm8H!X`+g@b_m~CsZKLry zWh`!3y~^Kt^OCHG*1i}YpL33uKm35V|t zfy1S-sBbX~_J>y?UOei%z3b6+#cRYwHD~aGb~L%8%Oj6WnA_iqXYboEcCrma=UP%r zIzwLeF`%aG2M*8gz{uHZj2@}XuJT>>j(726$WQ$K+?qY_Yg4DA8PARF%E`4(y#KQY-G6$pa+9#1 z13cI`qX+ZUoC#M)Ui#6Mxt`MPW^Kkw*>i0d|C@oPCa(@?&r7+zI9%mC>b8$Z-GV@z zjfz5zmN2^Je7dGj{T^# zB$&H|DN*(`h%SfvFt%F&iwFC1p0)Hkt^>8+xU-_aD_8w?;>kY_JW_iEM&`D-Kfngg zLxS<~`E)!xw+?@0@1XKXw7UJxxUa4q7Ys1q2`?LNUGBsSLwmAun-5<~ZgR0guy~3k zW0erb%_GBDWgE&Fy+Zg{@~Ew1`YR@hk3SxOyAQy-ZaNEyEysN!v)!maR%iVeI#dHkS z?f{MF);M-M8jB?dFsXhsD&9#?$d7k8y^Qbp5B6Esi|#Z0xwun5 zT8;?i!M+jnx)jMlq9f(Mi{Ql`;q;Zyy1qV`%O&IeLHOf_mVR{7@L_Y8UW_gAC7?C@GeXHcV-WA0Tl4Ugd5yqZd`qN7@nA5vU7qn#n+lt5Ipk$?$4wC*q zj8Hwk1wsoQ&{1tX3U5lEMEU_VZMuhLm4BrGx-Anvh}YNNh6k@X@$?_jfzI`&i7d|BQj zh;3p+`9!o5M+H`p1fI?4F^s0g0MN6>%PW4sj)&#jg1*|LujSNTZqe5EtPS4stAH(}Dw z58`g==?=_`WUR+P$=O9SY;!bMZyd-K;;CpSy@HPl!s!1mg#QftF{VQw>ZZSEti-J-xoVNGe@GUOl8f$sS`)7VSK^^bjN zk036K31#F!<{9>5hIzT7v+ zn`WoT>#mMCyt5U?mOEf#&RBfjz8nSNd$ICugY)4%(zgMNKP4G&wTP&4Wdj zKJ-5l$X{w9T&5n*Cz87jP#?ha$^*FQX(VTLh~Uh}p?o$Wga@|}_{#vm|W1H+^3a9MFU&c`jnoQf(Sryj#?K4PA5tByX?8g%ygBe#8%(uG2v2PMDOK1S?;^qD`_hw@*XmrC}^5L#@ zR_IPYD@RtCiucZTI^O;01*hwMu;K6oEcmb(o#Q31Htjr0oL`~C_!i7t)j{qhBQ8nq z%<^wev~KW_vyC@ZD*QM|*5LfSK6F&+!@~i>1!(YPKLsDoyG&kl>&f21Js2L?otZK( zOIEg}Myt+zQ!D$V>zlAYaR_GL4?@w)5%5l*g)95kqHOa)ERVa5u5Er``101wxTnq3 zHNxuKY0D5*>6}>CQ+lbrc)5FT-qH8viGbd`=px*c$z)vbo}9R=2k-6Z&VSJk!t3b5 ziUuqG>SoT?>L#pm@69K=jnF(f8$C>-p)z|grrr{U(fp+_U9uT>B*&CF=NYg*sDh0oK{>|5z9mqOD@{gfJ;yEP&9Nr6!J&GbM#c$aS_xzmciYv7QMdT z#^8-#vGu76r!3MCjn`$lcawh0O2%Q<7{@ zKQ*rUpE8F|P-Nh=Kj>TV9Y5|QFiAM!VWmn8RN9M>4s+0az&M=B7>`NKXTU`GxTX`g z;M%8?xIDQLEq#BY$E23LnbLuW?Dcrr!HCH+BW-JM%z_vCtQ)7pH_00OuUlJ|FA)!h zaI%%^WbeJnU(;t?X3uSZ3v9CY25fLm9S5qDrN zUanb<`)zjO&PeeK2}3$t@h7e(DN{?MEi+Gb;BDdY{OHwzA2r+Z(3_Sl7GLRIi{CIf z{{eGjoS3SKu|~3|73TMOJn-5?%Koqi`qwr?3NAN*PqnAIYP&()B^fQE7ZeV)`y51un$2 zwy9{eO2)gTGw`BuDf(>PfI$6ySR8c@+5_)kLQCmx%XkaT#cyD5{tP2$+z~cQJx)tU z?(LSf7<*$c)(zi=p2-`rZDASuPv0O6uM#ZWRe~8NnS9kIfp>?5(Lt}fFiq4MqHq_> zPs>`FJO^_>q~eE828PO9ga-MT;#>yR72+u|J|Q~xMHGCyg6=l;aM*qt1rrXVbirFTTt-k$K1-zZ|zqSD-pMn|E#|b1LKbJv@}7=Ja6D z1U-(e{EQs;Ln!T2fOpepVM&NEtFz>pc3Xr?g@v$GEx{q{N;rP2gm=4AxGERnM$3FS zXe>p6XwHvEFA~i;7smo~ail31&O2r89i4}o9GP*UGwehutf42hnzFmc77tOOGmPc zPdN8X2HR{hXVg>`ZcS~#O2uu+F)zUCS-Hq)p9}rLOQpMN1!ALCVgI+4xFz{?kMhNE z6aM%PxgUR*FT#%UrRY3cn1Hbbu;0H5M^3KA_LvPg)3y?&!CMf#F^4l6r!lTyGQA~F zV7Y4u?GN;&Vow)7h|{Oyp=L6xxQvgn+mO*|jj*FuAaQyDcB!sLp=6$`ME@Q7bERmM z`ABQC3>EX1bzqzdiTccV*EHQt=7!SgXEu*Lfv&V9awF0Zel{BaJCe9GdU zVd?bWnoN`H$S$y+VW z%jRa~X&k#Xg%|TD%DFy{CsU%O@}v*b>s;CTfGNY%+pv}KcWhQ~g#P=hi1U^G*P41v zvAK+yUoJ^*rVhWx)}igJdgLv=ftergVP)V`9BlC#-#tEJt(WuybpL}k;}vM$(40-e z7yYJ?!*;%zyzqG{Bg~TceBn5n?HJAw*%wZp+?z9mOX7Q4xVWv_uz2Y|eCzT7_rJV^ z!?zb$+fTBv5qQk2JI7 zl2ZBHy|jhPtI1;FRILaRHs4zf>YdbJ|CJr-wN;DFSLqUm_1U$=h@}>$tQR)Az4Wx^ zrdxA|XJ-bAK6vwZ4uiI4Q%Cf{Rg=^C+$V(tj!xpITww@C$Ix6=cvH>%d19a&m0xwG zL#7#B>kL@r+liLTB!?NR$4GTOu6?9StHpZ4Clswx--u=ErZl)D3@ckJ=K9#MlS>!j zX4%sGkR4aab%R9@jlM6r)OVRYsFuMb%TyZvPUJ9`acr77f>Fn#Sb#np@V6(mY8|NM zEBVrKqCHE_*GjU#tIrs4`hNzjm2=eD#|AuHXvBIeQzvBVGW-8%GL9v_BiKzka26*7vi70}zrbG3 z3Kl$7YsBl4Z+;9%oyiq$%C`4c_&Nuf5Hdf zuWLu2J@yZ5U7d3!{OrFJl{BIOm29 zo9P+z+;}}^80hfb9&PSDqb>K7HnYcQbE|?52iEDbd8#m|hZ(VL8&kd&oqAZDC7boP z<`v7%d~m4?pUM4vdtVNVGqV|Gn8~4KX*7B}h0diD_-UZLldA?%b#6Z%i5Gd~j^x=z z!@V#?oA<0c&?rQmY3}L_o~X`VJ=Hm(wK|(>)!Dkb1D7t3(?R;r&}q^di89c^PWb{9#h%x_++|I8^_|N;f!4u!PkYptSRWuG0V*u z7^=zm>#987U4_O9%8Ut9=Aja0J|8A)iKQ}?MF)Q0wG~%;$g`c;j=!Ti@XXzgj5F6} zpGaLQPUyr%-Uj?9>y_>}d5%)DxUo$J(NmtGV-q>Wcr-g+5WoC@AS&K-=Z$&dyO8HK zL0DW7Rll$)`wPDB{*3?3zG8m;R~(eg+0584SP<|H23o>{KHZGVHCym;8x`@^v|@C7 zHTH>b%ZFv^oSoN!`9E?PXeoP+wV8|_mqw>$Np!0n$GMTixonSk;#0h+^0O-o9klu2 zaC0Vne~m%2?#mhGHu|)<1Jkzmuz%V;>4LqBsV;XA=ynhP@ewA}JxA}s@8II}1;wL( zqTj&3uvn+ScORPz16+y!eahjyli3VNpC%dqR4TTg#8KIC;wKPRm`4CV{M2qck1YATQaWGPh{sDhcpXb&Zd(HD#x?lyL$sIj)xrW^hc1yD zaLKm<7W+jv)T=`I^&L>Sy$8dD6>_lY5Z3I^;rJ1mY+;+iu;G#E7$sSYaQ1)IlVyR% zRNwedx>8T#_^~x`domAk*)x$}JqK1Mi&4>E_T^L7;9lT5g#BEN{r}}-zs3?g7G2tY z)gnA^SPY%ZOK`7fDUvVcVcOy4IOtk{?X@fM{qkxIbk3pj)igeOG@kAYhlu~kpWD)f z8PTmZ|BSo}zu6_I6^3r=kW7r$nU2$~=HZa!JC|KqgQbJYFzIInuI(?yhw*EW5HF18 zj65W_TMEO;0(RXd-8Rx&Qh0e0+Q~lGzWY*a{F(>(Ca)b>ai4c`<5V~%TgRy$%Dm(JS4u&;%CDo`fM0Rmmj_9 zk!sF|xBdwK=m0u*T8!4B#ir^^Lf12yIQeG*?ubuxpV=0K3_gg+R;Tg$-vw;FeI7B> zPl!MFfMlL`;@9LYNQtO`&JfW)r8DZd)+*S%DnQns6_`}J0<+|NwDCYD>mQD1PebYJ zcInRMXWKAs=OuiZSBUF_(y*iXIO*t4#)Mq))QXNgDq*waCTp;~pdK0GA32--423pL z$a>ibz3(>=+_?^JFHd0Q{KHV}CthSZhjv@I5%ELHq%W%kPjuI#p4ddq(}e|{NOMpk0)tQu7QcMUpW&(ZzwHw45q zqfeOv)%X0x`pmC*-1RkD7Cu77{96c`P>*E^r!j13EeR-N#(&zGAs!ha|$1&*NcOo?J&JosV5n8X@gBpc;$QO#8W(wTjq{?Yy zI?~xjk2fyrGbBKt+q}g$__HIk2CH+F@MdbWm6@)kz$2;Oaq7lf%((pw?vVcIYfSNG>2cz8Yz*jAxo=2;k2Jr&Qh_>5xJ_`hFT z3?C@@AS+?Hb+%?|xHUf*2>*G3DSJ#YWJhg1T1ClzQ}*a@>szzqStb4&tw4{AUkL8= z6;uAjQ8&qkUZZYc>4pV3nl=p5@QV?*5|JtA*%Fy6{Cj@{=PmBTNdJH0H*U?cYAtHN zH0HI>(lfWoj>j4unX;f8r#E-x3_Uvz&hN}?2Q9fG-ITMB7*MxXhj%0&w@}{4Z`R9R zen3lRIt=37S8drZv>cD4#~^E97>aX7AWU-IM%Pwhc=ldw>VFM-nIBQ;sKPO$I@15W z5krNS(%Hk24M$zrTg{#AJ9u!uhC9;_xJv(9ce)L9{Qq9M&#^X4Z7}BtS^tKr>oX=& zeunm%Ja8o>kts~9+dT?)^C$*%fHqh9cNprn;b*-?En|ty~TX#9fy3j|_NxB&AcuhDql@~2J z%EyeSOpV!;=0xAPE%MxlW0I{8G&V+Kqx*lD7+r*Q_6KnM(k&be`hlNqRO!1;hnrTI z(^B?X4@S8%v^(e`v->wQ{h0pRpYx0SIaAq>J!^fas^`UUGce_>I|pmJGRoVDHw*0P z<=vGrLBf`~uf@$>r$hHf4_xc(jRCHs(eLs?1b^Cy?FY^x%l;*jy;^Ycfe!q()rii2 zJ9FQr?o`>^laUraEdA-vJn6=LTpC2zIYG>{>q}$PKu$m4$6bP{|H#l?44-@`$fnKsdMh{7W?|#wDjP@blrUAAOzvEz-D(CqLo2J-8 zvfARglw8qy9WO3%mfYm^zU=!(w8^CY(wEYoGXp~CGrAvp%g=Iky<|d9iT1t9o5#JS zd-$&hNB`=<{{mb@m)efXP3A~HZ-v3vLxj^e0|{0oXk&jI9`ByujUb;4Eez{u??cXaABqkFKq*S zxJ{=Q+t_+icaboa7tO#Y>khb*Z4JYI(RkQSI>XX8%lcb~dmG*%XuJv!wbkODmS%K_ zk-ff658)I@r)H8ro7(o{OJV;i`$f>9E|P0MMAGAvu!udw`AR&2Uv5aI(m6=9EYU+H zbA4Yje1CiPVrivh-!Hxqwx}=gwlgxW`ytml1wAgwtSn|9hAnKsjDbJUeT6C~F4X0( zCl-8LM?}||?oC~$3kFjW#0(Gj?dCJa^ zHsW#h=r29tZ9M5AEZ83QeRyMiF#oO(1syl>pphg&1diQ661bMccay&Te^AtQ)Z;z0+T`)d*C?;>tMex)r{I07@$zOVegX zKXYEl7M7=l8@;2wsAMZ#(k=aHd{Z>u(g+?9{@dDrk&>^A;MTukymn4_$-;J>xTr77 zrUlTJzSL^xO&49#by836Z5n{dZ`APpc~>mfipA-MCCGHz37^Z?u>Rp!c-(DG=We?E z*kHkHcO4iJ?aq>TZ@L4Q@nZ4B#?K64}5D;7~{Pn8M7dYWs{;LvmQx}pm1iY2wPkJd(X{% zX(`=4XXJZ6qU6mMt;jKhd-6|E4C1TXq4aoH%)Ax#Y%&3(8PA$r5dzcR^E5I9@5t!f)FRIC1|te1o3Cpsyll zcWck6MTWc`*qIKKyVFj?lgmAObLtPdw}cJgs2R#$!^0Th97a6rPu1tLMlKE#hH)U* zi1s`Aun%7h>P6#@qLqt}=})pN&C9Oiw~iOuX4u0@bV+TqX|T!{_QvgM{8@h+^*?^% zhGY<@z13x{o+SrVIWVQljm<2C39RQYo+k0%ObX`EgCWeG8N#V%!OYSOqT|*;&iNrs zhIc+}Dv;h^V^7J@yD{gC3r7h@Lq$b}`AHMtKT@~_)&bc0d4lY(mSXnMD(r4~5zkB| z5BO8|M%Ep9ufdovUUlKOq0Ss2ewWROK0MkkfF;U(*{5F+sVF(*d3|_OxX*at%jK#* zOukP#4DZRBvpwiCMEE3qyRkxinl9mHoUwT&ENc2;(;0u1b%{ky{2a+km0-fpT2u{r zfHx6;@YACW&n4-xgO(*HwRNC*L=UcdBFyc*J}jv9dZ*bCHiNl^q?o;20$17~;AK#5jBHh?y7&v!|7n|pMv;B83<{9tzB;bb#w>Pb(AnUg8^CI&O|&~H5Gw*qGuj=9qXgI(6+oATRm~5!Y_B$zW1P;ss|r7_26_p7rMRa z#s>rKSuKC>kE#`iT{7kFuZA4HqZ8AA>G1f`A+%Q4rI(d-(uU2z%x*DAcoHkU+l4Wt=fF$mPW&i3^X5g}c%p>^Kiusq zYqJgWhg#6>r!m(I)@QF*+ALD&$jfKdIm@vvqxy~Et0H%E$2peQpDPnqd-7vJVEUHLpFw0I!w{v$SwlfI6x-qsvV3trqR zOvIxGEE%WEz1upnzM&nL3{vH*JQaD5wV*;DMb>|mKDZbkhQDe@6^&C+-M$DWRuk}K z_Zak&-lRuLixE*(ibI1Aqi5nxO!NPM5yma(G_@^_3$>^$d64H7hU8d7K2+4_jdUI9 zI#QRQXHQH=Vo!15`-@T*#8T*u#|ul*aC#wGIs!sz_klJVm{!{vWl@K8K= zV-pjw^~^*Vo|=J|UirXg$;gHsL&3`1*naE-+;SDUA-6U6oNmW?ciZ#)TXhbRF2|ma zThViy5?e(p&~V{T%#fMUf#J_lYW-09Fz?{Vlm?Xg-oVQqY5cZtDC7OyshQWFKYL!s z%q2xwF*OS+CKI4CJr()&bMZHH6?)vL!k?`tV5oZ=Ha*|qu+J}C3sIEbhvwA#t3a*8 zzmXpO1&^%W;FI)TuD*E}bqB8DXK@`K$o}74m`r+6$8oUw7>pid@M!vI)-(ok;Xo_y z)>fc$RSj|;EQNM%n)u)+LpwAZ%ev%ZXN$EsaA`X_9y*Q^@m;G}KSJAv7g+Y_B}&>q z!;oJ0(HL?AyLbsEo6f-2;uzj4A3~bpKIu2z4cAjUv7&J&tWWR4?#+@9F`U5sZc%)4 zT-Iyxzq&|w(5`7!2#Q{Si)W`IWkM?UNG5jGfhFjxTZ|8y+wpT#4H^!eMa8H~7`p!= zwtqf@l=!1KcjEwdZ{LH1PCM{Ja>vKjHXtux}NYZuQjF! z17QqpHbwBAvnR^}jQMBxKgocc#Ew-(Xq+$y&u-?Rai#3(V;5k0uYB~Exy_mAGRX{7 zVCv>l=>A(P=kitfl2d@U_f`n5Yz3O%SRuN60nSQaRM75KP<3642AQ>ZOSUwnYYq)J zr7_WV0x#q(}(TP2$H-iorE z9G=~uLBFcWv{)F&y&6$G9^%dIi)%Iw<4_Gybfk7H$Z!6B_=d&#g8+)(3oD0 z8M||M-z}5-&160qo4|v2hVk3cU|#TWXJii(p4+d)0rj^ev#=Xi&#%Mh2?aR!OmgY> zSK?fwaHp570mXw@AewJ_ZUIW)ti;*WwJ5F>)^m9!K26vv-m#s~(BF^zAvK7;dIAdX z&!LxSDd)nn*yh4iW`CK)TNmPJymKJ^Z~JqE=wp9nZW3H0oFuKAaI31u$`P9c+*5{` z4Q06gssfceE78@p0&_IVuv|Pf<)bSxUQ^!P%Xj0&g9Dg(_6Xi?J%#A^7f@Do74xHI z-C5O$kS57n8Aw+0N;wig_}d(%p~H%KeC$bqHVI9m9a=b9hvE8999$V6^8x4mm%?zjv>s z)9a&ThrWxaLYTzcvS~VK8W#klvNU%R_naRsn#5pciHBLw(1RxzTk?8zdyd=o2b~Lq zK{EUX(u^;|WpzEW45cS-YduVBE@ArodYSuN13KTqk3WwPvF9c7WKG@d{~cNz|KLL_ zMf%z)@n(ybY$5u>70nzz|1Xn^F(HEShxyIX{kWoYewh7X0&K(M&FsjP@Sm6D)m+Q_Ua5&)Zpp^ zn)H^gg0nC5xZ2r(dqqF`-;x|o|ChzT5MRUasmx*$2hUAl^wn5a*AJjeS|7HFk-o4$ zUHC6kxF|b2@|K!-&q~$!<%25Yx2v*$f+|}`7FhMO8h6cY$FX-i$a!0f>qqG_Tu-07 zWgmQ_$(U|k%^1>C-qE6$t=XAF|D=jO4P?1~ zko2XHT~Bo5JIO0QoFg+pPa}T(Z9tV}lKJgxz)Z>i4p1{>w>L&iUtuce0t*IcTXSnv zXSVFul~ERUeDTqqj_Vv*BiCJUD~H4WWpkHH7K>(QFyd}1|C22Lgrx~=?LU$Wz6|7; zt3kYTn*20T@(&AbWS(F~?~_J6IalTnR)$>t!hnAygZy%xA)ie)rcXOF@%~w`QM-`+4_(~_Xl|RO-WFDMm$Va9oR8utP=@Ls` zxNpr)S7lDPxhp-=?bt>Hr*x&VT=(6f99|R8%DsM>4EmnN*AtR?YuH4#9yXd`t%mT8 zM+gl@^x*fHawPV`8c2qv8&dv(LFjmrHQ&(MC zoAfy>#gOkyjaehwyV-ejzTRp{6M2p*igP%iYc~62Ph;NNsXQqj!-<0vnB6IsZsL)D zmMcv0b1q!E$AVjaXfc1E8ufiu_-L#WT?Q-BqD+a^o0WLti4w<~tMFZUD=vPi#y7pB z`9d$S-OOM&&t!O%9$f>_`_%b<%2Ucfs-k=PQIh?|OGbZs&_c2`SGlbU= z%tcn7?A5F@M@swY8TFPly7C)w&p%_w!;dgM`3bfgKO=qnXS{Lwg8tgy@nYp~eGnIT7iQ)Z5B6C&o(G+)c|W5}#MhWLXg@g(*< z&e&f;(#Zzsue%SA|2@X#kQY$9`WE{BpHbiW2QI!6&dHE$dW8zxTQiwyYZJJ(|8QOi z3g!GWp7hyo!3PcIn@O>Mi z)%L)ngYbXkoqznE{2W`(;mEd27}erBY=;WdvdaT}y^}+ojahsr@4#m(6RDy-n)7uB zGBw|qdADq-cTM`OhP=kJ##*d1FGKjTmH6OSh}f#tuvRL@L&+gF-CK|L;p^c1c@4T& zti#*m8*s9?5`~6aQKGj4^V;lz=Y(p^>sQ0<}hh%CJ#SL;Zd#eRB0Z= z_T~LJ>b47~EYPFhitqTBbR2d2i;>=L32Ohh0NKA6!SLd8^s`ze+D;K%6Gd13AU!78 z`ABZN0vXQ=F>!DaUQSvknW_@BQxQ(>f=Udk*n-RNw?pCHZu~DlhvgyDXiy-20{_wM z`M*exoY;$B_M7v4W((=TuEPrXJoU~Cu>ai*9Q{2DIkgL6aAY~ob2aR&)?)XDBA5gg zV)yFhNUT|g>bCh<^LYh=rWc~5UKrCktKt7obX&LeIC!uW@BUQaNw9QTPfDi)Cvs%{ zFe+sCk=(I8r$x2nkI{FbtyhV#QPOGSnu&0eEbLr28^0@;;NzoJaA_)m>ArH9C6r-x zMltL&SEJ^AAr!I-;ZjkEz6%SH5mAW4TMDtd*(%&|EkbyM{G3D9<9=f{2REkDBY!ME z_l#tLXaL{dcao0K4^Yk6g%0BPJh68wR$ZQgrP5*izjF)W{c5G;hs!a0!FISe>_Ox3 zJy_CbJB)QVVMn!`L4r2m$-?#UC|-xryX4%SQG}>TMNmsy14YSY7oRM~_iouVA2o&J zoJMeGi$GrfZ6m$IO58i;1forr@PZW?hdSyTzjvEjd)XEj=ja|SgZLiDaB$)wluz6T?Wem?c7F%jj2jEJh)n#kUyR9bOW;Z|o<70$ zfj`l8xDpkGxxZTGGiJgL9%HR4&wvVt?`zJ)deMise@72p$?aZhf^T&rCi~oh8g9be z<{EmxNRez-1RF}N=;!_xDyvJ-y)FfFRbu4tj785*Gf@0%1@3IELhPI~!u5WHEmwuL zFhTa-{X|z!QO@Y;oti@=DL1_ z!U$y!KHY%>A`Ce2sPt^y?!pO~_7ngmwRhOFdz>w23%mHd^A&=B%NhkR(tpKMX+cEkq!yz?wFyw?BJ*xh#+7Kc4K0psMvyq zg@Ixxwqkd;82@!Y&x`Z%_~M)`m)HL7wb#si$9ZBh<5Uf~==FX~y*dVUHeGNoAPIG9 zV=;R0e2l2yg3M2Ku=xBOw-+>}ds|KJX=2EIDpvGSllkzK8{Y|sbLgl4u^qkHW3;ER zJl$y3!i9@YJJ3?kmeK93cu1=~_w+Mnq^$w7Q{}xu*pZ2n2k=@j4_1G=VZUht=KfB_ z58=73oxK`mOAp~r&|NgE{)MC(6>dJE%XRH6m_6Ttn}m6|w%V5xXXTaaWh zoHfF`Y3jp!s-A4W)s>;&9ckobN0+r77?p3q(MQdAyDEr7YR_PD#ZW8?3&-j+ozQ04 zXgEHYFCAXnanAn&wr+b1t>4YW>!`(;UZyNkv*Vs$Zgdu&j(bKRbzX+B%Y-l{_YLFr zpCKIcB#7U$W$(P-pLdt|utc&eFXsriP?)SQ^zB)iZo?33JH8mU9JR|Lu&N{wI)9Sz z?(Zbngs+0(n?u+y`L?uWf3SRw8t19VXH;*+ijUGG_@58U^#kc>7see^BH2{3SH*_W z{O^wBPhSW#bww!GiGK2^jp*NJ{g~F@o4<8EIMmoxy270pKS7gSW@IAB*&W%hf{+J}Ip2&q?%G_fZpN#j_WXClgPX;}ndTYF8tJF~6&6d^`*GBL z5l0@1WuFJp+$%iba_ew*?JU}YT_C^S^QXZ^ADZm<8e_X^?D}eh2L{$y9v+Xu zT_#}Wid86;8LFG_Bec(y9_u36t*RUI=6f5y_HyU9AO0LFdXSQQwxK2QbY9+(b4@#Q zu}f!pAL_)qh(wMxl+NF+F-&_eTvN$8)^7@?*l<}n-=FyG!@dSOY`-@T+f6;Oy=f5E zst-i8m*fs=Het&h(L{@1Vc;JnP9Ccz9mnF?sI=qAo*rzS9Kgs+Ask&1N!OP#bd?T; z;quRA`OFo1Me}>xNP5f&;|96j^wDI!fj>*`db8DY5B7?1qru-luzJ%O2EmT#Srvxk z>oc(>PI4%NcH@F*b(+>+(Mn61@wUPu+F-#~`A)1p;LTH#1$;6$oEDO?-5(jxOGgt} zwk3hb#2=nAIF`BE!nYP(x_o*lb)$l)@-TqbqH&Cs&)DgjyD+bA;If7v=4jbrt4jcJF~ldw=0Gx@a~IPHX2Cg_g6TdPLqzR4`ioIe{S{hVVfSF92dM679HEeVVSLT z26mC$jPT6^%8)t!BsT9BshINpHao^0@ZhOHLH2c&Gx1wAUpgeP!z85k@2`>7bdZiwg1Pt!xj@`Fb2}ySk%vj~yERT*&&1k^$;+))>-8b|h^liVl_}J_LpR2WDueCj7NkvH|Hc+j`ORd#xkaHD>qu$S7IF|B+7TyI4{ z525(tkc?)+A+@uVJkhKpNRNMj&olqxkL)!2yBhHKZ7Zr>cjgj9A6D-qt>%St|BeV| z$#ZGru}JQVh~Q@PFe=OWSpAUff5e}w6McmH<-zO;t_)n_#7YAPHrEub^u+;WZ5V)t zvysxB(p~a3c_?hU7$1fTFRI;T{ImRkwZEJ5Yn3)%KQ|=?*|FZtoe`RT+#%1nX+uJ2 ze>#+_vP1d(T`=p9$aC2TbnfTJBjO{PzrmfS+PToA&Vi1vY?(dYn#+nsuj=p`MI(w) z=hO{FPdWo1)6unYI(+3_r^WUfEKrm#so#GvO;?S*#miydU?EuoM;=M@;Qmv-9Bc*} zjw7YQi@u`*sNm$scJsXXdASF(f@RKF<;ecOZ8>RE2Tr+e!9f$+^M_Lp_UWKO#nu&g zQJEoKJ>8HjzLjQ=$HUvZ1YRe0LV5cIT=RP)TxKOEENabE2P0O$vf_#|M~(?{r{xYW z4*coOLuuYTAldB$e(p@q6=p_~BL`fw<)>D{?@Bb6E_V|y@i(G&O9MI$PT|<6R@^wM z9`O@qqhQ4V_zdlhQ&E|?a&$UcbXbeqPYz;u#4Ti(e}yMovN1wi{9VR8zg(EhBkdXb z(uw-w^NR>~=Adv#&fG4u(@|?)A7RPURc6dNX~ft6^!WRb4j)=*)7(<}&CaE&`tFg%!e+)dS~e#av6YozcTP&%rsPwU5-J9yOFu<9E>_Y!^LHPG4qS~SSGaL zh)KFkI%vQ@5r!OmR-bo$bh!s@`L;P8f6&9ZQM!h* z8TqR_BlMj4_JbnVZ8?BjN2cSE>oELG8G@P9^YCT$EX=;P25uYop=RPGJl*~rW?8=x zTH2hNLsc0QroowAH5i$u##1YlSu(!`%lb-h&*LU^8S)EXL|5Ls_bqmwdx^1gpUXS@ zGmM!smQU{XV|m{ocC3>5fgc#OsJoUX2Wg2tN+;Aj^SOi zWAHRPj#*|Wk>Hlk`N5LW@ajd&eqo=p#J8w2!N>zv&tn zG_1v2#SO?2Z+zX(?J($4g+sB0Jd-XNt1H79ema>Sp2dsq>&?AC4EWRMm-O%+hwZ+V zSl~Au-wLOAnX8Ww&?}x|Tv; z{ya=5!-eXN=oz*RU8h&VL#Y~#^M&ylB%IABBk44M5c691WK3E(KS{5JmgJ^-RK3PO z&4WlRSc&A_vk_S@dbiyy9Dg$p1Kby5dYdKivMhn2^&(gvl&t@|#pwTd8G;j6Vcz6* zD3?8DTh}eH*(sbv%YCSHIRfpdT0B22JMX3A_^4$zx4lZ|>JI(b>SsqTmPq%FzE&)< zQek78M@am!U%c*X;A&ffM4LsD$0$KvvG5U1S77DrL{`-Phr>e^O*MEHMBY`Y|PZhFgo!HnzDZ$6;r^CvAN6{FiP~FboQ9mpYfSp zS#~0nIsKivTKt4xg|Gkg@dGrUCw+!#)wr~LF9PrH#ehfq5Yc8oM(OUy>7NH+eBlT# zzO2Q{h6^ZMEI-$dd#E4t1e5x`M!U72@UZDG*cUe8+Rugbzfiy_a(;~an9Tzv8FWt^ z%vLjdbHL#Q%GZ+Lm)o+}*tRt9)s!a(y}+&eH}L1|C0y{oh`uW?!m;5Z$|MK8$w78? zS~oC2ykc+5o?*}KcapdIhMp&W!*!PeziT&Rd~QqDm8;NJ^aVe=Lbf?Ej?-S|&|*gx z&)rDp{#PjsOY6Z)&7$df-_gq7MXl24ye+zkgX(4mX3m85< zkD3q0@ZP_XY_?o}=GXx&t?9{7nURfMBCv*BhcWC$?iJ$yH_>NLOvK36DQ9<1Zfz?iy^xpK{#+`()QYtAHL! z<7n$Nma!j3adD3k{H{8f4b`#>5AV!ff5N!ptryQ+wdI9uQ~up4`CfA!Hrc4nT0Lzh zJ<;M5Pi_92*_NaK34h(pfUZ4^Y42;s^U|+l@W7G_esrLLn)F^YwP&{Ibjs%ndGVU` z;l3QlE84l#2$4>0&rD8VGL$#<`t#xZ?v&Gn1qzUiqYIl#o>x=Y6)lDvN`I2@pC$Wi zTC2w`U-dY@jR7-*jaXr2$|<(yT-qSBj-55TIts&2&z?t{IMV&16L-mVyWNm8OLW)- z&OaiXPI?ecuKBYd#8|xHKdf1?hE;1vQfFK`UrrgooVmR?uvR?9AA=Y>+k+)>b{ynu!OwxF;^j7`lXUX+ z%QvF_QkgSn81q1ppUWQ_WB4a%B(-0paa%-*mXUbl%H#mtHR9`DX=e zu9VMFj=79a$fkb#Olm|9<(C2dI6R^oXV^vYjrb$yY;j=um-hVk+<+!wnfa z-2I;p{mXUedAXf*g6ebqA43jkWy*wOvJX|aq?z!f0$bWJu+dg}s_khl*Da|jWR>`^ zW=tAK<*s8n_TMPpa~(muDT7!%zc-`oJMwwCun+y*IIK}J#Nr`nXCi0FZ4G){QD^5$ zb$&Ug&hijpZ*J1$J{v8n^wiY;i0|s6*Vzbt!{37Soqw$5T zIFV1Sow;llm(7>GGkAH}V9t5pn@Z`4Jbs=mcqaZJ@giHw|8w+Q(J)qjMsLFpFbw;E zkg$(HJ|Zt&38P%_Xp3*n{r=bb8es6ijR_28QVfbx*b}xx~F82Z|Jc1uR=DB zEM$suK2?Io((hvyN1jRNP2UuHtn9|6S;EY^E<0p<>C4{Gl3|BGV%V+wNIO}N@P60P zcXvGu^=`uD&rO6LyAA6v_fb{!6p%iXt?R$Q+P@J`r3dHLAVpr4KZj3XD|-G_;mNOs z?DwsJ0qWzV1APp&ZfEeK>0sFl{>T0nacr#cVP#2sCS{2y*I(Wne_p`t8Ao9tGf1~< z2hm0K2p-8Uv&66l9d;gr=FM7k?|vQ?b=Tne`VPj2NTy0xIs`tv#o9JsaJ7?YBDzg@ z;cX$yCl-hnkjwIW!W_AsF8$>F8FaBTvm{^fXQvIn`DyTj${TFctHqLl9WWbH4xOzV z5GONZH|;IbQ@RzMBe%k;+cu=N5cba3y(r#t7#o#NqV>)*7^HIv-ws_z>fJk77QQR9>i})Gm z;PBrSY>@1JlfUEGac4FUjT^>$GG8sMOyn2gTD>bYp{Xz~9A4Mq%>H$7v=>H=>2%Ee zGy?_R^RQssVl2{FhS$TE!BVu!e{M^l{zr6AnLmTSu7SOD4fnR)1hwid(E7Inla=?t zU}81?-aLx(vO<=g%%lI-ELvI(=GpT-s4Bf?v##3EOIw-G)8xA&>>qX$U9@5%g8U~T zy+bh)g z0oRlrSly}$o`rkmtSHbHB?U%(FR&O3!(C?2(K9(uXwj zMIHj;rGw?eLd+~$BUy(nP@T04p7xc!%b>2p(Je!EJT|@Aw zcN%ia3y|wL8+mV5;i%+wW==kV4zsReUHS-bwmrmx!h2YD=O%JrUxSVBWn5~1 z9=bKB@%7#bVWA&GqIQk=gKJ<`T7&+H|6^aI@mhvtO1?QT)u!G>hf2`^&7?FD!0fTZ`c`JpcBZ7JHnJ(L1xQ&htOM#Umq5VD? z24mCEBY6^T1un+M!NQ7nIf2A`cTltBBbMA$;GF?2xmQ7z$-7ng*++#tPq(D4ZF6?d zRAlqPP3R<9*4|@2qxtE#crx=Per&Sjx zi=M|drE5fufpRat8PS|4%oJ%qrU_l&{>H)QKM}Mui=j{A_$<$W9}=%%$Jja2wUi3a zElHR#Vi*SRod&OWEAe6cUf7?%f-i>euvT{AEsZr;5w6QyO-y(}*PJaDNN!X*>nHs% z<8qlfJ{!f`kp(FA zzzE}h7=5(>19vRMOPg)DW+d9>t!HS~Re?WSsCk zqhq_F+rDAgmp2{iQR^_X#}UkYbrFvL)F8R}p)n3f&~ zIIsQrR?>^JlTS4>w=&;NFd{ynEG&nn?~c?=FA03zkB$I!wC8 zh$TOga3N|k)Gx2XwoQl8yWu_#&H9V!M(VsZS-N$iJMdUv7oI*T*`o=8JUKs%w?Byn zxip$nR!1{#m~`$INAQB^THVeEv&%<`{vGpYWTp?T13kIAw;KmZ_jA!%Z9X&36`p`6 zW<~`eWx_yM6wE|f>&+PLbOzn;zCwV7_=;C(QM1^TB^T}3Ji~*tMh0;I{SaQQie%FJ z80qR3&39ZJ7X-y}c9&>gmHb!K6gZcTVlsVhFX zJ7Rv1Fhp2pLbd5aY|<7zRD5aYu782T+g9{zsY|aS;T;L1TYJ4XHM$2r~! z8_$Z~DqWpF+@hF#Uix46gfS*Ye29wz={G5WhYftWX`v^txw$b*JgNn`4s5^Almq3t z{`D8o%Q^_y+Yc+&7fHrtJ@n6>z?i6~&>7y8bF^DC-N%?3du>EubYonFAI~-if;Q!zSo^~-TCXWGpkh{x#OqI^mj+& z^L$T?UmOC9+e6?!e>RSd+JdG>&tbvHx450$oV|`{^ZE=kp4o2CNPAC4v;s?FLs|Mc zk^%Df>*)~3t;fV)pc})Jha!1HHJpx`lBcgCEtLEj=IYHmlifMl&V@Bjj$EFr#E15M zvCG^Qb2>_&c4h{2%okv<@V9rbyMoaJKVzFuD_-lr}kCx{|zeHjEna z(M+<9qlsvOYii?oN+DMAd{Jyt6Hdd8A-v=yzS@5Nyq+mqnVcg{U0gU>Pq^9t#LsEl z2`zs+AvZY^+NVchxY}aOkFJs)l6sV=f5(Hb%Dnkimwmfg(nQmlC+&O~UKz*>(ox+@ zUwjJ>;~4xwva#LcIdOa}7yXQ)u0aG<-V1*`C6I~}{JE;Yo1fOWQ^PS8|D@z$eFPRy;j|TZr>k>mHsS;sd`M5@07|!XV%Q|;R<0XS4sBs zcv1{Q^y0bRD1rWY@yuQ({%m1PH?@f18oBP0q(DZF^kgjy9SW?D^7>k4(I2a){hg z8^%)aXs%a?W4c;Ahi1m{;gT2**N&pEUpNbOq&sz?tp!uXe@(-*RDHM`eDCE@S1ZNhY6!1trCp*>C2uSyx4z=8xicxHUk}a zM&8RO{+NWSk^VT?IT%|nrr_q;Vhl~$fXO#c;;PFtd^y{casRdE;&fwr)!VSmTsN*= z;Ya^}LEJ2y<&WE%7Q$0o0_oK&fZGrF@JS0#1{J%~Zm6g1ZOQsJ2wGlh?RMRt6>r8cPP35ubv?y)h`1cl z8^kmDQ09)YUv2pNhZ9#GbCX=T2b1OV%-Ze7u^U}z9OFczMfMEHwc#EUD}KDyo}>Sm zFmJb!?3NAqF;iHk``vl--WN*8*VD+xG`IFGGMjj7S_GhX2ZMI zygWmL>XMl?Ij_od2S&;H-HEqMI&kan57-@84qN*ItkF)vys3jRbHjMV>Ij=Pa}!=0 z9fRZKI~er!GqhhS@y+a#%RS&T9@ z*0tc^UD9PXRf#S?6@{shL!A}9*=oF;xqUR)&hI=P4O)m#o+I%$Hx;*9jl{*TQ*q+R zQnZ<~9Tk^qk#Xl9o`3v=_5Mw%yR0QI_^YyIlp2@hsdB0?iNYSVq~Sr?nT0B`YMKIV zj{bqiq@U1y@(pqKzM`h(H;k11k<-^ysw{|PaH$11_Wz0aE4$ITwg`Kt4@2I%p*XN7 z4-3c7#EDm>*qpo@<$kB}eBT`?wtIt(cfVtU_g`cd|3mPcKiJ;y7v9OU({l1>s1JLO zgWq4_$`gU7g5`WWW^PvKoSjfE z!{8DkJ=4P{!c}3Oc%0XwqH85O&pe6u8rSe&^nLW~{TN*?KSn{tLwK}!fRwFwu_^jC z9^~9W^vG+Nr*uU+iZ5ZvrpuUk{wiW#)Z{m6Aty1f%U+8suxp|xnIdkxTNJl<-x7~kq7 zKAN7u#QVoEa#9Uk#cTb0SqtJwUJ>2w#X{Ns!>(yjFtk?##nY&PYX)k_S$~%bc4^O5E_tG_&c9SyrW=4wWL7mvM zFn}v0mlpO+QFhT6(6f0NZf>86>iS9ODLukw;l*-q&BYqa#qvJ06lImmU}C!r#Uqws zYv~GnwO);?#&rmvSdN(Yn^Ey*8_FfKQX$%VbWIH`$JC*0z2tw-IW{p6BH+3$bLX3E6j z>@+il(>ix$&TC;QU9)3hUV@KV}XKMSg zJlt_4Zx0$GxwhVnKPD`XMSgTk7Vh;IWe(mbUJUo+SnaY6$3B-~XRCFnO)o={Fc)=_ zrEhQaMr4NwFK@*T7@Xe=$Js|PA6n7{Cg@9vm|$lBdN}dOg^f>s(uK+q-Ajy!$L3d_RcU|LsM^(LES- zYcJ+~-6uU%2f*2fP}}D?tR2syp+fjOb$1Z!B3-}kuduB5M;!h31C!I5Fx+3fk+N^! z{;q)5TI2Xbc1qqmGCBLj5MDXjm#a5*;*==qc4%#9Ozd*Z!e+_Ff_H+$Wrh_=S(M=kNAbk#`5TV9gKl6!lkU zwCD@Nf@BxmDxdioV>zRE6q9v_Q?VkI$By^n!DewhkmJt--)(q%W;>RCYr&hcU%4In z4o%}<;;zvP#P@lD)G;q{I_eEl*MCH7>z^1~*M!SNV<{Qll4`k2x(DK8 zy3>x&MNg^DlTO@Y`5b*Ymp%uxIjCDEt%HZst#v=n8Pb)ZJ`qfB<;4k|tvJ7DJ1#Yl zjGkvRez~W>o|~I6YD5zT3xC{H_IGWpm6$lQ1^A~-2#4R3DYVnKnTjs_H7fd2?RA zqRSU!TeEJ88WS$7@CJna9HzqVnyMW7Q;iotYjShGw!Ftk?oUm6k}E_zY-!4%r1lK1 zvS86SnXg1I8-JvbNt+9}dztJ=SBw?U-)Odp$)MVQLxcsD%!xT&S-B*Vp@)4LHQs?m zE85G9VL(Umk5yD?v&%It9$h2+XG?7s>=yl3p2x|P45(Ud%nNhOSp319S}Il?Cce-2 zwzh1vlU_5quA%6IC9j1`Y?RN2j9m7;lFd)eM{;?!v}vWh?np#I=na@q>DZab?ZLMgs!kT?{3VUUoJtMpv z*`d*iORN6B?mXjOA=~LoU;3DQUi^{Eg|o)cuyG_;z7cPMN-C$BC9zLz0(XW5b9kNy zr?0W)fZOdk-@}+DuMIfkyFTYg-umJ<@etS_w!Xr1qyF&ByC9F-wOXBqN`^vMue~>2ObBI~I$6>HABV;u^w2Jl>uY6)lCiExXaz z*1YoAmJcu3Gg_{@{8S;Q9xvdBL*p1(J(hW;!dKPL;HJ}q`8~KVdvxq9{?0JYd?|DJ zzz)P-1L}VfcKKxu-mq4u#c?%u5U#rAMs;56tx0=BEglvv`uh<*?kO|mc`Xxm4l-l+ zLFRlW9Xk>GI&i(*zr%M(9(Gs(J1dW){j40WI-f<05yC$CHGuE`NpDqDEU(Es`-u+1 z8p<)?rqLQ4(!3>8Whb*bS&?H?6{&ttk=8QX+BP*rRJ6@lq z$9I1W*gDUc9u20<6mCHAEO`ch%$IkYJa&7P&8xAQTz6uKuzbdCt4A^S=P>?!Ie-Q)dN4XIip~Q)ne^S1 zX}?t1+wvPyXExx4Sv~Y@uVTdItCE$xj${Abz^(qGv21>TZOW37ig*uS^Y5VQUoS^;WWd`{eyNaHaH7t%82QYRTNLMSRExJhI(_!E<)u z{y6D~xpoMj435L&{b_U-KF9h^^+-Q_4_mJka!)7mZWfKWY+ zw1h?Xc?jrMh&Q!kVBdQ@${LH1a%Dc=NuSKOy=!ngstgX+W$-Uuhizla zaJ6g$%#}A`%A3td8L?lZt@?$X@^t6 z!G=G>gt0sMG}g-8T6ezyvqLlR`g%5kS4>9s;<*^FuoCu<%CT?$b~M=S7S`QvD7uQL z;EniIr3*K0-)>wMU3e}kF>i@kI(iTnULVDV!@}h^%jWq<1Nq%a7&48X zl4sIp<*WNRQ7xKP?}f;2P_noJy-)1Jv%bgRlY9!#El%Ud zF_~NQ#p~O60w(iLKutPDp4cD52E7`*lli;Flw&abbpkcNPr*;QfM@1Qe|GP_Y`0Z< zH)mTge`X_6hgYGcNipVyrepH4K}a_kBc6?7;rK0um+}^v)*iw)oX5&DH_@i!1Kj`J zfP?EFz;@j|1pK^%LlL*p_~QmPw7-s5Mbf#Xa2fTFE+OvAWlXEOihX~^@!S1jtU4iF zLtRgLsA{lf`UPmdUkJUynb@l-`-acMgo`ix%Ji*f!kDxyP5pF6A_-=F_bt?z) z+=?i6v@qvy;g79cAs*0g`Ouu$4^f$^IF~#QF$Ht*C8rDuW2&KeR<61E1&$;)VsNbz zd)u|5&tqkdD^cdVuClwm+JaRU%?TYPo?WlN5mA3Jztu0K>3xU!=&v}t_Y1y=C+GO~ zzEX51Gn=CtrwFIp{rGfreLDaSll$PrqzvRQEkfH3EAd0#uMQ|(gl*Fo`0?;BrnV8U zO_C-@{nqB`-#RpWsl%=^*HyIDmb_pa+Ba%2Rh|h$a+MkKvIP$dmw&r>m!@u(441Xc z%YD1@!08U$x%M>%UMNLaLI&Qh>xG?asW_!M0Z}=NaJOnJwCzvhuk&McZTSb!#6wxK zP8c%x^tnLv-p@PDxbln{n{6`X8YL5!w=-hSDSfVKrpufbZE5Es9R&3nJgzC{{Bc$O zljneTWF&vJYQvcIM-Uu48NDt46OBFz{b!8CZ0Y;%VO%EM^}{&pAUtf1?@%jf$!9XV z2Adc$$=HGec3X4a3t{OR+3}RHUdsnt(vH*S9eU}u z5|{zKhlHEtKL zq@l9~b4%Ir*qNc>uOdi6u(fH?eo&>IhaN{_1jQoQO}g%KOB;!@Bj_|&#wpjBJG zx^K$LL$nU04J(vxm?9PLuYyH(VLB@OCaA{`L% z?}Y6a=`5KeFJ=!3V4Xz}*M1j%sA`CKhlBXOS)ed@WCuM!_TDqS*-3V_I)<+N-NH#` zM0<{EwBf}WfzlaLi@)nrVevKuhtLV5gjd(&<^t@j5q3sX;agR_$LeM+Xz){;H9gH3 zsbJ54kKMVxi9eIN3VZjK^bbx6XUF_-I=h8&lYa8hfxayv2{RCcq>iB>ZdEvwR^Sv{thBc8=){kmHI0Btov*! z%t&V*k13Hn@BdS z^0hNdv>e%PVI+s)E+(erVBES$T=q+Z+x!eDm(9V(trgfh>(AQo`Bhh@H238=(SS#s3g%zYl1APO;aitr4!uKaxd%`e()0VslNv5=R3GBRiNgA< z*(~p(!cE#5;?HM`4nssqyo(IQ;exKnpO=Rth9#Iaxe{e9uVHDG@Nqu1ls-ot_ITNz zH+nj7y`m@ep7^uvmmsE_h4D$RaGD;MKGeG*+#x)+p##C*E`GeX(2Fxyi{8A*i3&Bs zq3UEUK3q$dH#*a(=SF-%M?^gjM#}eINH3j;ut6(Pak?5WYwzHEW+QHhel*cYpG|}l z{8&68)>+=tb4%frhH&7saK^ol;6}>`TB%AdGd!4n_T<~ceq8q4iw%3+#CPPx8-46J zxvB$mrQ>1IUdfzEFN5{p2yA^3j8n_|VnP3@_-(fq%Uo-)Kl>rVrZr({s0P1IGvstL zYx*dc6bl>PI9#%)k-WVjioI7v(j+jP#hxL&S5JnT`m@V#@w4Q+30uICt>@Uv zZby27OwD;*!I*Ea&A^MYKpcq-!Gd!sNZnM7il7ZpIeijKeV@VpZc|ofwdP;(HIIKO z*InX9n+1NHbX!;l14JLxiQ=mID0&@@WO$Eo`Obte^%mJx*Pne~d9l8~8@o#0rEst< zE9H4{?5n(+4lw3`12gfWO%SAK9Rp7F$By?!NbkEI*5^(LXYZ-(;+j(Fk1(g?K3ZI4 z!>@sE91`Nk=($0>ZzTENE28l)jAFa-k=$tzE5|``hF9l9eM~6cS_R`%U06xp++~ z;j8uPz<2Q$%!+a6ZrRmu?bH(^6hbhqc{ezp9EXZ~B}kB5RHtj#;FIwMs-dl@)1)1h zjm>!>&w-QGJz0IlpWBZH@$AV^DqjhspIR9IH3?x-r$Ek;_ec*-UwRMrWZ@`R2IV?( z-YHwz-AEUh=xF0w1hToyd2Ajz0?j%_pwOfv`Z$fm+(mQIef<^`b~%f+`(L4#%tcF< zwqcsCiDWTs`1hzQO)mIw(>Kso{3DtDBs(tW)_|9R6uTv7mkCRIzPAMBJXkx(g;OdV znCEND#ep68<)E;}-iRJQ>OETDEkc2IM-0=9M@;S@7%wlv*^_Hg;CmEP^&Vj0ia$`W z6pc+?pY;Kjd^*RGx?eqbKE;no7eNR2Kt4W2e%A-5UiPEBfwEOUPv%*=@ucvc^UTF> z@>zN~AboGbdGr0)jrTlMgqObt!@H*7-}!i)RP6)TiQ^IBwFD-gc4O0>%joMM8Q`+! zT=-Rs9afp}$v_)giwES)drw|3@#O`3@tVE$n?@ou!2nWx~hJ=biu z;n984HxX&UJ6(j$R*=kK0~400A4jX_<1y@UR~%Z?9c?~k;?)A_iELJm78%0Z(S3;X z34gKwoGS5FmntsOPcXzz@}Dl8n(9Gy8!y^wdr6;)2U8!oapN`@zOZ* zrWKu&%<1NAMz3kY5bQUY$!lGBweM@B%RAW9IRo)uzi!yPFcqtcC&5W~8ODjHb@Y`> z7+>}lZC1YD`D*V-DAJTeE#Ej;>@%n6URBay(^SiU)mAf8a zejWj8?qI&vcf>@uq;Y>O$(0+DcH-HY(t#J|+0dj~ybgPvgc*2m zg%Njr)TcqMF5~^%Nxnqp`>9z>lWgd%fs&u;-hksxmf_0p;c$BTA51n4#z0l^StpgC zTl{wX?OZ2%=wlqx`GbZFt>|E+#h=p8GGVcFG47WC{;er>WzXDtovrfeTtQdXy%&y%6o#c?|VkdND!GjmhnsGexZ$*1AOquS>wt_@e6zqrp|>RC3>xC$;Zo-d8b~P?w-mt3vb1tIxV?! zN^|xsRifRVrt}I_V4XZqN6u@)G6w~oyPij{dqX&IeS+|t9l5z_OE&3q5_ZStAxLi| zHl7@a8hKxdG@J@&5Aksx+>BGLkHB=w6^!ct7zcblAPGOPd|xAuTKvY2M0t0%`-R18 ze;_gC8-~pPjFay^pmU%1DC_?o<*z@WPqAb@ggN;$bOd`g@5R6u;1^+6ZeRKy(}f#W z*dV=DM~9Cz*!9K+=t^En`$r(4cJP2nbVp4`WW{SRfw@kqF^53x{s zlUH|oh?jm1|3}h!$MwAb?_Y_AqP_RtB?;lWjEv0aRQ8CnLuoJBBQq;xXOD;wl29UL zOJvK)9-+eT@%jG#a69LAZs(j=@7}N1>-l^>uIsvA=yp_CbxGB*l0OFW{u$){9QRm* zGUpU|e?~EWP$&bOrTatKlVw#JOg?@Wv7#@&nKuW|N~Vh^S^Uu7(@?u^IS#62AuIX- zRIE;6eC7r8u`a=?{-yAcIayu&RpdrpLtp zGbTztbMMi^!pA)Z)Ah$Ob@mBN)i^1L{L?rdbp{_y&mlCm05``K!eZ$~VE|l#e*1Ei z)ZIZ9Q+dfWp4&_!7#cU45!?E4d{1viNLFrha4p_WJdRUSR-sfa4epN;kTD?*Iev?! zXL&W0)^9}ks%^M+eJ74}$VSv&$$xj+2g?%&k=#B9Yr5y+t7aZN$DD>@%{g>>dJzHY z*U;nPEsS22%HQ9GW$G14rJ!)p_Cti{A$M7MOrL`> zDYEbTkPhvE83<9^fNy&@;(_gE6n5T%nCk8Dley*cu?LV{bOdS7b8$856ee32AkFDA zZu(uvlE-&ZlP7t$;3TdJkEYVKDZKdKXnxZh$lnUS{BhHQEjBCioh&#fX6{9oeXFrj z`ea6LUW|2ZmLa=x6~^uoE?~oY#1>`Z!v^UL6OQ&-n?1PL@esbfItGSi2ky9LcJZ%6E?J%}EC7_SbTKTv!PA(U|cBL(N~~ zp4&&v4*h|BrhlNiRguSpbC4ZY&WlRolOb^ozY)nb=~EchJd`yDgQ#lIlTU}V z;qVM|nhX`6*8Xo$I#!KQhPUuM=Q_;oucKGra?Ia$167aiKtJsv2K{;g2Zaw%O8J5H zjg1I;rog76pGM`X@b`b}?DnJ;hlswA6C!^Ck~yd$o(CpHQ*+#O?ueg29nX<8u?nD# zudu`{gr%{{gvzZ&_YC+ebNo-(Zuk~szSqFf@ip|Gyg|m|_b}f46$97RW71XydK_;? zn{KMyKU$qrgzvaERGW(y>+w;hA>&0)`LaYy*g9Q(|d1GLj2UOyNM4ar{^p z!cn98aFh7fS_xA%XoYlpi(Y+rd~@zIQe?OO(#=-<5BGzcu%$FIsv{3u-J z)2(R#Pn+Ma>QP~gA^ZL@W?qgNwFg^rujogs4yW?|j1&%WOX9{aaeT3L76*Qa;5Yfq zNgqbC*mN-e_3)GPXIq|7ao}W66OO6T<}%fmEIinPcgoagUZuvK4_Yw2L4(6hwWLQ? zX1by^M>(4C-)=Lyy|Q3a8_{p0C0Cj++%VD0a*I>t<(R^MhrI5EtK3vL-xXmxby zXvs%jqKOvTNmki`u3O}Kqor%Z+l?wJ9=s~oZU00*uf6aRHYH2z;m|%2#bth2JPT>{Pj~e;ohm&!Wkq z2=416909vgbh8WO&j&r(BgU6E(mZ(E-hodKNtR)a^w+2fZ+ey~U2;rWS!+gH@ePcT z%7S`x|e}VT#6_HQbn1$4$6dRkY|Y zqIo~C;l?5MR2bz*D-&n-yy3!c8{POO%;W#K(@UyS882F0<-}wjl1|TKlVdrrD3VuK zPNhTqc;4TaPd?C+om(!^nxKg+yAxVtTis$CcSscDT zg0t6!)5&gx_yz{DrFdYq65Y8h%$xyn(*N;7g(m%#cp*Vi&Pj^QJgdlUGS?rM(wqg4 zRQYAU2B!#*Q+=u~9b3v?J=Tb|?@U-7E4-gy((fbht7#iknQxiGpqYuBw=$MpCCmPz z^%PbYjAh6S>2PY^lbYh6*n7>A7s6Vx$zVnLN>6;w$PXCYs}>U<)?(|e57;>53mz-} zgwe{sm=n>IBZew7G+#~jsv7d&3AetZ4)s0s8K7-MEAh#!otDa=g~{R*N|0wYS~7&w zS$8~~AtQ%#`VQ&Y3-YDAaNqe={FS#G(5K4_6jzMc}nhhzqYnXwo0}B`NZUOLMOJCLPPSQ#fgNB6ZgaJ81oM z7COoLn=q7KdwR3lvJJJ@8gX*cUkpsUh43pUFg|V%bQE`D#EqTUwQM(c=Dp8VQtg1p;#xo@Td4s{9KjXo}Ul{vNzTcn}>UK`x?)OoA z(0>Xy9v#iIy5dhs?#O?qtz?Z>V!y6;G4I+D9GkQWZA&xIDLX^@eAnUbj*ZxQC<~Sy zwqe+To$zS74_hjapixoIQrpj=k1%$^qf4PwQ-S@y_fd`~I5Ye;Y#*jFc~3Gk$H(#d z{29^<6~+$pL)f#YACJO?Q^v@1^64?cmgHcL-&%YsSd4wni;$eV6pkNO!e`rBgx=gB z9M(J`A6^x$vKU3Qm#dFtO7`4AHxa5rsFAxaR?e7NyenbP^~3 zjb@~$46^@@q6z!+*T%NO|1jY8>mSi^`AMjRt-)ZGc}NeOjmDz6Pzzgv*o0O1>>-TI ze9@eb3wus{ecwm#5VqnT%zb?TmUcOadzp)G_fI1E#91U27U9RKQf$?@iRSt@Qe+(`Xdc?qhe$$mk9l_^D*Jda%>e~;J2$= zgtaHSuTeG@=I_S((fjbG;UG-6=V0y5T%57X18Yv<^v<)mvGW3&2~%dBRXIM4koV&S z@we4P(Ei+Lo|!G{taKYr{@s$Zg_XW;)pqO~I2Zp-pMm^tk;0#ujg56nv0wJ0`ZJ|CwrT!%vavPGpjooGQ1Mc zBT~30eikGA$8ok&Ki*AsW2v$VLzIf~#c?&7HjPB}<}jRZoDOC2scwkNkh8}wJPXK! zpLl=wEWHK&k}6c(cnov%r(syy;Rg|Sg8JbY4_gO)XC*%u{Nc{SyZ3ksb0 zPJD;CO{n`(`bSPAa8ucMmfY^nO?JktobwP7XEtE!=?M5f9fAL@O+fwaG-T?Lcsc%wECI7{b==mSp98hj&u zL$i0P3~*Fs%|jK22FFoRvZ?F!y{R=?iAQsDVKyfX7HdahP}(R=n;wntHR*7S*@>Eh zvv6{_k1jtx<0h0C=`FmrcY1s!d)2n}W*l8^E__qjYb-FM&OlRo95&|Rb%wO=tj}Fa zx{_(s=F^s1>^Vr23rwP^{Av*Q&$Qx@TaR#j>KeqwgyT*BA&AeI0PPm@Fz~AA9#*-~ zi!R41gLk;%Bx`P^2G^}JpofYDM;Y02+%0=HW{7qzToM1K_O#w=%l-GQsj=CTReIvN z+iJq}blHap<7VMYeTG-eVB-%zdMuT*#KIG}-60tR-iN?v{&39w7!6(P6)39OgU|^V zk#6=B6+If!&7lPo=IV1rr3Jr!x95;cE}S1I*}Na_RF=Og=($NhhzrMNI?=AP16QxI z<(W5DR9S1m{Ih1##b?TFOyTL|whV0X5gW&^L#k#dzRV3o*{1QB{$Vag4BCX_k$DKo zybX1=FBq{xm~~xsIKa`IAYf)E38_w(->o(x@60Kj-s2zx1j{xNST4((~ip5gt4avuF25gUU!W_M44{kPR@NejH`T zZ{bDo7YrVv%uj{7%$_aT&T%q_9qdJ4zxGse@5GW;poKZ8Cb{RW${l&)VSA1ph%N~fBvA&<1-E=IJ-;QL`DbKDTJ3Jn zEsJF)Z|KE4Yuu^Q$Aw;Dj$BY<#~(*}vF({E9GVo1CJTDwMRWkV3J>vo%?e!peE>cM zS8=raJJgO>qHmfuk84>__mva(y^-%!-&b+`f|^msEA)pAKS6euQNA&KMzBXv1s4Z4}ZOaaL+6eZ|<(c`L@DT z&AW}Vux|*sF0lz`v zSjnGl*&_PU{x;0s>cOE`UFbi|k=q8_QKjur+O-ouc19MO8;ruyfB{e)H5RVf(gU!3 zGrYTE__(1nm3s@XxMN3tS}Ix4 zm2H_K_qpCYclytCrhkY%`!2EOySCEZ^q(pFKMD}8em&}P(y?kpAX+aN2+!V=(6n5< zNaE*up;Cld=Fi}N>L1+QTk=42>8Fw0mR*PkFJ<{~R0UWiy6Wd6B z)S`nAw@R0|-DGzT`6~Q!=?1dsDg8EU%%p$Mm?tEoywc-^bX28buTy`_sUCzK;>n?O zY1kd#k6btDS6K80NhwX)XM+}}*qPJ*qa$k`wB~xbXY+(7op8Mi-_H}abh|FhUDR1L z9beWg_u-{FWj8KW7uL9~ zR3NPNW%My0s!sRfA?aD_pXtcgNj7wr^MA(}W1c;(&%a|q+o%#~znz3TZ+fBfaS+l! z#KURhTBI&GhNOaU(NV;xe=|8=y9G?CkCu9f>HBu3~%g--2Z}+ zD4pbn2Xi#Z{btwSLt|EVH-nzCU&ysc z${ZimE0^ah&xh|jdC^Ne+6wg!Y>}*=g*UxEwxTYK(XMU7ky5=nJFAR$q;XQ9Ua$VwL2$&ii9t zVF03IW}qTug>-5hz_QX(Y-=lhY?V!AU!lcOa(->o)`{D-ylCB|122D){_&zNygRZh zFMk)_vS(+ukTs&#%7^ZjUNkx8#sR+^St7b%NxcPEwKt_{y&*^I_hs>@_XrG>F5344 zQ0_DUUOnUF8#>ajx;^z0+A^-VHA93;HodR(&;PWgQgbT?$-E~o-Sc22Q*tUJ<*q!djDO$*hg^ww4Z~QnGMwmmI*` zTa#cPIY2zh!|5RTl2+= zD@b282g}rhus}Hon(HHwe`YCctg>-Xw3>qxpCImTBj&lNGh&<}o1CEHF@3qA1wOmJhDq0X$A-m%4ZZFsktyeBW2bH9?A_;w~TJjIK{ zpVy*LVLg(kjYD2mAnfOjgY}a+m?m@lnK8#P)aNFugFoW8K{Hwkr)+(+DgA8i*hglD z*9#~76JlRRm)&7IM!U0LDn%qP(f3{MaS+y*Niva#TfuEJH5nb*fynaO?X z!W&and1gc|6zh_ZGJFUi92>3G(bytqgmDeKG0CC`*CHNcPy2d&`=&}?OFb5MlpL0( zbnPy2V)YppYBqP}KvNfH>~UiMS_c|Gu;b!T@$((Eq?xrjJE)lQ;6h{Wl6O|DFdyfq z^=Ii@Q%>q~A9`7<@yI6}jl#niGFph|%Yb9k_38RgkE@o) z(9mEA)1Jsav{@6r2tIT(?n=|{35?}nUyR4!q4+bhR zE3_HyuOu-^@;)uc1=F)cdMLU}uH$bNjxXMV@#7K^86AdV=~lU-FdJ=`t-$mz+wp1X z32Zuh1y`RwfaU7fxX|@fa`+Zw;n zC8i!vYyP00m?fV;(7sDJUnTwVJs)7b>tn31eu{`|&vAUgODrpSg)JU6aQN^_7<#YZb+`tK zMsIQc+k2DJ^2B+4E zc=>Y?CQ8@T`18B)Y0ptqD4#`Z*Nf1MD?xVOt1$RdhG{t!Xx{1;ChoWm;~k4qDqCM~&l2q%F;d`hWsFzIp*KmR-WZTgAY{ zQcShIjvHaOVAxQJgF#QBCLB2f3z>)Zo5A9ic0a`4rar;aj)pL(g7`OmQ68vI2M|ZE~147 z9={;G>0+EWyN==IcaY>zjq}@6`T0q+x?wr zo4?m!vEzK1d8I);e=f$aUJMWIl_=c47KtY$>t44RvB$TeEIS)Hp$CwreiZYKh0k|J z*0cTR5q01))LLFgvvqf&uz2ULWCiXm&cIfO_1KZN5uHo7;PbLwXz&;9U+owUdY!`j z$LFvp{~}t?y^4iPZXs~ZLv(3)iGkIr9CBK6xD|06nm?1yQz!CE+Yvlx-;eLaM?BWs zoG)?}X}$9{VjdiU{l-lQ_07ODoz;j5UxS%v*W;7YCTv@^1?ML0gz@iv_fMFV@F~HWLYZMTxeKiU)#w*hBaDkrSXnE~%TX!JPD^0eA_n9Wpr zu)}E&Zu+ZFlM4;dID8Y<(@x-Yqi~NmA%5haV(JB zc`=^fF30Dl_fY@#5rn3TkrUrzU&kNNH2w>(@2RxCpTe0p6S=}QmaD$cpi#+0_Gmha z7yb_9^RCjP(BR0-%UYcA>nFSi-ov;47qF-KNenC$joITk?vKuc#;ena?R_3It`_6T znF_enR-)VJXY!7FgAPYO!Ske?uR{JJSoSvednj}8?^Nz`|KIF8iNjjO^WN7e_Gmke zoBkWm9_OX6^q)UZi~oJ+dK>=u*^)z+)ZuWptO}DJ-SQh<$e`D=&sJ2H(LE4hk2an3v1UWQ-5Isd-scB zz_A&u8aRn(_m1YlfIxbf`SD(|mvl{7@WKCht@9K(?xW1$>fazb{T1TJzCxR6uVtO8 zmA!#*8+O-2@q_|ToomLXuByCgq0UPmTXN=OEq4AXIawt`{t-RpR;F~*x}@;soeM$3LV~ zfeTwHvG$U(WD;9&&Fogp+@Z~aE_(d8&ww%Z#&j)_J>6L0vWb3Fa!R@_H>7Y$S`xKq z$Fpr*G>=%%V1v~p{xumRUAF&GQ@qp*?zgAz5NGzjYsPny+4-2;l1o>#;34T_*ygRq zSp!Z*s+efoPEr?!v{){M-tU_jMMu}g1<1}l=tTks!vMIy1 zNMA*{kvyO5 z0ov|Lq-Rzf4bx}Q+OUIn>BaA_efc5Kga4fDiH8mU1N?9 zmVCc3R;naNK4zvF^@mt;sH^agmk8HL^v@^19cdxH>-Ife*J|f0D}c4x&GA zNoGm6L{3$VqxX&|8kbGuVq58KoiKt%>jU_sTX!}+B)sd{PMn}6-oeY#H|U_x&r@`H z^_ws!m+CSJ`dsuzzUvmIjL@{;wslr)lWD`(A$DA4;y`ctJ-z$v%*AruXSY+SQlG+q zzR9dypFjgCms_tBNq_OJlr)60h0zdNeeT1`)}1(Zv>QK#S@Lp@K3^};?p;~tp_%iDr(HCA-hv*q(%_B562He5*M&)gJ# z&rYJ|#(4Toi{|fZ(^=X>*zprb(OBlwvmbP$zNeRDX|4G6pAMD8H~1h@i4KJdG-#qg z%LE0cC^qG$JHX|Qdc^zIhvaCcwvX>K*-Cz(N=TWdx`8%qYtd&7Q@_$UHW z=yoqr&K_~h4~U}b;i)vK7{_CtL%2T8pUuRV|EbZIOaExI_=sdEcGsb@^gVuRy~U@x zH`t(Ai=l3x@IC$qE+_s$zWfZi-<$JmKQ-=m)ZpamR@`+=n};O_cKWbz2SopEJ3Ezg zOQfazd;&*Jj$wo7gqo`-ve##MZdC&~V>LKS%bD&Ox=ctE7D9*Dc&Kn6ZLU-xM6|C< zGb`|H;VrmE-p9+h$5?0l8Uuyn)FJv8ax(wn>{ms)D$72)zbehIsPnV0CMVU4zAxPA zKdr=5Fesjp8ndYOb}Fs>#z{sim=SAxFe}fKUB4Q#VvPdpwpQcYipyBF_%yz4ki2P2 zVbG?FrwnJ2G5><>=dNJE@|&2yqYC;;&4jx>P=u zd#cV8Es`r>=osamxnK)Xs1mUF%aJPOD zUgj@Dbm3|oY+MI}0~_(-^A^-+iSI4s047@;Lk3R4uizX8_P&T#TS{@p@FtSxK7dA- z=V(_U>w$YRJJ`qaW#2S|j7#F7@ zXwDM6Pt8E2=O)K4?mz~4xkPA3}|1!4hy@tAV zl2O&F!qwDNE<7mSm?krMK5!gI{s`cpUhR1^&XD<5@1SmR81(}dquYl_^f?oOaS;i~ z^je6)`Kz(!RTd0-??ICBVQkub47P@O7;y76dVfBLu;L3iUUUhH>Lu`)T8jL0WvCV= zOGf+q7;&%~A0&7D{!$z@rAOq~so@MW?ZH4-2QK`pNQFy<_|Rn)swU3DjIxQ?{a_~U z`On3(tt+wi$rcY2L!P=69g&c^5~6?_tNN zO3XU*5c~T*#hF#FFr*}f4t_D5rZa)47)!^kIhu=qz6@-{q1tN!9azVrfv@?OF9U=5BJ)WBHI zpq1laLwUkmDDQfYfxD%v*Di&LicvgaF8ip}J*m~&j#X@seSNMl5*Hx56xv^m}d2o zd-;WszQ6FL{ug%E)MMlBKRDMhS=cPX>X{Y7jf2|rhPRfm+ixJWc8z31BCy4DG@4$R zgw?Opq&GPOzPu)3CR}BtV*I|&CtO@6o*lTHXP6$)x&Pruk|4^1L17*Gv zrs9sQW?Xz#iRTw8FanS`2U0 zWLQxv4(rpBN=-C4xScvvl+^iJ_U;GV6L{}RC_S!`Gk@yvNNPDoep-QUv%-b3Jshw6 zCZlcEJj^u9#G(f|@a}#U+YU?b|kSd zlYe&biUB+P)norb;yW}D&*2np{_@gh)%953P8Dyv_!Bj^{}msIc%8SzL#@*=XvpvQ zLtz|FWUjzU$&2i2egPZ*J(PDy9ac1xGmeE84>vOsUDsSXJY-*?B%N&PHtaInnn77s zTooYw9!t!*J5KgOv9iy6Xh`Q01De?yNGHN9zMmPuQRmF1ngdnbV zA|@xyhsM24IDhB_(&YJ5?E3+__nOjRc`NA`F=BnCCF9lX86{bpJ|CQTKU*}{Vwt~+ z2HmoWJ+*%cJ7>EUw+xbA%#mg+d~3qOSCXg5o53MV{rK>B3+^~CjI#O!7&iu^eDQD? z?2W;HB`eVK++Hk8yo5eM&v0(#9~f7(;6Nt>PD`|8@jB6fC%Q7TQSx%TJovV`2d!?n z@me2O=1vy9$HbB0iFRxwdWv0!_{_xfrasP$_hg=O=c5nbC47O($PM^4X&lTJ2cqu# zIGmk07gG*xlzhYqWL4e5?Y^Ip=GB~UUum;lnfP3@rEf-<)+L3lS=P{o?IpYA9N0!O zSgrZ+xd%HP5iWZx7b;pfa-ORl*Gu=a*C}rlWT;vNbT-U`rfo0=qy`~S zeLA+aOvm)i*`mW1VP)^9xcB4__7tfL^UR175`@w6$c5?VUi{|aLw#*w6)U&rqa4wz z&US04hyws%ds|mf7g<$PMLF1;%FXuVo1})$Kbj&0>6I`ghkOX z_>4(Ji`+H%+%gB_UR=l4k-`jlCC}ePZC3aAA0Nk&z1j+UUEhbTZ+2k(D9O&%`!cUb z2M+$`!$Be5^6$2$a!YqQj&X^62x2@NtA1`(ND*!@VPsK6(JU z9vF+5$8&M}_GWxuej5FpE77!~4*6fidtq+CVZwSl>flNgwqg2!4van9iQ^hO%lo!7 zRoiytj(MVoE)wo#doMP`OE*opGh6i+EmQgqo4Q-jqrWMYb`IjHlMSd0T86$m17R6D z5N|$Agw4vusA#fF@*o#*sr)HkPx*(A(og&TwK4shiD#m#2X`;?;o18gncs~3EqZ8^ zCS>{+`5D%CpjtN{X0GufqTTsh{ISWYGMf&Ne1Dc1{~IGd=ht!&TM-L7Gd%UrA~ zk$c&42o8UYgmct#tO(kR9$uHRyTc3d+B9LuKN>uzVa(vxw#<|BPKKfvFHddDEMd<4 zY2TK;_j_^A0uM%7xH3Teshb^bsgiBM_g*GcaWi22DIKo7sL3aXM>B1*e1=VV@TrP| zv34M4tQ>)_heoy%r>Sdm@MSYjTI#?R?L4TyPgujJ+6gDV zJ%=Cg;ej2Jja2mHHxpMbx$D5`dg7I6ZqAV{}O?ZJzvn&pp_) zZvyhO2SHzJEHG#eT94g`;OIR3*?9+w+TWnnLWM5vWOlyCk~^)OsXf0nHRS%C8r^|o zPjuk24_&=!q$Kyy z%#PgDrz5xClC1tSZ`u#?6gHd-O|0#?WPqjcv5i^aCO+BIE!p>nDuc?KaYOAO+Rs$r z<&X@_=`s|7uLFe7F#&5l7U6K?cKCKZkAzl_(RO(wKCV}1lj%m%+bm3yCB4`8 zK1eV9`wXn^HWW|V1t7UR9DXSaF{|}<)Q1(o#j_eu@*1$lTb+0GjOglOOGD9<`v!Qk z)Jps@i-gxZq9e~eZqM+y-s~~NljDTt+r5oFk4&|s&M0H9kJqJ=Z7WWlBi$&e&H3d` zAeZ)P#vIEH&=ajHI&dJmD2+#5+8zY9|pr=`OC;eS_b86^=Wq#|D{04PWTY zf#PZC`dQx1+vHv@?LeR0c2sU7pI4X!p5-naX(>J+1#520Hs#%DnJwLr{)hYW?wQqs zz1I!rPJ3OB+;#{d;Ztz6&p_zw3`a%RB=`od6^45b()X2PvB?LtZ`+I&;$Pd;+nnLo&Io3LVeOO`G- z=Kb}y?2z<7Jy5!lLfUe(xexs=c=M-*urY3nFFe$Rk9IoHeU1$+o#lPiXiSB{1~l>4 z<*70)8b1zWv-i?3ru+bf=a*n^{7{@W3`COpBpklF5Ti$LL;m2i2z&DYuSWgElqISh z@LZ3N`&-gYbY0tYcbV&XF{XDLjvVI2zLuV{uDj7?hBGU@9Qb9e4P7o-2$#f^c5;R` zo}kakVY*CvI+>69w4=^9@jCX(!ZfD|xVd*Q6l}&|M#5~-W!B+-$WhoeUdMO)T3D@W z%99;6X(e2~&~r9CCHVkbEqAuq;lTk;9{e=Tje}Z?Py3wA=l;nI>6x`KQ7mX9{(AdU zMpWJ_{yGbNMvR|M%eLLQ)K#5rSLUH}hZN~alD$>LP^dPEg5uz1Fh7}%te=HYo?ea6 zaleteQI*y=blFof5~p)*S<%XgM(QqHr{=<5lD+M(lx4zp z6O5=(ZotRFd3fX(MQ`E0p9nDHh{g9XYDEJ*Lk!diqR=jp5JzAw0cX_{~ubi0-x@5BtR<;_Xm;96w6DDzOOJz8pWdXG6jH zJhXQ`fWgpja2ni!M;_#@h!EvK0A_dgTD*_cBu4LPKj zJ}>>y1o-e({BbWVZO$kA}f8H=_PV{tnn zUHDboaYmS3rV169BTVDzWxrr4jM9M>s;m;m(b{`0c|&IG3VE&AYIQ3P{?(GN-pTny zx^wy!wcz!sEu`C}1)JL&x4GaQ5VffB(IIyD*M-_hKrDz-O zehui-!9z%U(jQ z^=q^#dxHzh-=eI?JDhz{i;-WY*K_PQM4b>u)?1mG$t)+}#w6Bh4dDjizvthRclo!! zXm526cB@w6$h~NZU`&?xbrkZG=3=h&iRE0(g7Lxqc#(ZVdYKB)tNA6&&AkFAjqBLc z?IxnC@1Rk*)2;&_K*i@FLQEb@*U)oxkj}J6CqH6MpH%)E7Ei0y5p35&&ipN;lSs*% zYv$;3^xc>6?|KloUzM%Gx#-PUxicGSMF-KwoR6;t>-b(&X(oa3*{^vI-YSzjPe5I>sSuJH0xc3QH9Tud@_|qLzDP*Vl)GVpBk+; zn)Y1>u(x@8=FpTDLG`E}Bdo!^O$Zvi5X$Xm*8%%?aIZa$F{Q#x2wIA58y4V>Xuu|F=@@=}C7KLhi@dLy zSUY|z0$c1x`dG=urX9!S%6ud(DnwcTVyOKt!|^wFq)WCM3p`$<^{Z5-X$p_=XuPnd zqd4itWM0@dn)Ay8_&%;PuMf9lu96y$OSZMq?<72nwqTs(oHP5ZM6ZjhFw0~uX6Oj7 zYF`$7Qg`Ckwf(sFsoJyu%#2)3U&wVZ{bK0M)@T>{1K4?l~ZDo#Iti~~tU9}V) zR%uQuZ#+t-y;dUQ-D3HqNhH-Brcl8?l}Jeut|e6212dn8P#ASFK8BOU3j z-(sfjOANMufdfvjPe(*qG4s5ybQoknuVrvTpU)lIw&a*qCN#c9Hhy|_|U zG+agA)%^#(RgKu+?ho$uZNiwgO<5p1O4%DVu1;u4{ZK8|p4H)n8a-xs8uIZ$W2(7I zZdLT7B}FoWKA*z>?j~`c<{{DW|qD=ewF zLZ8#JT5;jZ7BoMoO0O}hG?VU2?P7IyZKBDG3w2mpq|XWB_bQugA{-wxey|chu<)2x z53%7Yx$YXt(mN=na`d=lb`w7E`y;V*9TY{wrPH`kHJnw;M{tM9AgVa}u|=s5Ba)o? zM)oiTsYcW`(c@OhM^5;wMU55Od^}p0n`AEbv9B?sJNAeG^Nlex7nk&m6@X+JfZcOz#o>+>Z3+&z}FmxQq3hjjmxlV3*3 zdF-Aui>}%T$ zn}a=E*kEYOCzFJwu-}xWTX zLc8g1d?DA3d6P=r|Kv~e6qYn5(px^S(#UAO@SnkjS(Er%GO-C;g86V`A1;%5(HL7V zj{nb**Bz|*FV&QL9vHDQ&yXS2hSI5O%*&akOr2)Izh>67^|xi>J3ChNbm0EOj(pU^ znFqhRuw1S?QFQjtol|LiCz+!}cQz8;xo*KM=C+zHbHxcfxO)^UM+Nf6%%1f4+MeHK z9{IP8>I&!C6xAcKzto5W9H$n1?s)_u#Px_9HqeT0k%HH3{(`DZWW0T2}#T~;h@BWi6uAXcTA4cWbv8{;# z&Gm(cCEexS`zcUOx^u=`H=%is0_|oh@k2`$SyNkZ*{oK=3D@S8@?jJrvc4h2&<;B8HfB3Cc|Di*LP~ke}6Rj z_?Hg#WTv!U-p?&or}C8CQ^BI`d-aTE>|J5!#!jZ z%qtP%*{p5J-3_Tc8!9}Cuw*`O6b6T5G|#Px;H3j${Cs>U>;C%F)JtY#Y8HIfS@_it zMHi|ng~phClr}w%akj$B?QsGR0!|~T^Lb<&2zR*M4dKr|z~1xEFyi$a1ciUXnHF_; zK0-JcyA)}#xw&}Hg_CcY%9|^bXg4&Dd!r($Fmo~ot{Khje*vr!|9Vbb}Om&X{9Z4AbNA!FoMZS_k&g_UBA7{bvxl6g zl$zW_e~rgD^ScHoOoSOE?<%`qDXh{?VBLl&E;==tBMyyX#(x8tC9KRNdS+a%^9M^u z79*u}n{Zv1;h5e6Sfnk0dX#WxlUCr@v^6-;EfXD~OarZSusNcez%mQgPC;5lQC~MC1{)aPu z1u!xBe;l26T+i+Q#T(jtPf8_4*|+T+d+%Lj7HKK%E!lgIN>)ZhGE+8@1|gEj$|%_x zA^l$8-ya^2`*Hts@&3HWb-m6x&$AJl1rfMi=ZDc%!PpqK2m$TZpkK@F7+=2!Ctjx` zLhU$O4nK|B-8s^Gdl^Qmd2s)ckI%+Mc+lYvQl$IsvYb;UnLoqges7TCCH|bcNZuMc zpW1_Fuu*F``M4cbB6Jz?u@w8#cA_>t4&N{j+0FdKYaau*$d!2XeH-R<-jAEr$I%Ws zaA|)9<#HZbzpoI(BTDcuy%c`e%CO>7Ija6v;MJo_6e_(!@yvI)DEDYrL)lBd4QK6M zA3o?dnGuU68>eK$1#`Y)UUnw7O<9WjyXL{{gcs_JLQodE6o=Ec;CY46Q%Q~ zY)dhQj(dR2>c@DH_ze9IzQPTyw>UMV8d`}pxES*u_ga2Hcf-#>$2vSRuE$C7)#daJ zrS&6E-gY0!{=w~d_KPM{58c9?uN%?7Xg)5ioQV&uec|buAUfR!ymgn(mhY$0dfiP- zz4{OnlU`zQO$`w989z7G!STyajE?;+9Hu|`uiamm*!@L8+Fwi>_YV`26?h<6iPPRo z*TbtI4nOI^9W(n<=W#QZxz-{>7)OV8#^F$pnfTOdHbw=8qtmoi(zz+TWTlh%op)0_ zuoX!2dXHtbKQXk20&l%m;!EMoU9VE%>+`CTfmP$7%c^wkqspp!6}t3NrMu-SXS^`}Ioi43aMDq>Q1xsy=l0f6)rrbsmRQki9~}aT$P-g%hNQ}KRu5jkq=<= zyB70b%Qc&(!H8{o%ofjb{Sgy>*kwvTWix(I686PnQ(j0g5pS1&%cd&3dtPJTq z-H=!7BRQ*XCiRWM??Vl^>32Sq#;ry~x7lz>a)*!1HA)ngVy;m#6h@vzb@O7hd;SKq z?*Bt@o(7Mq8M3raG~F#$ygRHZUluoIp=52>jcUre`I0pFH_4HRlQZY&&j<0fxnz;>6!j~9 z!s(=F_cG5-IblJS?lzn;U$jo)iV3~U|5^YXnZ9xg54%A|g zWC@kqO3(6ZbJmN_xMzL~#)yAt*iR>3j&tHG$->my3KPiBf!Y1+B%^D??b11Nb($47 zJQ0nf$eeRigrj0Lh}U-OadnRzEY4Yo#|@LAUhj_faUnQsw-zZG2heDJ6>IV;h4=Xz z%AuNU$TwkU9UETNa^#4^t(f}RMe?4moG#j(`H@ykFO#m8!;VbW7C)U`b7qQ$T6)Wh zyYE?W=16mP%JgSn)t-#%(unfRJy{-@|F3N3%?b43jwzOl2j`*~Hy3k=)D;}0?RBnlbyn~vvPS`dD z`c|BqYR(aROqnot9>=X{N9S=bu_SgK4*vB(U#|%`<>G^G|CZs^1;p_lq9&_qY>}-EG5#o5b%Y{FPrFJJQRcJ)19SL+iBGyx}XkldBHgS8U6p z@2sisYr!QErW`1~z{-oBY&pIKwYwBR`=aEnk|&C{WHL&I1jG9HS}aI8guu8wL@a%U z9T5r~(@dAWWVX9Y@^OBm;m^|Tz{v~#quHF!(q-@;bM^|uG_EbDn7Xj3?CnQ2br8O^ z4eMK4G1T6S6OJ1(;k-WI_n*Pe^8a?MK8uk{Lh#6b9MUIF#ok8Areto$j-pJQ2`Glq z#rLRL*o5Y_GS5wKN(H%hG99HeIaN5;qTj@g=*qKMouzZAqjYh#;{?fan!adB8#{aH zPie+ImKJ=r*_ghEq|>3F4re^_=l0e;>13WC zfytNA)us}O1ODJ+SK)_sHshj>GH;L^u%k;G8Xs=Qk$u|n=>%71ihpFZt^?P-Z${0T z7ObCQ#A-(!UhAhuBNHVCPx}McvTw*x_M}H~3+YlX#+^q?@G#pQ`H$VvEk6X!G}pm@ z-yvM-k%t)#GTYNt;QFOH+$fpW=EBYWxkvhi_P3?U%k~`aBuv)|SFRInW#f5!(y19u zTAFi+^hbs%XbFF!2|K%qfAU)$23FVNww!I-N{(^+XYs!Y=ke*%$;dC9g5~?eF;4cW zfnASauwB0Jw%?%Z21UC5&}G;mOYsOeQ13)*MsI9Ky#pO+lhdA`q-&+Koiq2Zlrw&2 zQ>J*Cu{2emH?uT2ytOhnzxs{uZN7*f;~kcn&E)$BHk{;l6`NZw#J_%%5Lz(>-peD9 zRk;B*&KX$U^%m4SyhYQNO5C?pk3HK8f1*(C9a|S(_$D5XUmd9Xr#;tCZ6iHj&dgtK zPp9ooxulyJch1n~YA<1+K2##x{}x`*XDsac4kqnq@@^Mf+B;v9{)2^Ber+PgzIVsC zN#W?%ZUgrF9l`C8e7upff>*I3|Nf`PDbnknvgUukp>VWrO0Vvt4z!d$01FRcLX2=` z*jamyf7O&f|U7D3xu>B@`rs=j8wpTj__v#>dsrDS?)P~O{ zIdhzvyi?+uv2v3+n|2qU|1~Y%IIY4>3l#X$@E0cU{)~rQN#rN(m*P_1-!yZ_$|37&r%CoyZN}cJ4m6;by*00}emO+BWNu z8tZ{g+sDJB)jag^S%xoJyHF62i<~Eq@UHC-+<79sZ$>8EJ;a74-Nmba)P?>&Z5be* z_lb90=6)gjC( z8S{^ONp@cRGKsU0ghe>ndIwH?I*Hx4ORz-wBU(uoFn78EO~t?8zLf*>zBw~U-<2yw z2TcqSP2a(Zm0rTgNU){hIBWJ2pWA;nMoh^UeukA6Z#t=SNRfELzxeake-w6wD(@;} zAUh=(%IV|b-DN6njf%vB;td$dBdFCb3(63$h? znM!1|*@E1zUg*C=xD*{c@!V+<+(frfsyU8kCPi4i>Med<7w$-3ZFUfkL3OU=Xl2$} z*iL45X7=T|EXK2LrT2BM}LGf%YT znf{IFrICRqA(42zbuv75PetR&Pz=4j21OszV4HjqDc#DEt}c3nj1&Q&{QX4lr7XQRy|hhob2mByUyY{Y`22IN(J?ugZ=)&+fj{2E1H|2gb2 zOStb!cI@&+W}iX0sDw2(|e{NC-*dD|NRC$vR$8F-1S)JuFLX|qSdz2rT1H1=7?sMdeoO6 z7mlUo`S$F#NWA{-u3?h5aA|kV!3ANwtmy3v(_4wKAGH4;){H`O^bQ|GOYZr@R#^r>QR?mkJ_WZQESiuHBSZBA5~(q z@I947!f4muo8w1Lpz+wARH?S&`>P+&YsFE#|Gor+NBCgjqB$@a69P>$@zWpNj87(M zND;dRsagTJHf z{co&mCwfXw1PfaD^M0ZSKfD^und@EYutSUWk#})1DMfm$Jb2TLpb#2X(2uj?z& zGg&gg*oT?X$DncKEY?)ZGxk>)Z2JnZ^6MRJ4=lx&OJ&G!{uqJbPY`|f8HUZMl5cGd z8or9}G(vRPMNw?kG>DIdSu}gpI9@u@lM_ctMq|ToRQEp*N4<^kZWe`gRsmSuECf^Q z6L8gdHI^*eiU`HsFseU@PCJevW8Wzx9m~O+ZI|&aFAo#$-$I^xG0qJw!Oj(>*!-Vl zN1DDs)E>!DrpI&s>u4&E4q;&4Jmx-~!d;_=@Wt77%x-GPCmmlQD*Y%N6ISC(Q54Rs z2!Wku3|yR-AZf-L4DG%Jt4&j&(0VUMv=J}(x?`B8cp4@pIT*9<68bE^hP0hGf$Sn! z_qvacpTswB{|bw5313v^&V|pyxSj#LTt16)kBnnSA9BXP7QDGjmG>o&y7XNdv`kk) zx8Fi6x)FapNU5oiwHseK561FVajSJ#SoYMaoI-QYuU5gy)?YJl%&DY>4 zeYg&5WzKW^5z0eeAiOf3hef|hsE*+H${_w(>dm>>Q9B7Nv_C!(CP+S zB#3`w-5MM%TY^a;OYq{_a_nEY2A;_q@k?tvMh;KK=7jw?ZG99ry-%U<>>T{czJxrZ zJhUHQh*#n%Hknm{53W_xaV`Ak8F75FB#I9CAzWbN&*ycLJ@%ZyevA5X_B9u3DT&uB z=>xi5y^P>_`(WN@GfK+WqE34qIw@~NYQL=rj!j0w_dPg}o{l>ovQSrW4rbCvCl?0+ zy^7#;;-2)nJ;tMJFR^ykduZ0hb401=H@BiWu4y=*-VdZ>j5nW6^PpnZNWQ);{T){v zc&U%f?GxUhK>V@+4Ttfud>0hQq#)UHCxQm<#+DZQvHYBL(UqP=%+d4c*hc!(HWp#S z!+X#=A*^-nXPDyg2A71@VzA+d+>7xnbe7K48!^oP6~VNkU|#;<$6mK)v)AcK++jYL z@h3acre`zmv{vI5tv9&ob_+WS&aSJ@UQi_T!YyUUM;j^)&=$6~TDJ1FY@) z6wUUC2Y*a8jC+5A&(H5DpV)wSHzgWM&db$1o`KR?dMYrAUs{C;zaWr%s=ZlMGmVKq z$MD(3K5Qd=rH@BNWBjbbv~jga8d)ZD=R&-&lzh|o8yKWmfQ=``yS=guSxcT{lv_2D z|9r-W7C*6b&R_gjs=!g#m3i);_-4AuJ6rUHHe2FpXA{Sc+oE|{BZALG=YRgoPx@s% znO5M&gHFSlCfEDzc4xlylf8Jh8uw5Ag@Nid$i6KsPuUZm>@1vuYi|&l{Q*z4e_+s@ ze`tPJiJ2KH)NEAa)NYz=lBLZi{(3Bt?{JCeDLqe%ma{&N&%`5iEiRI)qeDd#3#6~) z>KDJB&g9|aIeO?o_6QLj$5r++bEN}Vz9~u4v#oaSA2JGlBlz-fc)n?X+dW0TiB;i} z!5ZALU7KTC>v5#;^IRJZ*-_a<_(8(7+-AXPq8}YDjAu+$9EY}yrCo9q=QxEkXWarm zjPzsbDo>fwxbyZ9@$O9N%>%dEbCv8PKYll7!V7&4Es}4YF!-ti)p=yR1`lk|Vjsx` z&o+>`mW~OZPL>R?wgm^QwB*Ou)-q=i?Xb~?8M1Dj%-rim+nBdMmKJ`JFP|O3mJ>rb z=0^Z8exA$uO4FHdHJ+-Q2eEQ&cj@RMC!_q3r}xGi54H)n!yPnHQArBv3fsFKX&5Xsg3iDlRy$^Z6<U2HNt2iyhC#*|Ue7rwUHAppQvQ#%&cYm8`q5Mzn6xkcPI6UJeVEHm|kJ* zk`yFxHeU`tHJe>6+^P0;r1$~)GHGrX;pMneU-abF4L0n)%$kp{TXJfGrJMt;_~f+A z2nX1*-yu8dMhYAItn|_zcVx^(>AKetmRx8new1~KYUBCXS9FHHaSZj3;hqNp@hN_zb5$aFl#;OI~m9%)`4|38Ofk_vKxwKR2E#o#WVHg0T9}iZ7~d z7~MmHXwcc0cU&bW+S`q-vPN*;kUktfKn@r$TCwD@_`ZiWrT23S4(@8sSH|XiDH)s( zE5u*;Uo$>FDh%X}!uNS=&rkIZeBt0o{k1K5+RT|1vhLEiqIr*t=K>9xy{p8~x@#nL zB@-JyF_3{N^QiM}2G3(6_Y@9cX-rSvmc4eb0tfb9Wz9=6?+q(3;!i(e#q2VqdOO*l zUod4+x&^mR5O1niGhTjW!<&xcV_0U#-z^=e___siWL@`X@l^Q#eMc;xi*ELAY6Oql zhwyt?0MFIUVc-xCR=JH~Sz4ZFUY&WSr`R+OFm-d*f*#buhHi0D#X8eha0Ot zVSLt4_%t@+)B$C#+^Wh&O*FW~LyPL)b(nfzpXoV9Y%TA}lY6DBwO<@(UX7-nO#~C{ zf|)thkFR>nX6u%dsFyy7szsf+=By1n8EEmGRvj9SKEb*0J6Kk53v0}8VbCq%(YC$| z%RJHH4vHSOtOlLNeMPVHzp!fXKNyRL{)CFKA$--T|3LUX@;$1KlIwgnmM8Z`Q4e9P z=@Cdn18;WF@!*`8kv#jl2k-4}L6b{5bZYe(+tf-hI{7>-hn+;;b~z`$&O*)o)42XC z7hA+vxYwx|16w`BDT_+UL{}l{VU1{~UvRSPFAVwHh^Q)Mw)!oMgnsc%Zj7OK$4KGu z1#`t(Kc?-O#pS&wFnv=$rVSBxWSo&;e!szCZh_3yGU3~NFEr%rkQ%TPhiC7`FT4F1 zw&4hxY(0skG3RkWzWZjCMR*i>qlLF2b~obc}nl6^`EPkbPh+v_@{gh?QH= zH!lerUH8Zy`w(t)&qB{}=im}?8HaLjV8Y`f_-5Y6*R%@UOQ?eLx(|5!QG5lGJDy(= z#g}@aR5I|F9?DtlJ#iehW|Av~aT$>*z4i+qV2;aStd|+y=y^*qXYpd(;xgR)v0ZsDw76TpV`vv|GRSoT-x$$q9beB%8FmLAgiQn(qrlH;MDwg9{o2JMmrtopVB zi<@o4=;$3X1K2IB6Y2LbJ&IEC+tE4)TLwx_cAs#zb@HL(T`YQTDSm#bK;{r(yqt-r z<}mqfpM@~o$cM-0P2n^DL3Ca3!vFr$qAu^@Nnk3}k0c=Zu|Mig2jJw-Xsp=2OtjRE zSgM$UR{PQr`{f8i)}Fvxt8<84aS3ul5$^ddJXuhTNa0&&FMoiQlPYk1^h@MDtwzOa zxz0_Z`Q};>->vcF$b)0KLzr;UizIJwrWOMS9}`dLGF+bKkAAbfVQ3QqqtL}T@OC|x z9}>@GU^)hep2YK=xmYFrzB_89hxFrZY%(v!7v)EIFWT$~>E<=Ge}VZmuc7a9PA7kKcq(qJyV+KOwJtFnB09`^u44#KeW-w_k%b zdOOivW-!_N&f$_$9xl`s!&LVHwzRH5`~A<6{$- zi94dpZIfr{JHu(@Fpm>9OyHTGJs9&)_;k~&&^1!7UvNCGm(Ip_B`=&+4oA|g6#didVAbH(dDi16uRyvGnY3=qLR_e3J(B+$8M! zgW}G)jhhaB-RjNiH$ zEzTcAkC0rfkxrq}>6O@YUvkI7Wzg;N7lvCD_-K_9w+dJ6TX_?nuu$PjbJ@#|P@%S_ zDzm4kF~vxOj|yVhzjqLyj+n~YTFK%zw_?V=R|sCa8!Z-uj~blY9++ ziBIKA#Tne2S%kiGp5ytlFVKD^9GezRxV4iSoqB1o{;cE?bF}!lg*J`;2_Jf`7B3Cf zV#XpZwjHI-c+m=UpU1G5FbWQS9>>K4JMc)3Ix{+mFGYVPvJ*TJ<>G-JBmA*4dkGXe z?SS5_Ou47U-@EcDs_<31q!l@9ggUduv%f}k=GJBUv?(x<*^VK<3^nAObD}pr5e`Ii znY&j?zRX3Rf8OeI+!5iq(VG`q4CPPh=xp3yi!LqqVU~FqUcQ@xdpfhQO)|1KudGAW z_5JYvCT#SqhnO_zBVMmpq^Fi95B1Y$bT?yu8g0r$l>NtUGX~3_72;=9vNx5ELSs(e zFZDL1{<6_F|oiMbw0j$cVP*JpGiXb z;}dY+QG{Z=L7Cw{B#zc#QkVfXdYLiqu=H!Rv}UIeYd+~}&9~`R;%Tzv%C;6vmM)`> z7fslEi81FtHsaj9Ml9+Q!5tT;u=`J}8THxwrx|U#H)V5~h1$KfrSmddW=TGBON?~32q$P(wIzLJ zcKFxdlwsmKJF?h_XNnAI`6-lfy~p#cbP766suJJycBDL*hvsJ|pY@ROIiA+B`YVRD7~cxp;^je@I8(nIjJTzSW+6RE4V}TFHDZ$*lNU zan4F}PQ4{e!g$g6^9`g+R`}kdg4w}&INRJYV@k(c=wH1Y8;Yg@0q&UgX+CcHuS9<3 z9t?^)k24z|;g+ZPr)A{F+A1X|=4W zbemeRaFhvE2O7{ePlr2RYqD;cI%kCXOGc&#Gc6kN(Q7YM(*1FL$0S(Ed-2v(>1NN~ z4ln(a=&<_^%68Y{d#ffKx?i8Odss1FJdu8)sSLR6!s&fn+1a->AO7dW`->fDnrh3@ zQ)GXgX2y`y(tFgX!?w>fnC`1elSF0lv-!|kvlCb8)?(7Yttig(#Ha6*5Z2NM6_(4; zrZ^RIt#WWrx+1El)!{*{n(*L_Ssl`hstMB5EX<=KlQXA=0T5}uiuCJDkve{TsKUFfh zL0A*D5|(j$;B_h&8eV0%aOEpntE#fHMxQ})4cErmNDr#qSJUm-`;rZvi-bS5So%Jj z8}Z#dU2>!b-`X@`>BWDDzWM|6M}EfX`Wkfi4PvDD8SJm?bA#bUyq&oi7V6XSxJsCp zYa$R`vJP#O58@aALO-_2gLyZ8YbpOSW8nyamt9Ysp2YTCm=sIlXPI zsGBG6C1F7rMyPS~Xa(l4`-&a!UZdHF3WTKI!!@@;1UdL|ZOiT)IO`ug1NXt_ru5Gr zcgJ>zxyam|2(1S@@TK9TWIakS_-iek4VrL_t3E4X$pKG=S9Z{mKg6rjMtsgqJeo62 z&cOkrjae0^!?1~}Jb%7H{578>FYyxAEy`fi^ESeYZ{T)oisMme|D$|zhBhn>l>Eb>}JP~rY)(y&WV$kIP#@< zt&2^>cPsk;xhXo_Z>7pLwhhSg{fPYGFQ7W(A(q4xx$j@I{6QFcHe`2 z+5s?db;F|tb1;5mBK`?8Cvf~J+`L(WD@#8hZ*~*zN))cf7E8|QAWW}6jC{mvM-Q(ORLZXJ><~XHf9X!;sR}&3 zEDbKkf#NNj49^yx@U2Zi(!XukB^?u;=G=z1!+Q+sE&4>b9>4CFF0EP3d3luFQ)*7U zH^!08uQca{^H%&fSMIBSx-=iE&I=(tQ&IP?1r8hR*lrZDOZznZbhOB)W+Y$3hu zEv5UV1vU0c2f4j9izZ7BGftlxGc=j-TUoTyKbYG4GfY(9B4f`pwAx&bY4hfDMEibh z6g^pY>uEgs5Q`KAHz;a);LG!HXbxJB{^seJwDUTSn7n}T(Z8s9tjUW(CR}ttwDhL- zT)wCUyB>F-W|!v7k<7fwFzF6-G~~c5+8k@5#^dpd%uuVBp6Jh5IMf>$qqwzE_2*@SEobciTWje_JVMxV`we-~yahmE&=X?}+FlTv%&EzTYR` zdSP(%pJ-2SYw@m3wWX$LzCOYqtUO}Gm!^8Ol(}T0mkOVmDYCZPA0)i^j;GmQF#K#V z%X*CB>VwVXxi7BO{Uk6P)*mf!dJAj#U_FlL^*%<%qcv)4Cu*0`(ia-tGb=PR)6Q3D>m z3Kiza1P(}b<{IhPYd3f|76%02;pK_Q+cN|GPQ)T)%qEOYJA&zLZeT^%7YGn8PX0r6 zE;ch{{2g-+ykO0zl}+XOvSv<@AAs6&8qw zc22ya*VLx+=bcW%Ls#Kw*V7pKXCdaTa>tVA?kLL*!e4_`=(Kka*1gEVLzf3gJtv(Q zRZ8p@rA-sruUXie)3@G&^D`~juD>~5mYZ_kQDe?oVMtYNeU1pzVPt159^a|K!n5i; z;HJ*rnUU=AaVA64$zB=8+*5i7_aCo=*EmnCZa*0xTFpgQ#Y9Y=w+)_>(d#1Kl?=xh zaO(RD<=<7JY~MbbW_{-J;e{b=y<5Ce{-5xq$$nhgABJb+-EpAvbe!lMj#$4n(D<+yof324 zwCmJh=zqUTJo#p#BL;ACvbXj#%o8CvYSm3I~ zg_C7&tfb8yelZ+d>r1WTv8*l26CeDOcV={Z;k6bE z8)@D(sCbm4yUGWI88@Ks+$P*-sZMJ%O?I%RCk2o@kS5a>OTvsdqu!=@M?sv z*oCzNPGQcwd{{rN5Z|t_J>UF9AAJP|h*v*DGGGU!FQ+k3{ITMV^D3*J30?kDaQT3|=PV?lf82<_x@vjju z^err&Rm0=P2N;MSDdpHNxXzQAw?z~$IRufxPe$ylF$0LeX%<2|yW(fbl~Zsnm(NgDzjRv2rX}tHB z=%5#%Vz>q0xe(sBrSth#DD*Eb#Ez-Uan5x;E`)6pZ~g!NZfUsu>Ijr~oq$?YHtLKo zz_$1@o^8JlPtjqgZIJ{;*thK7R?hGJA2}RfzRH?jzf@0^?=If7B_Co*yGQJ5#<*6MT8tWd>`V#_&(q zUW|DroLR}kr*wOO=VuRNRP+Yi)mbVtr9>!3FTv-+6}T$C!wv5?!)ik^it3~%F61y; z{g(ypOJ`6xEf+TxFH2XT%nH4Wpx>$##=1|0HxbY82gTDcESkqf%di|XpPd}%aCMm* zUBiYjbx?f<+b*C!9sc zNtf_dFCV&nO5mzl2K|{&&|CKf6inX0NTC+1q%XIjRXxlbq~AF~GK-SKZDAVCF)n!w@N1377eWDKrLHgcjCEgQaJXqO5s@+`ulX?f^*{~C(g-N1wO zg-C36A8ijyF690ztlM0JZMmOd|66=h6Q!#;xB^cz7bsmz3m8_(B zWds|Shj9L-09xwJ`?AxDS^RTklJp-ArLSfWTHR{Jo-?heyh)dyGJ~^LYs74s+dnlI z7GUacBz|ecCO>7a`k+e9ryA_npv6$>Blz#5E)B#Jbzy@cy^@XP94`7?sqkav-s&C} z%jNdb?EO1j7~8?@Vk7hDOLKU9f(K7sAH%5=`&0G(e{}iKk_R7IQDLVc6Fjv!_mMis zE)fp3wmMH~YjWFnZ4T7cXL6Py2W>Rw%?GmAsxf7CCt=8n2W?@36|cy;#m`0C=owGE zj^)26(M&XoWZA?}={*kQ!P0r$9X69eEhcfHlzl!CY$|@70uD*^Cd?+VHfZc<^Lhi)zv7X2dhP zeH?wJ#?YozdY=8l=>0N?=id3s{?C*B&%1N;j!`@u*^dT+UD*1d3zvVgXXLu3d^g8} zA3K^cM|8m*d1f40Zow|2teGfY)n2N$eEr3icT$_vEXa<9>+SiaT=a5T_tl4Zb_|v8 zl4QKs9*p6X{!xt34(EhR3uFf1Ps?jwe406x`<2GBa^gVl7uHo`+jg`}l3Z{1=G6Xa z&8Tgb>{??%&vTX>nQTp`_swXp(46j}c8niw&)6D!K06>BDLF5zzjfp{S@-Ivcm~DB z^Vs}2=3R;52k$7}>lw~3e?=pl=FfH3b9mC&gC_#U^2_ysEVS-H?al3I`lBVAuW8Pt zK23!mY{^0PmV6=eImJBD!1aZ9ZQq>tx7+bnsGKDX9oYY#1Jjc!j-Q<(aEGzZ@3&$}nO(!W&t;S@!uUo5Po zK9-y-*_fHL&1gARnBT%;d?~)lzpq6D4zT6c8O<4MYDed9cHGfSvLm4__*&K-FS@h4 zct>XSisKls7@FLTq`G9I=uMzC+c-W=)PkuHxVN1P!XMk{k3 zs5fG$bOpJbHW1#hA+<-CFhMfFM@L)IR?(W3R&qU*n(@1-^tQ-(s0gEY9`M2oX3ba+fW7VQ>Ew}`(9m){F`T=j#it{EA0y(uSlV(54-Oj1%+?aEc`hDyXB)abHef$Bd2TN& z(*MU_bawfJV&?{wsVlP5UG}oVmEU?z&cF5AjM}Hmsh#xMRKt*p?Tz_0z?A)D-O;z= zc};T1UmAoP-#v<1bHn(xV-RCpe0crd3|5>O&%7f8m})2dlaKnh}>KU1H?UflqOJ#bsQITA@tUSBDVmR@O=)ZGASl%^&I#0Z~G}oO;6NmHMitY?CcBHqZcwXFv%jh7ps3CXJ z@?kz^OuK>JR{3abQjCB>rD&G>1XmlXFi5EuWgEU={@5S5(NlVZ!xZQ)TJ(>{s$47I zqj&4YH_{}I!nvCKKqhZi@vaF>z^KOL05#RZRGwf{1d zZlA>C9Pty%to%qr7DhimE9{F)*yx{+$M;GQ@TUx(51+!^_7#2?zeBm#C$!G_fypwf z&y?J;%52F1&yD3o}-#-ZD3z--;IvaO_rGLfeCi)C2#@7e;uvxhrGkQPA(_!yW zC0)p?C3pPKKAtO*MOO)k;D-$hq|d{L6L(JM-W#L&pKUKt=Cp#T${CU`?~>-1GnMBvz>UJv>y&Pj-t8rQoR0|3zcnG5N>x9330_p(tLoo zQBUEmA(@14qW|6&-^P|`ZU_qFD`De5Ug^ah?Iz2-bRb*JZNs2_2K<@y7IjC?z+~+< zD0f{BlgZL+;;|Ta1~12=$TgD767HU15<0KjgM~xVF;e!@Zo)R4l$DF2jh9i1n{r;b zjo5)@uxxk+TX|;;5mwQJ!BOn~Gnk2ee0j)yIyH4i@rp?|rq)}tI`TI{%&*|q_8szl zUku~?D3ouCMvChqY-zt715(zaS<@}Bt4oH>;C<3tau}nhWZ}cdGiY@#7Z3KxIcCKT z9L+1nHqD2~POXGprSK@O#!_)#1f5?8(rB>oO|DO3?%RHB6(e(dHx2rCxrZ0M_P}lW z5=3+lf#bLZn7=a`GZdGg=FMvCU%45Y7n5<|@m_d19+oblEaV?Oi@?X{q1g5c`ftkv z>kIHW>n^n8%F&Sd0<8+o9|7mW-rjSZDhBn6dezYAkFx1_R@j z$l197>reY)rDX_4xh#Tx*(%8!Zh=Q^DpD35K&ZkoyqRkZXIzybuL7 zxA8{0a$X)T$AseN$eZ^L_8HQ>@GV;2@gd^9@n&i9WG-jb#-S zAHBiC0o9m2wH6nJ`+V5&2O^sOMQHDM-uM~8ML+%MXdzwfJqJ)t@>_d)DoT%IE;=+_ zjh;_@aR#&C+}j`Tbr)gQHObJ|>_&)27Jk3Dga>PjF!|a;#J#N){`PBlro2a~+h^1r z`i8|9z9Yc%Cv>*fix&PD{jwBjKR}tC--;jTMHmMT@us%wc>0~~%A;LP*fs147DT5a z^-cty51%RbzZX6ZjYLoPRcLl31%~pQXOv$?+UF8jj;=&P`a4Vs`HXoxe&Al_A3QwN zh!9&v@zg1@qMH&sS1L))t_ceyBXU+WnF2rQSsoF>?OC%}nmCjPCbr_6mr7i=IR}GW zR={lS9F!(b$8FO97@b=zH2kgD^E4e(v@YUnpS##)_Y!YpZ$7E89tnYxAyHT6cw-e# zNmk{jSotR9sqvxwd0D5%q=V}GRx51L?J~c+CAza~5I8+)yI-Ys$Za@`z|z? zh9T2!n)IxD!oxBeXGW~W>H&Lk^XXYAE-FU%9WPKWGmJ3Fo%I*q^+f4S8GBuWySoXG zqnkD_q-pbBv^I-1v{@@`j@v)ASSvHysXul2=35MNg8ex-ay*+Fbfo7|nKRwFg65j5 z@O-`(zD9XqW;cKASiJ=4!iV?#dkjxqZb<(32?mb%f({OfG>ubZY`PYkYw60qRgX?B zWNx-bpAJEi3Av>w{x4nT8i|%vDBb5~y3D_)%js{T8E-X@P45n)QavgcxGY4gGMy@UVl82t zj51~VcoW|LEE=ehAwB=-aijcu?N4a4`+F@;N)mpwX(UUsrqQTLS0-6#Fry*|X+}%X zr+zwC44#I7lfg(_wF(}W_h8YL92^s7*M)$O*k7c?UJJF^`=JrXE;pzDc1voA7xlGp zHSI5&@$L^3(V2|6wVMI+0wk06OnRk7myC#4r8=E6{Y*U;JnK3#~Q=GpKPmQ}zju_GmE* zB+GuL&1@tYc)-FV2;VYS;`xO=aQ>Z(K^bMZt?~^k98{^$#DHB=gv+$jhF4i!d5X9%n(gQQrfQ2uF z`F3eB3_46llS|VP*i9Jf+t#7_@dV>6@018Lr)v9a)Eet##SRY@j03$l;gH>DUP4Gg9`63HkeH0qH)gD zl0L)3ZhK)-ABv|D(=jY`9_*YGaUgaZ+V9AO^@UqlYF>rfg1@+~Cz|-pn+(ozI{OGfSOQ@G%J2c}=Dak;nnRC2YLf8K~QGvygl zw`BTOGhXRr$f3KmXrZOT>#Z7ap!O3U2n#Xd{sYmn3z0MbD&8K-#S4QoFfR#aaQ#Rg z&~L^k_K%R4ya@@Xe4uqh^zO1ig!fv3z2kRbYHl|EX57c^VIR?Fx-#QW>rj7@3D=80 zpC#^rk4f@AEfr;K7ygYdg9Z%)pRH}536&!_c!vdz< zlk>nWOEy1w7wvAXgJ0QP{HmUg%ase@|86A?#_Yjg$$FNNc z=vq9CfrBh)arQPE-%4iBaSl3dnGU0XV2n>#g|KmZMPoQGdsy)_jQEP&Nh%y7{LsYz z%;QF9o(w}iIIPJVwMvZ4{UJ=ww^%gt5%$k0z_fE0v2Md@>FYj*XB!W} zbH@UXt{=uXNfsRDatF50*WhZ5c*`=Tp(0=b`ZlaYo7;OZ0u{dAav5{2E>jr3H`SSx|({)33;UDH{tyvM}q` z5ya>ObMm+m)XB3F@9TXGEnJ6=f4yKi$OA(@&&PmN zyHVb`!rVG4`gXZ^gf&}P@LwwsPR@>O8Kigi1HsIF9Mnf@i~||3l7MevAFEWwtv?LgylNa3XS}w&VNYER_5yMC7G6 z_>|d*RdU8JD>J0NWY!leSaMtsbB;?h;>i~}Twt%xD{GaQf1w^-JAB0PsjuMH<1sq8 z+(&gqF>;SfCeuBPC+#QE-dbk8N&}>kNn_&>({U5VIUK=*fq6K8 z?-_o-{DrmR&s*_Lk9#u2OKvKCR>o#5@G)YyJ-Y1oK!a6s54g=#;E$>w2w46B1Jho^ zYyUF@%U-&V?jxw(2xo~fuTOpH$dP-RurBB%zU^Fygr8H;ZMO#|x`kqK`!zUNzYi0K zUBG?6GC0P4M(fG4f0KTWkAa5NjWS`UktTd{z>pd5b(!QKp1x2uwoFxKcc(_=&;E(3 z%FozmTnmRM?~rlw4feK)(}9@*8?A74B1Kzm@p=-%L!jb;lt$Ka9Py6mB{x zi0XF=*4f2qvR(R1v>MRkvh?41>9V!Fmlfq+_H>cG&T^S$6p99EuEFcl5fPy+`o8Rk zCf@%I$KgNmz5jRY8&rp&g;5-E-jl;`^=Ff}R+5u>iW7&nW8rS$x#+n`=Gzl~y%xef zYZJCb9>%pVS8>m+0?R7C;Fh-%>*6&ypj3x}*5#cF~_;WOx=Ksghd587fzHhvgmRS@<60*sN%=3!u`5-Hyw5Oy! zRf>j?22m*^m6m8}NU6}0nUv9%krFNSyMEulp2u-Gp8NUK`*XkV`#R6_b&431#{lnH z2WC?8gC1O4e-~%;T`;qg`SlHlF>#VLJk|YR$9b+jCh45>u7K61Mnw5^!zNpZa$?13 z?+88muN|x#(wBO5x;y`;X+X2xb=1_@K%PyA+(pzPzC0~v_C|coYjRXivJuFE<|->gz4bIq4bsY*(haK^0GKd zJBxLwTWAY&!DrJDO(8n5l+VdhKDfVMA9pejTgm2*4f<_R6*T$g5F$*@M4fbHD#y^~XHwe{h{QG1bR_c)ewMYnZ&j&M2-GwOBb|}TAvM=|a z4eg#{NSB26koJgWG`EF&=4O3B@8k#^e`kl+^4hpnri+z(?XaQS2O)7`=#aaM(8_FV zIQtyaj&lxJm-SC^W=Cj#gH}yDgxPm^Ecy?O^Zp{us~Z{}J-8Xsk8x`SXxgPAv}zo8 z@lE0$ymV9YeWpp#QtRkMloI{*7ohpsS;)KShXx5#6dck){XbL88taautIk8>Y6Lbt zNWhbW58?Qv2!rn`peFwos{hsEU&a?K8r}?no>oZbFne9~JH{CO#H!j3EO+~hA%*?W zBGJH=lfHx)$Q6eKEdSJcn0WB{F;8 z;6JZg1abyw@@5zEzhg(e=Z;h6eLV^qwvXbctfiSe+nzsHjQp0pfVxT)!hF4OVTd&r zOgn)e%#@p{;R%z+zO2CoLU9_iiRWC$>Qyl){+xhwW~tb4E*fK0^``X&f*;~&#hDot8NiuUxVdr-0GEk+h(R?0|?m^!7M{xUi1r{}@ zQJU$1|F%28(d87LG6#Ao`zthZ15tUNx%RiN!_F**JLnQH$Y+bR_xGWtlY#S{*_c}K z7#f~W;m_LPde#t zytjWSfr>{3)^)kj&nBL63^-AxvmFgRdV+4uHYUw+N9Z>Dqn6~aBrnUU^te-$q#SB+ zk+VORfnk_u8Gxf6zPO~}hhtR%*f2B*GQMHh-WrAKjj^arOoC>_eRx}C;NQY*M7(?i z_2?oP%`HQD&`aFtaij4nZq(W6Ogol1(5V^LELnq?9E zs-8idy^yaR-{6?*2khwQ46eBwwOw_g)!b>hnC~qopYfa$#uPm5D0#f!MLRyKQ7q>R zlDLC@>fC>jT=fnml6hEIo{Y7jaY!nSh1A2lh}OD?-?^F4ypxY3&r5J+S2;>|Rbg-T zYedFWBR#MVujeyA`Bp2MdEOoz>PBy7an38niM-ff(ByNH)Yh8Qi&{PEt~yAD30vvl z<`vA2noe#`l4N)K9|k{vLh!zFsF@XF*N;aika>((u|>$@InSNd?9t+`!7}!|e1F^s zlSeH$A^eT=N$t2J(FLVfJ$TLff-+}gj)l6C=mlpIzT-frvuw!R#**4hjmdD9HtpM_ zLCdePuHmdgxy+&+a&a`R=@+E$lFY{6{skIpAJC#-gYBp4@R+^oFRj=Mx2glO-hbgL z(T7mCf2bA|pdEJv$s%5eS!_f3e8IY8DQ7jmy3(4}E>v6MNcmQFbfVjeU7kYMQ%Bs;{J`NIhwMx?6)i_um^m$^h9dmb-#UME!5hL9!zoO+B5XuZ>6D*m{eq{eTgFQZqIB=>~%m`$fu(-dg6tQ@t^8AkzgWJsk{hJ4P9 zr$y|ooMJYSjNIhu!%um7IAaoN22Y~>#}sI--sJ!3unp~Q^uybYto=DN+3(EVWR4X0 z#*X6aPI4cx1syFjr0xwzX{^FtT2`@{ELFL#Q+qLY^30|@&X%N=%hU8sIr8+7O9$Ki*~x=NG>@ z(o;b`3&>bg=VuFAIoXJg-9F0R!M!vOTc|n)6ji>MJbXASx_K%EO`Ak+!mOu1mL<)} ztUvRft`a0qhU%P2=6o;`CsRiCWah^x()%t&`u2Sab@93d?3Fp`>qgNRU1^7a3(a}% zNHWju=;Nc4v`@!^cIz3^{cA@kXOadP8*QRqy;W2;cOj*54~>%AWGa-I!1Lenw6|G? zSvKRybA&7{m774hA12c6U-GotWD+SSa&Cn6OgGlui{%vQJFi>B?}kPv<_ECeUCDa) zH+}B~c z25Xa??-rb8OFIvqpyNMGDBvP@0FU5aT<5J+>b{z`ZdWF)9qhS|8%HS(qsWbC8ZX_& zC}DvZ>!spU@O2msc#kA8?mN0ap7SI>q$nVE42hn@QL3^%QzSN zhP!F+a~_GgCmQkAq(ApK^}aTwMeC37@0EL@8rA8junK)0JB=oPkfySD?%C=ZLQjA8 z!%>s?gx%PQ*w$&U{0AUN?)+9DReX zlzqU3qN=&?_?R7?xot%USDDegI9)p8a)1t<+(K4`%cx;-BRMBJ)RLNvjd*%@guO2Rk(9O?Eh?W&1z*O|OKsLCp0_fR^%dfm zKE+OxLY%!_2*1D*WD8dyJ?zc@`c0%$*&8&f1x@K~_?Y<SDV5c_R7T8cL;l^;q0q1l7R@ zkYO%w+@E+Xnam7kwNy+yng#Fmk0Es9DXPPHZnfbRYF(Lavbh%PSO-3r*^2cW+VO?o zzjAK;-7R*Z8Aa>~vbQCt082`+HKL2whnd5+gO*0Gr1wji8>cBjD|H%>7?KB*gga;$ z9nOAu<{+nC!PMj7DA;=guJ_}RfA${g?`7gG=QU5SFNE?d_7Kl5hx+YTaR16#Tweb= z`@X(zWe?T_7aDKsK<10BscM)xWvtdEi-x@f$#o<;pSu8fUf=)Y8#a69p)V~8q1_h| z`rZ#6PJU3=4!{@nOSl|)1>gQ%N7}hN5IvLxm8Sc6=9P&y6cS*8yP*PO@IpTStE7LG8#n=l)U!@lLoaJ+dRKUQWUzcv>SN}fRcQW;*E zy~YsUtH#B-(y3WaB!1n7sGZLlhPo8@LxUpJ*U$#98IE~5x>bTj8fX^w+<_SP;rvp4S zEFt&N8hz@gkmY#>D(qwIw+%r_LpVZpZlTZfE(W9Tq2^>dMwvgvn1DQ_WE5h)3bSl( zS3z|Ddz`O!BggYDlsDdvGL_y;9YPgKg=k?W(eeUkNHeoH zLyj}R7LFK5^TONr0nkcgCi9T%u>8(uN)pbf2(Fr$8xBeXDmD>H-_(e)HbYBa2ca`jD&a`8mv9@d-# zjqttaBwShFAFdn#XW1*LUw;G9R(Ij>_CCrovoW@vpP@fbk#o8n69->l<>A+u5m=2> z(T~VE!%wY`{0(PF#?Kp z(8A2q{i2>oS{R7S&!Vy8a4JOO9>azE(l0WXdEeVFc*@!BtlFO#;5{*QNH=HedwGxP z<8}YR)kT0N-5Ej#oSnGBccU|JY)N2_5nbVqE7{e{sKjY3*}Zy?8C%0~H`)%ThUs8x zj6QZKIzs)oA6A=%Bce753oG-1ju$w->m%+)wqkJqFDRS-MXqf>YL^HQ+y%*T9(S{c z^0T>&GtPR#B+J@UzPcD?+cIOs-kO$)=#seTR@yOa4lQ^fM85|f;-9oHT6{P+(Rma! z+X6L_r;#EQf}akt_sfDly^WcnPz zl-H&>)a8j?K_N(JkH<1*kq+#A2SdScI6bVNnYu#seGAX(`Z+6;CQbokC1}V@32J#H zP7ZIy$Tdxr7FUT-h$}O^O1O*Qv?#^xbEf83CX};gCtYO!TA%$OJS$Rg<&Ou(Ch4F` zpL^4KZJ>0;kLx+FLugbwbp6WV`l$i$j&x)AJ0Xf1B2F{nhtX?o_H<1cPL=+{xWj^T zDBI6%z{xA$9-*cTt?6i))>Sg zK`$1GQE&E8>f(9qh;{+`$1J7d#(wDk>%(MKM_POL7_AN=%H?z5X^(Ooe0%{vSnr-z zbp$@<7KktRMuEmv#D2SnMJ>;9)S&^tp7+B2h6uerFoMDqB+2oYBvo>ztWtO+mFW$m z?lduy$Q(+=T0ARm`-g!3zpUGLB2v2@kt#p1T*!fz8Xch}xvFHhP@MMsdx$mnypgS; zi;c!Qc;IP^nj)TuK90tK#Y4zQy+-m2)=7Q}(m=HYtsX5&LflW8FnlZ-xl7XS2F_v4 zmmmYq>FBx+Aw~Uuo_BRZ#_1c@#y4S&#uqgIsly^ZpU17!B>lq+sJCGdwwDty?4lcT z&9o6STOSsIt~jC@h=rc9_$yq1VA)#KeCk5^bJh=-+nF$JER745p#@RWv{*uls`qgJ zYk(LXR@fc?1oIQs@DO_itveOyxnWDwo%T^<q0T}0PCR+oI!X~k8S7KxA>QPLO8>m&CKj{)c&sV2gYO-`;%X_#^ zzrwymE9`JPhP@{(pzP#@YmJxTWRrwe-C{g^UWecZoe-JMKSw7~GFZX-$#&*Ia^7uF zxDU3I*xy*#45#IFm~fW$-k|4*+gOM%HMyv0$ikESbY|2#kkw8d$~(21jt5E7w2`Iw z^Zo*wipEO~`a|!z*V@xbw<9DO$GxK)hB31( zAKx2%Q7vnX38VCxv1gC*Q_ry_I~w-3S$MbXC03VxMbFAXykHG!Tq$RZRff|h3o)u- z@96%Azq#AI8Aq1Ahx3&(oIjioXYVwKeoDX^gBa8}MPrIp1cK%5DOg>LGkhwfdy4mz zWtlK3^oDSb0eX)ZzyGcg4}O7xCv;3_jm|1i25@81DWHmtPGbqm2@@O^s(H9h}FX z$qY|*L7JlZ3n$z^!<9XgmNyFFygVHS(eXHZJ{l#juHx^*5X`<92$NlQG)z#F#?D_z z59>Kw5SxNi2_87xuL~i0L!92>hNQ)TFn@ao#ku)7Gpq)4ay#(RN{AkQl%SJkBWbz$ zNHX;jrv>(cWM}yc-a9^Hk6{%8#S0PXm4>j!IQ)7Y1udnk%n%B}!GVi#+ha$cCuowD z`9kXFGu(>DDXd3%;Lmhj%u_W$vzaS?Gw(Cq>JD`D^3k359*W)_=s6}tTX@~wUq+JE z^^x?4Juv%trWI1qfey|wJYdJ-ly~seTRsupTZ{jKYI)8Lr#{RA#{0Ot7PvVEz z6TO&<6oo19^nI8d^+fo?V|eSPhimOl@a0VGh{D@Ab2Sf{zu)1`3qB)74xvbrpaZtt z>u`#FPW=9f%pO8>W_CedrU?`0yn|2bGsyhRfz!u(Fk2M|qvjikSs00E!>f>*Y)>CE zwJ1VnDK*!MlH=SgJSz2qw}dX%&O3%WL3=EF%$?$IZ(xsUHs-~@#{6sFkv>h39(Ie- zTYeAR?PvbM1_|2C@0rcpI1l3W6&@wE*rojfjZROHIOZXY+L%S!dl$!}?jT0_7N)+j zCxMYiNqzka?uZ&eKSmcoT84XJuIs@)@+gM1TH)eeUqly1VqarAPJXLI5OcxLR{z5n zK~c_845I~u!|9l;1UVjM&siS(jgx*N>2w2ReXC(zQ;vJOPmt7+gK<|gpi*@oXPFZ@ zCc%Ny>U1d8n7F%%v!nqPIQi@XB6k|WMei^&Mww&$Ja3Fn2}Aw#dvJgL3`zId``q>y z`MZT_T@ve_mc!_yvpA*m*=ESU0fcn@#1hD+D#*{k^h#NC>b{V!qr?*&Ag zD)ERpYGc)n=*0>4G<@bU$Pcmaw1s zW-}7ZdLSewL`CaFs7624^dc0h698>S~U!zZpDfA@aIcDqlg zWUo#)&u?noO_^t~i|CFr?dKWo(!eZ~a4(Fww+_7b9fnKnar8%@;d{eX1YJqN%f%(w zVOtAvUfWNE@t-5MK^&j`sT&Lw!=OIPfG0aabzl4roCQ!{ggfeCq*l-rPY3L=4OXEJr zeL0w|RD~ma_fHG#fakw zAKvG>QO+6NoA(8n_szVorS>FLXG(8dG)bdhopz5^qWD5SbAC*RIkEPvYJ?Xm+DKYv z2J-@UIO<-&Ooi*vv`xaV1vz-)RRSZ87Z6|a4y{A$V9$M%Q?@r_-K4KL!@1l+-6ufc9US9QyMhM7P; z7Q{7TR%k2wul<1i)ZaLDk>|j3nWM^i#naXXw137vl6ps!{fqhgJj3aFo{22+bCBw_ z!0&ImFn?%@Q>8A5Kj@1#;}BTZMPc92yEyUaKBib?W4BEK&gV13?^QWck5;iJ{R-_v zIY&5|`z&1RQ1!7Me&Ma~e!$woI%m2**@ngnm{RGX!}KC;3yt6V|DErn$xQnLe74;L zls$1G!vwEF4RL9+6}0}i+$y|SR5le*;9xreUjfL0_=q02^slx$Rh zJrshGM$R}zrdW|^h5#{ptZMc|>VAK$bq<2XPI^9Aio?E?7hI-or$XjHh^2bB0 zKbwoSvmWF5@M3(=FGb$zDiF_rqhGtw{E-gSB4$NXE*jDAv_mv}$`;zntfn9L#?rPJ z_N(zewM{kvS1cVMA#ohj!>rIg+XY|Nd12ame@s|?3I8n#L-*JyywZq)HhWL5#V4VB zGS8EW`0V7F0T0(4d|On2U6mzh<=li8^N_9F9cfRBHLZPRMw$zbk!<`P`Y54BE0}#$ zxs>O&?yn%V`!=+bnQ1=S84`KifBe+}x30LukwzQ)f01>r_-7>s8|!B+e>GPGlH zvosz9R*BfX#+!)->&0;wIA}wc|cg{6y|U6fRdj#a(TWJ#P=C*5oUs|jl@dTo7i?W z2FFF?U~q$dHyX*nQs!VMXQAjr9!_?<(GGcLd`CKye1bita9_JcuPKcjcZ@8uHONF< zoz~jSrM0XdT$1^Y4-t8AQHa8B+Y5N@>VrEc&mz9o7eUNP+;=z_!D832+$tJTQ8933 z53$va1YA%`Mib9{O7~`9R7eh&swZ>U8= zCEKaAV+E=In#yx>2~tgHzzRiXrw@;T$<{DjIS>p9onYA9yo_4O2oxsY#Ew1j$a|5D z-plvdqnVC_+*22QAsc)6eSe_z2_7AO#&=+zi~6|HwQlYc+v!Ade4g$Pu%e|;Oljkd zW7KC`ZJ&La*b)0Or3Z1eKi|6g5<5>@74QKQk&JIp@Ak7YID&KgV zcEuags*uAZ<^Lb4Zd4=RvCIYD&5T$d_WPu@pl@;&%*qNep*$Bam*qm*GaouDim{Bn zBdz`~k#qhnKEA3(>k;;A>DF?;!Y3p#1#ao5X3QJpp5+3bot5%Di8ZUstR0!w*wTI( zD{5j*GH9tT4O2Tn>b_g(+p-lDHjBGIC&*CJQV}{k_BVX{>M;P$Ae+DAci(#qc~pn5 z_ZpzA`V~jV{=lE{?I>3Kh4024FpvF>QG9PpE9^x8uN(53bxA4ihYfb2GHoYH)3T?D z+(Qxb&VuBl3@L0gYh@w3Dd5^VYU)uTi>_%@v0R2s7E4gVc|qEDrW;~KoruWk1f>5$ zd1^njE(p-T{votOh392XLR3-2c{cW6U5a5ZR)Yu?@VW|LnE$+yGlZh9^j?wkuwD-I zZNP@)*YkN`h$$^>;(pky19Zu68|CpfXLWM{DXg7J_6_5R8}A8D;*?}0Lh1b9w7n}r zJ@>_E$rlN_$sCk5i$^ePc?3D}8F^D8vq?O7PUSX=KJmKgU%6A0y#;LUp=kD>scSn@ zCfQL8dr8l^nA64e22`{7Flp@CO?SjMQ0}tjw5)$FvqPrPk`7s#HgqgqKRud0EgePk zQby74&63>jFou?fjwScS(zJJ`H2wE?EO*l}@0xv&_w8ipC$IY->#&t)+(;^zpDory zEB-jr*a`MT>#S+Roa0o$9l)*ZIV_pAkD4mB(Bw9drsHCsH_l?eg(CTEn?SJvte)-ieA|KGEYv8N#@-IcK2 zg|c=q^ZJTCjqS3gHxn&sZJ9CkzR{tfI6yNFZ>J|e*N}J9QhFmdkL=`>=$sPgHrd}% zXC_NicFNKueK}HZn?Rvc<*7QFSsI|{%{iQVaSL7P zl8y^mnX<<5$ewzp+tA%ZmeeR~LUR`$qct}VP;u#Y?)hCyRfkn6?EE|`JL znn0}?veX#B`e+#EVC5##iyiV5UnWnIcjW0l|NHsn@+5j$o|G_&x_RC8ysl9g`;xL< zsp>x$`m^1M%Fo+VRE;(LZaq%=CynX#Ivq07-cRGEZlji+Y81lrfvm#Wlj+pG>($3WodJo9Id4Z^o;%BCmQ7FVH<1GMiXfDtcm~A*TbZVFZd!coMC$$~ zO%1bIYYiJs`>MI4TV0B~h{nS}D6Mx>NobNWb z<4#;-&ctRp(*`X^s+?p;H?yoLvD%CVmGr4d=Md$d-$|pVuBA>X6{?AyPNl4AODjn+ zw`(}*`HRsk3sEwiCPrVd}rjpgMguQ##e+M^oi02f6hknx z%%XOq*aUtjSvt|`|Lmz+%$jOdE$E~P^RDM>(bYveX@dvR%Qy4Lx0rixjt`?-LjSN( zy^Y_!pYd?u1L{+2aX0ZZx>q%$DC!$BquMbjHkY=37(;?SUD)&UIrfJp z!Nr&T_Qsb{Fd-E60axI0DgybfHxY9+9wRJLaHjkLA`~+5>O?luGao@^X%V&>mNMJ8 zioJ+#bk>6R4bGIS?QtakG#mPM_&6=UZ9oY*n$(lMg*G-Up>(Z@B+tK>QmYEoIdVtN z!4SML@rO&K4;;t%!CUS;MwJDj+UzQtgt=SfdJLp$nD4SQ3AV-e;BA%;@9jCT`owHx zdWJ~$xouNr&HJ4*NjBP(!tRqK8EZ-*3OXd2v73(Z+~Y&Ubn?F`PTu?8BSJV49?5~Q z+wO&|6c zZ+cjn1-I8m&k9VO^n(l#YSk~Gw$mH#%;(^m^=f*ALA$$mr1oOGOB#{Rtp z9@t`GhiQN9aPWc~+W8Ef73hasM*|Ul<&AUl;*J{=+6e;|NAiW#GXMN6BhoyL6 zMf(X1s#_v#rai>gd*IPlAC#ZFfMu=(FzGp{c59qc+3t?%HtLlW#(J-{NLOx*mDjY_9S*f046 zT3?@`Gx`OrJl?_MjT_~fyHH)W9sBx^Q;M}N_hjy&S2C+%BP;g_+@?Oo#IvP% z!2aA+L*^njf54c;ddy_D%CIPB{;t@Nin%FSA3IFPM{l9Uxyt0AHIihl-=M~k&;LtK z;pa#rT)tw6u0(6x(m##ejpuRF=_-shZlhcw850CDIp3ZS)zc+7c(xo5d#g~k@eM@H zsxjnnEyj9%fmvb;gz*#Q%&MOg*>7rbs<} zOPQarmN{(A-#J6jiPf)rkj}cZxPc?xT5U;>a&)NW&vyDfWHIIK9ZgnoZxALE4$*W+ zES;(c!S@E3G{qhV9-hTy&Kb_zcpEUdk8Q6W!Cvh-HvWBy{eCt0wCD?K63s{nXoK8} zA1FN44ws{y7;VT|D8GJ)FmJblxI2Zr;LV?!k@o#Vl5RcdrCkR{-WMS_8N%QJ?O1ZF;6%g2*V%!o962T#LKD3WEJyOq10qB$$u@f&%< ze-U=88&Z|Mu%A4Dm~Db2QZGa;zg?NlY)9>-M%4F8gF0)L)1O2cim?3%?}F=ibj=C> zDvsf7yFRWSaDbhh9};9a7c`i_%!54i$CRUaN)28*H^5(`4JRM9W0zhR-c96Qt8;y@ zs_ci~^?$JcGk_&o0(6eE4yk3rG?LGdt|M(of1@62P1{L}=2HacPP#M-u%R>ni>H}k zwz>{Bt}(~J9#2&Ky@dC7Zln51ChD@DBX?2_e3?txs_+xD^nSyAZa21S^udq$QS-+R z;=`6f$jJ@ji|rsjx(m>;CPA8TUx>y`;V$VCEKjSX6mU`f} zkRGxn*;62I2a#-FI8ivpA7mYfeHj(eZ?Q1B2~Yi5=a|?HS<61iuH|{<%>m?{;m$d$ zLBt#xfJ09|wl3*ML{dM_Sqvb`Q-J2MXXNN`OA6s0oL!|WX_>Dy9eeu@gY(01;_FFl znWuyDm&W+A$sIG#2BJUmHZH$^h%|>vq_aO_^h{<+X7%ES;UE@}0130V%7?Qr_GbpL zyOf`yMLZ`o_=`D*x-jR)Z_H5bg7a;jp|v_w&;oN>I#h#F0vAz>JL_Lrj}iUTpS5*k zi1%}!_3;z@?m2@DgD|v|v7eyv2~4ZsV|?8={4?!EIOiDeF~fY-2SI8v7a$jb0UR3J zhXLomD4f*^-^w2_=dPvFW3A|T(1JyuTX1usGhJP1LfsR0khAeD()rSlGjEfT9^{GE zSUspJ=_5ec5t5PTkT)V4mZviixq^Epwl-i_eHW&$WSzm0eF}Gl>3qr%`q?pvkepuT zxp(2u*mjt;wc=ApJ+dc!gx#weBsW*1zU&>0Eu2Vxt|6`6vWcSaaj%SDD?A3G@yphc z&k4t1f72L)rtWwcb_upCVo~b-2n{iB(Q@}2Ug!NoWejKMxQ{7vu?U&94I%N;f6!3* zgObi4u-w~>tNT8(j`3S z$S>wTG~?RjI+Psej(4}`a8xbCjLKZ})nwslRyz7R9LOL-hbFeHX1*17H&&IxX+8Hi zj4+4Z8C@g87cneTZ=DrQo!#6KJh zzlTaiZ_NH`jGrPV7^?4zQ*P%`W6zm7WWbC;A?1NaDqtsW@yZ9cQhx zZGIfKDtchjZX+DDGQ)&ScO3j1i1ZsV2&>9r&g4scJKuzvPk$p}1kY@Pda;rjF{|f% z#uUzEo-=-e-zr&HaWRQ=(YJ8eA)NC`q41b>36CUM^RqvP8{Lj%ENwtaJsW7%6$Nr? zZNV?4o7j@(igDV8@G`Q%tGk|%ydI3gmvOkrI&#mFw@_NZx%P4VIdcYB6Yj@+?j2OU z_YpsqRN&C=e7MYi0MD1PxZuLGT9HshyuN^fv;OGN^1=5+=65jjV~2n~^-HfO?e7z* zcTGM1EV+)CqE5UI8=y4oIJCZ>MwwP9e*L(Mp*)k?daoLO$+yQr`kcd&y$r7}2~dAqh`6usvF6)%?7ZEF;GY7t;K~5h zbAQ3X;Ulg*D8o;lXC*hLpla|Irf6No)#rimjPS?Fqi1nwtrv#h;Tg>M;+=w_C-3bRK2<#LhelZxqev1tDh1(7A!a6#`fIv)q4{LUpL>p8J6z>s?e zw~*9YCA$Bi6B^%Q*`INW`#E&cy2}u0>8|Lj=Kt5qTiAXm8%KCfKTo*{Th+Q?kimT& z?f>9j-UZ_eEr=ag1Fe_}2*z_4iApwPkEPkjtZb>>0fvM-+@3?bL4jh+uD&^6~Qw%A|A)VgHsaw$T|-uF1C{}uA@ ze`5`^7B{`%e%|tL_%ovcXI^~39?smq8(WFc>Zcg$S%|V1k6`&F7pcg_kxjWs8sS3T zMivxhzMuLtmr-MwBqcak!tLiJ)bTU5a{5tZ%rHg&bWbQO3uaF$Gq*X24=XoA#&-2!e-%+xo8Ou4NGrzMIGaRbX^_Dq4Rm`GXS&q*C%5eN)8G;(hk#yOG zG%_q{wUid6sHxGU&pd1T+JIBTqhQc#2Lq3zI51TowIiL;dXBxY^-=$0)&2>23Y|6Y zv5NidhdH-?!0!j{$bZAAO73`7;yv_3JzlK-%qfdHbX=`L+`{*8Ir1KH4K)y7T8A$( zuACFJqLoqFBwf6bcE(R9&$s#6*r`EgpoRX;t>xTSBh0?Eu%^_C?lmo#`Jx$c;GAY?6SgTfqPC)*vuE|_C~Uy>{mn4VWFP5i zSMm_CA+f)Dv_fJh&39f%A-9Lo`uR^0y6_^LXPcp==?Fs8jPX^;9bVfnB4!Quq^doD z#q}o`_+EwRB_H5^;xnFps)wddBbt6SVb-1&SafpEs){*6aV@YjX+=gz8^(_LiSq~8 zL-pB}3|lx;)@wwI-|eN9jXaAxDN9y68gOO84ZhDgBVAPoX?nUyUTTY;***~Sxq>xD zaX7Ik3&R46;eVAG`piRkeftAK$A3c4;d;EA*93Es7P$Ut!MyxdWNNjcc+U?g1axpt z^e<}XxRGw4JwbtQ?>~_ErzN-YQ>!W|BEiUao zg98e|2z$?5_w7k2$LXt`~)I2|c((7>U!6z(psmG*q%`nLN zhL0v4$h_o68&^8fnp;-%_M8DpDeWWMb3~K5OJQzY4=nQ2FeKF%6&sig{r4EIE15xU zog0$v{BY?>D31Jh175lbuxfY!sdYJcE>eiwv!6kFK?PRiR6%LhYuK8<#ksTZp~9T1 zAKWn&EB+O8d1jVC&SW*(hTc_~(BxlQ^y}?r@_L}mos1*t=k7Nsof--KU^ncWZ-|{| z*$XCl5{<(+bA9(5#!Fs?vRf3At76%Am;%vb8PHG9#gu0S5ZO_Tvb3idBgGo%feMt> zy~N0^)!esG2ZJ)6!%Dl549{QvT`Xu+8TX|X>?V)IRW#L1f%;T?k=K-k<1f#{SmHgQb;u@R(eH zi*d!!?6jD6@2g5dG^Du#PRL&E+Jt`6MAqoyRtS2I6Q@IDT% zNrzqPLrg4rgt>{_;mTTy(pFbGx!#dE{nq3qYDNv!+GOOki^>C6(SLgt>93a{?YUC~ zmDgeTPul}e@+@&>jRh9-oJDTpDde-a<%H-tY>&BwxiVqs&xnNd=Np(~avPsUu}@7T z4pWs9@Y5`X{eJws#AmVQ%}i6)Gh+E%nP_fHaY4sPYNG+AcpoIy8Jj3jejyq8F~>f> z4a-H+afJCC6#?#8yTk??(rojt$hRfyJIlyL@b7D z#9_aBJUZ19pkSDc*+TabGBzDcc zv5k)8tf0gDl}LHfNZRnL9Xa0Rc<+>sD)ui_cihCK%QxYd7z5n7i|n=(ycOZO*J@_{ z{K>!)+f4X;%0h!v4%FC#nPpypK3=!@Av0AIT}i9ZnVGnJo?c^1Bbf1Ky32$f&eNeJ z&ZLe~*+A<@snCL41scf=Xg#q`=xu+8E0c?HU|cT5)E@FZHyii*AE9;d2@bO_Hd(h6 zfnT|Mu(1>e*|WOby#kW#pGx(9h2yL*_*QWCxR13XW^P(aJ8>tK9eG5YB;(iSv_a8; z9Ava;zur!oDWXO>{_|ruC*0l|;zF+Zgq^GX{~SJH$H4_lDJ>lW3r{=3wTGZZeg?S~W1V2$CnzYQ5&Wrn?| z2}xV)F!ysGrPyqyK9dzR%VjqGnZ=sQ?-7(;Ekxs(!4Wov{naYJSi@ijLiitu&gn(n zf__Z?kGtsh_aPvZxmm5fs9oNV8a{)*;dS3~25Go1KUXNhmv~?^EKvk zj!}kr_hY%+!b*(hRfy6y&M~+RijuG&_k8fW`9Iu9-IIB_7x;H9$;@1SuBT45Cu{C7 zZ{nUJxl26Do32d<+xBwj@D@r7UQJrF7Sf|;C9(~fNLOx1)1GWUWjjSF?N8XI0 z(Te=%pOO^)S(2`C-?5G6Xu8!gij>BTCR^Ll|I4MWV;y$g72ZGC1NNJHATGOb_eBHi zu(~!hJl&E`Ts2|#?lIbLevpcScF?wtwdBfOyc%!kQWG9C(!V=%w%M%5o(>G~BHihJfn z7xpmMjA;(J$Ik1pa72r^yMY}-Qn#lTS)0^>>lRuu`3(8VI|D0PmQ+F|*GoqX=*Nu0f zAN5X@V8h+Rb8YEN@d*+VHY2+PJ&MaXM4RdH=6#; zl@7DkaMIa{j9%K4jj9cG9krx1K@+O|q{E$Z`>D)w8x0w)MxyzPsnBgE&6HFi&z5oA zjWUL=F(*%I-Dr{)3WP54H+v& z4~hin%hDdaT=<7|fxlQF(+f5Ae<++Y$X?DtpmYExrv@;2{~*L81Sl(2kaBt5x$GNR zvw`)ieist^$J)NK1I^!WLu+I$DP6*t+_I0dZnlR)SFdN^n+h%TnZi79NvatmL~}p2 zL*P(7+^&9r_3RoL?5o9k73MNV>3U>+ ztw2CUA#{&2i}%qZ=*=qR8AS<}m6W1|Dj>S068i=#IkQxS?MGh2eP}iGd_G{>|8aCy z0ab2Y7q$yqP*hCpqu7a#se)bDSbz%B-Cas|Bi$e=9TF-bDV+v_*ouLQ{m<{e@Zwyp zvp4U4*IILo@eE$~0%sQ|Kd_)f&&_D4kO^t7GN3OHwdkO!GHncyr8cJ%B-6EvB79fT zqz}`njCIjhgI&mHZp&cwZB+e9fQDE+To2zudU^`}Acr|-w7l4ec`ABzAW@x`_Q{jMgmdJrc7QS+H`4gV zIdq;gWMgBxP??hlPtLun`US!^!53~)eh_#Z$X#%uShXe^16SkWJ1r6EbxFu8O~LW* z+tB4R9W5?^QDQN!vF<$7#ga+~%t>vrDftK+(i85U+ES-XizQ^KZ`5&WKLEN{!oAJqSh*(qx zCP49O3a;GFK$8d0&VN~wg&ccUWXvgtXTrkM^hx@H1{uv%AVVc-azC(-X1rfR(ed2V zDD?@U+lmp?7L1oUj>y%s#<(;~_VU?4=q%6vI-Jo{g{-`$#Le`;BXlaMTe0mgY zf5u|;q$D)+-$jpetJUo5YhS{AhDi2Lm>7{|s1BWSQ6Xz(SxOd_BApO%Dmc1;I|PQ4 znrRJwRYgFg&H;lRCKxqaA5(gbu)E$0L$vI$@wN-9#(QIJMF0-14#6+6a6~#r!bCm> zW6k&(;O~2jeS6%M1py)hP%DHVoaq*f|J*Yq@{iVrq_g-4oy^0odXVZe~ z-T05YkM3-Afmx~%u6@(Mp9XChEj7jBTw9#hcY>{oClq&b|L}!Si16Kcs&Xt^hH-}u z&ynOLnPZfm1&hK0tns;vT=uFwG`65yH`(9jq(=>N)F{AKj!aCXXhzUBTEA>AiQM~% z*UCj`96Fsd3kj8hqppu#-7D()o^fs^^K9(*q%`6^^w-Z+F;5M{N7goZL{CuTyplBzS$g30;|Si@G7!RE8%g;FbWiWXgjd}V9s3-jTV5dfIM2$) zyuqWYHz*5z!)Jgu7?$z|I_mE@2icFqoUJ&dXhBN6|MYCRLQ&-hXs-$P^2spwkG)LR z3%S3nMin>aDc}+7brtLEF=ufAoKkPWK%fwpjcRf0N+(9=_n^k&1(Y_v#0`O$=$ZBs zhkx+Al=EJfl%Jz2qX&{N*c)ci!`_bPc*t|KaevJzjD79lA1=|>tlfO?pGTsLKA@&8 z0fW13@N4iI7T733aE}?X6`wS0q?ji0&?5YMw2qv_@FUBU0JN#NH6+9-Vqb+3WSyL$6Ca9z@(gGWK0vhCBScHR#hq&d=zGr|v@iWo)#=03 zxK8+Ma_<1okq`VRLF1cTe0zQyD?^#6NG9)oc49)yz8N36Wk z3yIyWocpiF%2jt^>6-%;-4wJY#KNs59RD5$BlL41?)m#8<()ZM>1osNjEhwJY$wfN z->lr>w50H^NwL&3(JCO2u4Wzae9;oDB9 zhH=+-$P>g5jeyU8R%p;xg~1pd_^DinhMNz3X2zlROCDaoszP~a8@}f~#&4}ItdnZM z(bD_SZpwjQQ3_Tpk3-b=Fq}UTh$F*%V4~>(w?l5|f9V2)6lZ)A<>yL8gU=Xe=%v6G zD&_loBy(S6l6l@Y+Y~!@s^Z)@eT@HShpn&tap%u1^oSNgK=>gRsdm97q8t4+k6`Ft zkH}{wxVk3;TffDlOPyJy_UseT@rBjSt=NKpa z6q6Vr^ zS=5c<-}RWnTA1|K6a;OK#9n3!-(ZGy%y9>d$g%-@ev#;F49SxQDB7iqyS6+JU9Cov ze^1c!(d$U|-&pRuF2kl7o*1vIgUmr?=;ZKwE6|BMa)Z(NISpbJrT99c6@QoY;EdM` zI0`+(H=boHR~KVaA9Gy>qL91qCT7Qa;>{K()J9!L`v_}@g`4BJjS2j28gdV>8GQ^< zqtGWO>4NNfDsvN{ymhQkjQ4`Rj1GHylyGst07@>7P|e^R9rtNS#@>hj1opDoJcG0T z3!LT|S?!2s9MQOkpQ0J~&N-ZX3dPP&U;OcK!;1-yurcS(6B%24?6-z@gcVvJno$OK zOaIe5P3xLA(o2_#bY*cZKHB+U>v>&hs4Bv0ogRkVvB#f)K;&5_!|MWf&#+#lU;7k0 z?D`NY+=H(7EqL~%3}Lr(Ao4j0No%9gygL}edwr4m!5wp~Tu|-8&xkbNYmPc#A!`dZ z4jL4~oosUsZe?9#DqVWpjDGe*mUQSte&IE2i_^mS?KfZ>=!YTfv6%g~5PP{dUH|7J zh{yCIx%?Tz(%Y~xw-ORd3gPdYfmh{;IL5P5E74F4UmOUXKfY)`?}OjY-Uwt*Mmx_W z^ZGQYfi;DZp*z^KFHDJrPa%Ib3^yE2@R0YT+k9?~=AM>ybzTtZjloMfK4&ar2KC=| zWMp=uG>G^7Z|%5ZRm=12dszG@A63SgSRs>wvjy=ubt)Pf2@&ve4#%Mj;cztzXU2^= z$!KU(r1>Rk57|u(rt@e`@q2jZ#AEz_*7(?@fI)vHsA!r){=6G@-U)|dV+OL^%Mjes zgj2zf5O|q+i03;nF0&qfT^g*Hqw`btQ~nlanRc@;LN<#vZ+q15Q^FEq1&mu^fDTnh#BuIlTQ(UR-W5U6 zt`^Ei+Ti}S6VII6QPb224q|NFY~~HE_EDzeg$QMnwTVh9X8fJh+G$oBAFai zFDyq%c|FdFv?6A2D?+rJaL&0N(zSIs9A1l-)itEC)( z+~;*Uml_eINSC?SQCwsT{k_P2uknq6E13F33CDyjP~72)F?@gM6-s6-9VIObk%#N6$T_|(bF`<;!fUp3;8P!nSAH{$Ee5Tl zuJ;(KZ%LJ$H+mSQMK*jFd~igHh6Zk;jpc&$BIXIQ|8Y;|R7;#bqJV^**Z6O1fcOt~ z=-I)2p+_Pxv@Hpr=43;9aUneJ-NPPc;kv zi95&MV^t7qDcYu#KS!5Dk15h6<`TdEypwDj7m~+F?sVUkhp<)L6Z%aZrp78HQUvUEa%cn;6@nZG!}Q-dxv@^>A2f&?~frvoZ;NoMtTEWV!ub6Z#FKhefh z>1){fRu%hqnxJ)x9X4-vM@W%BdmqA}eJd6t?ULbJoB_$~T!_!VgLtbV$oH2({9-AB zhgagBUL6j-Yr^FxmLy$lPJe2RXr{F`9dlJAt;gr+!Hd0Qvvw7!44*)A{q1mk0P6h1g6Abo5KEZ?PJZCg5~4rD@Z zaV~Oh+=1N<_N$#N$L-gg@7~Qkx0NOoI#Q2J!c^(cY&lvZCPmj2w$atUb4Xd|Cw9#( zf}WTUT;7>t&{h?Dyj3xAsUEg$w!$f1dwi%XDmVME$NVOaf8iWQR}=!~#p0E99FB`8 zz;R9zPA^Eq$eWqCoS4fUZ?Fqa@5|)>lJX!gsWrUzDk~l{t(d$4K%&aXS5L zJ_RtB{o=tIWV-Pk^o#?0WsEV$SO*Eh`grebhRqvnQIh0IyoD^Bh@UhgRkRo+DZzfNQO=n5&!Cta+ zUqg+?LR7Tz6TVC=hQr}t)WtgBbFd{gt~ABLBNps=vw@SZJ#-yi;Favb_Y`jw{a|L{ zUvF%4_d$b_AO0Q-gqA`unlqUZ%{uHSZJrh0;(a*7gyy>$(D~O|bakCFX(q{#psf_y zzUAD8{}Osy$DF69JQr2X<*tlij2-TZJSjWupK=46XWB#VrxQxp-?l9(X7WR7s6W0hy$Si(!OZmw#fSgRV*@yYw9 zS@PO{g3dSXqUV*XC~4CST9U;4xz0|U{ZWKn1$;049*rvTNL?AA{OGJnY_X&LDIpxn3jO3imah|t7tK^wZALp&Madwc|u}KDo6#qeol&jR} zg}Xc*zk7~cz8<35@#1veZ7B_m=iW}{Ud%ej=jX{a(6TQ^|3Dt>rSq^jr2xxc7vb%u zyST^mPtBtx_>)(Jpv6V-?=RxnL$`p#z9Y)EO?_m=71W&KG!g2_EokclUR?`OG zfewTn?8Nj3yazIG^x?`Ds4ZxL!h}|IylO=?ubcOr+3{zXGrGote6`Fdb+`$Y<`_`Y zZXMD*tVWwmJavG@3GI{Kayq0Z1%kFY?5X`1$7( z(!1GDDK>!h?w@gKfH{FR?{VVuJ9KY-58>SRc+Tr)zGLR5sU^ACvbHgVHI-iG^bR+o zrStVD;=KmFo~=ZBO0u*&?-UL0+D}Wuw~Of}$Q7j0KaWO@7N&c?Q)z+c zBog$VNMesB(gBl6wBCOb$<|M#aM=kY@LYi2vM;>DWgJymjiVpDZWZgWgYlMR|JH)W zWtr29U#27pJ&9vyW|{>q7sr{w+#} z=FKIE%2`x+aV8nE&N?Y^CY{ZnNjLrolac!jGX5zf(N#J~QU{1ax;D_Si2vxt z>_wCoIG0v!m`!3iGwGSfOp;~(^qNmIsebiLiY{Ts?#k)3cR1&*jRnc`j3E8tbkdEC83-{Mb_jK_ZZvSA0sDXya- z-OFgqk_DveJeyuE16FToeG%+J?F)AGM>fXcf26&o;QsmI4__ZFopIzPWgY` zxq|m3lVsLmKU>hdd~@>WzpsRkF>N?zK#8xk>4LBt%@3C+F@+0M=E2%dj3lko-a=m= zucW%dg_Myni}F4Tk}-Qt3-?c^VFxDD%aqCVH+C|4KAc3BvJ>fU|9JYr9Vy4}3s7*I z!2kOSn%Gwm9Kt!mY71)BW&fyyDSvK$cO2x*nU)r1j8vh`WtYj%{VbjLI7GtIpgm1% zC`)(|9a%DyIna}7{{^!&AjfeSNuG)Z$wGLf*Fx!wD_MfRq=i}{gF2H*s9WYk*k!kPnz^Q4w8+4 zIHhDSCrcS&YG;ma=HwsT^YaR~B%VNMH~VF`bwY3ABj{&6M#GcG@UY@La#klIYCB+Z zsT2EtcOfS3F;aM4O+MSTZ{lwFE^``J!TgUdBl@pJkBm-flEiaGN*ygjW2BDL38!7O zeZp$?@XV%vwWI0jw0=xI(1xMs9-y#>XIH63IKUp}*OvDnYf_4&C#7imd>;?L-b2>R z`&e|l6mu3-;FC1pD|p>|_3UM2U(k`W%=~;}!aRIKl1tDbvn6UIe)cMNgmW*G&wdj5 zxRLoXobP)*mZphwukjDoaTRi~=vN{d&&F{lXdKq&B;ay;5`?Cu!ZMe+f+te3T8#5F z$I|$on2vGGd->we{wUsmhZpf)lxjhiX*}DxV@$VMOS1l=`$lx zG$snFTUcl6NPsST&UW#-x1udc_NWEXGgI0p%^o{#T{1{hqehjh)ZWQHq;-4A_{AE! zZ^HR^WCqtDGWdmKYGXgrKVpw*It(F>AmA^F453s5b@;d?2gD=Sl&8 zn2q8&fk-GaO(RjTAr@l4IH!4rbJpVKl$&TwV{dX6_oh0n>c2`2(PwDjmLx4$v5GG8 zZ}i18_BQXx#B3vP2#&Qu-5LX|ZqS3LH1l4ItTDL49;JyckXH2M46YC6Is4&OFK2UK z2jVE_@P?*DU>I|?rrf-RUHpz5F2WuTOYZg9z)b!nI+RhZOhL>s*Oxg;K|{CFqT6#w zjsC#C{5~c-+{F8xws>x>1NjmSO!=gXS25tLrum}E;4YO?a zXF%^#HkNqigw}lbQ8gihBJW zBmdKStJ--%`OB*5A(FPTJ&JgfSb8ji{3pwnYe9E4-`>Le&^b(a8AEf>1?41ajL@Sqd zz_vO9m)}`)o=gS$63Qq^(P#hlb+~xBB?k83uL3Evyk)HWeD9=jnZCb z8kjU;?#EVK*wF>urf%e0y~NYl4|v5sv)&$4`d-7F?a@l~{q#%%YZP zh32YGXbUhWDC;RS7Wbl?y9+FP-s7|L0HoyE^HpO^p3M4EmXxJ=a}QEb;3_isI+mQC zRb#A-Ka?$vF#DMz7Ejf{-#{y@cjj3~UJxV=6YxqU3li*S-+q?oqz9_7yrm8ouQWkh zqYdrKtet5*g504e7(3Vv{CN&t$v1eu=OYZ)^Et!Wh?vVx0w>PXNyFXD&09!HT|Zgl z%ZAc$zTc10#70LYoL#MtAxG?REYk-ewNY5l`kUwVLUb0EF@KZS)mn(|S<5_$AVRa39Y~L+avQ7{}SCsn32JwTKGSf3u%M z@k0zk6>ZRcOaUFh zo&ELLCES2NZuM9@zaB0b^_r+)JAmH*P#7<1tu7j zV!-(>680A1MovCvX69h+kZc?+X3npDHlE+hLGB_8a?8}AtgEurb7&9!wGgEGmjtzlKIBP1JJP+)z#d@JuRI0YY{yfitmsY_jtjKj70t_tMNFqT+3we-ulg4$!DA=M4vNrdhd1!} z8IMniJSW?%2Jc52SbW18Qu!V{H;Tk}!7Th5QV!AiO?bGy6EZm+P#oO|*%1}YGAl%K zY!-H&O-0wT1bAE zCJ~q|YlSlws?d9?4avRN@!w@1)W3?w_tHFc-K|38jW(ENK1Q>37lvpzKz-!9JB}?psU8!#jmo)^QK5fh8ErImHR~$tz2crhvF_`U)N*@PYdTfWk8|-lY^9>wwy8%^k3$neWOFiQ+({Q={)TF$O zUib~8MAl2*KVr7VY9r3Ca*i*>0#hftpu3F!PQpo$2+ikS?NYRMR3LxqJ0Sjv9PFz7}SxOzYm&VOu zmTUYUgmvX&vz06Ud&lpnhuXLhZw1?NZZO>zijOTR$W>#H)yWFz-K&IrId^&e&cL?5 zD1?6XN9ke@tP^v>zUu4nTVVx}8e>deX@F^(+>g0h7kQ@IsCjKpUw9^2V<$sz)g&pT zkGWF8gUGDPgzIK!q#oA8mzAtf+FE0(9A|-Ehe56+4GT1jA&*MbN>w8{_&$6ZSettg ziH=Qvcrwflg4Onbqz%rmH-$xx9zHDBhK!CT9F}R|3D1%T+RQ2BzBXkE$dGKR1PQb( zpukDrpnEeNV;?(0dj@Bm#_J%w)f%>c-Eou|6(J99V^RP=ORNp~u-|HP2Zxg{#WeLRAUVRo{z7s*G2y+Eu2c#fQ`O7@`_dQy_GqA<=QlHo(vsml;C}6 z0f~+O2FqRPICaGlp&N7{azPuv1FW&|kvonRhT}=kZ5VZy;8cGVp6;&2x`Hwsc#*{n z|7iX#`{TqRo>6{vV2#lhp*rUHKE@C;gLM%xMF(zwwQ$8i6Y`u{*fvFnrWDIiLbD`E zb8pmx@Ig2}%)qqfI6=bJ*L%owge*>7282+I-N9TuWPeLb?&khqsF zRV}6sPygVSdJZy=I72T(8|PFt@VL(cDJNX{y9|ZQ`c$l8-Qe!|O3V|k#haoEtTfK& zjC~?Rbiyz$!ymKw>?xSy0*7r5=v#OLvpj8aagr^5zOli^2pc{d@)`DuF0D4bOzIK` z$Ww7SX}%dop%)4fC*+2~03AfxsA1JnQ#Ac`g8HQ(IH@EdyOsMWR#w33Z8Z*Vdw~7y ziy9e_it;HjxHvx)2j2%EgxNx4W_xhAs4FD`#NfZYoX&tDBjp)?!58 z=BrcV!;7?m??nY|w>N|8+cX_rf0G2)wLH#-d9( z&}hDcD@H|-*jWT$e$Hjy6~pb$UD(aJkNXo#aq?&x;*u&*X;h7AYV{cLfj=3~ak5_- z(|l)58u>?#a%UZ(##YYSWKE>%y-n~I3Bu(z6a21K#8U3YnV4gV{hBT~^E43IQnApH zO2f91IoKSR4+Z{A`{oy8=b?Luaw|j5st4FCR)z5=Y7k|_o#dmMG3#yzo=@n;J_k!0 z_QHg;8?;ICn>-cDoupO5;#7Kn7Tu723){9tcr(w-)SNTkO-c|@GC=e!JIEV*AyXw3 z((cRxXI|=#vst)%G7qW-g}6EC9=!C+v2|z_2Fhw6;?7L9ecVZYu^r=hHvi&6AF4Qq z<4|czRj+j^HeQLg?mkC-ZzagOdkN9z;dGOqmm8U$c&Da^w?T^R3(!E&aw{aP;5@9f zAA-+>BS9k`m3b++!wf19_dFaFD~6;LcZ*$pfL9*XIG$05U3VJsTDTo^SYz+f=)V}Yhe+aA$$4`+s%u-Cou+$8EewqjEqs4IN+`!$ea)kFX1M4?u{$-nB!r!y6S~utF znVYx6oaUGr(zQvN)N6N@yBJRr6n4)#3MA8Sy4M z+`VFjZ(r>(VD63<7eADo3c;1{50p&AlnW`?>Yt9j6FE>y z=j;UQus+-)W`D?(X3jLAZ=W=1)LaF6Vs@75vLz`o_CKm<9Z$R2ccFMb4&M|U;8d-P zU8$<*ov4ZDe+~JLXbmrKdyETr#b-5dEI1v24Yz{ef1c0y?1!oe3}enhB;GW{U}!}G ze3$Tj`5VuFe_D{&dY&US8B!mg0rr_OUvq>E?bv#hn%``pso`_z_&jDuVN?AvPsrL{K54z$OM8W;9Idopju8{@ToAQA3^0iK=RH1Wp>Pxgn` zh2YcvaQyzkdC9Gu)f_OR4+o4XW{n=5+@wxY{#U8+!WnYbmZS$mR?!=e$@HbF58pRs z!d{6r)+%ebvj48pRR{JD3~<-j0tH>1V^DX%&tWczdg=<1DQ;-x@8DjqE5=&5qd~|E zZwh>nZ|%o_3!c}hGSkh+oC2?#(Ccjm+wiAZ$a>i+8 z*28bQqomsdc|4!=@wB9A%xNfeG9{5uhEy~{mlP${XdQP?1>878YrgNHgxWQ9-g*Yz zwEcmSeO1^xE*|xpz0n=ufV5xM69T@m@edF5nhq-8o| zs+KD*?s3P^O&*BhdF+Q2OM;~Zow;g8zb+Zm_ig(0)Kil@S{12C;Sz24I!0ENJE-&5 zQmRy+#2Jzg@X9R5;a|5fXm=Ce47_mRpF5&odBF3OH_p2H!fcNpyjS?*!BTI0$>Cm* z0#7Xe;l&IOAH4q07n^wQEmX!i+$IaUFUWH&XA^RsWJm{(=ulO?D$&%d)Ma&+uHD>E zt&2C2SkinxZ;qq&0dG(uRD}+nleXr^VgIs7WUpXPjejJ><)h))AC0j-Q5YB#!87Af zXpCkyPGl%%-3f#JX7-%%x&fTo*fZUdQlri3zO^YOFgwiyA_8Soxcs2Q_g--q$$0tn2`MdhhnYzWK4+!6VhXpswH!z^r9 zPRGr-G+fh7!|<`UF>?29l<>NDo>`LrLFVTPGjnr4YgXmV%|CBIvof{GN|*c7*IgwG zA8A@Ie~98s#p%7u3Nk4WrY46mq}K3_yKJ6AK&S(2emCL6hem7-ZQ@>;CQM>a<@b+u z5MENvGtCFsyN|v9;uZL|umVqbT?c+=4d`&5PS=9AzceGwBF=epAJgq>UCLM0pu?xw z!?KCz!hI(w+H4QazOjk=hAkzZW5RT%NPzXZVO0L|J0ebg!lwECm@d{2Z;5`ay!Rfl z(r=*0{`xSx9z@;fM#{sdFqLDD60iG!bBmh%OfSf@pufxq&a>co{~IH^sG?7EIS-p- zt3oH8uh9EAX}bFPFv&h4(%Zg{ii8)_dh3}qk2SKRZR|6hJBsf4j3D2WBPjYHcWwp^ zr(r*aQV%ob40L{B^W8xlH2jX^`P@0d>k6|Dn;6I30_JX-6q?iE1n$HwHm0r+1F}`- zuGYtD)Htl$lObU7TZWBLX2{Miqc_b0MF^fk!s89!e+vNUVr89KP<5Csj{Ni#=mr1%N{(c6NBlw~GDl7_-GLw7n=)CrMz zq!0;B5~90d)99@E6sp|AI-&Oj)}p!NY~1+&*J0iHT{*3UeL9>oi&KZLbHlz z(cs&elpZ#d2L1_?z9;)Vw1sHa`e`&$X9~F~P9`7DVgGO4_xyQ3aQ^Ada!Zn1Y(eJy zd1pqN&;aK(o#*ON?Q%^@yst#P)pAsJhINOM!;~Jki%K4BqFIktGPh$9?G~9!N_{iQ zGhqhRJe*E)`qQbmNQfqE5u{4)zHX|XMB`^pq#+B~&lNiU|8?gK)>QKInWH|#l4OQi zkeN1j*6lN)=2M2`evaSOQ#2@bha%n1m!ToSr)bsg1GMGL4$@CpOAjY3rMDaAQkBmP zYCbxR>^S3_z|Z%L-bu8ObLR5$6X>a^00nZNoTmI}TH-T`dR8*?oImds-h)H`r#pYM zpnx)SIycdbLQ`39V14D&R2{02P^0K#dD6IafihN0(WX!da=*5T$}N^tT;CjWQ|D*d zeFDvu8cW5*eSdbNh(tzF>(Y@_d~rBk7aq#r+g}*o_=VF-KT-Yh2R`$!i z9A|*p3soa+Mz8K0lihj)`X!}JL(^607UBk^N--YZIBai_)yilPQJyor&+C zF@vOzJ$Cmw$58~O+lA==T8N0uLKMF&z|gO`%$~|d@{deRJjbk?plk%*%t08hJHE@3 z(wA9MH@~YE`j}FflM$Ubr$+?=8nm+d8kt?YNHH&ulEuFrRQzr^v%9BJh2tMg670md zm=dh`nue=iub?%KPDkNc8WB+KW;w=S^`}_QBC}7W zhS~9)-T%0bvq<44)H>Rbmh-GdXs0S`?w4uHtj(+B7;4j8)mxWZJj4hrTO4eu^$9^ z-7lfsMZeX8nlnx5xquP1eAgj`FRC>9Ff)9_PLt6c2`ZCWMZa^W(tl$=AT+)RYqB_J z8|8#M(bn7-V*%-zoS|80i^{upFfVq5)B$JCTJa1;!Wmz9mNH+|4UH!~ksr$4WmEjo z!RN^;Gxh`*@w_n0gyPQ`&}8I6pxrfZh z3I4;)(c7(ub5jf;dfE)HENwV1ZU>`VP8ibQf@E)3%>L<$wc_qr{lpW`ulvI0P#|jd zbKZpC%@$JZ%~@tncWsTSj%PcYSEy4RyKoK!o~BnK5+wF{1sOl!o-&uGSZb1j?@eyl z&}W9Z{#vj4SkX?Z~}Eb{;Iu1xu%oz4LEgetKUvn1EDHe|H_DGql z3x!Bkj2Np0HGLCE-?K$ni4#&=JmJP{QjL^QaJm*6r?_7xoAU^6iC7Sv3jeN5n4~e6 z@?Z%{_+7Qrko_?pCiGTEk77Qk(&trjbmr|bDl6Jb@jpdKdC?#aOWuL}B5(XWWs0Ex z)L>(vj(rB)|CG&q%=1ndKhXy&cP-&=rTQS2{V z{(zY~b=bo1D$#NLe&q9dx|j~1*_CLT^?4$!=$TCZYvn zH&uv4>f*>ZD@@qqh@(lK*#008f8!%?`eFi(TBqTpVK#noZsW<#B3NCx2l1s9_{p|faIF2-ACS4s++;8LC{ zVtIa@TV{r(|JieH#sd=(h&{|-YF?Ruam}o)&*GehY!PN3xDP!(x9YB|M*TeIY|FRe z2Wy1qE&4F0il5~-X4G}rfG($~k?}cMT07zhb&lIW)>8z@F|`L|o4{-^d)S(5;qiVo z%r!8;oDa4z{pgC;2mVk=2#@AF1&0T^ zU@X;(l8tY%kk5028~9w~uSe-R%Jg{6dD^_6b+dKLX@2`C(t2Krk3|7kk#7z^&M~e$ z$sO7rmT;f!gu4UGM4A+a?ffpD9GHT1*4S0Y7hoOdgceV&z`ulQJUa0ZPdP)8ez^_U z_82RUJcsa*cMuHWGnIlVsRT1u?X3d!@_8b4)DDszK9{;(n02@y3x`cyu%cfV%?WCl zQ(%A-zxn=W;epccK`i{7SWoO8@T)Ye>VomL2?UH6!2R)M0IRp@$Fi_1;*kWFqz z*qaW#z1EGbZZ8qe`|f&66Y4e4qVmO8s5#>(1*>nM(>+ruvFS0oeWJ0b-xd>xXkwMW zCbZ%#kimCN`QLu{Qyz)Ox=Hx5ECYKNCR3lF^E`8?t{ao=8Vxd8Cqu@*`zavrKYG1m427^4;n5p^-b>8bPpyur zvU>QGV~c@}tkdx9B9?vY(R-NjZI^?0=ki#8%ftPUe5|v-gOM&p$m}l0QPw{e+AyEM z=sxrt%5mvPHU3?!$HjH*zhW-K)mBydyZk&|Q`t?+){4^9&%YrsDxc?TZWxoOkL5ZV zDBf)X(|r!uRO5@C+RUxnnu=8ZZ7$~SZqq~gn9f?=>YLe6<$oUsve8t|8Ir}>2B0o$CTBvus{qG%Ci!!DJ_K1Oy?vUy;5Ae6cToynnkV6!?=2{@ zm)>Yz0y+-eLfX+Jy!1`M<8}*jlrbQ)HHviY*FFJHaQIC z?ZJ5bDF|Y_f?*^cf<4DVaW08{W|#FTbCLp`sX9u5Hf!m{u?b|nq5&3qL5O*2iSa#} z$i8jJoqu*{808B;&uD!6l8%*)g&6nv9&S37U{h}nq&1SUKrIGOBsg=sHyExb17PUE z`%#G}Dvx_$_Q?Nbdpr=%dT1}7A@^D7(V^8>>EZ1|G)-U?34P{n-G>$MeBp~RxuzH; zt%YezOrdtk5sd}`kkg9C(2v=0;aP1w>!IV!?m|#I8?8wRXc!2?3;#f@PVqs(a1Z2^ zIOF$C2Q2HhL+?gAjQVl|$?i9BZ>uFuG&iDW)v9E@;5^Cs?xuU(p>=2Z5c0;=cOf^-Fie9dkkEJ4aMI z*dzPc4elJZMMZ=yo(NsX&iL!_=)8`WcNS#z!+`$FQKEUGrzv-*IDM6x#hu*mu-iTr zCj*?(Zf3&Y6VF4NuS0%_C#tMNaf37bv+C19YcsKbcQV%R2#13c=Lj75b4lC7(C#`m zHCjWc#R8)`O%XfN6l89SJ>N}{>R<{h)^W5|4X7?yk#uBE(uc$?H0{T98ojX(29dWg zW4w;1E1AS2t)T`PdI5gprez|0r#!o zTWN|59R}Fcsf+SK)+Bf8VzL73zm7Uc&bOeQV-4tk-QfdbCrGuQyVA1+X-Ggfjva}^ zS4Df2>@tFEi7A>F+hc}=51eEop;wWD!;#sTJuC;Gm!+VBvr)%?x#NAc9h^Q}BU9cS zBIbtpXQzWb?1$S>r~%(!8kja*6XS@+?2VX_Q>h?LBO>rTwKJ@u~aS=4(8&+-Zbn@4~K4yC*`-0miCWP(YmcX3dwv0$j2Hi#^&!Nh>_gF; z6jW~&-b_nBJZ&L)?xU`l8RLX~C3YD8#2Vp?C2RcE8skH3pgYY5E@vHiPR@~&FYn+_ zVNGqBG>O+ncBX&VCpfk?3pe(8qGqNg7B9BJXd4e0NGA5tie!A!&&Mj^9$a@2Ptvy{ z^e#z9%;+e%uMWbJ2+8?6i1)5kviF-^VcXILHV>SUZs>vo&0L`hSHwJa zx5`&jyV+C@IMj`6Zok5ptUN?+@kXMWC62VULf8>ElpPF4eAh&TEz86BsuCP{aS#?Q z_hZG;EJPiP!`Y)@IOZCRoT~x2vegeGm-+~s!5cHCdn06!58BN3m5x|H+-cy%uTj<< zw8cQ?5bLNNL29p4XSdOxQP8CrBWLLaCl zJYhe^X63>?Eg6x|VzA035``||D9jE;?n0^M@$Y-F`B@r5uKQihTkMY9SUeh5CErpkKPqNH~gHzD!eHuQ^lDSG> zTZW$9Epx(+{4Ji9y(@>Z>w58X*PO;U;RjT|Q{dSPV;F9=g<_f)zRU?j6_auCeJ)0A zDT3RkVq8xykY1V`yxW$E8EF}~{U8JHl``>NH51=XWa2g`naIJ#2ClYEYQPL@@ug{*fb^%ciV`6 z!5|-9ChU_w!$Q>h?ZrvYJd8Hn3k?7T)LHymaXK=@JU?Lsw-XYzkrH!F8n8Yp;3`31|2m)t3^&2lY!SRFOE7xpL71i=#_ADA@#%C0{+3ojx1btM9j>5> z+_}R%6il9FPBW_=>@4@O#b2j0@_8R_{q+YQe5J#qEC}bnSYe5=5w=XXL5my@=v@p( zmVPXjzY&(Ge>TRYrWy*;S7u>T|m`K$%wYT ziyx9LTRO#_rQa-AV{gFQJvQ+8sQDcHVHl0}w_wwhit&(l-BIS*vwMoKP|%^EjiD2I0=ShA%Y*(*=LfTvNeD&FI-wNYWcy-5JSM%nO>D2e_%cqMPa$(JJ?0qKhq3!biUu%Tf z+eIUDbi)1*UUj|pGl zB%0)%6UN#lB)@iI6D`T%?69NOb_>=CXKZ_p4ttj@r+vBfXa4QUru%+il^FS9!&F z6?hQ&ZH^&j#R>G3nP@<V*X~n8I;Vwj{+Chy5TW>5W6%8PZx(HIX(tg+!OFx7&7roQxGNn6~l$M zvUO1|3Tg|m^I0*0XexeURih~U zst|vadIFY~B%^mk8ovLKjMYS$iGHwW%ye7M>T1E&O1qgPT9r=fYFdt;#hg|{sdK+I zV=g?9cUl3WD}wM&GVX00<(#Bx3Kv-6jfNw7o^XTn8gC5zPz{u$>hTv^QbF*&U49uz@C*7BhU@MAj`; z}f%n3Dz zGxm6U_7H~Fj#lM3+dTo-Z^(XW=7Isi_OSPKfNvic^#0?9k;$~mT|3jNiHZS2$ z+sRa!+LucvHfM$2LwwkM5D#QVsb&-iHw`a%^z^_|;p5CZA!p8>J{Wk@8%9q(@nD-f z!Y{j`h2)aeK1nxLKM#29@IW7#0|$tf^0-R;$lvWbXT2Q_j#_cbSyP!a8!)t5SGc;X zsh2*7-WNwRB&{3wY!hGSoLl%X_aMyfreO8E2wW6~OKa<3#EsYk$9v*&>%IpSF@d7#SrC(Y;qxuJ7{U2iK?`lkWTaN4Ue%+>g5Gmr(7<;1xRqir_ zxV;x|jkBRh$UtOpIx4rNmuWB-iF?;KadtjvQ2D&kJF8boaAn#SU|R6kS7m z`*sFaZ{YfhJ+U(w2bK3>SR2d6S2ak#L4)!5gl9X-pMveVYTDBbi0%l^KHd$)IZv8f(q zonFD=!xMB9j?u(XxACRx4fr};$K`p~@I}_`Ck)E1X-;hL(vfuEq;#m_asv1-6 zwGzFI8qx0eKMZd66DeVzajpD4?t8pN!Ou7NDeKyc4x21}w0IsG zMYZAKS*>|jsU_d_XhtjHQHS+WqQ&`!-1(}(fBIm)?3IH>4^@*4g#n~D?yk%YKTFnH zTYBs6Sa84u6Kan!;HexvcFI^U+2mymcAm`RiJCgB11DKw!&el)5v8G=a)?etv zm>(Vask{S&GupH7O%)o8uj%B$7JNUWId?B<_Maa5MP~g)G9TSKM>MQW@@{J5!2VDW zRW|HwY{|zBOgUnfA=TP%KHR&43ph{opQ*g@Zw#kf4WWO2U)I*EQKv@_+7@-= z@M~RpD5(pBB0ACjsw#JVY0FNcgXeW`&EIOR*r0(kt$vCAd)|rj7fU9{z>%xe9jGhc z;}l_DABeJK3k}K4cpFkle=CFTYV+ctm0VUmkB7w@rkpgEv64N0FuNaDS*WvDP!Aqz z+nqbqyRqa>7gk9(@2u~tT&>zpdVbsR$>`QRFz)}?rK~$cw5zuBWmc){$O*!p+~3nw`x?Einyi{$sLeM-K=@;yz^R@Y`>oJwHmyP5E?Y>06uH>S2Tq2#nUN4%_ zNizK>?l$3)A4>c-P5OBI{YE$Qudr5p#Dj_NVR8KJe?E#F@louOchYj{Ky*Fnz;U8G z7vwiC0WS*bvHHz(SeHIR`=q<*C|z(-?{47Y)7t-hBQD}~yA$Tb+9QsAl?`9|HGyJqg@0?F=cP0(iglBJxIDRZ^bVQjp&D-%qWIj#A7JRK=Lou=pD zJgX9F>yJTer1S%bH)uqhgDWWu<2J84j^!?fd?vIos!!t}9>RH(6TP;==Fd@Y@==TUg@5GK6Y ziz$P%5d9z>mow9_I4=d^^AoXjVJsGpibB+j2oxa_o#mgaBcc!{>#n^inpscDS*JU2 zuz`XV^8P*Z!JK1%?Pk^eZS1M6%?=M2bDq)^K1`6FaP>AE`Q|OgWS+%UmjX;GNJJ}P z9c~p3xWi59u1X3;6Ri+TIue9?r~I*_&KF%oPuum?53Owjuv{Y$b7bA!XT`s9&ymkN zO9zvw9h=K}qiwwOEC(5I%U2zq?5)XG`m^}fYy=yA>&Vx`f1rC_jdPx#@_BpUeRp?El9}YZBc2E=@kYZlJ~$=ktDF+Ks{}eS z#?YR1b41&h^VPL0M$ER^&YW)AEP1_%+a^w=?ss*5|K6Cj*J_1{DO!~0>({g^cwR0j% z9ax-d$F7sC*tDB;vM24J`)6$?by>_Q-8K05Y!8;^|HUuK42&3@gfVj@Yi4PSfvwF@ zmT!)NL3Z%nGFzBzeObGDCDJ4@DdENX%IugU_-@{*z}} zV}Jvd9c}qzzGMnI?3RwNE!?tkHH%Noq=wB97EEkS6}S7?bZRd)xCG+$ReKDnGm*|& z6QuXFhLZU2{m)4M+X%^8><@xjX^8O6!vLdjSS*u_j^venYZFjrkP4kGnV2VgWL7sh z*Vzh-s=suJNH(W$_)h+Nw2>pFd%o+?iHwcxLDPpnvH$jQJlGzMqU)}3NjF1~tufTT zTOj+qBWza5**V7#H(u{S>^8}Bo5jj(B>|zTNocb-1@S={Fin#A&!~N9wY3zca>hv) zO=WYwbPz;XF~!uFy~4IJuYN5@Oq#7=7+fpLeO|hG*klsh*4Lruj(!Da_=v2XSgeIkb;gp|hOn zMteIlu|m#}H5Q!ew2Mny>ax?s6->;T!p6DkjL-gy)p3;w5MAMk-o<7?1!|5aR) zPLCr4r4K;f1142wRFS>7t=vsKPcC5el+j#b*q&Y5)Isw}9v;fsU-^eMR8E>eanTB& zGF>oE)kl8bU{u|S#G(F)NKVYao0og>W9fe6njC`3fuq>ct5P};Pea@20=}%Ng>CqK zjJYqqK6#dttRy!vc*Szsy`9MA`?@h|`v>GUD#eWKAcS^RAX0KrO&Zx^ z|Al{y95-eG^#E{R1Z{&`<3%K;=NI6re@ao0ecom-xjfP&R&9p?eqXka4 zl8(UR-mp*(!MtTLu)Qoy`jhF{wIm11js-Zes098G4&z;LIlP=sz)5i$r*_o17q)&$9(CFcw{8_+`mU zb#cekI{|oX8woR+v6fd0->H2Ta?a%7g>Jquc8l=PP_i$>%CL6kF}$@tj=G#GteaMi zi9au6?Sfmv8xtneTm>JlG~-ayZJhc^`W+TdWk1iJjQ#x?KNlT9P_;06!xfV0w7_Bm zN2Ka_W7De;)TPDY)@$)LJ;{{(em171W=mcr7kL}=;U2UPc|G1s4SyH8Cy4hrFT#W&`g&YGZ5bPQjOVa>9ay-%4tZ8NP#x?8$PBvTpA|Z2 zx#HbEf1DMcQY-}E#i{TCO;5%e54D<_- z^0){Gryamo@q-Q6A$s?DTL#`S;_scCX_K;$dmE4B+o!Gg#poJ#I3>X`$Q>EWtZ-(h z9pd6V(4a0D?mBVUcp)8DQ*#j*lZ&>mGSR{{6^;{1j@m7r&kYLoMLKNcVhdg)t8CR%#aAJMj4duj*1RCGAK9bd z4$0F0aK*q~fe7}9Mf2oLEbCQ>yNV)=nz$EMqf!vvIua?LgJ3J~zO;`%_!;PhD0%;f ziN|EM=%EWcxI;bD9c6O%80G84_E~mRoHe22aXq$pF8Ppa6F6#$oKeThz4`S%?Ccqg z`~6+fYlI7ugGCpbz6Z^^#NyeS6wHrF!?b-#SYsCjmq)><7#@IQLH^h#KijS={%9KO z56g%C$WjW#zVkt-$(O82uTarR#ADu3a!DRVwA4Eo}YVc|Y9o0Yp@q%GHY8PTHYX2vdCNQ1T`Iik8X-)h~2b#*#= zukpnP(YjZvOWx7Q5AjnXu+dN$B5gBq)hrV(OhU5EXw)KoU^dnreSKZg?3YXrXGm_v z%^By{J7el+XQYZw+VGGo#>y<~(gNuL8f(j-gS&ZESVUTD=hJ7+Fs}4z!Sx1LaQKDz zsRE_PR`Q{p9=pNijz1c;jlzW;DHvLp1>dYJXsk*?yN4kdy~GRe^juI}=ZM344tSrZ zK!S=LHe}i2YpI>|w%eoL(gDv^958H}1g~R7e@$4J$G~b2jF7YOo*Fl7Eentyv}p9dmWHu^a?xsybePUdLsn1({FeG6tK1Evcevn^ zg)@RWJK>JM1Nvw=h)>NCs`}2DE@%5;b6lVyvlP2wwp=RRY>k_264u;&eh(YQm|-p0 zXwoH=^i9MYb?Id3?uZ@FTv3@Qtc18Igx^fXQ`1~DNX~=W)pS`a8lIm5kY4PC5;?aI z&3DIvXKwJH<_7B-ZctKlM~R6COkzEuJHQJKHRV3&W6Q`aBc3rIl%gW3vLRJ!~DAJwZl^2&KxW?$wTJkOuP+`gUXf=j5;n^%BQ}# zqvwl#PkhkR&IeBq`Cz}?%`T?y%~XJny6RU&d{MhZVY1g|&k zLDBX-(Av32@~L|;v$Na{4~62L>@962`|I{F2JyD?JUiQQ`h63w#}=j>T26;p4eCzs z%==N#an@-s!m|AEZHWWc-*P}lx%Yet*n@KE9<)80j@|O^t3-Xq6s-SF$krk`#RDY|h`3Sse9dEj63T z{qIaacCTy5nZ}j4yhAiV4;Li**uq!a1!+OPSlTfhb;goqIwbGEHyJ3bNJrnF!Ydt; zhD){?=ogla;@x?e{WBl?AMeA38u5hPJB(eT1?5&s{=L|Vp`-1Ych`c!vkcg?Oq<8^ z=JAW2@F8TLcy{+?{Lqp;OEii^*y8O&+2>AqO0Fyz29KhlR-1&2RFn&_k;GsFbrfItcNAzv`QLwhly`j zGY1+w3Q&G-KkSnafsc;DcTy#0Se+Ih-gzwFd>t?Q+(W&b+iy*iPC8+Gs(v)$HjOQu z{COoSt*6rMl$x+rzhmb5GVC83D*LlDIySdKAJIa`iv~U|DhS6qMPkf@I3$i2PltB8 zaN}~YJgyM!DoZhb$PwW*S3uvb3LecRb5(sA%e`+2m+29#?#Vs0*n!~CtXDS!bAjmd!SCv&wb|G;mt)?Jf7~0%l>=NLOjt62FAn7 zE(N*KS;$*o0Oy+}*tM+;kw+OR{7>WBKlM2mXHl3^gZ&3GguhH%n#Dv$RHN8%O-;C3pJ=e&{zP1c|*w z100ZmF>;3I(Oi7*RfJEQ4xnp68J2u6$K?K1IOki9it|_Sc!1nN#Q%Iv{Lf_v?fCPn z1;72?#dq&?*~oJx>lRI8?e^ZhcB=uuHWH3NKn!k=aK~Tq*O;c*;7EvY_BVQ>zmY#K zTZAC(a1?HvCL(k|1}f#;eA}ZC0Uf0;wD$yiXW?>f2d$Y90Mo%;m1g;oQ|;g(GY4^3FLbk`kF-Nin?Gur6D+&ECq@qOb7M;bnc}9GB(|$^BLe3mF z8aPOApDk}rvEW1T^B-Tgg$<7hFKNOo`n{CAU~+3d^?e|{+y%&&oNX5;;c;kN<5am7 zPJB_owWBL~fAGYRmwp%|{JP>ZA#kb*#ozW}X!9}*t(rw5bZiV_0^*_BFA39Rjx<6v zD~BKlF0is=bSEp$X=B12g*&L-V*}?dTgO1uS_hi_my?{FqRS&wH2HhYOt#rPlxhRo@_fh({Ahg``WDjPWiI!^ zw@#?pFS+9x_Gp>rjK*PZ2rTwM``(_2sdvXFxS@%yXrW`=(Ce)`;!`~FdxIBTWlp$1 zTmBfz?0KR+o3*v&=VA-)&Nrf-&URL}lG)kQMO-PF4V~7#=s&FqUw*%VSv7?S))w~h z1Rv2}T@lvW1wExpLS>&DO0KxWahW@Qj&jAI zUGBuG$3)L~ZqLQim73#aMJwq>IJMJ&cDHr7qOB%ZZ<)o9E+d#Xs1r@T|G=ud=V7*U zFZO;C?`UlR7K+x|@~#(NEbxY*g*U!+_rj5Z?l>cU#@8?H5niRh)XomLoa%@JI~`#u z>snqCzGy??e%T1~dc1;0@?2l4FlW)_-83xU#&KV?cu#UF|GG}$h%5aWXWE7thu`6h z@j1*850>w(MAT?TV$Uh*OA$}O+9#5`sgazHia#8VNzTH`U3esJNbV?SFKaj4>F$OC zS$D4NmCYwgKJYneZ2ypMrKdN`Jj*&{Q|Kx}v!fJqkukSqN0 zE3$5#?3G?y zd_Vq|;~Oq}r9+Yv*F1KlVxI$NG_Ys5uqf5+t@wSYw!X^k=tTUDz^SSPK7?n0MqiLg#(JwXpbj8Lx$j z5p?i8e#p8H#Itf}zjREhN_Ip16chY~5A0~i_yX}6)`<>#)rj`c=kE8LY4mV4Kh0T4 zMd~yb934ft_oM2ouGG$I$C3uExG=F9qw|{bys)5CZZ_fsoqzbH{2lR)MgU-4y zv3AgN{ErW8B06l~3DJ)NZ%whAY4ws^d8Li`;Ac`m)I|HBPVU#&y+Q*i=h;e|H-h)g2GXie9}X>1W8=ae)Sud&q1IhlD!S-f7ghdh+m;bBvoAzT zHmGa<-@5j)?#5clcyE`_IYIKj)($MVtDs+wE$$sU)zpduW zR?_J$x%%!sCi44=k$lv45ZmnQL#2`G{N27M?|OHqPC-{jH;`Ozlq$D%Z_CfJZl*>{ zZkH_h|8(b7vhL*TlIPQt{W4O1&pU+?bxJfsFIyf|2-n=ej3ayO=1K8W#LU{nIgd1{ zo4Jq}IGt0QPN4VeVT|6{pG|M8Q-5p^zWyXRV07Wbk)61CfhtFqsIa_OD^^UFPFyH+ z*Tet+Jx9ssYa)AP!!=Iq!%yeCZ6c2Nsq(x zxUbV>E^IK2Z#~rLa-u!k`AJuI7bQ;r`3L{JzoT{NCp_I)kL&SI;q7-1N;hjURyt+w zpSXyc>*w)R)_we0w8<DA+R4hA9;mObRvT!gPJM4%y5iWlj1% znZq=_@%*%?Fa5f;qxt>@Op-pl2aT>{WB6G(+!dzW(L$aIhn?ck1D+!wQ$ zex0XM{na39FYCa2Rt;!i_dt3EPGNs9=@@O34}0mpv-g$EpII{MI>%wURs@_hAIVHsJQSzJBjerOo^E=!)Q+>{yPYOQn(|@x{FhK7qo~|j!`=042Q-tHRQvgm+@`Z|c*gwM?I^#TrmFNl!Khb9|DiA8U zo-v&r;l9=x9V%pQDC?HVcc{MBk$KG=xPOfur|z+0lXa$C+}}Xh@;dxGZWVX8naTa0 z!}v!>l{v3ILgRN8%;onvDsB%7hI`?Az0Ch*Zu;%2D-I2H!?@9|IHu%`u35rUthGhm z1_jv23DHYsHq+M?<#L|Xmv>-|k0Z5o>=`{;_T=LhG>J3fqJGfG2!9CfVJK4)Szf71=d+j)Uo)y;)H({Hy9cgVD~j7ml?H|gFPW{=lXZKV&}2{~HgIrW#UTeA?XcpHUI%af!VFB4Z5hpbOAVK>@0am> zRNjRllRjXJ^C6sm9flT_u25NIhYR`+sF#_qy|OR1Zx6<|%m`ec9FNBqX=wK)7gJvr zq44}ccs~7K7Wf3-&p(SBk7T|p{_Sz{oE)ta{_`iS}``(<{0*ab@(d&8|X5I1jxA$L~{G$$ru#2x8!Yn6}R z_51PQ(P6l>t3dzdRXF2bjqBa7!u|do{OJA^Mz>|Y;q5@jzt+5)Y|JS2t>T~1WQTWP z#J4^ix1kYTqfeu4tZ1LAo*41kUbvG^__*8y{f+%mJ1i9YvZK&TU1kae$uL}&iL^6& z(PgkO3gsC(DEaQGekY*0`z$);UV>k{TTuLX1YDGSitv~;YphscwwrM&a_=0ljLr9G z@RKlkVzS@kUG4$gS{aHmQ&+TIAm7tTuIN5W{OmQs(3072#^G2@OG`k&9?@QeC8jeu z7a>Oqp%+jBM_GGlOX<*Casq$upM|OSWw__u!gSHQ?IWb?(a(~>qYQbl%|=GuT0nJ; zQEX@3mJ0E`W`D>Lu7*E6Wxp$McEZw=u@~k@Vu~z#UHCGexTVFtbd_TecSa6kD5*8 zj)U2}do$+TyNJK7laa0NjbBbq@HKTo>N;!$3DbxbF+YmhExq9-V}>^4+;u8HawWB*X6} z9jTw9G2w^!y}aWw=3oLs<|Lt~WU{l6fq2mdW=_nsZ7Op)6y|1At8Y`@8U1z)u^}v__>CWmVJcO`B>GKudR+4Z6^JAds6pp?} z#IxT#6xA|6nszV@3+2BJdJ>NDagkWkCk9VH#Ua)`5ue?KlOJZs%W}7W94R_`!7?uK zo50_7ouotOCA6&fVQpOy%49w@L_^r|8D4m#wFiGj#Nx>o$!v)q=1*c0?rez?Jvta; z2g_a=Eb}wbL+@Pn$CVg=;ITjAB$t*US%Qhhd(dCHST(K1uQ|kyHw%rq{h1DTbXvls zx-s0dr9DSHeuyJBxj6e%zPI)+u<=P)P|MIW>i5AGXbF=uV( zr9Nx8xbt*AEAPvTqopg|@*I4`FX1{iK)T<((O2e=*Z+oMMyTYsb!1)`BtC`Pv9MI$ z1IX+uUd}x~C8O{`G8&gwh{vT~`lv5>qseIb`<|A+*Kc`mtqQ;sd7eLyQ&4}k8GGh! zW1n@O}+=X;&5O^b=4iz5v(W(rtUqAK|9qP??i}=D$-Bn4F5Pb+H(F zH3&+hJfW)Tiv1fUpJC~SoGI=|I_8emGN+$^#uFa>y;0H32kO7%8IalJ@Wu-2y*1_Q zM%%bQZ55BWO{JoxI-^_v#`Ranku)m~njQRbrjO(wECMjVJ_4it5|KJD9U>ySU1zM9Gu?zBSFCI9COl@z>NM~`^=wb{Y~+PKRi2VHlsVxmJ034H zWn9r#ZXdFWK{ZpjxTP9DmHxo8yXBbjA_lbvzUUzDpQ%R!u(iE#dBoo}c1i}GCCc5P zOFV{2uDed&70q8cV^5AV7Dc)sN4V2!$6QhD?uP6#cZ4;RGu}~8%yRNXs<$x4UWy;K z+?4N*Ze`}=ReW7Lg&%(RWS(@^ADvPzoPiiveU|T!_%gNA1JGz_B&KDHZgegK362>s zoR@%;Ger|j^TeZGuBfSY!KngQXm50bp4<`k3BS_dh#6 z{$|Ps{k943dKFdDrZV=48n@p3iQTFd_+%b~Ud2B6wa5$gvM=%+ucTJ2#Z<@Hq=lc*)$i@mX&qZuW(*mmda;_vC*p#&ts# zJfmmE=?}ND$7fAyjg&b}WiM(Q{e_BSB`#VCxJ8KL>P%NPP{B$ z+?b!z+x2QY^#j)M&hF{_5!{yz+X{av{S+KU^G@g_IoK|qXw$WTFfK_L%+q2JCGVql zhZWR3XwJB_hdN$^8|6u;Y2=3rH_@>(L_@X?!7O>km;aCs z$wrxI^C=bQcE=&tFdQGmo6}Jx6j!c=LER<-+g?TD(oxZ5g!R{O^EXqe}Mv?S3 zJJHnIp7p}$x8EVX^nn|BKX@TOej7!n=Iv-u`2gDU^U%~S2s<=9&|#-1o`nWrePsmn zek5SB{H$+n(qJOlSZ~t=lnzV4c=1TBY@7zAnVHBOlY{oV_u{YpJ_H+=!bx}?)q{>B zT<)NQBsb9B*^1FacXNZb4g*V<(z|#92TN~Ozb5tadnv&b;aneA^285ex;(1&MXqkB z%(M$*4aI)sapGJH#Z3|tRtzwJnsf<6QPUEY;aYp`ro4Uqe?HlQ>Y$o56e%|PE zJO}{-k!a950VXlYD61DARKGL~9FvIw>UkJgv=1Ynlp<|U8Gfy-fcf1j95k=NoXb}c zzUmJC-Em^KH0cgg*wFTr3D5Q3&d}{^ncQX;4ZaOzw5{}zsa!(sM>$vQ^24qtZkT+; z0~g!)BTyKz3*`)W}$0(0j>#;SoCmWr?(gqh` zD_z7N+uak7%~KqZ*_CsGa7Ftnc;~PsL&g}1H+T!Ljb2Uf!87={d;mA;E3=1mH?Mz^ z4o_3jMBck$&}=spfAz$a#s0YeVh{BEBeB0Q9@h`1;mwphoS$2axS@woE&10W=PS{E z)fue#eE~O3YH{l6J+zX0==g_@93$O4=LT4FldlOk>FnU}Xf4LSo6j+`M@vpdm7T+0 zB5F%1f)+&J;U~Fw%H82$wVQC`y%6?Ed~5;2HmQ!n;mSl5)QO(~d(mQg5wa(gg3@_1 zNjf%H+&T$OMK#8Lxh!We`Fy)bHq}o!ea1FypJztP0t2Q7Y^Fx<rKk&aAvM9MvXlb-O?mWAT}1?dbMo`i>rbQI5%tb40GIK7m< z@Y6*IK6d~CT1PP{s1nD2%Dh`VAA2{5W-6AY#m9v|X%3FCQXAQ%xXA^P5Sa@HB zZGJsLXY&%oj};C=r5BF=bit}Z7o1z+E_VZOLgKff* z+BE_zm7=gwDHb|Y669W7>Z z^67r~An#8%(?GP%mv>h45cta6EF(vBSR=_SPO}%zr7bfgW2F*r#53czb76@VeHJZ} z9z_jKJ*`IV2TII-d>yfG3J_Kx=N)5Tgon6;)7{X>(j7(Naz`HKjie#onC0$?-D}(t zBbkjlT{rlicZd6YPXyhN_xL~YiC1P+e&$DEs9SoPNv$6~#(^R1UK2tD!cn;VQO zoiS>L1B^`UF}kfI_PubzlwVFTlXbJ@?mg+Rcqt^uzq`MJhdx{Lf}c4ftafwVplw{F zq(yrJ$tr%D$VPYjux(WfPONy2z!R1DG9(+y@_to5x(APT3e!h30K2!yS-qz(7R{54 zaT8Z0+;>2hLV*!46^OcL4-;Xq=gGR0?mE$Dtdn?D9XS27g8ir3@|2w=OO~4O#`K+R zS-qK4wyol!v{_v5G>S_ryK#zhW9st}?xa`3ZE!wf`lrCBK^!VuMq~c92)O!(qJ427 zp0Dshc4zUoy>Z1{BUhZabj6>?F1RS`UU({O#7$0Ixj;IAlkI7>Oy-5%rO&%q^qZnx z!Zer82Ym zJaGE#UbtD@sG3ZeBy8Y2t>g`C8F@#GXNFGKX} z{cza47fp6$pihYG-7<@KTq@eHVH`flx*LQY@F7n6aPNpFnd!j!?-hKv$CjT~Sn<>| zQ~vpF$a-#NV*@SbZCc9K3#K#mz$o!)^`e_!J8mp%%r#TK;-K|QXh?QWCrr95gKuDV zgUdMbLb%U&Di9qlGo_=YSRb?>Cn}2YUDlobRhX(p;&WFKhHMi@hDsjxrFc17q*znq ztU2$d8qw*6^ww!_76z&&E%(l6(}7diy324r^;4%(hAMRzwBYj!B^u2Bhv(0KAhyL9 z^u72Nn$^!Rd-8o0zpsV<+sn9j>jL(;)!?_Rt0FpVqRe%cOq6w(Idbe@VHXS$etotL zm(7%1Lx~A@RvK{ZxGmKCzFzqK%b2}r7So-c6qUr^7aoS#Z<^C#H!`H`>dQtJga4bD@GN zau(>ZL72JI&3Nk3E}mA|PK~mS93py7&WZ&z915n-8_Q(vz*)ggE zXLz>ftHy1)CbktP`!%ChOk=7mB$F-KsVND+@IPI0uI$lwMccS)DBmT?S{sEsaI0h{ z=Y5rKS~cnXSZ>a4Ym6AFq|aXqb=Ya?T29-xg#S9s5^l#tDxVt36|D#H#6-!7E>L6Q z$?oiXu`BC^_x^HC2S%By0_JhZyse+D?u-qRa7p*V8g<8RO5+VR-kcw3;%O_GLpww;RAA9r|#l zeCD1}J$YKIJMYfz!ebU4IMKH)->@~~>ss)W=+FQAd6&z&wfAHm>>&E&c}HG4<-mVy z?Admc@B>BT|1DYm=G%<9(qJbWb=PBD#Cl#8KJg>nx$Iazl`|iY<>C#)xF}=*?*;ea zym#`f)%B#ke|P%mc43NL2VQUxMsmN_JY3d-7Rvv>?r2%JUiPePJ6Zd*WHm)=7&+UX zVbkqs;%3b+H_W-TnK36wmz=SoE_L^=qg}&g%y}}K!`@D2XvJvG_&k`Cg~v9dLY>#% z^H*V|Dh4)iCaI&j>&bqC5=SMTP4ryATNrB}9ODB)g-@BWYcp90@jmcZ(jJ|&w(KI%ljLU-m8nA?fn;a!sKZ7IcKDr0ACl20r2;wma99I4wQ@63hT0(wB3$6CllYT5hmHWbaNt7MDsd@&B1Dhbzy$>#rh z%9^w`B5DmVdH4h)O>aQ4$$7kJRu1^q{eujAWldN(O2<4Z&&>bivB)meY!JobUdYx)s39E5HtXKDL)qf~u2_7aC3QR|F+ zuKVf{`5QC2huNQfYs=quj;&N#GMjyfxx8VY<}`1;JPgp4L7S&bR^SL`i+7hK&cq*{ zy@iAsCvkv!r($Z&Y^Qjmf;ykXRjx3?h0<$QIPlmSKHI5JqqfI| z>ps8XEGIh?uk&z}Ez|6!U2_|mewg!)mD{ECB|QoPS4mvuJn8mxyv)^8mCGKj#Ovx~ z96U>&@YzTt8~Y&G#|48|F~4QfL3GwUfNm}K;fSp>{_9|i`;J!l8eoNMJ#EmT!VVLn z?J26GP$x#=X4EgWT{uANvOwwBe*LbPqWRWe5C%gZ0S((cSOiRw2@3cnGvoA?Pc zg31x+9E;7I)tDYV1TD`0XJ6S1_ha-9qRuFJ#y za5S-Y*3XMikkjwxB>!!65T{AD;(mcMmK0NQnz&Kw@>WY%eI1E#7%yuS`$|RkW@3Hd z24bwzV6W$gRh(seKG}nr=XaxI!Cq`Q&3Pa_?*?ccLX^`%{5W_3v%(Kx)vQA>ZsdX- zYLb3bd7uZ^ofyWNW2}=DB-xAgH}V)R?T~pXCi3R?1_^XtDp@7d#a(fj44B?Y!u9LO z^)A5frlI&+<$~qQ_TpoU-B4Jz2L<%(F?66$Rt;xd*WB@rb**8r2mBH|5j4aHs+asw zo<%$`s2F0W0DBxuPBX_Ycdtg}YQq_3R?ZQfuX z_ly^@*5hN2VEorTxbco_(vwO3F|}lF?l@6NKKhA3{BR0GuUC=y6-lk?<#=kZQc%z= z3xQTAVU$sb44%ywShG7-k;_+TBZ^He<$zXy@O)D7AxKFCcDWE~X&Yvy%MT$GIWNQcv|Y*gGiiOK$F zFl1~c%5PspbDqs6jh#gXJIFBJwP|0?rL1m~e6wC75AEm4`9I`o_EwTV&A-6^#TneY z7J}J}U2tphK0Ip0UW*=_YHeH*SLKCq8~xGYau|X*n|rk)6-|3)Ao>(_ximwo8V-|>{0#4aYrP)1H!wTOF4f9W{BIeFff z@V+JnPHQ|cHI)5&^dY#@`_}854^BMvN3>rMHsuDxXKpAO-RA6~GqvT($rzfIfm)Rl zco~<6Lyei;ZCj4u>hm~Aop&wkswxd;#Io-UZ)Pg?{u^Xm{$i2-^pLaeEhiLy;&XZl zdT$Fwrwy*C&OL}{@keOY?FrYh$Iy6h5QZ_!wCjr?d_E9_4jv)QD2-rtTr6G=O~m2E z6y(pRM&`_M=#9#U&Rlx4?Wn-Lvz+tabdWJ4=*zHjn^=C*lQB2ui&@zi@w%)i3JGtq zC-fvXz3@l1_Yqv=d+vXixuREna7!}~PI)2du{H$1mj&Yer(>Ar=nvD4L3sQ=l=;08 zc;Oa}j*H?j=T;&j z2taLd5Y$RTFv~t1uWF)@u_+FfUWxdTOTO0-a=aawdAxXwC^M^V>Z@6DVDnI!Rn%5I z+A`P3DG3SGEIcI6rp#wvq(N_qApz*E8-~rjBG7St2>x{NLwth=f{2HDYj|J|J;3v- zy)o^N5A#`>H?jX1^1Ctrm$UMtheDxEeBmZ%lTlUVD|Xy0mt>_FE6$Ljgh5iXr=@&N zyMp1jV^L3EWK#w5<#S!|s~53P@-;S?Mc}MyBnA%+flfU$uGYF@JNYbIdmq8zWEb4H z>54B2)Q?bexOcrLMz8XQf)agD=s9-sHtz)1=?8Z1l2&1+a?H~}p3K*kYuQsof2XSC z|862v`;setH4&He{n0+i2c47r;B!3~$*UuAWqCB#8%Cg+Mi6r5(F@z>D0(XSqc?d- z7SZ&k>>U9+-za*1#=fQQ`;m zmafXbF-pA(^Df6C>7yS;@wuL>`(r}w|Gw{NXh+51*@Ot3o)&;pKfUqI)(fqie6aqo zA13z;fYm7Ov56u05*&uEsIy;6BD7B$qH98RN`++0uik8nryZ4dfD zc;ovEUkuD74sPX-8+!sVG&~q0p>VzxiXogYtc-DxMCKkIqSyAwiF%^dc)lohA0roU zD9X64A27qE5dDd5`;I*dhf-g7QLnDpkNWr+Vv_gckUx*w_B20uo4O`y7M)Cx1K`9*C(r%%(i;EWP(S$kn}8QZe5|7VKIl-Ji`9EjJCh z!al2;`y1R2K1J{IQwmi96r(hPTX4 znc;zG9WQjDW<+hTFO1gtq4_Xo0j>_fUvfQc3h2q?;2?btT8Y^P6FC^TP7XKHk^gKo zG29fPJ84>I$j$yRZ|qhxVK1|a0)uqTboVsK}od_euo8O$@D0!`;!2pgd{w1i$$ONtTmEZ(-DjH zf98#3L;jwL$57@PfQ9>m@S!FI&6b5zZyt#syP|NejQBn|z(!&%kJHKhjb1PPUoMcf z)94*=ntFh>pD>jk8lUKWQ?Kd6e##qPEdt=%IuhQ!5>OnTL|+SPb&JEG819e1WBFZ2 z9z!VoZghVIqevqR#s?$lQBIH7HRK~RJ3UvGv(t0VQuNb7##UL2Y20?P7`Q>~lo!k9 zu@gm2tCy^A(*U)HC8*=wa+c>ol!`YB==(TlX*h1D#UZ785{7<~RBpU65Umrs zGQMJp)B5@0_s4l(qTsh<#w z{1*u@%}&MUg<1G(lY`$``PhH>490XT$J-z0pcZi%-yX1LrPCACT6E_O;#92#nCTjihUNan9t`c81h)1zWJW3toQS>em zmBndjdFVL$x6a4h=4bHzM=3Fvvv~6K5)PK$#GS+U|K-JeYr||wYRk{if9i$i7BMFu zw?tu{?0i02_A*1Iy~1m_MxMl_eC9FC^TOK)p4i#c51Mz#=ern*jXtrMWfu<{r$p#F zroziE8$HJ4qsEn4+fo6u%hka9E5tylk2*+hmD)>OC%0XTv)_^#cA}uLQ}T$l@7zIM zJM*l*9GW1G1A0l{x(2+PT82?CA~CF=5BA>lfR2_A(q=JNWo8(rypBT6T7F|)laOwo zjyrpEaHK^cI{hib=;75^(VHAX&cSnDJwVEr=de#d~Dd%KvG z7|4<0CGvMJ%ezzkVijDnpph3)8bA375?x=?Pz$^H={uW00Jis4%1*KTM!&lDun>3=gsW^zv^(enNV*#U)C%gC+hTWy|}ai9Uqh6p!&nz=zN!gQtjDrDm30dj3vE9u+u4wOtX;N3qE>sx!`7P-gs$V2n0rH@O8 z08Cm+-|L82+|y1)AoD$yF6LqQ5N2@AV%?xziQ$be;GFk0=3m@_CHwBxD>(C}PUF;D zTN$EiB_DU0%B)ew@@&aUd3kfT^!zqbj_Y-m?&selzpe;9k3^vd&*?$u-SEuV4fk7k zGVj3`<-F5R)`UYhA`WibQUKP!bC`cTEHnp$x94Nzsxx?dqYORj&!YAhac5gPVczUQ9;VN^hQBTY8#H%!|#UJz>3A?ba3DWz(f!-Ux9BRuVE)C4S6(IPqKxxtfd@ z77-YlAB5fXxEbB;7;}bvF<}#HS*;_8pK<`p={eHo?Or@p-Uo}+y|~Ed-9Q}n@>*&% zXEJMpI)({T?PSysYss?RA(M2sNecxdF&MT=Y!~Xt2;E5%k~mQEPjr&HBR}E2wgz3a zOR)EG7QV0ssF)UuYk84)%YOMJy&)g<^Tvd0uE@$i%#1#IpiDUgD@SJKaNS*Rh=)4S z8M4fR@P1smN_q@jD5pZFiH`mlW)%*S zdh4Dtxww& zv3=sOvsXIHSI(hx3>>AV#9qD~q&}27&F|F8__y3nPo0etQ9%CP#U;|`#2j&Jm?Ejm zM$5>^fub7UTRuKgl*5xdi*kn!(q)B$jO)`{Tz@u`8~-(uDUIqebn{0Xt$%~z#jpP1 zuv2(OzrIDUUJLpmpK_A9X^s*#)5YknYSiT|2|S4l?)UWJrz-U)l*v5DoHmZMLEB>i~K6=AWe1KihrY)@^gPvx!Jj~ zct-sBx9(`3(Qd5oUhSpk>z%*UUTd z^ZQRSPHW4X4^t)X@)$YMFif=X4U`2gYH};6uQ+H?Cl#P9DN5a?_d}lRLG9&ptG433 zvz6EfwfKii7I9rK&h6aDN$&Q6wMP?jSB^9Dn0#F=`X!n$uXV*ebFo@(A|^+SB>%`- zITXA^o(`QSpG&luD?VQ0){K%B>xRhjt^=j@z<$!@ii$)}>?2FB(Bm*sQJx!jmby}A z3iePCl~HZ}t=qtLhdm|E?(HlyzB|drc21&IMo#4)J5lkokq^tQ=rv~{zd}r;#}gyz zRlk!W3IcCx}2q+&eH!A&yJ=}63XW_du}JQ&23~bYk_;^<`Su4B0F@A z#AN(hd91cXzBtb%heKpt={V_oZ-i|2Q5Um6{iI`&ifDG~BbU4Mlv45yzBB(*VMKeG zr{6|ALt2VX`2XYVHe7eZBWKxvkot0R-V`_|HRA6(D4h8OEp4R1b*C)pl(^hspiK;=evpU!*JxGP{fK)ULAZO$QnEtBq8S zYawZ?n@FR{f3ZvZ|9J(O{GKanoaM1Kvj~gH56W_swfvrS2ieKaRrDwJwUlnu6MmwW ze!~WRN$s?Pyps76xMG^vH6JIl>5*}Apqg~MuPmwiyGgEFCpp=uy@Y&iEe>s)%lkEd zQF7ob-h6wDi{w%4Pna@XNfouDcQ9lMK*cL#Ag*IHCen#)eT#xlaX0ayLr!BGDR z49o9e+~cdTF(#j}NA*AbekN<@Wf%?q>$(P%P z5>T^JHl3R%zUC7ps_y_fKD>jtj{JfKg)5kqmJiR=iJ%=-2LK3}^M4tx*PmEUxhpsu_Fz3Ed=kJC@YI4Zm>nYFc@{i%^G z$XF$6@8?P5Sreq|OI7)9(^}y86!&|b!6stBUc~TkMR_18!j&^m7o<#M&ItLVgX0gt zGifhYH|(Zg;~uOyv=?{j^-|VA%`n&PM2z7yxn@5*If{*uo#?S29YsthoBm%*_vp*O zT}!0gX}Zi^G+Y)GbQMwhj<>eeNV}5^^Z5bDYvVzU%n=L-K7uiiTyU?m8!B4y8%lD4 zT~m5KZas{fyN)1?8Hi>xJELqc1hFaYx@01=>Z_=Nlxonfp#|@;g zY?(B~%oN)@Bjl2DH(5qc*~gnJG1xpFaoOxU@?7z^;}Nu8M4i(k>WU8f;8-(yfc+N) zt6?DsToVe9Ug79IISNyw;?TxB8KxaFV9q;1lY1zr&QU&6SEu;GT6Fta$mn5PWEu3N zt>Yr`>^((}e(NW7g{@?6?0pOlKY`l@A<&@rXqmPP^6C54Zn`_BkMMyPpJ^t!_Fdmc zVbqp57<5gF5O6%){idNQX1H{{-Rfj#HH>6~$2e|QXsE`(z13wo># zPevZIy@JOc$1byz*j;!U{SC@c|D+l-4X@z@Yq~qy&XPTT5NB1FbgfsV*qijUK3p07*w#ekp zdU9^XLUHk&C{5%0O81{lB>f4!>~d2OL_SK@GY^cM<&H%=J)z;q9GIj4?CwIpqKY^y zR!>LMF}dhCp%D8Qm*MY%b7TA4a@{6?Jpos&efA8XrwmxzYBsFZD# zyj!c}#)dg^&YyGlMvBsR+b8NvN^o)!HD$+{eeCAJd@fJ)VrJ*FTgT9@I2dsSkct9@)BiD|Tjbz$PzW6NdSff+3Yl5n=%4Nfv#Wt9`%7O}S3`^ZT$q{2X<)@6myG!p_~SZQ|`@y#G#lrM^vS zBlM-$hsENiFh$CbtIC?8&7_n5RSe;bVxvL;9&mqde(Hrx+oNcf8-S`AA*k05$1~n_ zHJ zBnHf67N1KBS}n}R!F_q?-0BR>ca}kG?>QLnyFw4{J2=|;F>bNYa8aR=D%or=c+pHy^tnT7WOxrm;^T7wy~)4VJ3X4FL#6<zF-<6-4 zr1vorqp+p&%>1FDJUo(!vrz8*xoIkI=5(x+Tm$YgX zf@N2kQ7|zO+QfrfQ=7Y=GkDK6@sMpvX!a)+`dzcIT`LEd>+%rGJ8(5;buJrriN`)u zId5VhXM5?A4>?5+sjG^8MiU8+yNGXYiFh^3ACrlP9?$i|=cmD#W*mvYWyG_W6MLY> zYsL&eJo7t>EERv83k$-kq)_^?(yJ~g3gQ%t@>2;2zmbgFa;Z45h~?Yb;Z0%`OlApZ|6=hE{NPM*fDVJISGNTd51$ zA&Kud%db0Y#qJp5ITi0l;Tc8JCu`jz(Gntr<5QoeD_MJp~bRA}77DJcW)L|LS3g`4H=$P%sxsaXL75J zG<9U|Ppq-rZN5UR17=98r$fZsyuIj}J;!P07@Q+6XR{*;Q6K3~KbrpO#mvk6l?k`+ zDe$-vjn@Z5&^S2==?x(WONqeo7BNug@1aEusDphnHkYQt-7p;~73nZwKUU0ouAhdD zWcN0grk#yBBU>)p45y3Xyus2pKtX!9dxFhf3+UG!kJ>3w#Ft~yzF!K?y~sq%HrcTA zPQ|C9Xf$wtpNSzC^h_|c7KYP1JPPO;i$Bb@IGDz@&nJ^Jn~HPuQ}N+8J;wGsN;&5V zJKmegf@=Cg6fYBtvT5?sL|tYbX)Ae+A7hFS^T>b4A@gq}+9c95hxcJXeiowYv(a;V z8p_r%TU$K@(RqPz%M6CzkZ?Gxq9*fc37QG4*4o>j*>Xq zM!W`_OY_0TvS`|Jaa~2t(RX!Gqo0+__eYr5G#?HUhxtn*p~ji_%lH%&(J$q<_6da0 zdt%VRcs$Gq$HwWQ$Q~Nb8F3U$J~Kl$F#%5pCnK~Z6+7o-V1h;#w!O}RHP1=k2uEqW znLJz%bE!OQEa%i$$X2%*a>!_i+@0M{lDa)XX8Tk0?u^6ew-NYEe_Y#!WW2nRh30#i z#ao|=hqn{)H$Mv3$O#&m5QVu5xWCUP!aP3(+0N<6G|j^JjVI8&HWzO5@-UwBqcilp z8l7V!C;BpvE`6h9?OG`li4Q*Y9VVx}I!JKoa|EfK#_eYG7xsw2l89(j-b=#XH<_?r zdICub+1O>2f_h?RKk4II$us0$pJe8IrlGrJq4gHlpO?rZ)Gxq;Fy@JTDu(}KXYr|J z@2NzM#7zsayShn^^jj@Ucy8D^jFkDOI!o97Z}4tg5jsp`&X8^dF7JuLja7+AJeYy3 z#N*g@E*s~c^4#JK{H9+bYVnCz1F4GzR3BqR*2m{JMVuTQ#rYF>xLc8slLqqvshl-n=GfOJGoiYRJ@v9$IsSTD4!jL$FsxnXIwNg zk0nA&(qYDLrgTL%3bl`8r&}&=&OVK^9m{Ye_#9d~T!v9M<`y-*hhW!Qyt8o7=$nzoUN=?`#}dA?I}iMPKBgQapL zhDWpCS&_o}hWn^pHV$7nf%&fa7__DsZT?g-Z}>7yvub#SbM%zo{2x1EUW^(tgvHUSwDI2=Xme-mi#Nc@+IbBu<{4Ih_N&-3Eg*lk6pK~Pqus-6=$u-EzP}#v-}3@T&%MK3zi$X$ z{0F|Q+t)WZ$vt%kv5F)Id76cE+PYb62d$OnLL}Xs9`hUc7@Sa$Z$=Z--oazEy7BTrZC!7E7zeQ<%f6CL0d7kVV=x>^E|u z7)Z`lmoSVo2}AkUNNh=rgYxTSyy~BYx?6ep6kdWp`qg-$d<}yG??A`qAtr@CL%R1H z^c?sFBU68(jD6XEbLhL(n_8#6wxVIQQ&hC+f3?9-RP&a}u_ffl<`0zsU20qO>M%Z{ z1gRU7;JG&(mp_D{sxk~i3Zl@qE*{Yv(olgM_`E5Ek#;5A3@&2Cfa^$Ha0e+j9-w~e zGX%|lgHzK!|MLg7-oo1Mh@)Vwoh%|>CTg#lbkEx;yS!J6LHBtwE@X_X$WWFIQ~#oy z?-hEMXCt{wG;D5%V6Qi)I4A5w@YGtEo8;@EfTA=UWV8&mfxCFB?SY-IKGYS z?)nsicAvoy+eG-28v}BH`!g#@Yd^U+?_*$-myCiZS?v4rkpJj3bIFU*qgfgB=&|%> z>Ujh?T!F#%n;6A8!>KT82&Os7^_%wce7TLBx2BhS$~IXw!%!w~UM}Ys&5|i0Bjw{V zMNz!jfTCOU`?k!68S53QW%M%D&@z*QL`)i_8FQ<|t13c5*A-S}NTwWVhK?>65N6Ujmm%jsJ9U z^BK;bwVU{>{DPA8b!=F65?(6t&>0(sZ}mZ#V#vJhQ=z!NFam>PnJKOwiE#IDyqFUP z6{9e`xE>D2{7C$191SO~TTgD#*%tK6cXpJ33HIW`8M~XmrTj48E>GyWVv@Q_`rqZ; zY}y1lrLQKbgWJj)pE`W~%vndDY(#~{V(p(Ws6J;uO3$xQ_QD0nIAbE`!g6*1N~!U; zxpWl8YyEg99>eiF$B@o-8(3407)OmLzpbCt!%i8)cdc(F-MX5~q07|bv|cZ>#^_4u zCt}ZrBP2dlN!EXBBH_Cp!DDk7>z6DHbcsi%C+p_a2qXpa-R}*>Na}MRsL>y3CAocD zJ)u9CIe7;?@j}n@pB^@WcVYE9=84U6k_6(gcP7|Lz+39P^*CqmyG=TsFqBp?E5+M& z9yMSS-9C*-M);OK_~FoDiymXCE_vpziu&6n7o1Y)-r!oPGa8k zL{BUlNIdwpJB)SR|KYG(-q1_ViL-VeCvw{y#Y@RvQl{I;{pCC5<#$stqIaM27(Iz? zsViSP&yb~|qovxnj~K+X70aBj*ywzh+TOFsm|Fc6 zR!WGyj<`&iBHxHK?l{pyW=w4@%C+AxY*Q^3RbQuG{XACns-Q3Q8MM|qiQ)Uw(QRKm zdftje>G3eMP!7e~)Zl+O>^|bKpVO$xY0drglNvs2`U9`DlMCB+$>Hsm^01AmSbd~6 z^v*i*4xASB|Ue z3UF!42^e3`fZftG{yF6z4r|W4VN?<4z+=eem`|yG+n%$6BemtRwarGPnU>XMVzt&aY8F?iC4 zKH{wBOKg7s7*B)l!)sT~KRi_B7ddvfoyE_RKHubVjL+v;J;y;@=ki^9?qW`;rMR~> zlUCO^%PUo8E;n5%_5};%0QEtIx#MM^-f%J9)n8U0>MedpmE=vAZerTCv$Tk9FM0p9 zky+cD%Z4kB#B$pYs9AkQ_2ZBK*458yZNf((a>9ZjPEI*h_}2O z(N)f|2JgTe;O?_p%GqPhBqV`&9rds;x$Y14-$ln*TQ+m1PK7>+?;ON~S-cS+Y$WA| zl{}+w)5cxf#OQ;u7;EZ@ncXtc9Xwwi@|hpr=h@v^Ll)i~DyOvv$@Gr>eMQM^(pk=oX)mKUx0UIJ|Br|2@@^PKOvNkQS@gRRCHaiPI1nG;fgn$m}nF zAFIk0>X6rd>LELAx`|e$=kPjmU3=EizoN(;>B;$hEoZ`ve@9rWc;88f(Q7s~kBS=8-J<=;0}nt6?s`1&C-K53Bnf9)^# zTB=D5@zJOEddLOsZlblMlf>!qnJ54My4$$!8$PdUBz5iGInUzUFsVD|?>6?*e7ddp z&a#$%OLnlb-Y%)!#|=Byi_Z7u^0sDyva7%6Wr5O%24>Xw~s zVEs|}b*H>>G8f09t+F-5P=+Y25sgj%$;&$8&YDwX(%aGEcv)Q@8L7&&lAh9PM|ZLC z=_1{xFju@;8!4FAOfp{mM*6|8$)Wu)h8d@WWw>QunY^UCJRQQUV?6~K z|F)&Xylx_2Res>|tG5UnP5=0rchTtA4NM(;^&jrM?5DHz>230Wvz}L1*2;@%x^ideY*A@3QAT?W6`!+xM2VdAKkwSe`nApE zocma1Ha@`NOC&)4|jHIARnomTqq%5hPd;g=8jUN zP9F|(8EiXR$yi-;nbTpbOlx8wKYp!{!e2Vfe}E({94UtesL0J$on@+e3+c{hbGZ73 zK2Ei`I`uY8KAwl?sUqan>XegajUG=%XWr+a}A%n8Bj@ly&WoW}@a>2jA1z;n%wyDna>} z%DyyzViF2lkju_2z+FAP5HZgc)9Cr4@%#|{nYZ#k+*y$rL+ug%#>Ab!PjHmVVfHdi z*G3#R?36oyO=WS}CehW>V;;D!?EX1jhN+I0Ci8lW_2AZWz1KTz4Y~q{^3wkez%-q(pFU)ihpZ|z) zG^L-7B|Up5Zj*Np45jY*N-=mbPx@p}l=F!Llgcd~2`cLZ1Q*SwcP6glMF; z48!xjLB!iRiza_}@&XsAD<4FY0sFD1ejiFrsIy$VAB9{uhO?C(#LV_Lr&m3_MsswT zbKz_yniI_BbBirvX|!IHt8^u?a=J8I%gnrWJ>>D{CgL;Z0r2|_s#>Q&$1MWm?*}1o zWdJN=nGdCO6d&e$;lgw`?6*3CrxAw{t9Jws0$p%_k}HmJ-4?ui$M9TE33rsf>&Pb@ zZX*|pEv1%zMN>8y%bbO)WymAu1D%>Ehlcc*pCt-n+~qAkbUBM-Dj9gDL2onGtmpbs zr`d(tZYSoOygY__>UTo+`d~Y~s-DmFh99+86Nmc2u<#gGaowNQJ8-#P%dqd&P1V90a#}$Kc1c zd|L(rR>3H|O1v~T1pD6vW8(Q>sEiJU_YZn0Q^%0fCk~CdZlBY9-XtgSj&cwQvz1jF ztz;&15cj)ok^^aLW$K)T^8E5-SxfK#G5HGOM%?Mw^l~JuN@1R681}sPM>u-`2gN}0 zG(r(#6bbdj7%bc#hb-NAED1@#xCzN99i0xnYuV80MJ+n-g!akwDl&8uV{!`4``Sn` z-|-8j?V|AAP!>n6kk&@pQgc&7uE!|LEP94iw7P|&6}eb2JchV_5bBQlBcz15WPS)j zb)(RZ^R-+Z`p7)W0N6LC49&seX8Bm4c?RDGmqY8@Io#o$VCDD!zWg`%+m7a0k@C@k z8Jk<>y@I~n8>lN?|B9@$94t*gcaS3{@9^$V1u8YEhnN_SSQ}!49|O>b@A3lotI4%^ z9PgP9$BYxOc##k7)@N{CxfI3B1l2a8PSf=U7E+h+nRh~G2l`N*bd+upb}~+#*-OkW zN#C(aQm(C$L38Iz*jV}#J?$$a@|(-!z4xHMBOgt5;xN60y0KD~3;uo(26#a!B` zR4loF9M;aKF>ZAQb9*l0lFJQrno9qmHBaDSP={X{U(k)bR$bmjH_0Kq{E0mvy&$^q zjOucoy-=IgGP-V#EWD#3sl=}O-28#VO)sKqVkTBjj7Gl!#MJ3W>_+@cKQ0cJZzf_- zk2KugavbJc3!qN#wxa0sus(DR>aA;F{_YVDCcH#s*N^Z#)_}XbivlW`i?f5=_8!*a z9&Rd+HW`Vk%X0bAoxF|6Ve)5tC;1Za7Q5}(+o+}BQ%NK)_oZK2IJ1>oQOg&TfTOO| zbkl?B`@SqJiO<8$&Bgc{UPVv-OE^O>Ne$b(P>QKV2ZPtB==ll#*@LPDJIU>jb}~=R zO48bHm!HiIB%#j|>1rxsM&G;`^76JVti_;@1^D2efN7H=k?XEl~%&Z49 zjh*6j&_t9@uNU2WizKoY^ShU*N~3wrW%sk&aO*+;Ikh-+UKWXu#KRsKakfAV%~*}v z=HrP7Q%yv8R5C7I&OpkW97u2hH95t2F}?zYt0g5p^@ZGxPsMjSy<~EgLmwqzch};q?kl#?_$1W zSu$GfOTf!9aj>_JhgvZ)NouTW+7av3&PRn5VhsIJH}$T>urt-jtiFVktk*lKI7x(; ztrQ*DAuq3P7LDvRa;X128Dgd>HB*&D*R388m8voSzjP=^MkAK`*22O#?9EBWqMm8c z7?c8^KJhT6rZ9drb%oF4@I5>kcfO~iEFc?wzU3h6a6Wb~D}=`TV$_gpu#xrpaMsHS znzqvMmW5>HZIaLqtL4dcZMkhcN}l9&l}*n+GLy9eD;}iaj#4zf8AQWiO#;k_;rDS% zN0w;{`c}qb*!BqgN({#Xes9$evwU>T$i;eUK(aR$kk4O&UfU{hyyrZ8 zcQ6x2=b4=>ChyAY{x)fRaf1x~sVm{}A`c!9lGJnUq;TCEoVKopRee4hy~{zv**wfV zRERe803EaP3=;q5p~CD0e)K(o@u?>fcI7lQi#UgVQh_xG&S8S$MeK>bg7;O|5&8Be zJo!D}vt=I^KwN_wZ-eIBMAL(r8sq+x%DGeJ(U1O;cczWZ(0u{p^m@5auBlHF>7^>SGFpgr?nOc>pit5=@$qi#O)E z^gB2Wo2*h~I#-~PdMRH2%4e2JHcqY1#QtRBh0k+|j}<`a@EQC*QbLVcInH;i!r~V6 z;Yc}$gIu@U0w>8IWhVnJTT0Hatzz{@PkM5Oknnb*G(W5&uL_$>mc;|yTwH<;uXFIE zOD=31pGJLrDXN}U;MDLk^h=;7BK-tb9L>hmPbXkEH6MX{3URh)3H0ZdLqorcT)%2; zin)NK(-+~x-*-Lvsm3epBtdzn^sq7!pS1PjeR+|LQJgI0=T+s)RfrX){I-tgpkiho-W8li)VLB{YhQsm`_5v(hx3@V z=rTr*yoMKFuA@8g!A7R^8}YOg_X9hnOm&+aOw|{SQUA$^+f(H9#eNc3(n@-G*5YlC zGAz88gYYLO;J5k|?o2H~m}dnPZj>YKWD$a*`2XDVVeWVu{^Itrj_6qHK<=}3UO;)3EU$pvG>(^^bWWJ-MSl? zru%;+U3XZ{`x|aANmhhJ$jFw>b0a$|vyc@c?Y*IDU~EGw57rC zK7N0k>+;9BuBZ2Xzu(X2d7gW{7v219IWX~)bog8#O?X}vo(sgKp46jU^FZz|2WG?C zz^UCl6f2BlE?!R@uGYczzNPe0jFyee2f2HK+Q7d#iSUXv%)=f6mHW~mhUmy0RS=PS>d@wmkg3+M@$V(0{` zUaMfXZ2{^gPrxdTzL+?*C44<9W!$$onV=Oag~VHL6g{AyYPejpdnSRyUdl@PZryLf zOw?sj(&kvKtR9gpW65bRJe4DrTAyXm?qXTLwoFu3*2w&ddYQ}JbACSWea_nV^}SFz z-x)2a0nQw_5Es8rhVP|;2wB?}5&IkDs#=Q7ObnMO7iOR?4wk3hp)!Q$QA`B;3dfhy zUNuSvkxO?$<*gVyWr*v!9EomPAo>+0l2}kKcgEJrw)eke=RqaR`9VJ!Vkr-%QA?lA zJ@*M0*jKJWs){w9q)j7kGZaI{b)=q-SzV2p5=PDE(dEH%q3wO?_2Z!|8T?qb9gdV1 zw_ixR?@=m*Q|SwGoD+@%}+qyd_^>sbzuKCccu(PV}!{bx%AG1WS=osC12c zBDFuCN!ycA(yBgAO53E#mgH>NURxlCG`>lv#Z?k?qfR9_#|`%FQzz16 ztR-_6*ZAN`Z+F-f*&{V$1r9Zv2g3*B5O~@IwMUyXhp$3z#>UHBl`s)v`bs<_Jd_^F zx4)R}xQJO{DzBu|%s5GJlE%D)92v0ii`|__UX>+$u#@$Sr>momb4pJe z>iAp&u%w$W9M*ZD`8(zhKU#(GK34FxnvBrIfvCEmkIn}baK!AL_#S#H-@6doo_$vi zQp1*~Kpt#kIQ5CoMMXPCJe#M;uudODrSoTT4KI<0KIO6kKji1xM%gr337djdp~-pB zGJw0^XyzYr&)x1HPfVu2X5RWW_}tGL74#6W(i;KYK}HydrJ>+p}ta+FM(gHhwg^8f3F2h&@i+_OgZ>`#_K>=6so z@5!OPx24a{JF;eUuv}^qCRux)$-U1!_al>JK&N-oE&qc=W`2~m?Fz(<{ISnPWnvIi zBkMVLf4I!7^p)huwkMAA!W-9kPFwxsfWJ@_?1F@pu^0x7N4B3qRJNG9j$Tc&I}lg9fjdzjM4wC4mO!o z%5S}7DLqZ!h70^P;bJ4M7f79u;=q`+w9+P6kWnR2| z)J`JrB2{K`zY+7AnIbvdt8}BUIkQxnP@|bp?Scym>rlx|FQX^qaZVYJ34Qyc=7Bz1 zK2gT0>OA^tMT^d?N0L7Aj%Yu>DVud}GxO)3oUS29W?mRMR^jp&&zZOl;qt2cW2tBz zAp>tm%K2~4|L7O8Q@HbPPMs$GOm(c7d11=$I@t}Xkq+>6Tm@a~oI)O%W5DI%7!%$V znH4(dx~@hX&Jz=heI}m!A4m;3gW2KirDxxf$k2P@GB}vM*L``b!hds*xAov{8RT$B zhSR5LZqNJjfS>n%Hoxc5L9qG}fYs{M5}5m7NdWJr!PNiMuECfQOYz)qHu`#uLnlY( z|6Oc{W2?A-H7}M87ZS-WjG*rLfo!~fmpMImM#neIufdxA5=;pNnA5ThG;+8?1~eetHs6TOzZV!OUQT+Z2G@?VxX?K&0v zT!%xAzq6Wq3k(KxdlWy(gT^>{k@Zxrzb6+iqY`5Vcmt|KTXaVb6^6M#MyStW0-g_N^^Uv#fDtuLUgj zj6*M>M)FEWxNgwKbc>&IDlJz&8YPMudka7Jr}C2gfvqn>&6kCge+CV7g?SO_1LG-^ln|FJ$ty$8y&#Smu*kca{2} z3hwr5=u2g2diD(9IJ#!X@y8r&K3H}1z>N5E8H9GmS0{V{GZf69D>fYmtI*BY2mZ8 zUVy=g=6Go@h#S}+2iqFKjF{Ao7g`vistiNNdKnO1Ne{tN>138KYUK269g-;XRA0+C z&NN47M#%Ys@IN@L2k)f5ar})J5Dz7f!^$QA4G2Kk_1wI{v|b;VG6zWD89&dHO- za50;W!gUkrM?M@QmHHsQduL{GQcvgD3;_yS*xXzVfm@W&E~Y^&k}72Uj6#Vt%@MEQ z3`yUUBAdvy`45*=x258~ND^(aDR=}PsKV;jw zGO^xXBJtXVf9e|ZZcxtUyR@C3nYfi1_uqM;{&3;0WlJD;Y5nNg5bKOE<8^2>T!|Nv z+#gPuji=2g!}`H!l;jM;h_NQnT-p_TXB#5#T6^g4Ym2@5t)UjBi?-=a;H#&OrQ?|| zw4XcV$&G*Nma%SgzQadO@cY)~?@XSby-on8Ir!oB03S5;_QdbruFMs6z{?}6@%!5n z=3&f7Vcs;%USfv)$s;ge>_Gf}W`ft{#!&L=f-46)LCxL(Z9MhSE}<0)yX#`&Sb9OP z(ZqOanEv~oJ$N@v=X**3G+y|TZ$KSXwkM7nxxvVS{P&q_@UrW2 zbX2gy!{M{gZ1N=BDFx{L<-%rguzWSkVL5)Qu59Ud`Ql1AJPikAEFovyswvDn`E@OC9M5G!8R`;a~l5T(>vA5AKPNeT-qU#t3Sk4RC*$KJFT|hJjWK zTz%8*Pu(=ub!J_;O}*Dpdb;{j|LYKdg|&W&;peR%?}fh1zcOm-gnYBL=*(Q)2#1B_ z3Rz%J(W)X|FlEI{nnV|%YEZ?9mFd9 zA9wzjbvJV^w7yE86%+Q-s{(;(0rV~MLo9#a8JV8A(%uc{%p9?8$QpWMF2ka$me9L6 z6Z;oWME0W5_;=)B?C|P~@)12zHliDBo^?hhzu#Lc^l_wpD};U2L3)rD`c6{EpvwRI zP7h>V5dwOC=lx8~1# zw4LwKQ{v7+d)d=c-#xc!Ag)CE)1%iHKZ3jw*4qPbUN~cKt91w@k76gaeScrHKw9}k zSdq(mP=5d}*mno)3~{TUK3;fAz8};+w3z_E84!N|T%PpX|z+a>=8Ye{g5i zlUMMTd%Ii9R@Sr zhr8MtU>CL3edcSR!(v5T-Ty=Ci%TU~{i6gHC6jLzCC{fnmZhg2h{9HK4R_s=M&A4D zI3ryo?z|e*GFWoQq7s0w-TcsYuQ#S}ho>^u8ILO0B0POL7Cv5p;#20Zc{Ccsj`qb# zQ$x6A>7sR%3ToAUNSb4j^u3-fwiP)Dyv9laL$In>s5vi8N-RbJS$fcn&r^hLhD8Y=eI$Stuzhd$%+V!bJpuXjO1 zV+;7{s-WknGMS!}EmLpNGj%|;s1`+t$F^Wmx_wnf;fxHvd0es-n8ncQkhEpi$$viE zk2rg`p;q%X=iq?r0cca`hdd`AxTtyJ!(tcQqQBg;v=wk0Z;4lZ%wc(MBy79&!h+N` zSnjL_760#|{J-4XlP%Mu@VN{YuE6}2Vwm}#UpJ zuVk`rOWq0IPMk+Mt7ts&M^_tPjPBqCt0S&h;%JYIJ}a?gxfQamo8#-x5$N}{JGNN1 z#OQCoCGmQWz#4VWi7&#m*H&a9PAo99&>LF!2H4v zs3}v&v#4VEYYyLyCDiq7B7c9=HPNcOA|>RX-`sFn);3&_T@TMnyWZS~FDJKv`l#B5 zGZMMroSbFdGw(TfcjMmf0d@Nu{cv*?vHjNWIA-jKZf#d%_>DyQ4gUO#c^Urif?nT}KvzFlgEoL4p03@&%?pYpL&7e5sSbNs;t6^qElK5;`{p1dnv9YRE7_hSj|6Dh8Ho=bi_z1>x!1$*L% z@g<3tMaCaKu+e>Hq;Cm?$qav-SM$M5H+MYm>WFKbSHZf(3Q6=Ra=J7WYv*>xEgdbK zO`*S=#L5hlM-n{mmRza7CS6YrS3GJYNGy01Xk$a$FQJ|2C-`(ccEThxwH zWR^-6GbNu((9Qc&YjaZ$*x!)ACe&W72o~KJJf93-img$cM14q>>`CwBZU1a>IrT|A zIu=VFcO7rP*2-z#Mftmk#a1)_sJ$P8tLfD<#05{-&m2Fp1O~HakvldT9bR;Y)#YZ; z&Z&?KbKa6WA1)WUS8*=9DbqRkUBW#HZvRNGR6mu|<9&9U2hD3+J=j=M~5 z(~i%wX7_73xFSSC_TCYnRk!8Rg1b^P;en)d3Kz?t&&8%&oUE`-lO1=n<>a%^V%4Km z*1WEe>x1jWpr0Zh%u>Z`-bD+ogV2rrMeoNxa5(3V$J}igDzC&+`g=TSLG46*f0%i+ zr4KguUR$XpX!=ZgdN~Q^i@C;{!7?K$M56jW5r@MsrTfl!dHgnARHx>05Aj8k z#(oo9ooczTt3kS2DIwfV4Tm|SWQ?T-xXd5E%08%CM~>!W{m=0;?Q^K(1TEE-#tRkJbft|z2oJ4r*!dh%#m-* zm)JY;tC%%b%KBOL((|Yy!v9jmaNdF0U4vk^pS~uOyx}{LTHFt7ae60xk)38@*0+&3 zx40`-b!mcQ!9~mxd?QY?LnZOheMybGFDvm-q8~nzy}4mBi#|-x4!n@B!(zqmWQt4+ zBsRA4ljz(nlH|5!@<#oK=&$}Iim6Hn`@tDOoqfA0=b@cmNNIG&HuW`d$zO!v=jNE# zoqN&Hj@Z0O6${_y%D*lzW!kb3Y1bxL#yxo;XO8h*3JjC9`QfrqiTvkj&tw#}>&oGY za#txsa+haIXVZLfGAI^9pE7yT>4)sz-YC6z_lA(0dOE`oZQgleOR5tTCNc-u+6o(| zaSwcdAljVL$Mo|G7_#r3bfkWGuK7bLP=6r9qeGri}sNxQjqjOIt_g&cN0P-?e$}M%zahZ`7pVy9x9cu!=&=T zQ+l7jlKb5A58=LQ{4MrYi!x=tS*~2{_*qUZE|MO+^YiWnVveaFbE`dYHQE8Y`r07- z`&?v99)|_Y4|{~>xHYmu{08&v*&Hrekq@K|xp@v@;rt8{vg*+jX_^@-ovuHWzUCp) z)tz&7D`H08QF2fxmfWjEQ5lvZM^ta$fICiqi#)ihpik(c(F+U~rfG+EE*S-TUlI&#Z@(5tWiEf<_> z$s8PiYYZAR1DijL!04e}(WZG*WQ0^o)tXGv&rg)m5ec$6ElF}xQl+#ymD$^=PX`&PCTrHb88KCe&HMS;)m zlrwR1A~;r#)+ETpQK@oy=X;sbB1iU=qyN&N)i}A#3f9LaW5mNj=-R9UY7p^_y_`6h zK@a^TX^4&$2ku?td&JAs9?5ciB>$$rvt)0tTsao-NmlIR{#T<&UQRETy?l>;rgKKP zO+S|do;bYD3B`9-LGJ?f83!k!z-A!+{i_`YW+|htZGoH~o+AHjO^|>7O_FZSGsI+8 zru4IaC*@^HGGc$M4>5?VzDGnX0=EXpVBnBqMa#?e`iZ%@JE?^EMGPy6v&i_FVe{O zcT{2^3Qqb#(}$TuX54qXufpqbmgxI)BH9KFfbv3O8poBe?Dr?>`#Fgkt^_%onIs#^ zGNgb#fKiuBv5nySIWnFakyx1*K%SRLvSg`Kk2x?)W<}%>m;Xdut3X1R7s~ndMKX$? zw@(W9H`o2RfA&N_Hz!mQOEEfS36(XI;8;EY?Zfra@2wKdo8-%uYe{nTX1x5`!E-1f zLk5>-$-v@Fa%9qEf=;58w2K$}{b zbsMq*@iod1LNE2>aZczxW;I^Y6Y$EM$xu}qglp05aOJcz8uRkSH9AT3XU2oM5iUlm=T^y-|Uzo`A_5I zPK!k8ppqsp2WQIPu~{rS=(-k;PS%(kH65cvnAxIZgh%RH2$=mD0Ye(rz|N{We9^ z&sM?3&gwW6q6tIxSC6?X8&*v3cugM+*yM)79CE)_@J=$FgEJ4uKzBedST{7sM&F+@ z?MIgE`YSEj#4YmXv! z*{UF!dr5s8E$EMHilOW$uUe3EZOT0;xv@RS(Fs^ej_Ls$tR6of+0VygK-+$}qu&M_ zl$i-;ohOF(-bzC#bF4PUix%JKyzc3u@A95GQ_StE$dk6+i^a;SLfW%$UH-QM_U%$e zgSQ%PZ_z~VZEYw$ZH7VYpN-dW9~u~d0#jcM_VqyfB1eR$t%9Sc71VT?yXRmE^P&zo z)lUs$1B#dzl`129#7XOqu`-@mT&!)1DEnr}geD(krP(LhPh>8;q*9)0HOSQU?3cHw zV5}pv9Jr76e9#oFxew9dyl^6b?^09%2EXyeEiX^hxH+T2a1HK1T#UZ{)8X8F1UxOe z;2Zb2*A7+4y>aiQ>%Dl{k`^Pq&hgo{erDAq#Zzigo|X4zJW^Ug{!xKs$M1c&!WKSJ=U-W+{$WS>Vk_Aiqy9 zobS~V(TR;ReKE5fBi~BPXE74BDMnt5h!;!Vxz4jPBseBV3ilVvyMS_ex#B0iAb!iT z14@|HQx#8tsv|^=n#U$OSj@TXId>{oXHdiUnLPd_-k9vq@|xesv{f-OSUZ8f0MX#-`pL5+sVFAS>T0fey+%jScgu#R{%5T zL-o`|EDjz7pX!e2!r$n@4Eifh%9g)oCyKLuw0Pf(lDDs-rJh{eHl`_(>OgJv_bfSd zHd_{N%8`Mox#X98l8!41s0A-uV=XdeOj~?!rI?NG`daLnZ z$RhmgKArxABVj$g2c|4+g`6`=xbcI!m*eT8Qy(h{Hm}8d{A=0JE`~lI31V@HK2D2M zL~%-rT-uu~dfSs_D>;1~^3tTuhYWejy4|Q>I8QFcUClsjF`|#EGr9FQ>6_=}jNQy~ zv3s}_8N=tm@|GE*b`QYk5gjp^z8v=(YUT5iJn~(WW%6g@*}bDh>wUCLu8xr|+v22V zcATiKrmkH*T2`KklHnmy;=;W0_m83_opn2L=Y7wX`=kZ*g}UXBZ0>(Ont0&=zw4*N z>~Sc@2G3qvqP}=47C4VU`luebS=k!?TU2p1xlG>8WG-oSk{n;mGy77ElwOVz4YL>- zVi+yc7QGb9g^{w0^Ns4>$I|w~WA^`#j zb-~VSb}%ztfoAp#aDMVs-YFxY^@?|edpq1b(*$3d|CSHyzsm06c~WpAQ}*slm!m<+ z(mRh_sK{3`=WK*T4-Az8m4`Bf9w1?z@BhJJ@3ZGvcZ(by&JAa40+6-VpV+c55=VI9 zsGb`P0~|1ed%MNw7IWWYflhvAc(U6RZ~Aw`lOuXaAQl-lK>;J$SIT$toubH}|1c&; zWKMc`J;zj#XI<61of~3iC>Xhm_eWK5KZPwuJuM0 zv+9aVoX7=Ri$T4Y<3jm-oS`nKkI!hVS=0|Nw|0hWo7ULWs0HVLmEkzCQNB_gtP%BH zjJVU&KbkF1%jsG0BUUz1!=JkNh1~od`3Hw};T`-lmKxmoLHK(+GZC!m)BC^=UDnat z+{_a_f4LxFo;@x!T?HqN#YkUhfi{*CV6k@?e#Dp{p}`QQntE8|p##&-n&?5Ec4CSm z&K|B8qnL8ZbNwPALD^E#F+=u|yJKbYe;oEdZqPsJ)LOd{kL9`5GmILFN%Vjo=L;); zFPPJpxZ@K?gw0xmS97T0NSP1qw$l)DXbj#}nIcZH7vApZgo%dwsEufe@oCLaWuuK8 zdI2Y&SH{WL4bo>^r8w!8%HWw_q*?K&KX^(9-buf6s2}9+JCfY~lj_V&ee93Q3SZpz zXYXU|0n-p?^!;szXCG{kzjzU>9$4V;>`6EpISQqB2XbERi38C_ShUvw{hsJ!in|`# zyl#QTV|1WrtO;k{30r*?5OJzr;*)>;sr!|6-H5}U|2GH&TL-~qJUyV>lgkn0hkNu} zTQS)aV(yAnMh@t4XEhczE`iPP`B?Id`8{pT`1#3aJuv{^i+kZfQ8(0z5q|YBg!`NJ zX#A%QhD5c1in0!#G}l7tFm>GP`+uL_Al^x%OL!MvV^7V`c-RCGoApEXC*Dt8 zy$~DVhPTfgVVFUW@yO+vcFhVIk68TMn>VaL0bh~H$5Yuo2RZ{Bo#u%3Xwy9;)79)@m#gYc?pe;kuO$eh&+yYCs} z%qgN)t|ckS=WX870oB)aLnV*;U@P=5doO$N(Vp7WMigVrj4d?PPvy`eV_GoxU{aq>7O ztivpR-m~7;*lRZrb6ZWvl=AUdx^*<-Ck%xa=ky}(yb5-i;H6m)WRB?ytu2O_L#@_% z)^zRE63H+Ak9Ys#?|Xt6QtJ@vH7E1gI>K}MIK3%G_~Ye#UnIZs!Y^xggqAzu*h)Km zzGs7bii@Fkcn%i6n~Lqz$00Xp1fQot)JFBi+R$DopV%F14>5<>z!1|a^x+@T3I;QD zaj-!f8xQ~AXZtqmz9+u_H_zx3L#RRaA;w`BfPvKLv}5*#eSa@J*y)DS6i3`;=CMT- zc?El|;J=%GGXo~!qKd%e{9tJ8>x-%NJ@D7DuJ8`-gaDroC|=zLHMg7Nxsf)!cB;d6 zD0|vu#XrCAP}U9M9bChGe|ZP;D5eJ@nB23y+x=iN-v`B3o;bb274N9Cj!jsN_f40= z&0s#dPoIV!t;Zp%#V{Wl5vvJV_eKD7ieQfuN@8{hb&T+t6(_jp5Pj0fB@>BHG)9Xxec!fgE_ zOgU?Tk1-Q*XVXYT1@=ez7GoSU?FiLtZLnuu3)uhF6w$3U@YgFv^qN~EWk-r-XxvB9 zE9Aa|^Xh;6jlG;z{_{sP_2(_M0ha(-}6xI^Gv$bK)`OCeV5;|O|<=(19&Wz2LZk>}wi9N_) zBVWnB+t1`TpXL8>=bE>~y2&{UI!6C?rvMmH?|fmi4~~-;G;@F}rk}Ql7xyg*n-(K5 z#sYSM6L3Ix81{|p1-!u={oNJ|tIz6vf=1JAYRJlRjz_cL|^xzDU zt!M7bq1Si*+=Bt!QC1Rnjs80d^ux&P-uMO|J1TY68c zXN`uPLO3H&6&T`L%zRl$W>_)2&YZ{*ht*jgt+t3d2YCz?ErV;&Z|7ApKr!&+#Tcb zb@E`8>UBf(kk$}Gb=*w(Aw~DU$klUMqB%81Mh}P;|AJ>S_;rXl9ls?8KQ7CTE*E4- zuXAE{?Tl3Od+y1dj_w8WCkpB3${uiZR8M`g9G}IiW zS4N?Ir3ropx5JCAT4?q0r(~V@EJHOiL}guq_?ShDZLjC@J0VOioV_QXXHcg*;S%@4 z=jF=vbCQ{JR!UiSCeP+Md+6tSI}pRj2kK!)yxZ9u=XbhegE*mSmo+F$TY|Pt=it_z z@u<@?Wkx$^C2L(`{0hikTPPdN(j_!Dmfp^qUC5Q45w& z{~nFdD`NThisZ7c3-3TL51tv$)X=}B|3wg=V{(CBrntg5+8(9BD^W0DA?8+3LpROQ z^wBqgw@mnIUCG8lY6qj z?;i8R@5^_3-o>%*%Fon@F5sTSm^}9>+;g1s!L1q(*l-tZ#>``n!evNrIS+g0OvHy= zQ)Y>DM*c+|^re?bYVk*Twd}#hux^9aQ z%NtK6|Kl@h--3Q}tXubiGw>ktHS-}J_Z4esb0<%ngfwpjFdF`jIm38PWK zfz`e7A+RlslT@MB@~ik+q)U)8XXIEUc2kHg%%Ix=o*O7ko7kmdE`OzQqTQ!^kO^=7h;3Y;kGbV&pHJfrhkE zkS#q>zosSc3kCT6n=ifGlH^zl^^nV`6`9Vwusv^N+l;r=_`Z`)qdv;fj4z_`vs9wy zluJ=nwKUsUPv1QSY%o>9UEW0>dH+?o(9gW1Kj!ea&CYkjCkqD@tXPS^{P-ProPvyr zLr|;L89o}?aBr-VcmHHb;l_B;(SI%Baj#^YQnXYn#mk8WsZy*>ef^69Sx9W`%Z4gB z82eLN4*V^t*Ogd<{%mLH!_K-jr+6O?4#c$+erR~(1()@%I5%b;rZ_FfHv73Sd1J;r zsD7v@*2hC04`|xYg$bBun-o!`;#kXSUk|BCFA4P3jvCIgmkoavs z#ogn#nBP=_dbuimxH~)9ttoCd5`*2%KDH_Vom%)IBAZzNy3C;Ov=-;=mcVT0EbQ_U zl&b(R+Z9^v`Vi3 ztd|4>MFbC2!NkAS5o*)~IgvWhV%?K|uT`^1LG#e!7qXItXD8r&w4Y}N#UE&P-8tf(W{jvEYb1h8VaCFf+XvQqVv=`J| z#f(AH5EED#wnCrQza^BPr7f|*E5z`#cO=l`Gl@Q~$uf9)vJAORtm3H| z7+FHxr(9xlYou+T2B}F_z|&*O(C(#%WZwBHy@RmffPtTd8f5Aj^g^&q53oKsC&;IqB-!@^BC+%=5tUxl7k&CB{)%74t*BVKT`dvS z&)=ncs~WMiZ;+0S3h=X5#)Ms}ICfhd-yUk>{#x!C`8;ptUL&bz08ZQZAg0a@R)?8c zetHEC_|8LJSfZZ^H+JCUnIYu z7fCL?5Da=&%9^gs4qWn!na+xM;-HKvOH|prsN-p%1}gbHyK^S``J3K^EqzeLyp*LW z>+p5kaxAW!gPn2XaB*8-q_5J)m?7%$SX(2Pua`(6Jvf|-ze?|J<+8R(xm=p?RlYYX zlCu7VGIBu)ITmG-=};pdR@6({{@-%yx+2V7m2u&h3I>w@K9_X|jt)Yr9DnGid1Ds$ zRqJIP?OB&0E`BylCXB_P`@PY6V;e-BSH*nKDw)(!EN60xW$EW{GID-}nADatr}V45 zJWwb;oxaFNe(ob`-^DGUO1h5uDG6g5#nMgz_jHx8(@hy_vsCbr&vVzo{O`3tBCmNP zG|3egGwd+iU>WxQvVdOyF$lIXfnFCqSOluTUWvOg*J9DsFOm6Yze})rr5qz?|8IqF zvTIkNJiAZde^dVU-M-0T&r0dmu2$y%u9rg3-!grcBKLwy=xoIu59{vg&n&Mne;6EQ z-XgO+gD%)1B5^79?6yGJ0b&dP^kT+kYal`yfeID$Q!JL9Jxjz;uS^#Fs+6*0m1244 zo5X1sF$eyOtg|YT@%_F@kwc{%RjrjfpX$YVK5>T8iuiZ45;~1irUx45&wfFiS(&v= zKcsheT~;JXlpF%9E}m6$PK_F{E|1dLWdSf(Pz4OGUR24&o2T@CIBUp)6m zzpLKRPj!V-rX5lmxLACZ%jC9q74^$i zqJFVVKJ6-zYtM?Ls9&j!Y*{WgYigv>s9v`8`z=w9ibyi%_jOVQar;$K&$<~y$(>C1 z$MYm_tkmbeh`SR@&VAz#%*NHbV{rJQ35GNKA-PT&9qD%)-Kj|W3@R4!{4QR7tK?SX`e2v!>(V->nbwSG2*m8WpH*sgREPMbdp&k?0=$Cf(Lo zN{=?x(&KrBJX!T!mOW=S=m-85!^wGSXpoC76|rTWGHyAkVw$fyS})ebgmf*`uzGZ1GXU@%B{`v!F-{(+cS$`;~gw z3W+98bYWYiETX@B_Y39n%A#6^wXB!!)Q2R^QD(-BD)fGm9_$dWrOGGXBi*)cQ zmPP~i(Yq>SS9GPQ+f>Qc%o<5nYaowD5gH#y>ck4Cl5CO>Ctg*a+?6t4K&BTS1obvupX@D z>0=H1_7CUDL*~v({k$GRBc41X){%sSbK zQtAJyKz5J|`gL}Zlye>_zD3@GX00ex{FW2_RS?#Ubtg38uD2;RuhzxBh875a+#04H zZJ8~_J?vff9Mr;en(B{j3d9svyL0|@#GSFLVC8LvZiA*G_TQnnJG=`Xx70-^V##s# zWzr$~i?rVGS$?sf_^nec+sV&*Ox*F`h(=Lrp$wk{HKZhHB4GsI9UmP`{HTl2pq4n- zP7fV8mko-bS1&nsSAzZVBAgy$`#gvb(W^hk7Mqo<(O5hKrT(Mw{) zKC|v8ze&@j1?;sxNzJBwYEcWt(D|Fx=T=L$8MVzfc<;SeMh{0-%q~^qoFTP%$}5`SR>lo+zCsH*9dBJU(6`7kPf~n9kmFM5!#=Q6gJ4OXb3r zuX3pKcgbemrp5F_bmCn|z0_4B(Cb1 zVA;NQc-%{yyB-C4BUXq*e4)(a^EKZiUlw&LkkY$fB<$)JF(v=GlA76nb8{uOXD)N= za;0R%M=56AI$|l|fAcq{KSb@|0A@c>D>>DhUQF(AOyz9oX^Ydz)@T)Kfqxg7p@Zij z44%e3N6S{&l%@{Nf(8j&UnXH;e2=?-k#2mimbw?n@;&*oG$mI~*(v|QhZ!Kfp&8U{R5?4<%TBWj(Odpimedu?p_oD`==51eUds`+58!<98evy3R6soDR8$xYOooW zB$?u$rN*%M*%mj?HATG&u}HgLV(;`rO#4(yI6aDI8WzizF7)L;M4#rNoTKU!rOWPk ziAsq5gTwyA8ENiQe%I`?9+=UW_%rvhwZ3R=?hS_%?g;wV37RT=zV+F z0$j!;<^51Jqb60SXD8G=(Syy^W-xxO1-&Ke_+_Pnr>_*S;#94Kmy+LVQ6TwRxuUI* zC9UZv@ZX+afp@S+I{Edk^frwM#G-r5(;ViHgdV=AweZ45FZ#?_IU*|37WRjhVsObk zd~0Ej(A~gg4O1MP)Du0MbwX@zTSRczudStvl0@dc7-*vJRu$+L{*o^zs>O!<*nx>f za@4cnPu+IB3zKt#Fnc?-sow&TrW6SM`~LXyk?+$|Z)DU^d+g|p?fdO;dy)-qomhwk z39~TBVj|w(8;R8y25`pcfthbRW1ep$rXCL23>D1hwU~i z=slW+b$2GAO`|}!-%yNx+8;wcn80sr4=9c8h7nDSu(DN0>~qk^?XXt({a6>*>1p-r zdXqnOi&%HT_aKCYkRQaJcEssG9J>+#I}3ki8T;a3xfjfsy>Zpl8AT85FrvdsMD1J* z%%6*{pUko2t{DofW`-!05Kf*rWHS zt|jZ9tmHZUig*fl4$0*A|6LY<6`T>LHYa8l=M5!#6Hj2SOGJ)6bU&^}UfoiBceccY zv(z3dPC;Ldak$-O6r#5c1&BY~pD+Lh^tNA{-3v))jbYc*2s65M#9Q_DuzTM2Pu;bw zJGCkZ?O!t^&yE~F@^oG50`Mc6Iw)g594;YN$Nu*3X09+daln%?w(xUYhU98XJQ`#{ z?d=paKQ#`1cj-~AH4LvdnZonm0q{827d~nx*m~C(#nheKX?Dcc%yziC{r`1sShor5 z4vgZSat-gMMD{$D0q}_MN1mP^{8PPQMV(IOXBX_KB+gN@8Xirlb@Q^s$^EnNG;uN_ zZ;r*U<0J7kb_iA*4uYk2f85XNjWDI2=<3=PYi=2$(!D*>2DZhFl-7Sf??l%9z`J4E zbLKxREN=K)_ z804Z2!{~%T`0%+OX7BEe&RRX;9N!hwvJJ`Yrbp;HJ=|*55`C}!A9wz5=YF@0-*tEp zHuC;YBfdY0XIAee{)qUSS?x`{(e{x$PTM+D|H@t7%9XhPW-)d;&V}cFbM)(AhS7sY zqKmsJ*0$=$?BiZ&->W;C4eo;5UL8@Vt&c^QTjHBuGio%NV9j z^BhHu_r^dR_YS~ERexw}_+tEA=Bk9bAxXyx?fb1oKfM*W^T`Trb7pZLFo`*hf?hWV z!^E{Op4{vKt@iZx{%wdQf3>GRS`U+_P{V&v3;9K=__A9G1AG79?|K;PHWtz|m;L)P zLtQw!!q(Ezqz-8yDF3TF+I$E7u>gQm0%r_Hh5jd*N@MlZM2dC-0|s zZawb>;?C#zEc^Us&hmEZ+=hCfj(!CX@em0WM=tUU^0K#eA|&QBY&X@eX>rQ zcf{4yHkf&>1*)@~qUqlnxG`4=^UwWY4pOPKqZTQZKKc9h{{KEY+Q7_*OwNV=>}lfz zVMoqt#%zDQr$0m7-(K*gzuYkPJFA`7V4l}9^e>)|FF(!Ey6G4+JR5}cZ{6|C#Q+C*>jq9c8fB9j|s?y_=vk>-nS&O0&gk|h81C7u@={RelpTz>`@)uIRPVo}A59_-n;tOnySX+qntoLmt%ni6*dp+7UmiTO!`K z3Bm@dqKeE(6#l#7)v*t$ag@q@YPuGuQjY4>7P4)0<;o_px5p0rt>p zP(%Sm5D=uh)9&u>ICdQsk9zIyZbd~=1Qje4gzr&*d_T?@_l`U6op){4dfz#pnir1= zd(IwV5C5O=jddUJtkU|2dIsM4`iCo^*;0lz7Ywih|-ERmQ|6UYU-Z&$; zyPXow`koXH9%G&q>!$H8@?6L><3H}ijo|Kuh%yYFRSfNe`7kS|kAm4yn^R-)<<}6z z_3VxH`@InqWQT$XDKt}6aKWHeQ1WOH6iQzS>4TpOCF7q8U9LSAmjCxaXaTdoHeMIx z6RrsN^e+ojscE1&NqEh=E<7{7@m=0iRSDmB<=C~JXCrrL?=a6pgii*ZM<*h<^9aOD z3W38UVA@+(Ol>#ClTg7WSMg3i!K!ade)uOsgq&s>NA`X@5W5%Gw={XWe4h|OU}O)9p3j6>3z zVMy847tN(U+_hkj97`$tWn~QUY7s8By%jLNMtCyzsjzRvGa=Oah0yfum5}Q3S}6Kj zE128Y3e)v#h1SB?!UegvLiDYA;Uw#xrapFf5U~~BL7m<(Pp-WfsXy`&Qk=#8V#(Md z#9-3#!DyZph`iv=IGbsXj!#;cIEDCM-AAE@oXz)D&jb(UXF>;aaeJbAto}u~vHQCqx2RRv#JY)(=%u&d+_0_!TLzaQSGyPv+$-&wmVsaP ziTE&hICUezD4-Mn7X9-|T@4^ztAc|Yeh9Vt^}^+uFNB(f&xEKi&xPJ)wSvk{asW?0 z2@A>TS!w?e#uO8~GjHc!Z*s5>+(|Y^5tXca_a^z@pUhD2Q-OrNr7+blf|phaVgR+UM{8uBC-*_(MZ+OX!K4R`GJ_;9q zFc-|}m(c&U4D1fcVb5#@cwbV&nhaIMIy2vhb#>Thh38z!3iJe|6xOZmu?&TNEsNsJ0m+yrUk3I`F zifzJ#`LZyNSHS2UO1Sx!S)t51lo_Ul>mefizI!c?@qAuSE%11HQ-_q`?yG!eE@wh_ zVG_KzjKH~d194=oKVB3%;^a~(CLB{@zw=#iE5?oEdH7|T zfftUW;T;(XgV26x>faSMyRA{k{R*)&<=|@3EJSsFE6mTW70z?kS0KLfrhxm{ZhaEE zwYQKjlfl9#3TW_DL6x#PMjz0G^;{iH*sO;hr`gN1ZYb~F>GXGJZ7+lPX%Re^=HfSgtt~hqQaa&DR4DhL|B~j0PeHvu4s|30LR67Df+!BYdZ3<%ME{5c!}< z*!~Z>S94i3SSuoMwhFctsbfP|EgUis;W_tSD0k6^66?MuPPkXI5`w_Hz@iA5xjAsH zPJvByET-KX44?1a;hgINg7oK^Rrr zB>b|XUVpC)My}-!2y(_}>(vmwMHA6wI(U+$hsGgNEaROYIE~oZV`}Pem10eDA*#Z& zaZsF$<=P`rdWtxXK>++z9pPs$L2<1Dv@Q~NJzOt{wBHJ@&)y0Z{h1>l{$427Y!p^X znuH;1z6+Ns#S$CwQyB6ROBJ&R1#>LR=aIhwTl5 z@@n!babJb|>)M3luVoRZsEDXT%DCgNhQtfR3nR4g=zs`^*=tx4%c&`^z-0|)eM*_Z z)GrhHhKaDKi$pm3p{l<(PEWH$Oc(Cinb#qBrGDgn_D&exjheYiYKvzz3eWa52cxg@!%v1;5W9g}SS-{;;-sL_#S zE}{-|iB8ZXdMg`I*Lfdwih)(xAQ)Z-s(f5{-x?z}LKBC{Y5sFo7N?HOBK?va`t+7Z z##VZLsK*@QD2Mmv3fN+zjNqPXFrsg`hFpeZK6QVq_0W5s1l|GqILr6y+l(so;9T$w)hoOuTGssjCxLX|?n>F$7n>GeN7om8)9z3)ph-TexY!bHLA2HK3~ZvZZ*FGo?T0k zGNJ%=`!gXvnuyw$QMhoWKa4N?Aaj;21`QOWcc2RH6v&~Pzo{;NPuUIy%oh~#aHc%g zQUe=NEQ5_}ct&a{;NDp!w0fyxS`hC9{+$DPFN`qL!53{2HnHxs9L`5mD{zweqIXe% zj%k^Qsz`+Flqk#@&>tRFJ{Wk(2Bq`#@MWJelI`S>Jx&%iZ&Q;NQFlmI;(9jy$@Mp=Z9mW64AJh>W047s5Qmo5sJ-7E`vHCfo2 z$zgT50?s)rqhXmU9JlgbDAU9adu{GS)4{HJBJ5z@KE&x7XH_7hODSeY6+nJ(CU;gP z!t6Nr|5fxyayK7bJ8T1maB9t>lyUTuENXrEOrG-iFhUXSUlbvktAOxv%zHQ_i?qk& zwD~(2wNbl}r-m7GG~k<~g)#0r_*E^!qFcIucsi3p>KRs4;6G~jx6LX*l}r{I%#-M$ z8;)o01E4GGi<(?pq&?EZ$Hmmv(Hr+#K^8YZ%AphedDDIE~*Ib zrh%kZEo8@t@M9+5$*mH&?bJu7BtvXv-HMsSVH=py-^%xDPa$R;&ql8e$uRyOIjec0 zXrniw>uD#RlLi=*rUo0%eKE{DIaMKx)pO+0O__hoK8o;AS7aWSB1F9R)*e)Y$!kse zv~*D4Lk}aSOR(~sKBQj^iEkUDopndE--ywvL`GlkAMz}M*~}cQyP5)-=vXxG9gIiE zyTf*o3%B-1E{D=^1-KJ$AIx_wrA!H9o~z)!fd=mxZA4RZ zzNB4@6T|f}cQ<_#GW2X)F<+VY!t=F!-mc8Nrk?QfX=d}K<)X7R4Oi{y0hWYg+o2$Q zd+m-^Z8Ml{)WYCK1+)>^(KnK`-nJ2BAEeDr#m0X;MB)Xs^*X6~~bcGm~D zMRv&QE5T8oL7`=Gc%UVNuZ}Xv94w2fBXU@ut$>$rlyD1buxDmR6muxMpAutws6M)^ zHpGmt##l4P3^A82;K07!??e@P?X85#X5zjJOE9c2J&VJ#kl`@~pMs+?vS$eM5Cf3^ z*a<;P4e>5i1NsgMkfh1rl5K~u)0cMvz0|FR@+fmxA}+0p(j}U3a~I*KgBVwczq(c% zV0M!cwhuBz^H%deeCU4{$sawa#E)O)$RvL=ABD_U%tk`L6b$!>#d&>ZTq+0RtFAlp zXPH6NMF#;hl;A#0mYU^u!GC7E;5wLl4gKVx!~1X7Ruych(ttnj+^L6kx%X0x@as~{ ze_+5&5n^eqJ%!JkMJ(s^uS)C=t3c8DQv4*op5>K`IN}S2;&>D`4~2S7Z}_kFLZyKX zbzow=Ij+jSS`G&TJB0iAB^>GBE_{9aMFr-(gkIFEd66CTl@Kd6nG7-lXxLJ_N0tKeI%I=<&Jp5?1A^VA69{yx5|K?cwT34Aa1p z-J0lRqmADs&V=~4-I<`&qVF2NDALU`$MZ@OD5>SW{4IbbMxrr!9V)s_3y z?9s^g#4$+|V^S5dh(0+h_cmc7YYSfOLg7$Z*wDYXC0G%HsuHH#C?VdPbKzbkJX);` zs~sw+WnH@lW)&6jZ_Jsn^)_eVar7Jo79sLfF7Z3=l#CjUkiAjJUJ=asHvoZ(u6Q@Z z95wa2m|3EVLz?n9bF5uxS7{S2y=xQnA9M&wJVUP%)5;kphwt=`eAbo4yPw4IQe|;N zK@J~SH;&kI-K;8X-_C48p#tyR=wApa#z20@TDLQyUN8ptpO3`3mch&r=!wtv%(yJE zgUkp+jG~4m{<#uPK9t3O+uDVe3?)9F33lI z7Y?oZ&bgU+$mBG(uOnymjCshbDzJ=Mj6z@us;di_Q=5&=p{dyNdnT2MKuz!8eSkmC zymx`Y1Pkc?5~FX6I!tycLa&v5-6$E@E3wBq(;@V5X&27y`YEh8`X;n|X%Y(ie-sqO z9|Y69M&SqR?)=8QuYK$()#!aDrgMLIIi8!8!mpisCHLnd;Y~V%`y@hT(+FyF2cerz z5XL8aVULLeK3JJxvY-pq7izdEQDV<4kFRcWc;dnN#HB;17yT4AQ>*kQ@`I3HSTA^V ze=De8ee;KhZ6psXIZ6LQXL1@sB{UzDqx*C2r?)7^v50)+`Eb{BVKRC=jfHvEP~5EQ zhZlSOG2($64&JuLkz7N>3>0C|3Uy4Bs^Cg1_hK-kAhtmk)ser112?`3O^qJ}?YHlQ zb!XoQW0%+d#bbYxOSwv%eF*15Vs@{n#qy@dtH!(pE|&|iuy+o&4M@X4Pvq zr(VeKTlWWV*JW=f+$UzozB0F(d2*X7aF5#l;`S1BC8o4VmwUC(XJFFLB#4&ApzK^Y z2EOfwt+NEE$9KVAGe_*ow!qnQhVU_#P@^V7Uvq7Id#{cYH;hT`_{zOsI-N z(}Ga+7xzNn#eO(;(jB$*nop>*#Np*8kTo_$k&P5aS9Nh?G5eeoswiO3r8YtqVP(Gr z;|o9jJnuW!HG5M9^OaRl5E1KVF7Z#*3XF&?!w(TN-aZx}|6~r{M5IH1St57IjfCrz za6~@tkJ;iL7=DRZ&r&yB|KTEHaP6n7*>*#Aj_3l=;d<29gEq=N066yZlM z_&2Xn$vZgVJ#%Y!@T~M84xP`vi}4k(d|if&yCn!zD#DFkdE`1XF==2j4$H+s=f6lq zUkioe!QNQ)GXSQ}-o*ReAbshC=mGY)6=H)eMwYl@Ws1=y2IzEF57pQW1VF0g6u1nD}xC z!V{RQ=M;q7$GTzBY%i?3>W+22T+m~r6K;pvWBF`rbQqZ9=vX7%87sx&ay{y6{`z6P z`3;Y6CLVj5++-X%q%-uYyrBPg1NB(%si#;_jN#b@coLa|%+Bff^&=7c$kT=CMX=`z z#g)K5+&$f$c_e;Fo8g7&dLB3|xZ(T|XV{(QuD1#s$SPPsM$H5^Z3cLC?k^Afn-ASc ztiOZz|5{=!!B6SsQy~Wateo0WW^!nkpu?mP1AFEoC?x|a)5c)H*->~M8wGdzt?zpG zM?+)~hR~a8_RR;Krh4J}|2#11of|$BJ7ZS315SOkW=5zvG+&!Q;~%3xf8WWh`{gTl zXZ2i*QYPju(r*tL^f(UIze1T{PApRHj2*$isCO#akez`EmjRvO-5 zc0oD!@HSWCp(Ar-b?C*NQ3eUUbz|AjyK3dZq97Bk&y(>pg;~s@BM|g<2tKV0#+I61 zs4(u1`pMl;s_FwdSN10TJy3VU6=fTpaL&UHd&oVRpST~Gy$6T+%lv2J| zA1irhGHY9<0`^thPcVjAn9B;W?qe?ByG(rAmW&vic)a*B9OLqapiwOtPVVHf1_Gy7 z_+h?GSBT|3@y|DRIEA<%huPORBW;KwSYUOF3Cvdi{hdo$cQk+JE?20r&*Pq+SCyE< z`Qb7-j;~>5Sm{uLt$~F&S(1x=dos{mGX~W*ahO*fg_U0iBe#El7(Sa%oi!enx0YGZzDW|()~SOIzJ z{N5^-LGMU0R&Fc6((O5DSd$LvxFp1r*V(;1f*z$%B#rEYPj7(0DqlQ_@x*2FXlv@6 zv9y;XCO6pN_6u_;pD;q(04bUqb@6oUU*1{Fx~5IM6OL8kVK_6~cUHpfNCi5Q%b~uc z6jM!$@$-58w-h~$ysL%N#N4;5{^gwy^W0e6$TN#~@JjzGxJ{_ULh{aM70RKeSc>i; zMOeKj53}1dnU9cy{Vj2ju*Xb4HV8W+IK!9-+!5{tjRIHZ=sRFWP!_PjFC1) zirG7Lkglo@+YBXGo5>^Q{9oR=8|!`}@4RCJ`yEsEmmz=mFqluFmoTIR76S_LqaQPS zjWTd?MH1FtiNUfvLm^TM#->L-@V&7s{;hX~-3WWwr&yx&pb7Tmb3Qn&heGb`o@l`w z>Kr+k*0&04wtp462LIhxK9~1`_!YJF^N69fkgL@)nB=I0iJadO z{k{qz%q2PLRVRG!_cs?}XB$15k9h|ZZ_vD4iI5W&2=85vM&D9=NiE{Ll!q%TGO^5o z``PVBVgJTqSo?GU#J##>L4_wO205Z>x;b7}>%)~;U>W<{vcamr0R<>bX%|MTeHJFX zek(MlzYvbheIgw5{>yidZ)M$E)aHz$UWL0ZJ|z-EJ6wiTezyz!3gPiN2SR8%UP%*i z$1NJBoW;`{d-1pP#jSQ{)D~Mp>7YK0Jan+&CVAkeoC%4Q>hx(B;^{L`)vOm9O=zY&35Ba$gwMe(!ngrHg?X#n1i7QX1l=yb zgh3Zug`QV`3h&KZg`(eogUjv0Z=dzCdUCND{GEqXqIh08?rD(|T~vssTREsnN<%|q zJo=e&uk^S6Q2*WyUfrEBV7MuKh}XPTS4OvGGRUz0DR|R+6T0Px(478D&^a!H37zGT zd{YjY#HnYx$m8`VdE!vi@}5`14c1*lpS3)D_F1}>cukJ*SN{?$N-IEyWj3E91*W@V zQTc2Ll=k<+qk1m{eY3^Hx%_U_G;o{Pp#HynCTfgDH7$avew$#sSq8g_$HAi}oI3vA8-5CXHI>gR;kLq@pbp#`ThMc8*|_Fqw}u%POv@aT4%aQGYleWn89O;lhWNu1%X zCbXGXS3Ow-gBf~w^GbrD!woQwb@Ol1JHc-#+Pwma#AGiYDkRpLgGWTHN97mYETb$hUn+{PS=+^6M8uP?#kD+L(aiMfpPlJR~*3?}Rx zi~|mV82o{{dL`~Bc_KpVYh{G!%HhZEcIG?%5{ebug=>?3zf1CXep3my#B)~DFVQqb zgz55PeCsPEA7+4Iy^P`V&J>+kx9S9ElWmo_GqoJ!!b_0&JRf^5Wuk6*5`6EDK)PE9 zj=y7GKrd$;UTF;1L@li49+-ncvY0OJ5DNck7q;E&5d1=z?|6atYhPt7?WPW)MGI!; zx~QRMGucd^`grc3a5urOt7iDlyEpC-vEYfE13k-;db}7_U-Ga#GXvvxj;7W+3ZJj{ z$B37{n9ypE^@sH_V7fZC`6!^SmKt*z8NB5=RdGZHXYR_vxkDb0dXft;R7dJnEzJ5M z!jSc1SZV7+GRY9;QWNN{FvD@ywckZvlDMeni!ubbGCM{m53*+Ih`SY!rw4}N`>H;u zpcd4O`nBOPVtn^k#YHQ5jCd$R?S%~9E0fbnkwuHK918edX+B_IQKg1shJ0Q>5%iz& z{q4oyd6xl{Y>m-#yeTHJt}(MHR5=6c&M#xma}hTCupanXUeY`#2>87VG&UPTX6`wg_i(axnc!3eFeD!rzzP+~J1eeWsY| zpp7~?B~1Ax%iJwl+!4$1ypn??N)AD4a(MNa`{$e#@ne(<`t4Ci5xL$M5)lRt(Zjd# z63pDJ554(@Fl62Oh1?a_N}ON047c|4d_0j&tUDP?G-99_J_z=!826>(f)W>FEcfI0 z|A|>Nr)1H1QWh_+$U%{J{HM)w@VZ5Q+gT2q?G$jI88RxtYS^K#iJ67k7)@ToeS;q4 z7E4gyop>~RtFqbjF1#c^c%u{_h7{tp6FtC>#$floX!JQ0f-S28aO$QLhT;-M{yxbss5KmSn&8p($?X=BbK5&Dsf z^;DO@nBUP9zR!7uJey_8c}^Fjzrbvh*mNlQClF(fpl_-#by+@;xoMBf+MHv>B4(^= z@C?u(N2!5hQJR?dMFUs+X`u6B4K&BmpYJWgP->2MD(gda1aoi47(+Ia?^73Zyt`w8 zC;Xik?Pk9b#a-Ie@)dq8A{LN~rQ^~NrWB9I*SLT9ac{`$cw^=wTkPy@fMPjn%&6bK zo~?mi^!D7V*TldeP5kSvf#;nxknx`;<|pcKr-~k)$VrhDWB~gBBkblkboGWQrU#kR zkIOTX8n^j9hXcclO~#&snF)~wZl z>o6_6zoUf(=QW{js{xgT>Ue%y1LkhpnEaVqjS?}=uaM$!kpWztji9s77{U`1YP?u? zO%?v*oZU|CmDsHac~v=3xRU~n?Xk$(6b7G{ftWRm{gelBqMKrB+%&OWq=7&cP5dIS z^QDtERw?nB>F4qqua3(LHSplQ7VfPR!QWKOy)sgM#|D^k&=BfwtlQfJL9F|X->(M0 z-z4V9Y}PNr-Vr&dTAhMj(XqJa9ftLxffzxL_b1Mh-@1umvzj>#4aDRmn)u0Qy_2pD zo8DSHOEhqKojTk%X+YYfg^D#IYIwz{aFHTS#sKQu3~|cb7w zcq%D^C@KdBmvWCtXetcfQt`J4;2kus6> ztNAW}rZ(%l1`hYo#`TXP7$%G9hm_(L@7(ulMp!h}7_+G1>dm^2tLTd{<~^!Q4N5+~^ z$k%{8-?7j~+E_S;zCCRTworexNJ@R@9V6UeMpgM@Qw(R_1N{5iF~g=Tp$wW~MW{QF zgHguJVEh@2E%jlDlMBMaw9dGE(gN9bV$AHJ35Rrbyk4P!TXVFKH(eVm>a?(Dh9)lX ze2VPPthhZoxRy`P9>1eyo(AY@Z-kN-V??bpMc`L6l(6nLzR$auH*$AU8JcDlVVW*E zoVBU2*fk19*N0$(W=}L#c_3n!CGwAnaUoX|ZM}JJ1@hTk`JAEJFlf_)Ppc->ZfIg~ zXL2Xh{w$a+#zR+q^dL?=rpOpsv8GsSXHE|&aSqnK#qalZze)_=SB8|wML0e^7lT65 z;IuLhRdPd7T+s_f^uG+&;{3i>g1wJ5i6^S#@k4dQus76h)Z+dVEqooWh5x4Utm2L| z`EV4R0D7U}px zjn3!FVfY%{2Uq32G3kvB`fC2Z_qCA9ej<>+bvWOx`=hjwc|;3~s0}IPcdMu;Lc=>f zbZ?X50W<3)ZYGG1FoS8D1roBXa7D1eSJpj#fEt~ON}RVY$J%wp@IH};NxB)ht34Xm z{UhmN><1y!2Vs8pD0*T5?Kymph*RjOsH1U^2EOm0x0mO~e0MEmzbmZuqD3UwZ^~;wy@;A@Z%`YjpfWVPh`(7E{Q***ja=5B(I!OAcj;-vCix#LO_`U|79@B(N6*Y?Qv~f65m-~Dr==z3d z`AuWE_cO!c3JW-HCT@1q21}pXp+D=!oS~lZFf-h?(VsD=1gE?U;O|dP!DI|J4vR*@ z<`Br{378P>g3w>4m`lIp`m-9`?WYD2`HWlj>iG3p1D$lVu!7&u`jxt{X_eqsgdrSy znV|mxGh*fz=r3+ZYEY?lX>Xlg2l0%xNg)NLLblMUf~GIyH9;MQX1 z^nUAsR-MkcIl>YPhDji+tp!mHbCn0HVM>b{++}$eMQb7|UmL@+baAs@f@Q&mnCfBz zl__S3UuJ=W+pOsOuwg%JhdHc!nixaN_ew1At>7;DQs`ePgwpaHEF{*P@M9Dv@1o!5 zXfM3&;swK=wkV1>fNHHa^2e&Ve4J$QKGqfaGwF(=nZ+i zk>BiWbF8eiM6bEla9(4J?W~*pkbWoW-#KNnW*OX;79rtkF0B00Q9CLDw$CCkd2c`D z?(jwME=OF@GNETw7uj1i5XC(`O=GB0swXcySREGpu551UU?Vj|v&OQY(KO<|R}<*( zGR5t5b99Na#Pk8yILW$`o)a7FLhSiW1)fhU!;DeISk#sW$L0)JUPwfWbOeT-9>{%6 z0tU=*B|l+-(Gw+jP^^WEYt^8vsfxEUs>rZYgTQ;)j=%L!W^88sPr|(@`gpM15MJMn z@FT|rsSak?-e`{7tgFm9)FgyBp(5vB*>d!>DuLLVyW!Skp~Z$7FMVUs_;fI8`UF9G zwhN2{Y+;&Wh$s~i-aJ-^75xg^=;eN`r;0FYQqMfp#D>>8C>XDY&`1eBPvU)4N)P)g z18Pl-@N2#?e$U;n-;r}B9@-_S5_Qa+?4aJ;B!Ku)Kn~_+rlNOp911=Rg{EI$xOMl% z(X~$KuVjX4DPmN<(1fcDzo#G-%ow18!)H}-%3T9$x2Spgrh|UHb>TwZ)GwC47zs57 zP7<7rlj658)${{r{tEuq<<#bKHmjrF?O-4FS!X$mu1ZH%%4j^G4`#ddK%96=-$JrG zB*E4=vCaTvM#aQc=53e1W&<+}dgxw>d|A863l|7M??Trdk zN9^5fio2V|m>Zylp?#>yoS=f%X(|ZYr-~!S>ahOF-~5w0qWpM=WbIgdUkp0liiGkO{3+m&GU^a2Df&W6^96y~hN;onKa;49xB@2G_y7v_eK z(bhOakKRIJgHPB~NxrM%w!SKo^Hfo1OpeG`4ZbO=Q2L;ZeVI!5UZjX|*3_7rD&i;W z`hI6mxt}|MREY=Ib0;3L$w@6Gczdc4^^v(4otRF4?r6+>G#t5Mp=ezn2=}92NL=QK zRf^{5Gg*qqBdMFC=jZ1!HK_McgMqmkA`6ITJW<9|;v8e*}7-3*fvf8_`9n%o2=e=0GIGdLi)42!zjhZ^%FCggaqY zu-MaQ4#eZ#tuJhO%`Wk#Yh zeJT!>cuNjvMW-?}oGeCdQ2`<(IhdT9hA+&Pn`#h^x(8us(eKAhM&LhxZ?yHKKc|-W zo3}L#o}0tZ*c6%SM(ANHMaeV~&M>23u0)mlLzR#+^zR*2$9xcO&MI~6hiMZ_HT-;xmfggRi>-k75)cqn*;Shp5GMt_I`@?Pyef@$fMpimP ziThc_j@H?U;D(a%^;%hY_88yC8e2}8VdF_7?yRNXuKcf0Go5uG@yt4NiC&}}`o4H>nW=MsX-zp)1Iy@_ zuh;6R9SlBCr`#Q&b`yo*!2xq+l5VO$-?w(%Ert`qt*=~?G za)wWnJsPH3BYd7YHk6z2fAcRtIFEH#ke9qdK2*1o9>Q^U7dY!F-#& z%<)sr!zXg>qb?*vH#na8Mk6rg)(~u76pXU>y)e(WJCdurAwJXxeaJumbJ_#Wscsk? z+X*u?>@ko0Szkp93_NH0=kL3Pb@%YBoO_4*g30716_}OSvl5rOR^aq0a#v?cpd>9q z*!(;sn`WUiIS!+A@<>yn=@T4^wLu|JJJbiWq=8Ua8GwFW{Ky-2#f%-E*fNqn;8+(# ziya}gwMAc5D+p)J|J2>Xy7uHY9C>yWP2>Hd zu<}p>LLWpUc*{_@Obemjv=3~(195#_01}M-5arbsub7!}`j$JsZFR;`Uk8M^*kE^? z1q`;D{i(Z}b^G#nZed^PORf9zwo2^v;q%(koBB@~2CprF@P<2?$mzIU%|uLp>iuft z(Q;=5&SZz9)O;WYGxPLjes>)EUpL4Y`XFB03nyDTqjx`7Tzbo1u)i%dEG^;m!W1O# z{?r}Mx^KUdXTL_iy@>lo-c@3NQ6=)eGp8Yu+SMKml@b@l8hN?am>;g zjsZi$P`|NAMQ4CDrw?+%iA~(6POYZRI1`-pV7}9}zr6EM z)_w4qwa+m#a13YgtJE@3YuCWcf@*Sgb0Kz&}Gm z5#Of|`aJ-WihXfB))PlYGgthFGsN+ZxFO~}$-if6hY=Pplwypa$20c-d1pEHI=;u4 zbrMFc>{iYW`c?)@?vGHeE|ahU;i_%!r1rYdGeXGC%2KAVwQ> zLwg#rCUZBawL0NwtUc5xSws5T48GKQP6}a0$ZgI}6Ex6Z|5vY4&2w@9dFOk3h(r5O zi&aK$KeYnCK9s?(wgf}WikM&cn@h^XJnECnGvbir7=?DbL0ECGH!htDK-xGj;tH<# zWbBBg&unnU+LC-fadlq<n?iB9P@STEe)6(7+48; z-3n|bZ#JC0w5$gG&doX4zBwH`)RT~AAA@zBhO$Q(fGL$d5L??71NyjOsl5ZdIRjbc znPJgkBXnCM#hvXUgjs2z^ggv%i{voJxkKpQ`t#2{xSaQ3^-JQivw0^qQ9mKW=lzFT z*~_I^7r-417xMAUD;s@=q{3b~0oAd?VX|@%oE>^2c0o6oPU(!A-40kW(-N-KEUb4o zK-(7LNBebf>=p6;ndFeB$Qm$p@V!+SufUui?xc?!z^qW78+*w+*YVqZ7(>tGoJw?xq?dhN8O8?j z`%&UBmxU{L|MBVPCiG0&{G-lkX1WKI>*?=*SUMiYSKB zzI-%Xqkld%1wwcnKF)}MYj`lyy}M(}hc0j! zN*H`l9?v|eg>U~M>`nhFcwFWlBhdTCZ+8{X$g-o<1(~tG(qivk#jL_XrEm%5e&?=v z7;roTYt|;gaqmc6k{Lq(MsJ+p+5him7dW|Dp>Uo7Cej0GJcan+J5{XRtAb@ul_5*5 z^7J?bEI2HSS1a3v9dWI~lBYie>z+S^-+lXUP0;9Fzt2kU13XoM4IRYoX3+24iCN}L za&Vr!bM?{$>RF;t5J_TjVfbSq^pgWa-V5!#CD#=2_KsHQ%kv%qhs=e)K%n-CqpJhkTs6l?6M^ z6zV8PL1Fb!DBAZ!>W*&EWiK`3yCtd*=+i%{jox?E@YGoa{>;LcxT~PrR+Su=DjtF} zdAJJp7BgeUMH#|0>T-Ql@QHP6KXLY*!Mk?__dAO!F!fO>CbNHST+Hmvh74@UNrLI? zXguyozOEw(8Oq);S>=FPZ%y!(T*WCZO=QxSw2SYq%)iPw%$oOKs==0c^wuRBI6}-a zHJCf9mTKS%vA&@Th$FJ@K%P}2c^{Z2k*nLmd_?b39JyAAy$-q9R+mQqD0h4HkHVNs z!AQ9w;N1*&JWHlmbcz9YAuzK>Lmje?D(FLA$t;bTy39CNsMo-=hgujSL*Ea%_|GyT z)L$22zEls#mWc6;bsgSuE~9qio^vHO^&nR_r3Ch83sAH<81swG;A=n!)P@@kd$W?YGz$lczL#sf@w7s)*Cnz@%m^7-x#0;i!jueZ<_r$N8y5 zis1&tTHYA`;kyfKS-Wo)HEOs7U_#AE{Bwm`k;6e)e#{fQ zxC7uZHFA@_iV!i3cz=K@#+j<1(|Q#go&S3d)`asJ`nk61Vd?`30@C$ieZ>HA(MGr) zV1lL~@`HR{;W=|ZT*z(guE6x+WyJ7{ads{_!lDd#3{OP$tKmph3_+KEfU~MQ_S#!v zBfSvWhqWg1V&Q?Bh|W{t=FmdzoeO)f*{W>|y@N2n+Oe zaZgqgx;xZxwMi8VjMQ*Qmp)DtO%}r zC}v$;6u;-g{1b-;ObZU(-(Ep(x7WC7{;BH2Eve^VTE}P-)M+^MwWDQB2EuvYs zf&I<=qnu4PGV8Or1P0VwSxd5zBN_vXrP0VL2t|NKcisVR@O)>1vHwZ1Q;ymM&R7%t z)iFDi95eYFQzZ@j7{|R9&N>)M4T0(hDb7DPL}?iD=SnlYSx5iDO)H#jw1L37C&;sp zTt{sVap2M11z0mGAMdm?(Xu8H|85)(h3~;IHx9t29nNTQHN%q2dhplO!US?T7mo3_ zRpz-Yq2I_~16?9DaqS3solm;R=_Z9g^C~vY63g4UND5$3%?UnT({Q1j&y#J$dmPhY6we*$=@aPwTF9-TU(b)$^TkX-O*$|ZqBFx|%T}?emWQ_)%kPqEU ze)M@?4ZP=B*fm)TZvuEe5gWDB(#P~shWv(&xxdI1x<2Mu_JDj)JM|lT*}uk+@1}pC z{bvz|AIRmdrZhBJ#^Y$yP@GffgNxm~xr@dYM~MOd&|#j`It?^sYT!;PxwH4oHtS7H z(nkXy95t}(UrlJm=s$cU|Fp#}YZth8Srk!Kzp>UXTx4G)F?u0demBVvOr2 zM#ypQ*$Os*B6)dL^1ZEU7TEFA5~k;^ky~twOT^nlSl5-9+4g0Xc-p5N^Dh>o%XfN! z3Nx_!_Gmoc9fiHJ!MHKWAM&Oy_$QvdmK zZAPQd=}5%h?T?#|-O%S)CuI1Wqoqn8i;Kl*TOfwxc?qu5FPFYu0$ZM)J>Tk4>nwqE zls?uN86kAN3FclfgWDzx@|{-jd2J1A&cQaU>%i}*uCfveI`rbC72|jk?}YWt%h|%c zKZcRqx!Dif9Q~km$O*4Un!&J9iemDyGuMbwCnKe$X{qb`-J1S-#g?u@74=cv3-ppXiNk?*A0!AK; zz|q2f@S5riXLTp2o;Af->f_HSh_P*%7=C6_s6LS5qMj7aOU2B`(!*o=sr*^j{H7tY zf=zHb)eNaUEbz3^lDpom;l11jQr10ph}aSJ4A-b+omp#* zz+Sc}S!)MR{?0GivscclM1prYF0g;ye4pntGH`-9h#P-JV#V|R$gA;#tY#-H_cp^Y zKHD<#6vc=2u$!~V)pRLBL#XARLj6rPzvI*V|KM&2lQv^mEHNW)VnHvq73OZWLD$iC zkbP>8zN|Zm_mRu0O74whcFyNweDupl&~@%<_e?}$dlWpR!HDkSkJ0^{k!4|yxbsp7 zBlPI&)uUI0m@Ci1Fmox)$PJILlOQ%r9|7%#IQQHHcc+_U;W|f&3XP?BLZ3ND*9{5LZc7ZziKytWg{C?k-=wX7Q z9y6)M5RxVEyDY)YBNA-xCxy8-H4|sJd;5Q;Feo85$}=+Rg$;&1utOd7;2R6*C1%~~ zi^PA=R^opLi8(J}X1#v_eEMa)3>nO2p zdZ!2eD~3)8`OlRSsF7=kn`M9rX~wu-ZibaTEr~5zqieM-HXgD^C3!xjs!o{6x^IZ% z?0-W4Zv%ZH-%H?pw*a>6b7oc2+ch!kCiT(^W@CmU1>b~r6K;D3S>Zdf{F9qV=_K4{Ip z?hf&)%9v6NC@w^W8nsqQvn|gXZ`fb&y z1A5ALTF(d{^vE%kn`1|?CA`>2u8pxln#2xIZrkHI>o(R=+dhoDE<}~su2GJK&LtS& zT7a#4vf$Dy8KpB~Fm4Pn{fM3zxX%lo(GIxy$`n@Dr5L-9-nA3t83*dXMoolgHM;1z zlzACj4CqraM(9j>j8)CyCQxHG#S+@8);P7o=Fc4Z?LBio)2VOoTgly)xJ8 z{A-&7xv*4(4vyoDJ(RlQzPLQl4}~ZBUW~JZ9r^pR4n5?lQnN?S;IpL;@^^`_o!;tY zpP5gcWQcT2W7H8_GLbih$v0EHb2o>^R119M^G+kD@pLS|;d1V@yjzYz1@wk+x8v#A zxiA>ay~k>!ac+DRw!RO>0(0QmoX%)CWQ!zX_F24tWtorOUPF#-vNqgT>mctSxeN9| zA}f7#k2T zmI>pXNqG2^dysqwqmy<|WbXFHz#=Eaj~;As0XX*I*3oEp5&D_-WQ0loagA* zJM?3hvFG8LJ-|Z(od9Y&4l!3eS&AR5JBS#t!Nw}gc}NY%lnQ9CEJNRV?r?D~fV*Zk zlsc!tl6ylJ=M2Rt`F<#u_s7sJZp0vLFe=J~bCwi?n42>3hc-4G(nk139aJ=kaP2fT zI>&WUK<>uWS_GFhI(WEF2W{{959`kQPEYDSa{O}Cj^3-lIrDN994tZOh(g>vk%Q%P z(x5ml0e1YY>t7ATuk;?Mr3ctZ#R-RZSwJPtkiDE3Nr}|p4rZ>$I2~9s`{|9p2wkdl za9)+Z_Cihi9yRdl0X4nXt5H(c8(5?h_%tk znc^L1MwQ1Z*m6%9KLeG~!MgRt>l`ju;Y05#oc%wNt}-mjb&J{sisUd&cc+9bKuQ`3 zK`8+#k?t7m?(Pok!ocos46wW7*sW(N&wb8w?>&DuGxL4#yZ5Tixxv3Zy@Qe}G#8X( z`q5(UAI*b}M;4BLO-3ipzUW8)PydmgDC|ex!pY$;H_#Y zdM_=2(vd;P4NAkbUj5O%RRr1yxI@j^4e=dqF;(3JN2#l-9zxzgOB>Vqdq+*yz@YVN z*z-UIUh&HCIok;}e&ZS`)cq~dn#L8&cWCo$%4^O=P;slM?mPm->eZzTv`~nFscVJ)jTYz&5HIYwU zOJ;00>kmK$`5tt+;QNxJsw+Gz*x@+_fP9W>x>MW5+!h3DTINl7TWHk zW}0*H_*I=T>str3XKuw`95|CRVI(oEwB`J737@S%<}ylZaDsTDrEvvBZA)Qxi5b1A zIk>lKAinQNK-K*y*v$^Y%D&$4TJ3^69qr(A&m8gL#u%$)fC-7Zuq%`yWQPdu>~ROK z;IrM(mEN{4ke=-HuWl3T-X~_4%Xi7zotYTyE%l0P(2n!)QtJClL}gHm;m-E_T>Ob- zFVKAed>+N(nr%3q=lH`yjn=Bg&M;QkV#PHJWajX$Txf)Ti}c}lTaH-?V(i$(T!_tT zc(GOmeTZ-U{k&#;riYPFIk=x3A@7Gfn-rMz_czY0Mi1(uj<%zJV_`9dS&&ben~fo_ zQ>o#L$6V(q^m-kH2mkd#)*g2}sCUA9WjidKVTH51&8g8h;qF93++D4M*slEE`I{@A zYH&Y!OCEAM>&ATK{Y&io+@M;RU8Se}TMgD7u7-m;Kl2f02-KCJ)0_e%mFHkVF#A1^ zB=nrumv>PZoOA-Hk@bRWpc~{oLk*qnVUcWuNhOwWVHS1qQqJpgT?`_>a^brWXZH#I zecmaoTS5$4bg{*ks{Wok;*II$qd)XiC3A^mE0|kEA1-l#%6EeyvE&Zm@!V4s-VbN! zzZy3x2sY<^uvHZ>8RH5gfiv3lc0i{hTa4*og~LxcXACuj)+k-%X30=z+Tu%H&$=IZ zRt_aEsXT@}VF!9*c_&_St3j*dRcL#R*@3EMuo+c^X&v&Qn4HDFJ{8VU@emqB!&xm9 zp=14#_0kJd3h1xs=!$G_XWXiBKwC>&#=XO6{YJY#Nh4cuTX=%Z2~j%``2Lorz&*pQi&v zOl4l-z#8PVtHv^&O2}uH;hTFgb+F9TJCltzN7B$BO5~o~z6f*+hxfVOC~;->+kRkS zZ+G-9bAe#36N+Q((Kf~!y*rv?jS797&-D?zpyl)KWZk*^p7&oT&p(0vC41T{Q3|{d zt-+;-+*|&hx`LQ8R6HX`I57{=9@%hKPeU8u1dKC`!88GL6GjKXtH~Q*d_B;6zbo>t zIpg*#2YfEE#e{JFz7I^0bKL-ASLwn!vgP;v+q;+Yxfo4+jF2p;GY7kae zh15m#S>G>(Ur-Sw2Xe93iCogz6ujTvAKhZ3aH>lP+R#(@J=+s+8r-NI?}6d7>mEog;y#@)TP#Va*y?NV;~_Z>llIp4lIl%b`5Kgxo;^#5V?^pYlLF`yG!v z?;59oqJ3+4-Gr^{}!G^mn z{_b->`3~{Sh%Vx@yn`O=IW-93c_F7>M5Cz`=9dfcc{H`MXVd9bN<_U%G`WO8Y`72D z40eRd8*}{R`4{(5ilS?r3v>kJdx>SGsIeF7#ti;0=+(I+&eNxvewUduLt5&dCGV_x znfOW^?<3y%Z+kMky{rn~+mLr&Q$lTN0W1RsVe<7&izf*XlQ2tEG);%xMDOj3+vB?OyZHr=yg2?o394pPft%gX>@>Tj2TvB@QzW4 zAUdmw0eb3~^niH0Toq=_p&I$AD;%ShP2sBEIl2pk7dj(wPv?JgiVn}nF`KwsgBZvC z2F^#+ZKUT?gVd@N#_O|w4@0<!-8gS) zbjM%|RY=yVQ5!+NM_mK;`_wU{H?s|$)wx4j9pkGs{>`C*?}@8S;l78(3e>6BAdtMX z&51Hho57szH-j;nIS^HX1hhR8!9D4|^tbmw{}<+1zFrsSIn$a(X#CromHL>zoWb;6 z>!@L=qXx#l(!|?)TF9~y;8&you15qYAP!JbZu!cD!c?#^>Sq*WoN-Vg? z-NVc^@zct|)YfU3+!P1RlObpt=ZUN&dqgZW#!O2Y=0yu|{D~T~)4F3`H|Fp%v*u&4 zI&ec1;cD946(z*B!6K|uqp$5P=WTN-b}u4l$LH!XaR!rYW~#a?@F}Jm_g7RP@V^qY zewL4uR?Lqn7=WsYG4Q(38z&9ik;Faa6Hn{oZ6GlOVxb=7C?AgC&H?h;w}`V?4bi~1 zfdag0&FqW+#8C1iR~9CN>j*isk##YpUhf~SqW`QG@@Qt(cTk}AeHF@DS3vNs7`KOU zFOEkh=kY{Dct@hoGheL#&lv^3%&@kSda73R*${&sa7>kFOLv@fqLzZ53ZEI87`csp zvN>X`yd_1uQF8VfI`B%>gOAt%o8KAYDnGBvBk~%chwwrT8pcyU*uR{!OA#zYxo8!X zj`O|Zv7{|~C_QiDv-aqmLi}r^6r;QZh@Gj9+(uPs_2v8XMimho)RD;NOum5{{zLRk z^pfM7wJtMh^>DM6`!3>*5EE>IldQYvF0&|1sE?gp1AhZz8}(%nv@1fvwjB5-q(NtQ z9DdvmhT8w4YfyQ|81KZ)5U`zztty$osH3b8b55N=hc;DBFW zRNn55T3a_<>1v7HMY?bpE5svxO?bJfe7Y)Qu94^hNE z{h^Sshmo40R4T`20pEv(8ZaKLj&xVl_hJ~z~!aHFLd#bgcm@EFP0#>(5m`j?EM*3nlv`fP2(|u4<=7Tp; zjyN;U7!ORPX!^z7AbmBkglEGma@$@_>d1J{OuXkB7$Fc~i9m$$u2ML)A@>=fho4af zaO-7+3$`ZQvulRQtQ$!D`T8Yhxg4*C_iExeeM<1El(~qjGH|1sIc`_NG0ealF8Azk zk(o4OCyQ~*oLUS|4a8b$Ah=8ejoh)-&Oifej%(n1JE5mGs1sLpHE2 z*T;)qB2-V)M4~^>J91{1$={f=-YO3bEF7-^5$Ei!FSTLuPz;lqGOWKrKg$))2;|9{ zOO24<&E%i1VA%=wH(R(Pro0+8q6#P%7UQ9DE~3s1M9=U2aAipd_gi?N{|_rv{zsnN zU5L|srk*5f;2d$^Jo2F_dfc0a_Y%R1p79;TQl9ZFZa+l|<8)$q9rSUL9*4c=#&Ba!k`X;M_F{{F z&+`X9R|k0q>Tjt=Koxh>DVLx@Ef0Q%8T6*bQ~wqY#c40>k+VNuYJ@YM%=(TL(L*i7 znJn_>UqpzUAi}U!#C;Xys}740XdtEbOpXmTdg!p(fEsBdxJ8=a^*d7-@-zQ^p0^RJ zGdrNbj_J%L>Ro}1{l)mUBo|sE)6u0{Ji1thL1nZj9$vIXg~AYL%cZ#dmEP9%LP*!L zKkhB2KT3pY+lAPWOt0;55o|B;ULGk&YIi*tx*NbsVg#4V#;8m)MQXhnMzU@&`J=Gy zyaN^07;adB38RalAD#<4HSPr{=#QF|P~^-4MqaVO_{9b|b5nxv>-c^0^My2uAR8{m z_Rk_ToDkw_6o21jKJR>o!UE(7J4>GXgFXx|8e-f~W86_PW!}FTYFYO@pXWiuaF!Kv zH;rlqmSz^?k!~(-oFB+6kpAfTC8e`Y$|?)LOw9@RN0Aep zFr2!W5h8TnMeHO|jxl$25z9Cj`pSBU#( zMCjX5%8W-jgf4n;O*cSjkP$w2FhSq~Qw)A#h7qj$n!R%e&IuPrR72gN0y7pBK*zY=)u@=KpjfZTUQ(=kwf==j4_I&h~fdgE^H8=^D-@ z!gvg67lt89p7`<32KvDU@Ya!F`37y6&*2Wh7eXxKd-U>y2#fhQB=Y?_z~5WmkvP>^ z9fW1*qu(k+C@nI^`FuXhhUVzHkXc~-yk|J)H}d`6znd7!j0#jcmf)Lj98*`-(&KLR6R`h2iU#~#)B4(-I$%BV+CjL`SWX3@R z9O}Ix9%+w*;YOIo`|a~-ZK%H#pzB3#JipE7*i(eIJ*o9OD?;4}2{|M=K5FX0jXCH} zVq-+3#1m#a8~)^ZeN%)>=jjEREl2zuJ;cWt;>TKJ z=xry*z1$q>qbzZ}7xApW&lPdH5OOJdU6^B2QHh@mOQFX(yTmaY6{nMNw>b*I+x&2D zku#K=Oi?;bj>xS-n2iNW?}pv{#^Jg_Q*kyRlg1`a}}fhpXp5`$L;y>Z>e z6`frypdLw{kH2GwMFQ+pVs9~&S=L6{@OY;UqZAR$k5Er{nz<9KYq{JIvZuyqxMhm9 z`^}jVY>CJ+>wkT0PtN*j7OcCNxq!qLlsTg~s#5z^mjf4R8ezRSw7Sk6PZmAV*v=Zs zBlKbWL`*-K08iIxL2IiP_lpW3VePY=TSB5FxScDZHuW$FI$7{bAeHvG4U_GM>MXBYu{i6tdz7EE@7XuN=y#zz{Q@1&o+)WQV{IoE_ z=~c{e-pwr2@yyU>KieaK`f^7B-hbx1HCv2nJ7kz~P#3rO-gOQ!!sVaFxUtd{RT<{Y z4zm+-t5Fz}WsnvWc!`rVq=$xa6i&OP6^%yyCepgBK-hE)* zCeDKwIS-y3Os(5iW^6iDLgQd5a=sT}^3XxVs!}l0Ar`lW1wrLWPc*#d9s7w;!@oNdDFQ%t_tsFntM|Ta(4oA)PD$YJ!85y)4EKM^@K@h%OH6dj zV(yqcslc}68k~z}Zb3ykRE8C!zaS5n?quM@yhQ9yio~AN+-uU>750JbEhljYvA&eg zp8x^ZG%$)9(F5$?hxa1B(w(1qs5bhT3W=N2OGj|Blo{?wi zml)TP+2u>rQ8k#E?aQf6DyJr_ig*ii8kXMDfSf(#U&^D7hDyn zKSJJlOC=7TE5n$}MQ|(R4yMcu{Fs#pPrE2oTnd10Xitc$nd?S>;s#?~Jo6J{k7y0fXwfO?|En(NfyP98enp&MSLQR^9~gpyS)KA@Y#>keIHo_Z8<*;4jdoC~dq zcOTzdj$P#1@5Sch+OI4O{yYG!l44=DE(8Trys`MaGnTcp!tFJN(9@E`w}SW4ZUK_w zwD7uG6DPd5_s2jDj=n179F>U=)9+6IxvgUh9($emWMDJ36uIno)^T6@NaB;sq_UVv zoFk)@T3_ahCbZ-t<^{#WAfgW(_W7gEr6;B*+rwd-8LrLO$L>TKCVXT+{hW7XmH-Di zzvx_1gMT!&4*QkhdYt~F`VP?h>-=TKwcj&~C|eo13B9=EQ-hbrzT=v;!w z`T3BI$VQ!WDik~VL6H)H^LzdAJjeqdr#hl_XDdwNoR`_sPLfuk8i3Z?352uA=w^OZFQksN+mgK<(f(-obZOC3rtUfB{d{ zpg|A+$~Q{LXl(J}cI7j@o@eE-o#e@U+1E~?Ze=X>BU;q0s&PJKu4($5BKRH6#il)( z@S8FK%ZA3m@m)CHh6Uh}h9^4g>Vbkqb_jcKiK$iOP>8GTn__^Chjn1?z<#S*fMRL^ zfB7;0OTF8_&l}FVZ+KRIJWL!ok-XJjW)=}s+D6R1<7Vb@*p%ZwvC6{*`P95*qdqDP z>P_)bKOcplMZs`v@WGa8%&eX4j1Xc1hH=aW8*BmXWK$$9F@&=Lb7)HF+v%f?eY`uW zn49ny=O4&q^cmW%0v^q5Y(X-%$ZaX$|hqtipQErlaV!-ubeS=W8zdR%OEY*#Lan$Ly8p zNVL%mM*DPMRILSu2;J~!9x=VW4j6LP7X4$bATu$8&Nf4wAkOi0hz#A_TKM0;7}kDb z->&TMHce!BfLJbIB-s?m+u`)l&B+q{dh=0_fO9L#gj1dp&xqV za-I*?`~ebXy5q??7iP6MBlfgCQf^q|@mzD9TyKn@6Pd5lyM>3nz`8$J_r)z{Qt%yq zp+|jCF*RT{HK;YL#^_-Bk$6@+jVOj*R6atS=}Gb*h`POraDCbr2lXSkr#%pJ5`0nr z!4vTldm`bqD_*JgfOWG2hJWHd)dLpH31;qf1O1}K`v0D*^Q`-Vbyu)%!E8R$y3`mH zD#-1#ztkgt6;O#0^j}#HDaPz@o*xc_py)LalAVd%vD=r~_z`$?lzWl_d=Y)q6E%fB zadEvXF<@uN>m8tZ)&_g0TA=%C6F84Hge6+)9%J2f)?IdkbJ8@P)k13R2GVDpUIXX$ z)v(j4MD4UPd~zzrgm-yp-jmJUd}%n@E)i?GQXg#^&U>#n+--djwF_A9=ZiV0%l4KTz;7cTc({LXV(_xWdX?VQi5*xz+w2If!dC?Ys7%~4cg;sECKk1xfH zD}~rV{Gl6o`T3A$}!VWJ<7%EAEHW_Y14n>vQAPB<^J#|Am`!IUhqnZ35pZT2LNI^;wo z=zWiU*0dJhxhLxy^PFrvKtEp$zvGSc1nsVY;lOIVoLGra<#J4*Hu+s#J~SR@V{vR6 zitfgv@MIK5R0Sh8-v<*6dP1w6GtxZmxaY_U-#OFoA5Tu7`0opS9n8-nPVOUsj*bQv zabLmT-}fW$qkV6vHQUBqMt8nbwd^_a=}qPA(Crod8^&c=P*#NgTXUg$AQQ(|4}kg7 zIQ(fB!ClP(==1=vPU?ZfrgnIeV~N=>c}BEhHpNqYjHXv3xJZn<`?T;Mbzs9*sGw_A zi!Nw7>kfTUiykX^_wvlRPpwL*kh??+>4~+hKs!+>GTRs8!I2#9gG@)zq(uIm(Xc)h z44wJCkn-9U4gvP?7g(aX+5~$nmBX>LGybMm1{mCMdSUHm|t;=OC37s>pG0Q7M4z^-5i=nXSRJ@e=X zO6e;)E=FpT5MwK~k$Q@AF?&j{w&V+UcEu%SC6xMi#tnlOz1dp!u}&BG+lKOUXDCpq zQiB3Z`X*;^&$3bpQp56LWRQ*6^b}k(iNpMpq40Rni~4aFgoIdQ{WT->wWrr-z8H=l zwW*00;6*C|+I(g{m!<~R*Q!E#ts8ar^q-X|!DV+#|6IVn-S-&t5pAd^G$SWkN8FC@ z&$&A4uN#W6tuhx!H)cTDH3@dTqG9Y8i2Hv$5bf#!1@YwE&3b6uOKjn>Hm3g3!mdg! zysOZ{A{{Nbd?W_1LC*p8qq+0QLpPF#9n}qg=Y?-PBcC_${pm&x$1`fZdQ@`{VFixW zmEu5fA>IvR_NK}}bht-s)-eK0H~C_}r7J?lS)(Y^2&c^DnAbssTGrIKqlsi{2nC0< zkfbfZ=%vi7EZ4*@~z2=;eDsUG8G;$2(AkDa3=#W|cs6 zA|EdXWkWwG1;Ht?@Ky}Ff4|53i>%qUTgnRz9@ugW4uQ{iD3ZvJ1nE>k+ zY2#ocalJTg=6#Wa?MqMT7-|kwwDE&=E!bP#T2zZshxrag)X)pZc`&gYpU7{VG|$6O z*Gz27PC~`mC}i9ZfMl^d<`~&xe}7|4-XzCqd(H^wweT!f6K`hHH_}dkw-2>pTPWg= zIWfMZkyi;7<36(j6VxRb`$_U|?!L@3(v5TX4(7!oND zD}(xB8HV;F?jNp$`dZz8ID^H1)DtQ=17|Dnay7jZhLsr09NEJ>qb?pDgw&;}SX~~6 zMH4uW)qCQ#mjkX14q>QR=H)n37+%!Rpr?{C5}3`E()A{$G-JvIb`pPkzbRCPF~b3zE5PvWF&S>@I$Gs z3nob{QMp(jM~tLs8zjVS@}GTpw>iF`4me#Kj!#8Mdnv_KYJ~e3>m%X00e7k!;UamB z;4m!qjW^SxbjAsLZQe_K2(H;1FAZ!avm=!jY^ zGtBR&3q_+C7B{sa9zlFmPk@AJe4aWB(Vp|z?Kv{|&)3CKIk9#vBTU+3jJSAHR7!~z zKDWRP*6sO__pTkey45uhx>jLDV;M9Zitz1s4vsXY;Y~q5Bz*}%x}7J?_S)f2ficvF z$f5I4gh#tt@ZiG-1+X;JhA;Vr4ll@;eU@Q#2hOFdxwGZ45sY~kzxOx8V4($CzqG_< z);)5Ue5joNT~>o;`m9&`mf=iNA(mVm1QX7kC2_I1nH2=Lgr0EuW{o`Fa~<2usHr2@ zN{q%iR~r`+wIQq1W}hX*3g-EB?kq=&h&=Zx14RBX!pNKCo9oQbWrzjN1vBe~byL~D z)^+6_9LRlblXyNJDCM3b_B8d`h~GW{m(pX9-@P}Q&bp!PMmVKOYO)uYAY z{HVseU)%vqY%<4#d%R9(qO~*$qg08v3w$wtq7&>^m|{~b`@&#iljJ2I?-62;rU-?P zh17zvF2CnPgTaA)N z9j(BisuBq6^Ptxu6Z6#)VX-#?$-{dg^pOK~56oo;OhnDpcA#aut zcbAi2=}JAx7%BcF>fquZ=8(7=qMNZX&b~H5rv+xXEU-W$>)zlz7D$}m$FK%_dsQOS zuN0r3Rg~s&I(35Up&fU5ehO<;LLyCShvstO8m}L!*tMomIUr<5-fd5pY3odZfi?XZArYD&-pwvDJIp( zvCmZxLoaa8RtIAodTWAT8>knd_c3#I3vYGiJpBvAfPbv4#Y9Pq z&G9G+3CE9i-pmlQhitMj=C7AyhNlEtZ6sLTS&9?pUJ_l z9K?)WdI}W7O;GP{2B-7pP!U;T5$j$$%f9)b0$Q`Gu`sj(E@!y=XLl|XTezPsn*8AG zFsMnruw;Q9pG70?v6tb-dND3<<+=2Zd-3+rzZNBhEQHTB|DMDSQq+0N@q(I*9yJCi z-E4#ob4~ClihECJ&{=Nn5U)}PVU-P>dFI(r~h;jye4sy2c6pjubD z4(bz)pgKbafuk7zNvS1E+l$Y46+ioVInu{dm!r?U?y1Zbjy1(K zeREXtd7fu$g|)2vgFXA*i_8|=M;*ni3cPVC!SCoijHx018k&Gho5E3T=M71PJ)FN9 zQG-CdSDD%V2SmtzCdRe<5;Q*~9+x12pily<3)Efp;JMLG9~a0E9(_xGkUq=2qx@|- zhiiRm;X`vT6K}tzfWeJwn4aRkqp{ShoXEof?M(F0O2pH}oHt#1!K1(dSGRDFld>GY zzKhVizX(ln)UvUE^BN<;eP0Pe<}r`ZK!&Spbuf)td5@iq@NtX@J$PmaEwI2RXDjHO zw8mb3-ho%?0eY#xo1fLF?#P`jFG^q(oe!rQncPR2gxafppfbY;2aY--{-6n>m&@U% zB&J?Rh*7)9Gj9|_RY?LTzRz7Z^8CnQ9y-&kD9UEHOpsb}8g;cUE}%iVi_qEYJ_0M(%`IQ-EZL-y)o*ay~Yw5ldVx1ZHp$>6%pfq zL9FcYkQ(&(T!}lW)b-UC!sG5B>OoTJOY2LGZ6G>7bweuyEA)xh$AM|g_^%e?z&UM9 zG^b|t5IvhCMcB>v_e_KgbNcGyaIpbGdKg24pLZ-h&Q7V82=uo`Cw1n7v2HW_y7!L! z+*@mKHJtf;ABhRl=d?#`Qs$`%v9mujbD2G{ z@)$Yq)$BRQ=d3s_#kvPN*z#K+>RXLa|HuT>j+wz|h6MyUR(NY~^N)ue@{~D@fy}ab zRD&xUs-QovoVn`EjSJ1ig3NRrv5QB$`Y^0qN>cg!yJ|EhVVPUOgZ+=)A;@#q^_WBfz`ix;T!w*TM68=C}3v2 z5`DP3ob!bx_%S07Kg=>Qc0?lafj-Qe@_~65XS8QN#M5+L$O^>p_hOdFTP@r;u7x+F z1xWNH?|)Ma(Q_HvU*#O+%>C-j{S=R+p0<@KcSxIIbPtPv^O4;fW~t}Xt3ez`GqW1$ zyDBjGR|&H3S4(IP&#*EFw%`WP0X!+!3V3Q5J@x>#l}1>@ja4DKZ{>uyd(Cj_wl0!J5o^p9z~6=1QF7juV>F>$ z!2HxbLhRio=59F&4$hIn{_d!n{_(IKiHSxoCpUOO!Mu_h>`1MG)%bGs zpIeMgj(NoH>D{}K$o&P8*kIrf6uTm+$qL%kDqOrJ#nU~U6&8{6j?{oJb&cwsIqx?K zFfW|?e(HpDI&q#kD?}&$)7h4px~E0|c-Y^>86vj)%^2n9-AQfkqbmH2px1hT2}U;M zZC#5gKhX)}|4xI0Ak2^Z1xc5khep#LwTgdBA*T7y24ZQ!Vf!j7(xUa-) zZDJfVaUrs%-v{}qJcl<)UUQ>MzNJ9jttd+ zznEG5V^wiEzB`mps-S36OMYx8;#cQRQq$l{@3E%>ey4dK?yAIn&V_Z?im;UY@0pYg zj2xGUZL%mR8Um2+)Dwa%J9K|#f>}Dmvb~vypUKb0XKU?#4Gf_lb~!!gIn%gPEtB=6 z-LNWz`!{TrkTtpmzcT&FtbE>uTl(_6tW;o=6?XtfSK+_`?zJZW8`CErq0GE)b8`TM zPO;cf5`se!-e?}@jHLaR@El->xu;~)Hw@GXa zA_9VONY$HtRS$$`aR2ZNV|38hW9}w(5~*USxN&yc!~BcM-eFe66^PSZQ0S-m|n zBU^kx{eSZguqtrhpyl$2xl_hO{65BMCD4cW&O^oi?-$uGiD zV&+d>NKd$(aljL5Ztkj3FStMt*~M~dfF<1DDMUWCGr8+k5iH`)4|GOEK63+0T0ZY- z*7bVD{jlrE!_Lp zu=ebthryp-vuO%qkKCuNK^~UxQ;BC8UMQILGA|d`s<;=)X#fr@#la;p9CnNSq2mh- zI^hiAcw1a;u)uDiDNZO+2Xssi^$jwddn-i$&YI{;zWv?Gt~j2p^p9UiVBI*LmA{X0 zpLH^M(f!0JFEVp;NHyx#R-)Q~d(-z9V?|ItY>7RXwMoOv?0Dw*Mq%7H?xOh83oEv` zqlR;N@ornxDzP_sX9mp;#u&I+AD6z$@KH!T&2>#2UZjSTF)cZ+<5~9-G0L{=-wliT zE^`LgCzfT^RE_NxRh&1NAA76>d$%$}U}6rc`=w*5ZxZ|$^hLw|FpSm;fTo%kQZ?M* z6XArdtL*S*x-~G+0-qda2zujzgT06k z*U<;N+8*QQ*x(H@1=$pG9Ut}Cr*r?pTnV|jmi*^Ktow%;bQ|I;N|X7y)p%~j5+e<( z!LBt5p-?ZLh>ZWj4pIdxIgV*MnC)T=vZ zCbLQy&nZKeLoxM_dDyf#8>2?2Vbt{mjJO|z6K%qAdQor8QS-s9*?>UD9hU-J;9bg{ zD^hzHYgtEvUvWm&bO-v% zZ4jMpfutG}Y>KBIS*wMI{fpUIv+gw3^=I8;1LoJ3DwqpY1DSC(qFDF#S@NOee-**` z=wdzyXPpMZYHcF=U+s$%8WD&+6NtmnzRa8Pg!$1XT&;Y=-{J%)C)LI1Oy$sQPE=n(^C zSGVx&e`loAe6D0o^utczbKR8~X&(g^^kxobGyOT_IGUWf!<*d7sK7jIe4K@eb*boi zDIQbnqp`F;6rDIv{8{0Rp;jK)chnVqemi5Xt0PvtvW5JuB}|u_Qp?KzWD>Kt{91Ch z|IWZEe2@0KH4U(^a8lj}q=QSwa5SjCz$D%+i~X4`;(cXlBndDlh?d(b1^43*k;aU*_a{ z;CWyV?g_KUT0a|%;(T2=o%wG+jnIkN%uO;Gf_O)~`bk~sL5+WPRake+TY6J>kiYVx zhGPu9NQxTF@67X=^YGf>GW1+iguCx^p{tgKMzs`ZTl7O&Lj-q{@x5*3g{FEJJlbN9 z=X%yiy}ByLb=QzoZ7Mj=Bwf>SQCS zu`{i~>~U4}!I9$#D5X!V5GyqX!-(4RP|YN?Ul@a1bs?y$_ko>{I}-W6oVaO);kjmb z5y(AuM-8x-8Szs`%V12*bkH?T+-8DgukKD-jj zW6KbAst6g)ybmTbKmAcMc65%#xpeLa9O=UyU9M;d<{scS=BOw%MAtRCSmr55uL3D@ z`f(;5tc}(CH1PEmeX85LV#6RM6kcoLyIZlYn0?)_Tzvl8q<`PvZ9JRX?x#0Zn|`40)bz(xV{unznZGPW)%rpxb>g1o!vm3bHXcT) zeXxzQnbAu(=1Mzd+Wq{P1JoG2vM*^8&8u3nDd5Suv;4N z9!Py`G-nyUySiW5TX8OX|A~3X^);A&zX~7YDscE=34)ypFrvdC92cj;rYMe?E1_6& zz#Ha8&ghb8iFTcdb!?VlO_~T(McNoUj~Qd+k}MYp5j2XKQ>(b|hU*0*GbLf&4)b()={p&mKK9KuM=9MGiEd5XG@|Z7|iIM%15z{UPqcZ|Aq|gJO z-0fkx%oNMJ=|MJAg7G>+SdypPxP|j)TkexQ#T+tE39fvR!00DA-~Ghl_efx2EQMRM z^xxcV_mN!tO!gV86&N96J_0#2qswK?elJ3rYA(DFrsL4)1ei|fgWd&x(2H<|WxO?J zry8N-200drxihwec+La?CR#IZ?Sl}DOeK&Di5Eu7VRBy%ZO+15+LveHU8`yw@;$y_pTiKM-x9AtB1Kgsk>sHPZROk%^A#fV_ip{ z%^@Y!=Y%rrvw-~X8DTg3F4FZ8JNAz#)+&r>nqy}xyzQzIHdF6Uhh zb{?!k>kZ}T%pT72W-dyyGZ6AP0bYIjpeDo@do6n)dzA%DztS_7Cc_>27Z*ICkD*YA ziC2hcxYNVEN{ZjhbueXyJ~nkS!qwNts8^W6t`l`Fn=O!&H8x96TcrQ$vwe zMZZ=#a@Q8&#m2#CxG)f@XPJ4mFAS;!yfN2}d!B^kzD)F(2PDOEdaJesiLgUM1k*AR zmh3?+CF zFT)Oc71~A`VCQ!us2P}&(=kKTZVM=lAja=wgJynSBjT}h*hA`M*P!-TCF-}h!GP0wxEI{XR)LStJ z=|=AuxM}uArMeqZU97OX!T=fTWXN#iJ9=EqnV8t`WidYQAx@Sqfi8WVM}Bbs9{q}; zE9voiVgkSGW@uk$fp>XU&=A|;?~HPVcVOdh&VTY6w05h+B-c`CU(bi@#w_egNydNY zqA(`HAFk9gbgZ@DZfJcx|4GhMUxF1P2{|nZcJ-3r`hVngcursYCB^i4I(Rl*9}8{C zgOspe&oqOac<9g1mN>@u;WXA^&<}|V={qfiJ1Aj4}v_YZ>GL?(llmP z?jd#|mSV(FF{-IukQPhO;3C2KLt+f&S>>{nTp{_RJBG|YxnqbhddYRj?LX^g0sBkD zyV9(Q8PZ>NiCIjSsVmr9ja}5p#JiQiB_|Ju4$?n(m$T-U2#nm<3q5NcG04#rV9(Zn zExB&iEs5bi?h6w1DVMa#i)qj-jqvnC^qRKFv$>A#2YrPH>GdG9IDQiaN2H# z1+07XTrKwsDzJ4?HC&UZmwZLN@wZ&W=w!gFa{`o?hr@H2H}3UtfUSW1Vk;eN>Mg_N@9!IGMV*b|K&@(4D|4J*Yk>?ca!9P3 z%lV{vw}L*jYMgVSrhF^&K33#n4tr(Qtav0}BknxK3+Zn5*yBL|V7wg3@0lOEj_>SJ z<}o@-VYHi^0?+Dhv&1O7OrFr1dX}%cSUuMOC+?7ky=Fr1yBThHSzzp9@<^@YxT1WEDV{A%*cL1NUElJhl{qU%lx zHt!R|m34ctZebTWKG5sDioA0Y`QL|iCde-|!xU8u?8@M~#JVM{yKtKVmi)X<2INEe zd7E;$A10i5(f$6+qzXgwI8XFzWrx$x4Y9>ahC_cuJUc}6Hb@ZqiMa&x$dm0B!+kpU z3SXj*%~H;PuZ!S01{i(c2mH8OZfD0m>-796%jh8&A&H*Jj(nDU$q9O8NwA*$x929_i|wU& zn=ePWw;uYvHb7SoW0dzYg{#CI@6TDli+YB?KHSo?e3!^~uUJ^kY+UX^;PbrobuLbP zO-J>{cue~mhMZh4`0ucT39}tc)MQ93VMe-%2%$b=q_CGNkx8&%i5P(w#8@35h2?2_ zNI8FwNoKalZex66R(1jP8&mQuvG%PMma(q#1@f@xd0y>irpt^9+;lC$mH0fYotOc; zK?%5WJsj2v-Y}SAkD!}II2a&$$&T8S?_Bh;d}Q82hWlm`WacAbYDc3mquQ zd2T!~L~DN&;*VzN8exF|eJkkgw*IHzu;YwUd7E>>oocA^yfR(Dop8_cP!^bp?um(L z{24)>q!)f3c0jMA#%O9FFXSjfYa1a{U(!c@RYczjGb&v9EWZ)s>-Kx<}f+2Sa_=ViPXl&K5qK+ln9ZYBIDF>l1L00%9yar1wct}-mjwd+a< z(lA4JcM3?!ERaw_KoOAc?v5djSjX;u?e0d7Vt03UcVdhBE#>>+<@t5CGxI$6z4zKH zE_3fJ7lR7T{!s4g$xJ}z($h;uW4KftOPv$OhIH?pm-61v(d0~)(}X=o*l+06;<=y#odXIX5tRY^ zC$YE|5Q1T?#D{p!HQqNtzdm{pduwC<0`7=pxd&az*{?|ls-%k<}~4fnV4R>wakO8!2X6Jo>Q67>Kuxr z+rkiH;)?^C)cpn7)HhA0B}9 zX1ohGS;I7){^+g5p{PmK@Y8^vC-;#CnlSC6M_mV>^*07^>@tRIs~Pb~3uJ0q;mv0H zuI18yL(?AbSoaj?h1`7RExI=&Ij|AYP4!4!LT{HidXBi~gbu>z;77 z!pk<^O|1Kg^HEu23tr}s+qbxpS%VGOF{TDnSCpexwh%6n8JIXE4!a6MF=+ucFXNnG zwbmT(HtR#@A1$UsQ5PL?_XIp|C>oP4!K^O7I=t;545Ide2Vc2zJT)k(4(>zOj zip}wrb&nFObDBuJZYDkYf6yCvH8pgFwaja)K*psaT-}`oznTOrb&eo@%}mArxuVdF zo+U0uD0`rdw8!dLc~yYqYT{GFg$SQX|G1voXf@@#M$T$|An)c5_S%WegAX)B@c?p$ zSXcCkdYA>|BJ+%nB8Ov$2Q?{y%mlKp!ps>Z@G|9ZOdRL3^k8hz38D_t6U8R>@QyJ> zS*I@jfIca#>+7*f9=;>f zU{%h+FPmWM)d(e?)7}j=cyOm2>T3(}d{!oP1@XxFI*?wF{%9+3$4C0?U+iy!91C6M zKy$C`BY>H;8cbu=u;Mj+`t$|Jte~fyhAL*wV!rkWB~jt$8@x$l~Fx&vLQ?#(U zo!>7#xSpD-LGrySzC_b=`l>SZ2#U-%lE=ffy->GU7Vm#`?_uu{zfwKNy+#y0Pm1W@ zquzx5HewvDXJ*ZQ_B>nJODp8#8-4SpPKkwbVi-Pk00FITcsbY>!Sq)<_mo^LXJT{d zLgc+v!!hE5_gYnO)LjWC9`ac2CW~Gdr6KT_!hbH3fAOp9UE~vTR>?0UHoCMKX%$V- z+apHX(0cgwtbt!sIaC@8A-bGNkNX6eX+~k-mq5II>4l!NoaoPBiC4Nt@LjHhDrIs= zE-s#qjc(EBCN63Tj z$836I`nuWF;+9?|eT|Eeemn<)=oAFZpck@gDALk^y@u{czG??w8FO6uP2S`o9n83^ z3F`)R#B5Y!&#a8}C33hF(-ZTrNFaLOFOhEOkH2;M{^Ik@nZJWOt6M*t@lA!?v-U>J zz0iPL4%Avrtb!!HHYeBQqtE_yyg4)!9?v5n-X4f7@>hCPyAac|L7a*if;e->eWJ&V zsy0+lsl#BUDkfY|z!^1JsM<lGik3ELh))AR?kR6Ssr~a4}~Dh){j159$28q`+TA`zD^|{<%$s!=j)?Q zSqHMcDeWvQWfQw;M-x#UTwLbIOdW>Ew-h6Vl2)wcQ14Zq3b{w%aXqn!n+j+@|vfgzd} z>)_vVdQk?dp!qqou>O-p;6L3u>p`r0jA!MY{p4?OuMyALp>Z!gYq{f(Z)(81JGIyn zTLl5X>xd`$xbilG+?zzyOdE{RA42eCoNzRV>!}-mZe`sxo|R!Isr4$Qp6@y{t$#G3b8jQ0WyA-PW$o;SobU9kAfHZRD=`&fw)d?#OlLfISBlh`4{U~?h}@(D)C6rOC;CkdCI`{Go^wR&=RCAk zWS}fI5!ToV?GbZKb{HeO-T*1q z-FnGCd&z}7qX*pNEIf^QfV$*WWRX9b-h>%n#8^Sh!zZN{uJ0}Y-nG_PNiPQ@R}*q5yZ6rL zSyz!~^vhc&me1 zdDabQ-4fQdVcpO_b%l*MrUYJ{No0D4t~!s zwOcSalm1Zrp7(wiV<9;l_sVJ!D^UgSC8anKS%4ANS(w1t$)GqE0plZK?GS>Rbpe>w z4`{pR1F@+WiVOQ7*vSFLD$shej`J%&u z{FQtUXl-(1POuAx);OS@v&x}B3(O;)+W4?rSH6gKUA_?0A+9sInRP!lBm)=ib2HS-^oW|Y1054Uf_>g7krS&UE%JRt~l!80-ZER)R4Eb z@wg@0QcNMtG=yxZ9>4SM{ckAi{*UKm51x~^hZ1+EXQ0cTCW!JIF>DHb5|rxTxryFm zk!84Xj=Rae^hr3Fgpch*aN*x@3|SS3ACr7>W`zer^<8oAfD^7OQ1|9(i>+;zxLCuQ z#4{s)>p+hjPp7u-{clg!jpV)HwVnGqv|#UWdiC;N$m|jWo!lci)MBEzlHMmJ@Ndt< z*!k%=J}(~q1%vRrY5-dO{UGt#6MMe*MqE8J-8}5DV1_l?Ptfa*Si`xm1{l{!4b3>_ zvriJBjl04>&(&JqgC4vGyH;>t#&>8n&nidbCYl6A0rPKH0a}~KOWhh=&fR5HoauyO{@Ou#A z%BbsB1oBe);M5C8{Nf!L{EpcsmgGNup(p1k1E|usa4>W7+9U<|Sg4GW;|h5GqDb79sU^}ng~J0%Y5?Q4i*#7YmPGQ*l&(goXvsJ)?rRkIZl?=MeZqi!AU9rks4 z*LnXHQWLa+ZU5*vB4XAG?sd zoGS8xZt2i#gc{Kq>*`=l9^I@C^?jFEl}>= z4A<@CA}=Q1aJUw>4=V9{Qwdr==HkrcR79?i!Re!+n6Usj`ppfen(c6Mxf#4Uuc*GJ zmQIH5axKhutk*)9h8ChPQkQ#5h(z)z)2^yu7Pv+4FZ$>!Ot)j^JokQ|+NP4tSU zFXNX#IifX)Xs$rO5$XtYve6uo1l!$%us=E&L*IMDyx18nS;XS{7{gVAT32~;-S080 zua_1&==B!*N(Yh!7vDBx-s&b~pA+K@ zdyFl6t05yrjtX~#2X?CJoD(f!b-KWrm@RJP)*}{9CXVis@ z#eDB_;H8;Nt>z#ov;|>xxfho4?5&Qo!035~a8TC8>`E}(#$b?V+$4TJJ;}T@gROQF-g#@RE`na@L*!!w={v^g zAiw?#c|pVJ_aO>l#s+=#-?MJ`8*;A8sMQQ0?>e~=*hp+tx`zJ5(TCV!WU@5!4b*(Yc6Pk#I}SnO2zi z-5PTmsr3@r;|c3(JR|QbjGX?5P2~EJi%Z>vPec`Vv1g6c%Z1UfWW-z<1mFJxF?)mu z{N7QkB_c;|qXCw$)P;YL4s+$TF(gR`KDW7dPBK7t9y13s%rJl&&F92$kH*;`>H+Pn{`blO zN}MOI^YfP9=j>}ijp*4%(GK5~t z=EPgW=&QO%7xSa&VLO0x$XZ>vztY2AA46Q@9@c?a*_xhKsG>LFd~JF_-nYk*EsjVb z{`{wR?tOD3(OZ7 zLHWBLzDxhngy=Gho7lIp9`}*dc|{t+jy&YI1I;l^(Td)mHhh=KS2A=!A~jye+NrH! zT`SHB22w4|duYPbcyhceYf$M{jw0Uz{O6pGq6aa^Up4?r3%oJukOSDorymBR_@F25Y1re~X;XTx8sPUKJp{MY z4_;Cq(GT^=JJExbqdtDVGvFLZ&Hf@Yw5D1@W1BU~sikq8Nk7v}eqMQJ?B(aRy+VHK z(`Gz-MsDb>dYoQU%`BWU9Gahp3sR{#Y!rFWOndhj}K!^7iDNI$$n+eE#&#PP}^)&NGXi zz0LQT^@tl#jTNK*@N#mIXpn-vUk72%`v9!Vb;tQp)(`~Iv$U@+B*$yx&`fPaKGw#w zW7@=2wILZz?HA8wGfP8i08Ef!We(qYmRP^r8v46zv35B*(K+PdvF`uaHz#v1d4Fsp zCikny=A%{AgqF}3B?ms|=vR{%1!J9l5RGxAZk#$zH$(c4>cE?`Y-ATP{zM&Y)+Cpl zp0X#&54!$L7q;sSkhjMehi91KS5JBySX$#8z0j|Gx8vOCfPB{7O{{MGie@CHk$3gI z4kigzXml>YD9s$)+M0-M-y^Vklpn@)x*&fRbNB`s;C>-}iRl6PAY2D&+jKB7LOwf`0T(OStnz>?=po#t_42O3)#QihCDSpd^>B8A*{QT{dzGmU?ou_GqvmR zfb&=2=3?fjWfKcZz?@!kheTJDV74d&E-=)*ya`gwB7t`c(}{Eq&u**cJ(tc`{I z4(f^jrv690^?D-&mzy$^%^VA#lly$m8e!BgE|jxJAnR`3L)?2Tu?6!+^qom==)Ov< z7+j1m519F*pMb6(;i&T#Va@-XaD(&fUEZnt4{75X^<-vqb+~J1Rv58e<2r5J&!_hz z?_Lw$v7>{H5HCgTaXM#xW`8`2vPQKucfdt<@Mhh;>~S8nav!Qeti7?0IRcg37Zu~@ z_$-K4#A9G)IF1bWh1?BtO&*xSHBS#^rP}x`)<)+f9ZY83|FO5w%g{zVG3F~vnU_Y6 z#dsO=8MzNwtIND!YfH#WTf=Cx4YrUI;la8ucG0`Gr5R&X8kv{L-0*)Z=|^9Le{-_1 zxhNi=)WWeq#}^Ni9r16bDPrkuJ%&8GSzg-6$)L{btq!gYX0CUXHg36Ug+n@M{tGlU=s(VqJHB-kHP5qvvcN(NKrLY4k#pD@J8m7M7O9 zW2;;^6o2)Ft*s+2X`AB4JYC%1ss))_^h{A@cH?FpJh0cnoPOHy<9)b*yR*qwcKtU7`&mBkt-7w6TQzLBE~sb2_;9n8YlC zo}3FE>4|J-gSKC`1DHsyx4epnSH2(e{dC4e;1=`9sRxEBw*3zftWi- zg!9=>s98kcnqXbLo32TIgeGDFwQ$c|8~ypa+I*w$#|ZY4ar96y;_S)3x=n|9&hwci zaEUX;eH&<;u!CZW1Ik%fm9xSM&f0;)8gaP~IprJZXFjEbzO6ZEdz=W{OA$~!18fmH zV_UKrwhyO&cN9PGBn^BkWc^qzawM4Py@Ni*6UghEp^F>C3@~UD=OyYDQr#`Fz1|uH zGi^~sjp*PZju^?hqfT-@B5!2S#zvTou17I@t8IRz7@U-gVVuJXMNue=_h(=40)-0R zffDqIcVLhHTpd+S8q6@~&g=m_VNl0Ydb9V zbb!9V33FICj`zabn|zNRG(z(}{Ta4Z2HGEC}ujpqbAnR23loySTVr? zA~C(3{?xtPf|gEVu$%*@_G`e2uo?tCEW?R8`6%g;#_Y{#tSAnG+HZG!W{-CLt3Hy( za}Ri52+1)*tTW_(=pOTE?rETXu@lW1I}bP zVK?h05idKZz}`K&38~!CJ!79!d8Qnj?-k(TmUP5+#US~0e~g&xiD?2`giKoupjDnxYd(BXsr8! ze8RZCZr?A z!^7dKSbba-AAYOB-<97v->p@<$-P!I#NM;?fbL<2Q+LfVd z7;dbL(l%w>ovI3#1R?Hn5B|kR2e;z%P)A-=YfpNwha17_kTKHxGN+w&*Ap|7YNj@( zgnr0dsY!LECUt8Kyl<6ba&RH`P0S$IJ`M(hLviGX4;E6l7NKT};@9ji=u0=~r7|jA zl%VFUgxZD5=q0BHnY}^?H0jf{i5{-YwV}ir=7X0mo+OYfHk!O4*4;yl@c3lze|c8k zR%=4k8D?oOV(wobdW$<2L(hV`4Oud&fKJagat;+bxy0eIH+}^}okuS}-u!}m$ z=VFW*LhSdyYRJOx=>fl>NJRE+L*{#U5+90rwd)&D3Hr z-$R@EWw+>sOx^MIxFVPqWKpA=fVGDss6+IJj+#5B%h6ALp&{~vG~waSom7z`9?z4< za-lqqJeEV0ryTaLlEt<9o~VkE!mwi!2!7TBdlS3!+TIes+H{;6QSjLcX@=Ex=9dxU zIP|m@cb`-uq^1N#A-S;YlR{luG=ly5BX_emwoP$_&k{4}Rx#5fM~IHu%1}S4fX2bh z81|9Fvn8@HwCst+YLbY_6adtSDSBA+S@}a*y9TFXJ$jA+cN}3-o6u3djpI#HKMrc?=T;;PGDn!bBR~7K< zxg73Q%3{2>H1~!*aO>k&(U0Ix(fG!vjv@;3D?zdF5K13^Q{R>eu~M@ ztH+B`HMnA5fugh`=)B27%Y{Uk3?77pk3o2xPE4OWi+g_LqukTS!+K4Gb}?&;yLDqj zMYKQbg*j)W@yfafQs#aURb6@~I_>^SWZwDgZ{2IG`|$yL+41CmJ#U8WD`IC&jo4Jv zfPa?M;tTufj!DG4Z{*_oRQeV@i-A^17>ok^pjGaUo+WnBKV^zB@AaVnP!sda$>F%H z0&5jTWMs=?g_9(P%=|8rc=|!~c=l_N%;V>O>lXat&Y5TBzJ>JQQsaNS(5L?uv+<0? zaGpo6GxKU(ZZ3m!bpiPr8T0^7z<%}uyYB=cakvj$cT*dq#~DP)6!TZ>$IAO2P^)&r{zCSt&rGpk4t-Ac@m?6A zN$f}sh5so4{xbAnmw}*&Ec}bDfPY}1e#@{&Bj=WXN2IjBS zGRwUZ|7|S6v*tX!)kufcrlAyc zkRkUeXNmO}T+BqeQWBo8jK)fNX7H};hm#-qUM0C@>LA`tKaP3j z*qvI09|}2Ge<}sH-^L>JN(Ag*1jBqD==JA?hwg5Wc67#%jrRB{P44k>a|H7|eN#%l zCGp@3%LPb}>fWbpVBOa|EB9aG4tNB2p}(kG?9+^u?6;ndp^tldJsuKkc{Hm6n>H2W z)1F-Dy-1@5jJ%oCgYY0I6tWQk^nmh(RhB1KjdP>dqzgjsJHY$84P1;Zu*%06bD!xk ze^46_1G@Ll`&jq=SKj@^SJEc%b1RW+AIaaC=ZDl`F~01mhy1%53}-Ha`W)s2Q5WQq zm4Ow3i5T@H8b{ZJGlz)yhm{{xs<@Lp;(;*k23L-E!MAhFq|LIy9r`52jyGn;EIpjQ zbnBguvF@KUI+y+3xGCJ%X;8nB#GK_JO}P6|%)D;y*s&K#;dIoLMwSe;n6C|zd-iPjC-Ix5HEBHN| z@O!?+@A*X(G4Me8Y@eZ?{X{+PsMX>Ead3^-^h?OfM_=ho>{g@R?C}unzDsUYPJg&w z_D4sCFH*a_u-nrEf?78y+PWag-T{xuLvDFxj)NynFn&aL-F2+{n9tQ*K3Bu}TutM1 zwbPS(Gvg-q%;bQz*31`9t|&|kBJKH+(A`JN7!s|iqViiX0VaPE?Wa9Ga|(^vSw zG{_Tequnv^SZ~;qAF5VuhwT>3dUE65MVqtQ#cmz$MAjY5_virMqXxc54SbK})R_

q5*;AAl1bd!nvu3`WhLiCZ4)HNBat#`B00a{S~!L2!F?KBQ| z9wr$OcfRY}g4I*@!llUsoci7yM?wamjp0P;yIq7)(|6$f>hst>p&AkT(oy+TiOuF| zP-UeqZEp)-yr&WG#JcTe5}S>Rj@IymTbjiEZX)I z-HKi!0PnClQ0Agrrtmsl1*MDe9`<=@zIRus>Wzx3H0EgSu&@SOt{aBaEr*ZRn#6s3x-WXWScKJC*%v<%4y$Dc&8$r5zbn2N3@ z6Ob)Ag1w6K@mO-lhX&{2-I09!@Z2x6lmg^u79i}@0c^N_7#rFhhq`EaUXRYBci0ABYAUick43L&s>LI9$VoiIR?#%`FNVU8$CSsA-v)sY>yp7yv|wV9=M2| z=dVdGZw1VUSD|eDOFVw|5!+-gI=)RNqx(&x%XjH!6*liek0|bC5J$|hVVGr0>YuuU zv7&3+nJ-daSY*se zhnw)qhf>VEzZ2so*?#l6K_5b!%6k7aIGW_%bsmq^d?x>M+V`~tM>ui;ah zJ9zq{O8OY3x1snGo*VwchP~3?aH9ne%iQ~dn)K3kNvFz^6wZcB0Z5|sxi#F z+>aSGzHA%bk#C2KH(WBk0i`!_qTxK|h;Lx_PRV)&Nx$E6`EUE~V(FYmSh@WPu5WyU zum_)E(f1F!%Guj`yrSqpEy-2t+%r`e7%$rKjp!OB3$r-nSq62gQ(2reh3h$k-i`Mu;->ac6+51e*v zz_sbksn=YYr~O3#wUQne*}MCV(Bb}T!cDp^eH5bO+&&=r-;tRt`;^9WoXQV66KVb{ zfq~vJRQ8Tw=e2T9ec{NYILUo}YRPDgMr3{chGBod;>6{jsC@Ys)5@E$R7Enm#})Wm z*rnD!YD^j-oFd6ETs^BReQo;uAakYR&CQs-&72cNw`wmQj?ou0*|sp9$AoQ~nK{Gtb0jj{i9pR-ka{K`;p4{Dz zy$9=3Q+#}-qED^wV#c%0EvP+RJbqOjS!>aW54`R9P}Z%Lj9IF98Y)+0h?hbbn2V)b z`tSta&`RW2_Yv%HegLa3hRPhMGmWH^wk*n++nVX}k>tRNM3-FssU7Es>9A*>uwMGL zXOmL~Y;xL|<6fKbk(_s~4Y#IEo(=aEccSSdd)~X^$d|J26VaU?brDYg$qeQWNMp|b zSfl+Xa=PMJK3*Bif;~|j(62j#6C_J0Uha-Y9XKRUpSeAC`LR-m!^CHQ`Hda}TkF$% znE~CzFWGssDSvJ?=aNJ#rpYYp{U=*S2|Kjcdk30dcH$FR*GAS=>@57tBf@Fuk;XOe zQfMADkquuHd1Y2C?ed~HtiSNoXZUc7ngd()@4(ZK^f^9DmmB`+uy?sGeQNah;6r=X zMjFszhRjDp<$fUfQ>{OiRP5K0!6R(hFvyNIeI2On=)|kC?x`BdJPWhFU`7V*ehBk3 zHibi{PoQ>n0#g-4!`e20zix*zXLe`SmDn;=!GvcEbm=v(9mhMgV@f-1o|WwJ8+Tn6 z?9^lHYJCnnY9xL~Q(?%6ccI#nzkgZNxv4EDwX)-~277t;W%0qCEROh|$v;LJ9NQz6 zABIk0zlQNVcVskIDGq1x=6-x%5XfCah3nPSLeA-WEVR=U=3i@O&uGm{Tf|c**%KSl zUMJ0L%gU$P{PRMOyVVR?TxZO%ab~=>*PJ`5MEll|+_R&t^!a2_t2m1$`!hLCbfn7A zX~OTF#Ed;jTsv$ef5i-v-)#>n9`C}LPi=VdKzkm1C-rC;Ad@*5*F&(&Bxa1RM-!R@K8SAc@+_gWA zcGW4gTQh-?qZ7E{P&9`>3FqkDfxNQCiJRUTa{oV7;gdAwU)gidpZU#-Stucc$_o))*P5N?m;k$zjs{riz*!)2e`CVYlbCo(vF zekuoloy=A{#__r8D0WUBL{IzfQk(3>^Nt;aGbH^m!Sx6;dWq81YFzzLEncvvcog*t z-3s60?&%NGdG{HEM*c)&(?;YBYR;5riZsw{$v38ITpHMhd1*2WmCum>Te_XQN_Me6 zogf_}Sk~)Zp{F@7Oc<9?o7Uk$%?;*eG+CIj>9M zmU{!Q9^FRKjk_=keS|n+)V&ig(yg_1$b2Pw_^t-56@O4)dqozN%N)6{8smOQAHmN| zX4a>1u0{&O-Ny6By^%brHjslWLzukLnI;`|g(dVG{#S2cXm}xV`|riJL3(YgK{;=C?l+V*SM9V%STKaY^tNixGKZ=s-FhojqnV#2CMjNjj!I<=DR zo}0<_J5yOOZxTOhBy#lZ;Y`2Kn^*REGq;r~HwuHHVdpL6eb@)jUF$J#*J^~Bu18eP z7T6xyB|Thw5%*?40?$hhcJvYK;z=CvIggQ{SMcO}Ij%0a3)NLs$R7FvZw`IHo8RAX z=b7|+yvyL#4^#OgMEY=VjAZDnNV>EP;>k!`<_SZxF{ctertA}z=?e4?Sper%3o-NU zaVyA(BY$Wdx+O_^wE3_+qfUy)*pe%&XahUEc1e)C0N?zIv%dRgKzJu za5M5H_LWPvTPuV0d#13p`B<7+4d<`1J^47%odzX3JU{INhDgWS^uNpCH)SUJ*3Q7I zb#oCPwG``Y*NFbL3B{AQVzBcz%naX#`#Qo#uFl7)ZTnF-Mb19P$1zF!jN~&f;7|Qk zw0M3C*EJu(W54j9<-L3P+(Zt3GD=uw1K3dJ$J7<(+ybdlx z*{DkhSXPF69q*t%HH*P9sr>bCoUrbqxxuRkwZ*eO>0oObAH0VyEjA&kVLCEDjYZJd z379L4;0@Ooqm%l2Jo&X#W)cVRLYOPdwF+^x{s?}_-?vT0Q6#-QhTy*^FsS|%n#nVF zq4@=jm%hai$-fYfAAis9d;h6)>xLfB}ay`ESu=Zc&tbSx8TA_~O7>q9raKUxKuWOL4bsEDprQ zpv{ORD1M%an9LP;e`W{DFCWL?<5#f%-W_cCT8XLdLdpYKbr_ag*csfNGTGjwYrUWQM}wDTE46$d{a-)F#&onOINxH#n_ree6;Z^Q5P z>YSOT#-TT)r}l;P(tcHvjyFXnXDZMnT!F4KbNG@vjYo$LW8dzc#4%;+t&n@1Q8s># z7>I|5qTuZ@9;tojBT6w3z2==nuYA?4J3XYE z?u0J)37a_EtQ~{CXmaZAHo}Y%?I=u*>(;4qdWI^k3Z~FzYXt9QSu%O;OMJImjXs_u z(YtrJbazFg>FeompRo%4?i3&`;2I8SzeZ_M6Mj!trIETe-*=Ha`aM%_dDnqE-gn^C zgJzNsG-VenWAQh>pVs;_7#cJK zF3R%6Mn<1_Xf<% zjYGIiA5n7_w{2}^Hd*sIx? zvn>M3&H72Rh`h(PJRwxzVPJKEprju zZXklgdtz&C3~rs6h0>E7&=`9X4<0;#d+1L*6rSo)nOP^#GGp_fwj3VgLhAxg7GLt_ z{2g7`O|o9?qP^H>raL?Ia^d*14s!3ZW$Y{~j_%um5p$&A z1YC{lhGXmcqj}U+I9&J-UCtNaiAOo?x_pG*G6fD37Sh^WV@}YuVfZ;`={fOYtt~j> zpf4}~^JV@ZlF41z^Pwk~R!EL;fFrL**>XY)OIjQ;;pb9)zMG-LzK^w7pYF}44M#D# zArd;_0n(2(0He2MBDQ)pOhXINdiPxz>U@V*h6?AJ>d-~Bpd-`mnA}o2pd>4qx!RBW zQv$gnB8YCI19`QJKkY}50ajf&ZhR!$_YeC`oIS9y(#bEYs%|3oAM(IWgGIi+`7pTZ{r3 za5k6+hIgfvcQD)B3*d+%Uq<-&aAR-bR(+6MawkV-547dOGE3G8yK9uUG5cSa%+xpO zoO2|uWCo*iztI@|VIex?cg|5i|W5|<;qL)EG&?0=W+`^(lceWqoHuH+c0|Y zB%D_6jN%ocxcFo|+Sn{dADQ_p7Tv(FnpzwV6hBWNVP@VnWzn%tRQ&G784G+EJ~4n} za=VH@Ig}yVVSK$Pg!LDKXyW6~`<=nk<6dmK$CdRC4jhr&ktq=Wb(}HJ8R@fexH3Ka z$6}I!2M)IAjSrAu?oK@NYBmj4pbD6>}6q{U9AMwHU_eB za|rE|hiu5fVF=PB#VKx-%ioZlf zI_#ri^~4G8bwQ8=0}Agi$Gg!7(eV2gG?#rrk5kHo^hj+F>cCAJ_Ox>G;H16aBl+E) zo*p81)G#_-3Zsr*D7RV#Q!UM3{8ixAd@tHfa%Hr5+Px$b{yn_|Lr2RURQw4V8f{s- zU?Q|Ii$DfgM2CW`z$uq|^9g_-5hiQN~u@r(E+w%P^o zOJrA0SQ9Ef?NCOChH%&9AjZA-qryiYhEDZj@9C~QYU057Z5`QqggIZze)p!YJ~_sW zMwvNKHT1%V(k_UdJQP)9=3-$l>CbL=5wA;L;J$A&dVFa^4>^B@%tgI9!&kk(#;eY#GQry93Vc2k{WNWf9k>JlU!(9AiQMZWIXq`V4(i(}inbHY2}g^sb~ktfzHUT04IauY7d zFl65AJ`8&y-0Ryj@H#dC!oWz~68F*l{KT3+&TzcW|zFvW{T;`3iX3e8p_xgCAAa=DO3S zR8A7zS;~Z1_yR1w&m%3E4E)FyyPdw>>>A& z)xY)lVa6yvmVWin5m&I>cq)$G2!Y+X?wHIm=+|ulVz>j-%r3&w@;Q7?Hev8dnT1OJ z>Y4moHR)}~*_tl&ob1l;Q$2V**`2HOTVEx&ALHb#6F3l>*F!NgcM$GsO-G93YK+l2g750LQ5aB%ITICm zpIVGB%pOT+RNx52nTY<|X{hvWW}?%WmB@E0z-hH> zxM%ws|J@MY(jZkH+Src8uENlKZp=XkO}O8|m@gCLyYz=h;TA zo^~96J5}I^>PPg@ZbrYfR=jB;o{pk+@{DM6)||G~UL$?O>FP{OX~kUeKdcaK{Mx8y zd|2OzxUGM&w&*u39@OKEumLra#_-ep9z4C=QZkcuSQN7bgH)5@C3<^q;xL2;r$RqW z`e2*y#JD9zxUBvF`RUT1lp%QtWhK6eZ^@tATe8V2VL|08aQOCSob|d93lIE8*y8W_ z_p=V=Yd&JZjrVX%6h85j_vk1y=kx8yvmiK%b>&`+m-C0(YWaJvSd3Y6zzF~s~28i znXYttnmiZI$p$Fg!BX=Y1WV`abGzrL+58wbTONu|eiyD56*%O7ix}EbxVAmd zW6IBS`1S7`o?kzYAJ2;Myzgb{thy#=@|&V#${gwSG=8f}VxYn>LOkw65^cqIsz~jn z#aOy|8Sdv!!lpxGpdL5{c13fbT)zTe`fbIZ^};E9bqv#5oWjBRC(zro5Y2xd#>SI} z(ee5bsEjB?MepP2U4IfMO3z|;c`>vXUd1Kn8#u5>G@LDI>>?e!Qzr?VS0{pr=04&f z6aVyopE1nj5Kg5oz{OrEsI!|4Kk-8s=r4j=BlqM4|SD&$i2TCxt({z z`Nkf+DcTQXr$cD<=_neyokafLA`H_gL213r&E{2Lp3Di4zmz$<`xJKWlE{Z+hEls{ zH^zlI@pVcY_UT=Pk)w8^B56LV7EOmsd^U2%%*WMzD{!dCMkttV$9?5p7{b-4W>aJ&6sj!q9tNg59IaQRQ?`a`@F)Aepw(8ySq-CS9e+$Fu$6 zI1ZcKpY7)O(ddJ8I>st6bpCZTcHN4jdlutRVGcS9zbQt26^b5g#J0WLvC}Od_5bn_ zKYllq@9)KxwFgmJaSTd}&LSf266Vg5+^WkRj7zG*zT+>D>G%OZWDfi~C6m9?QaR0W z5(9@Nu*QBEBS-cWPo4+0bEH@3Xgww>lwh&L4zyHUgR<_can5uj8Vz?ymU|zRH%d-q z>tW;#I*K#jkK>?t9o%9{@G`s{=l$+sm&;?!i+_btFF&D87x9a)Yk>Jr$+gvH@_kL3 zJX=$kT{4dM>Lp*E7Rhy9!F0~G=an)|Mtam@$EFgTt3H4(BX?t&lgy%9A3+39Vby|j za64Lzg47ax`c#U!yKiBQye(V>MqZM>vUElGt%kzX+2;}a}%s6sCT*>_sK z#G?!EP@qtY7X@DsoADF9eHsw%CVk4n6P>=O6|H-INiqrqtD)6vY-qJGFMeP%I*I^9}=0>!p zp?H~xeiA-hww7eWM3OL!I4+|o%u-C)sS_M zTW66ulK=fDo`%E}#w?t`f1eYm&}jq>-w)vXo?%q&?@c>(;fqz9Qbp!{k7r8eWQAma z59{$tsq|^=)u&#S0nb;+ylA@22F95)%gKsd(^0yYY-xDTj?SkXxL}239Aw>BmGYg{ zW&J)GG}xHR9uFlWdL)@%=f-gDylB2X7r`$wyEtR!$>h}?`D%>(e`|DjBcmOakBbKU zTFz2`bh!7nF5fq4Pn%{2?0d$D{f^6>B->or+m`hFAMeIRW-GpS94#}<>9X#s8`4Qo zmC5Tb(nWKh#*3#Xv%AK44jvQFk3ELccVZuAnfi0v0Vnp~WX4mOI>DrE zi{`DL$xFS{SS}fQy{*YCRT#r{mxoa+sxKSn`mwoqQYU4#r)J-l{Ak*QGsG`qD(r$s zE&kw{>?8m4bIx1el#hi|aJN8#>y=eFQ|^$cXwAOoHJLa=y00nzil+KJv(}JP{t7c- z{Qu8t&TpN{0Tz>~H#UjuyT$Q8?ujqa_DIl4}T7fruGUHo)fA=RkzsX_d= z=g6~u1HUyguNCg_?Ul8tGW?EVZh!IbT2pFyEAZW4i^1URS02{7-1;cpDu8FQGB37+aQI z#werf$nAa$Ln10*RCWg?jrTE6&Wy94ynsgd2i))R6*_n8@m%_R4$p7SuF_?)I;tkQ#*&bYIOsXk-mzF#Mtvq`%7INNC*suI>< zp36q~nrz2l#XWe`dOtEp?Z@WC{fOUqK-e=!Q8)7xp58f+n@=ud@B8b>9)AZ8+bXfC ztMoHpdWTL`S#;Q#!Bm5(>~L)yFHDN%{?-xP=Hp8@XA8bRrobA{oATb?A$pf||Mpz~ zwIPdfe%(rRmve4*pRG_ivK^~+x8s8GHuM>^9WV9r;V5UlxqT1g<-p_Uwe}3Wmlwk( zsT5sa+``U558)*qieB5(`E1E#ZnBd7Nqq>{eh6hmPe)cHx8XAJTMquP8=IFbLfGbP zoEk6_$NJ4jmm@NtUbr58HgCbF{o7y_z7>eb6EA5Vlpk-!=_Na{DqgbK&IQOAei*80 z$I)}=Sqxlx36cBDP`y?dC-VF0gbD2D9mnwf(ue-dn|B*TCrtPX^@~T*qRkTIj7&#| zyD89A&BnDM3-RezF1mfq!-q4wF!1kg%ohzftj`Y2(%O!WdOPr?`!4j@yBh~u@58(L z0(jRS#zdzR_}i-p3(sG|-GQ=SO`pb(bz_-$a0uVadnc+;X14bf*~0!Z?iH*=?&(y_ zcTGZk&lE(t2$yUAf4J(JhffOou)NJtIA|P~?!ZDcTs{OB?*pR46rgvH1F+CPC|=2f zNSb^YPb>=Y+4&?A!isRP@*+Bkmtm~D_bg-MS!No}Svp>vV5!aW2{l-!v;}@0q_5?D zJjxmqv3>G%Oe|fDNaaoF*|q>{U!2C=|4L9;RtnP-SMl=PCGikmfVMD$x;LD~O2@MZ zm~uuumuGNgRs&O8i5!CpDn0|4}sanTxe< z@pyPM7AuM;qi1Li>=f2Pr`>+6op%nuV{W3Vv=SKi6f?U$#ijAp!g6_loQ-#J)~o_e zZrp@Hb~#dSl%YPM47)a7M}Xx`;Vo96BsGJ*rw9+(I+E5+ym(3Up{lJnu~NJX1(U`i z!eJN&Ya~G5WF{&HNO!fvZfI5)VL{3rEKrk7?3*vxC->-ymw)5G#Gi25`vsv1pI{(a ziyn@zg@OAVD|$S^=Khbxhggjiy&C*y{uD~K!sIU&Ue3Di(m86$Lp9$p`pGV=9+(Ez z{=z~|iiV^0G*pjVjL$!|!gt9jY)h!X>b$oYI^-{`wPY`Dp~5G>ROq{1naBDm(&cY+ z`d@F#(Wb&w81q~B?cb5I=?jeS)gfYW9ZDlU!z(zIwF6{Yq%ZT;W*TgD?=tRcE{E_{!q>Q0-hhH8qQ`5A-rH4pbi%|Pp4XP4>6$z( zY$N9)HO2?D;?ot%>>-@LqNB~}?ADBrCN^byKvS;FO`%swBnv}2kz>E2tTqpceaFEy zU5K|YhoSUAI(mLtj)XCL(9ER-Z&y8q+4(;xnyx|(Wi9SgZO>NWMofGv=WA!tJ$&Vy zBaEDgY+c5@YsX08s*SK{!}s&m_;^_>(a~BmW3vj^Jf6srYr3-mI_$T<1ZCpsK3@}s z+MxbuwL-qDNe-glZHD{fQy3loK<0Pfuw8GISTTMjKEEnJZeba`E#5)byE(&!|FgHPJ}gbvVr^o=;lZvh$bQh&VJG6UxKU z!@4iB=Z-;PVh)mp4{Rjwna}dRbyNI)u{N+TgR0lSS=CgB?a5sBca;AC*y1JY2U#%f0yBToAjA0ZQDBUmbLAb{#luhf7 zX(my~ElWk)ODpi-hy&O&vRoKjAF;?!fjzHjGEC-lT_oS;JJXTdL*4o4sTb>ec{8W7 zGy9r*vEyxb7I?T)JKTv2wIvt4S{ODRtk@t~j_j$XY?W-x1$V-^uUQisyjg_XpZY-M zS2t*C#bV*CIdGk{8Ckc_pv&1RtR3?QE;hpFbJOGD+2+*#Vn?fyZtP#tnF=;w_!y}` zy9@qk;zPrQomo}l!I6ht8AB&3%&}wM^^Uwb%94(AIxw=6Db;;Lm|l7hRqH08-nlFK zjqi;>&tz15UxJmd_8?$-DRw`6i;jlk(Xf?!@F63vOYO+Q1ZU0^{a}Dd)2s6RSR`}y zJ3strP){aA`0#OLXYoLK(4&V7qe2{5< z_$@Y&7Va|VA4JYI>cZ_aJh?f=l?(J7>0KZk%!O9mD$kcn61>#+5M@I8+Q zz*ToRsyGK&oQJtO=kV!$4IbDvVuF{tXuJ9hlHZJ0o#boVdx@4y4s936wkLy`5#5z_ z3}!%e03ToW6@HPA^bUG4|CeN;z2rF%-L>|<%#ANg_C3IuX$vH?HG3Ya#(87uAzwUM z5`(cpIS8!YjtQZcuxrgrtV@@>5{Ce4spC$#fVL&LG#>pA` zLMWYHcBP+u_lg*QcGD$4$9uDDtOwPDgd04-fqJ)XdE}BcJ4jF7bGJC0%yE{^wIGyl zNP=F0oM)aEAj0k@iZ<1uOR*B|C07=aZpLTh?1Z~7UbO$?=)B`{e*Z775J@RTMWre2 zl+tpJ_TGt)J+pTkWoAdR8z>b*Qk0dwXZFsPy|?sxeSd#=^p8jH?)QD)@9Vl==bYzB z1Pzvm&ipHY&9()zWNC80Nx@l2hz=*^Vm}tZA~n zi*!kS!pY%*n19t4iax=(Bt8Zot#x=mxf-g0chGP9FFbD1S~PN9>VG$5tGV{f<&tdrK@8YUxb8mqkXXS=mhYvt}pE;dh%^#54vA-p{wYGk1Fi=Ime1= zf6Zz1=sM~XT@ZNK1}zRoz&2|Z@>^_xzVl%y4St052+8pTwBe(JdR#7Cr;-~E!V7cf z!3r-9?i0ZAqe3|1Y#8^%g|lHu80*YKc;7LQ#pir!`o)uN3VLwjD;JJY>c;-^xqkY_ zl3iPx^Y*c$STofY+AD4FF;e>GB-auB^`q9PFi)Gb4m{;qS-(OwoKSDX%HiG$u5zI∨lbR>putbd8{9ue7xvY)`Q2kxllF1kzEJb zN-u+jbfcK@#)jR9^6HEsmex2J6pO`+GH`5R5uWT3?oV(%l-!!}rgT;n=^Ap!IxCvY zbz*9L4>q~t!{e?&%x@pYqJRjFtcu{-+2O*&2<86wfs6|h-P_KKowItd-5eL1ddqc< zZFyJLiwAAZsNlL6N|!rB^50lDFdAOc72=y&2;=O|3p{8>1(d;=6m9Ki&&CkTmDV0wRbnqSkW*(Uj0 zFF7)Ij0asLN8YqmAjdZeWfxz0Uyp?I_{uQxRENmyBao}6`?6EACwIxYuBobvu<{*v zp}~gb(sPyA!%RB3?jgpZJGyr_$A(M((qowd9$$xs=KJwH^BxwB`U}lD!bJ`0D2y?4 zp1IeR1r=_>@emDmn?L8*1T$VYj5{SGRx~e^xmLkEsVkh)H9nkQ+LNw<9;}_!on5mW zxa5Ql3x!?RQszp%hBcviPADF%>jLY$lK&k)9H&2lNG;WWv(&}&WQIZ z)Rdn^LoNt%q5D-baGx)gzXkH@xDd8h4Q1-@V2&>hWJrl0&$x)r8b`i1b>~p=37t9A zl^%vRob;p%*1UGNIr(u{9?G+F3p zNS!PzS{pm@{y*`-bn#}m%v(;_266q_scI;-4CGrkcvEMf?~(M(fZz1jxJI9T7z__^Le%^*VXT7=I9tCDYfmPXk={tJ23` zcxrmme`e8@!^XQxmYmdH>cd+>{!EuWzk`~baZdX1z4&W~%5!w&88@!3bY`3|*N0rR z;fpyIY`58rPOip0B3-3-YEIy4Og{{bbVKTdAh?ee58>u4obZu6lig)_O#LAIsuq0Y z-Hz=mj93;X-Dk?8@r$M%d8#LqwspL{VQ)Cb3d2gATS1GA)iD5LEW#Mjs1ic=$! zyUSYWD_p$kooOeYuHrw=vN!bLSs6PHY9Ke<2U`diA|}|CvmQ8cqLBmBTF9PHo)zc) z%vj`X#L!}WK3Ls>snvto@LhUGo^QkHMT4;WuMgr@Mcpri77XarmG`rpS-Z=X9TMDl$wIi|@)@dhcjV?$JFbv5&tj7~b)3Xs zbKijR^&L2Vp7bGXm)?`^V}$!*&*|^)V#q|vLw1S4`qU5%4H|(}7W0LTx)EQl9KyH& z(b&fRhLW!e%geNxt!~U{J4-%#VaJi}9r?1_k@Kt^r~*62AF!fwgK(f5Ot`j8xcaV= zTag|+pDvml)J&aEr>OFDyUA=(AfHPW$uT=@MfsL77;r5Vhg|!h%grQch&Sn0&URQX zIExDNmpCM>hIPVIZoOESr!>X)B0b02-L3d@gB80}S#ZO|&b(4$%xhH!yu7jlcQ{IR zLh?vUOH|p%vK3>8x1g?bbDDikpu*EAK5;N&iS;d<|FZ-EDgAI@T$FHs<1oL?JorT9 zV%yujc+l+{?uLDU<<4fbdeMfJlEbl-`)BdZi0;b9EZfnEV-(~J@k9E;pGl{2Wg8|9 zR$+=s3x*DA%3*z*uxYoy82Iof>UyUzWBx$8E*AfxcxGP;H+@^bRCM?pi}E{t(Q9cu zG)5AHNYEKpQQlDGAT-rV3x5Zo#sH zP1$CkbahKduJ*7`NZ$DtD>Gi<-^7=gv~?P9MvbK2CVy7`)2Fg_9kh?F!T$Quh^Xp= zWQ@R&F|%Noyb1%(Y)8M`lQ^yX5HnkT#kdEGymYB0yY*6G)zns8q~AigVT!yyw*mO` z4atY!!}|D3R4#dp6Q?8>TYU$A^lw8k?iNP7%%JC+@!aw?hC?kKnQ)~gHTu`U-)sTy zC=Ww62clE@B;jE#K$<}=<_JHw%`Rb^3**>d?L8)G{X&g?12zTxMk`^{rL27o1DB`R zB6)$ErneAvxfbn5Uxe$NbBOhn?h2o?X#4go9y3ikN~bWj#~|kW%bs+NAZJrq=EPzpKNWz zc*9y;p0NhYXC}a)LmZU1Pek1QIl!cBcrD+IjV~)v+2fe-0M9|!@dB1KJp&b^qp&wR zfY;ttNO)9^p+vE|aGyBo%-mGGKU1)YrjSl&CGU(Hk4Xg8h>=+Eh0e7W4V zGxZKM;oC(A&{QQ8e@;(@@zn8X@iZ0NXD&p|)ioIUstEe=<)Xn=qbO{j_!0L&VMG~D z^e9F3?5)V~-iqWETQS;e8(nrHA2uQgUsJrc*sSf%?10 z@QGf3Y8nM_Wq=JoTDN9U<_#pa*oZNeJgO&C735Nm!GBeiKMO5Mtk`nUoweD>jSufyoI{3O$(g{YZqh5sO7jQK34q^n~?Zl2}Z#xbj#;sLpjmhaN39Gb?Y zrm4)1naV@0$Ivu#00Z6!i$2zkUg0`?JNhf$_#8*=?;@ODzaGoKuE&w^O&Aff71L_U zu%~V})b%QaEm?uH;d^jK=7IjZkHBt+^yKDTf|}%jo>tyRa(e-?2>DKB=qD@|?)1{D zY3#pX8ckginR8+SUo0FhjQLo8yX4KA;-4|?DlC>gZ*Ws)sh|FoL#c5)?iKIC@Jm%_ zS5yOa$zFX(mOS;klc?Tv2D3#^H#>0yb2dDXKIZ4BJ@OU_(s72Re{eBaftLcBvEOgW zXxgV!t8_ZMjZfw?ne$IH7{mJW19@v{I7^Sn{Gir^7dEsKKeurC{=0(j`(@7GN_rK~ zod>dnACPqm6KsT?-|0TySJWY6?{nD2i9co4H%v`ykh#1f6Eu}LWVJFIp0?)IvNmic z`tO~|lK-7OLwuRS__j`<&hQCLHyFWJxqUdqBY+3KJ5qIk^cncbGg0j`60)D*vu_<9 zD?ODyxK}WWd5>K)KjH3!FNmEhjQU{>*fUk0p@~WyF7vAE4^&y>q(S|$qFajJb>uHy zCWwx+J702qozhq>+0gc0Ni@oxEWIJ4IXHO`O^-$L__CfXKW5DWWnGT-5H3`CBM#~P z!ou~xk+!4}FS|=-?Lku>*J{q4!ql2ROPNv4RcR!>^P@&<(Xc{WI7S^fI!B+5A%;}S zHD-6wtxg@3uJmPTJcH>ReL9(0Kc=#B_;}u)J)Dcr_F-S?3{qI;!VR-KbC-fP)oNO^ ztZPeNd(wi3rT511mN2wOsB)b2(Jxn3=LVS--mYlJsgBaAwMdTxmvrJQe`7YeYf2*@ zbMePoaJ^ht`${_V@20Wz!E_$Godnxv`r3bWn>yqBU!_(dKGbUB+4H(LJscSELzB=Gly@vVR?>VaY$< z);!>3`~SJrUZP7*Xqir9$sHd}O{GCzBKyWp7M9l-hMpMA;C(S%7wX5U`<$hht&7ZV z_4)RI+&h_j`CMzyw*z#ToTE$Uln!jj>BxJ_^tmX&ka|%jTg4r`FX{Z?85~uaO6y;V(l?->lZ%^r`&Xi8JRUO&YO^>%t^%?ZSfaOz+s2y!eD@~al&9q>`R+(9r z3;Sh@9hWC`j~+H~8}jwZr8npLXJGetTymdyF-4jq{OMDoAWgwGgf#Ba8yd|23-S8sNq zzT{?>f3=qEh;-}7byJ1I+~Q;!f9#mfZ*x=FqH_YbJ5S_nkC8M!(qD2B;liIIn@p6< zi?Ea>vCPRY)cJf{8=l>+F1^cbX=I=&OeRgPU7$s^G3^;-Dr^@UJ0 zD&x-5t5hD}eot#A6#| z#-454bFuV~Jx^)DR|>*)nc0%Bty{5wM-|pCQ{ntMtywEM_tVLuOABwjt9*C;q)Xf4 ziJq(j(lN8$h<_iL@RU5So@B|I^htQgKTrMFT0A6icQ0hbAqMauXKPn{;z{oI6Z;R>vZROF&X&FH_a1>1>F(#=hc&vZ5D zeo>Q-DeZYoPnRv0bmZkfk|!S|8Bw1ZG+duT)1mQv_=F$@i{IGc!gFwUSqG;8>E}QL&d{SNE-7KFGe*Y;-7SB zHB+L-2l1CYQsLlBYMeDkgT6n-7o?reg#V_~^;{C)j-1TKL!!(CzA9hFl=Kt$*H<{TqRaX`yo5=q%4h(WUrGtCV_>E#?$igQ0_^IVfJvzUG%k} zgTCA=vsdsRege0`%JI;6J4V}=;mXuKh|@g)t7C@{-tQ15CLe_LpF=R8bPOx9&tPn` z%NS*PL;6eaVRetk7<2X|`c8Wfv(sM@_)=Kr`e}?!O_6@`DU8V)#fby@3F9hId?gNy z6c)iP&rj&|_b49jFA^LM?4xc?Y{{$bR0_0bCJAy+MSEH zbmE$?jo9FQ8jUg+w%fK0HTM={u>EorD6d7r!wq=(do%uAFTx75LL_(Hh#p@zU`~Dk z*6b@r8|6~0@+iaZ@(OgkQw`HoHOTIC61fq=D6W#{r1NwtCCAgub+oX+`qFQ!c=ESd zvQkmJrxUIr=0hQ}cPv8k(%JC)H5Z5Lm!S8y)kuuofZnq=SC@NT*rYKN-OFX0f3GSd02ES1v_Pm;6vP{{#@^MUxM zuN$@Jn)>H>+Iu?`RxZNV{8U^^Peo3}T>Lw-40(w;h=?k}&d#N9&M8H5LJ8JNcJxcT z4REf`$IXKqpetF&JL3w_WzuG>m|ud0)us58w+nl=igxQFT=DhEocCg^bdC4rmEB$} zIBY^Kr+-*iw;vx`EyKowL@2$QDBU#EQS?lhiTBqa<4_Tv`IKW$Y&CYy-G_^@qMvG) z;n;|6qAzbj`tTB*|5%KfmBpy(UV{2xTacr$9sV!N&`G}%b~S0de>#C}b4PG&MmRkO zJMhmLRW4Axg2J@5xOgEMzcfc9@cd*%M9xJ{ZZ>*N-He#M(&5ZwNNhNV{D~KF_wrc` zQ9X`tuMZ;bK{Zxe??vv@3M4!(NB=#$@aXL>Y+kz?r=#{rCbk--ifK&vJ(;6Z2S}%y zCr8XMU<;GCC~PJjG_7X|b8CcjMvsDP(hTt=WMZ(+CVZ@^#KsY)vEutpWF$Stuqn@_ zi|GmexJzcd^9_7icp3V7(oxgqG<3X=<3q?1cwDZ5;l~=xj6Q-Dqa-8nayl13A4ltE zLgxKu&nB5I*rNSm>>a-d#=}Np$e0$$u+F@XsB`Zy;>0gh zbZf+gpAFdF{3mL5engn|YdH0PhB*%(;!ye>G`V#hMftUurhf%rT3cQ&3}vMXI^5@foC}2_7v~VJi*b2 z$$S_vfE&+C7rL+jM~|z(wwjq>Z{a?-7zp1h3D8PkiUXrIBlq(m3>|wL1Al&k?ucgG zUam^3-dc3|txdB`9iH6Yp3{~Juf|E8LGN1AK2w>$rHiif8bxM=HDbk|-{|8k&+T>J z(PnZ2uT({HuBjpa`*j;He6!$te<-Y#dSi*{D7Q?x#@@`V6PMn$Xrv zRdfO!)~E@~B-xy;f~}}1ouw)BtmtcP!O4-GsS##E7fnN90!iQbQyqSoEdIYG;)jY> z=eV6}RQo=fuS7@l-1`uJW0s)F!vFACISQ*YMxy(d40K(x0Trtc;Cu8POjh^-r8H%_ zG)O1Y3Pb54k={FFI~F~5;7|8%d^5|DDU-T#iezD9K3Z|q4e^$ZG~<#}My&QW;2Q^F zZXVEK?bY_ejv36MG6O4^QI5~|$KsT0C??kS7Ozt>Y<04+bka_whFyTpv)6cpW*mB7 zgSTt-|9_vgpw^DJ3!ONclGiK{9a2L)>UJ)Cv&%`=TL*4?^}j5dHGRKIpGchWpq-2v zyG+=~6ZQGsv=<%Igwv{#3DuZL7--j?XM-l8#zPlWWZaWZYenFVEZ=j^poYoeatk^Dl zP9eaY15OAgDg7|yFhFyJUL-oFp*Y-gePNRIfY_rXTk0~teq!DfRBHwo{kq>~w^ z^bxjnZFh<+&x8ryoI1^q_M81#qvg+;0lu7a*^B%Cko|9pS1r<&iNYFPzq%{m)Z5Tx zt0mK)nsc?M%&S(rhn*(<9~)?$-;$rWm$};IVW-N&2!8vRAh*q1`B%_uw;wm zU*a5jB*C4XZhCS~$@WM@{^h*GlI7Sv`4WhX)-dxbUTFH*UXU z$9=7=sax5Fhnh9z^W-SR^|6Nja3AcMG6@5-v*Dhy8yj7(A^)jx2}UZ@Xofbc{+cjD zxJor0MNc_H=0EineK&|#1421GA&ddaVJwad5pJ1eR3rUq?Bl~WcY5-Zu4r}LT{&79 ztj~o1xkf|!x_%1d@TmiSs9Rvfm=JvYlLm98JbVm4g!Gb!;>~G<|1LFl4$|Xa-!9B9 zabWs1caFXwdU}3h;)?N%^#R`AwfA;2{Z=}*i$r~Q* z&f4FOytL7trQ?_4T4{Tfjj=+RL4S0cv=|3vKQcb&B2@c-z_N`kxK+PBxBM_>t%LB{ zpLVCZeov;p^`nJ!^(cf#FzaL#R~ALnMb^@eu919M8OC`>f|)Tkfcv-mP$#n|?J2qM zW-jz--;I4T?K!AxDyDSPz){-Z^u!@>cV3429Y>3hB>C-(DT z&EM`kRqse6xBq#dN5S7*4UNjS*du-sH{&b>XYaJ8Qb|;b-&11Ofag6X}r6vbL5Z^{*eFE_L&b==l5j& z7k4h5*qvV;C3Dzc&faBl2nuK|j7%%M?Kc?a%FA$Y$S%zIBprI+zGBmmRy^n1%t<)iAuy9FZ}7F~DRo7JV!gp6exCAN~;=y<4*Ia(h;m zn(#`U6EKs%gTZ;C;IQTSId4^3tip|;;?Xw7(q%%;sbK==ZI z>PB3*%9=A~I39L46FBW2zY&fV`rIO}X6-${3)d$bp8 z6u{v#U8%3-!~4;bQAWD>4Z=Q^pTJu524zyP7%qQLK*+=G;S2~b=E_(BxWcL5W_%qwbk9(eZv-y>t@?Gh{2U)Is zAw2zy5A20aXifJg=CpZdBGGk2`s@jy<;xqGa&HW7R|0ANfv|oz77A{e=-8wbwu8<| zXTWPDmCAnkWn2E6B|Q@XmYiYk$R}EEbW@~s-izOBwikc8d9vL)P{Y?<*7WYYwZf4d zEv2(w*we@M%5%rph{cim%zmuPdjtNHJbFuBUA_u$wnd_Uy)PED>yN<>v!SVyizBo5 zLHp2cym|NoV+&fbm*geR2b=KDK5O1o@5WiKu58!NgFjYEpLK&9&o*(PU+-?5-dEmB zS!-WQ4)T}u{P`!e$-ma@S->k2dT%U4=>gTJmMXcsA+^o7?{v4o{TX#izjt=^75#QG?J9 zv#?B9W<{dIbh53*vm@`ZepGXw6XtQ;R6S;_H|3_zmRwzD&Ew)(oNj8#I-|~fu~NDX zZ|ZUVU2PiYw&mGgs`Q-Ll8zqDSXkSH@%n%9;pbGU-3eiP(XiKjJBh28=Hl7xKG?A+ z7Wu6wV6gupME5H|@1#S}e|86!r@x_>%*0wPmhSU^9r$^pA^#YeP$R*ZZJj&)zjhxl z=M%k!>YUeKg~!#E*!6uAwmJM0-D6~~wN`p;&0fO3eKHjS`?80de4jV}LTPdlX7ri@ zi-Wy!(swZS=1s?f%oVt~ZyPiW#8>^k4(0B@ps}t6>nzpz`%pV7K9vmC^7j0(sx9Y^ zRORK4!WI3k$fxK3;MvhnI3Hh+lUpCjY~wCwrrZ>-@ikP`rLxt3!)fT`#o49px$|W$ zn*CmmUG;G&ozoA724fM9d6;)+9g-8v@hj>a>}NcIzx0S}4{XMb+m$IL>cZ+&=8ARA zXl(fp<|SV+NJ0G1*B;@U=)EtJuHd%lu&>4lZ%;C!!*q@z;lOmZyg8OD8Y0BgYQwr% zMSkkC2aBi8#rB4wh-*0%Z^9EXC2ujbUu;0Mo^ZaSE+B67Bg}OFh~lfikbde9HhX_Z zk?vcpb9#)rid(p}=(1e%3PIUu%d-q$pOs>s$~HXA-HPGux8dHMZ8&u>jq!;IR8}6r zVN=7HJHVd(_qF1zWoNNZV zokdH@{2J}A#*dr3usypJb&VysYgUA@4<)lLGq-M1%%CP&hspnKhD+OG1bd_lyCGQ^ zFk|`X%zu1W<;i}(jJRd^FIXk*MWbXDs*lBES>Ys@Xr^QMp~a|pvko!J;&K3XLs~(fiz(r5cEpIEb9T#Hnq*VAvOvlvl3}_tAg44u2;SU#M zcTg$1z1xZ%twisw&cgwl_3&7_4!0-E&(k@`kIh5TRQms3MvrDs!@dlf>B~J`Ea~4#S$w{=IJmL^Z6)(t89N8ta^);;whS|^*1@-E z6SnFVBOyj+fn^)A%zZGr zlg@>)GuSvXiF=kz;N2-hx#C|WSKoG*Zu3rb>)eRylv7wE-AIeGmg8B?Qe1ek0xzO- zkUp#sH|2NkUB3keQ;Og*yZ|~Z5WYYWqQ-4S7wetKe!d%vj#uHN;vw{1dJHB{PQ$Oq z1=zh#XOvkQyRJ^5P1Y1Tyd1?Bhx_s7r$EVEI`F1WdoCIz49#K3(BpeCo}bK>9-BOL z6b9m(w%hTiqFnUc3h_&nL&tI#qFl;lFS-XeUhhYO@==5+pT^1`7vREMkX#s>{Cyf5>+sgEsKnytny*01Hea0=HNEAaMj z8M+%*;*sA0q%Am#x0xrPE;IPMt)+kD`$>#FaSjLj)gpSzZIr~+LH*-%JfHFw`|3X9 z{mq})P~3n8ztTnT6(-~4>1^GY#KZe1)7pA8cVrIWt5#wBBx`ItU-9`AivPLvH3ojJ zMSk8H94kAEnr)XcU*{&88r;R#zV$L#3Rx99Oj^k2#BReq2zW6ggUu}nccvBFLL(v0Wxo3rU%B}Ora=AI5<^!V)M8XVWQ1WXiKvR?YQcc4i%f|G2)d0 zcPufI-c?hENoQ-cdlxQ|>-M}L_xVg3Yb&O+^O6*1iq>B1Hj!!n)9Lp0r}flux{6=& z?+*ogY#Td`A&Qd$!8{UTGD8$T_41v+-PQY z^ktKcPP`)hn#cR=Q(sMoCZhcwd8Ey+@w)squLA=m`>XM`Bli}{Tuu7YoXsVxq1c%V zmkWpFmL)%mCb{90E&X>(c1^B3|3Ny#BoF)0Zw4#ortGN{kf%nr{q@2zOj~QG%C6%LAB(hz^WX|FkI(`|<4pzNr zX5mja$>Cm)vE*+fL&k6Hz+J|=-1kM7(Xky_L!@6>_%p+#Bgo>1J&WbKXRoL8f%u^ee@~~4qUhbv6F4Gz620z@;>A{jcywzN z2dD_MZGo`aGNh+@ydLMj(^mYP>8uiOP^@TmFQ-kXqD=}-^5fZc>I4q%8Yexv{iG97 zywob7dX5d(dvv1F;C9s5q0YQ-ZTOxVY^-U^$|OyO=4nc2ofb7NwBw6`I(!f*jJ?hV zJi9~s;AB2B#>|Yzdzf>pkvs$C`>U}doeGlGbl*3P{a++8aI0{bSBb7;IF!3;d-1|t zKTiDIjf$=&{G6`E6mu0;M73aK{}z%hQs#m($^QDQ@S%7s`(KgyK$a@)<@b}vdJL7rd+R~#4>qin#@&UyEm#-|B=q$jcMH9W;%;plUO)o5|f6Fq>)KKc6l4b zph#zqwKAanspi}j_6$x>FCfC{1Ri;xL__&G%xW*4qzkU%zua2bzrBq2Td!bd!F4#w zKF{^w#c=W5 zD*BJ~UC2Dj)TIhq_Ya`$>mvvncMAKOU%+BxIU8J(?u8aJxF|D`G5^NVw&P&tpATdA z)7?4$ZU;`D_Z@yCCGXiLAHO9V+Ol{FDqmz_vCewbRBl4$>mnRBD8kD=8{sE^PoDaE zJh+_$SZzSa?gBLJRE&s=Tk%hCCw3}V;HPCZdVe^GZQCUOdu|#F#!qF_Q6rfiF8$Rf zJUMKO8BJF<y#AN+#K5%o-HOuY>da9N4MnWBLU78?%bAQh6In19#%@q;y`?Oy%atiCpD8SUe9Q zyewLbf|WXz{@lffzD3X~nvc$5sc5ob1{Cb)BkJG^C}-ru^m_?jm2bz9o~1CB>~Wz* z9;`$|)%mp=-x^oLLwI4Y>et{!r*%jeoP%#!`MA-e0O#)(VZBy5uk1_W&5xtm@^>t6 zzwE(RayIYiEqj!dax_g_gqW&$B#fAX1=nXH)N?6X73Cl;dn-b=OQ&?fe(7o4hp9#t zm^)~j%zg^7uERzcYv<#`(p=OG%)ud*9E|yugGa6N5p!-Mv`ZvsIwy^-=Ebw+yP=%Z zC72fW)|@m)iM!4oN5$SOJWY$o!nI>?{AwbUKg~z}*EQ(kxD|~ue+b)n3hE)3ka_nK zhUA~Y5&gr`d$AX-ukFJ4yzSW3ME3j-ir{d#5G{2IG4*;O3KNUaXF>_m3S?hpGl|bG z{YN!34|bQC;QrsjM)JtVIQLY{xG@6pJ;vkXwAt`?$i}_J#VA|7AN0A1W+U&T?&)*% z*RRKV^{2R+cL%3_Uq+V88RUIBia?)(c%xN?nqRwdC~X&x7L}pxk1~wAzYAwVrgMF- zFt3VMo$^T$fq$5w%Nr)OZ@*)*i=(#v{0U^)Pa#9hNSo zRDPT{lnM{PsON2okc+t7dbxCjjDlkPK%|eGfVrX*zFnDvY>x^AIbXn_&C-*+`Zo$q zmH5fDH4~*LGgUl;%Ys|+P(gD#Ybfyi`d^sk@=18R_2}jI82wD{q4wrYEHV~$)%2^_ zG$4uk=6yKoh-e2dzd++tA*@~Fk^iSJii(H9^G-SnwN|3Lg?MjgNR~V1F@|3Kjm>#7 zlUb#~x&NZJ6sNUYsvVY&uVo-f4ePD9nZr<3B)Q@EaVN zQ;!h6sr)fBh;vuS`A@X?t$XL;=kWe0>eUyoVkTqe_Qg0izYyIg*5K&wyI8RBJF4_s zQb#(i7Rp|6nDoEiTqGTRGLO%wGok7TL+(w}W7<3&{u!i2+lDsWuv&%jt6H!mNV=^~ zC@`tI5m_qZS$f`u`Ci{Jr*=K=EE|C>*P=0`(=a?NnvEkn*CA!Ou+cMXk@(_0awj+E z&{l2v;eZ}Joy_<`dOjXk$i8W&9d9Vw@}jO4GmFi+P{~yC42FCs{{Een+8kV<$*aN& zIufSJJz6R(T{N8G;x#W4)aX7yWbKlqt@1l8n(w@taKh=g86z_PkkW!?*1$dB1CCu52&ebi$Af z%;>-knc8e>){ZeZ`>@7P`nVpf!K#^ku<&p&2HOor!Gt+Dur?PK)Apg0(`_94`W=SS zqwqRJn=zM-IdHUek~HncDT`gH`ChV3MqtZ9J?QN%Y~tB2)J<{XT1yA!2rFmAdP~mV z+nIG;OjzB?kgmf0XwyM5cSmkxQ~eZSi~3@BRv3ObPm~UtO!3o{!tTgfy!C&DG4Gpl zpzPK2_8M^iY716ubl~!Hu3XcKJQ3l=o7vt>SMcTnBTxFT274TEXWz9hoUi9Z|2|#0 zV37@*HL;X=cxPS@GZh|;c$^JN@M~HK*8P)Z#Je|+Zk&lFjyc#javx?y-a%lCU&v2W z;pYzGQ4^+Wtmq+EA4$fi5Zs*S&3;Gy7}!jhxl8%m3c+-M?M znQMm}c-qE}I!e~;8D+s`(z!8Y*(~(C?JNu_V&A}FxD>odY9CAB(eX6ma$kyfs3~(b z+Ok-;6N{!>^11M>A9{Mon&ri$q5dow5=4vj!F(y%vWc?;d8D)Gz>>F{Hp-LRTY!&Z z-RQr_nUm@rM1Qd3t6FQG&yyKz|43{bWsS&VKCu5X3HcARvEck}*koSAGN-TTzf_qv z8??E;gDLsdmP5L_@a`S5`zK$Hof5=`&``Q3hH;KZ4;iAq^Av=ECsLK}?Q%h|#zHVccRh z4iuklKVi={WIFJAoI6)akL7<^0qplAgr#|+Q&mLJ#yx@~vci~b7s9|1fee-Td+0GQ z`qm5k@VOg5{d4BiF^=3KYN1l)SuF6iMo>a$d|nicduHhvpPUDc7Bz^>tb=#ECcN+5 zhSgW}*hTnCi!6kN+~7{-Xm8rt2hwvzD7TJ{p#6?WMvRK2%cpQ|JrXK2gJ4<*iKeFR zOY3=_EE9cWm4h2+)j3n?o#cnxuSC#*_W0{#jslZdyuLFZZ);>dPd$$S4<)8eT!ABT`rB!{`^NiZ8VB3N`Tit&G=nIL*uwqF!$^dm(74rQkuLHs?{ zpSxaobBDF?N9TC3TPIgq>Ns<;%Mb)ws=(l&C9Gx)hPBf&6fWL{*Gku9z5R-|y;{*w z=H+wz&1e^D&v(L3ta|Fnuj>QY^-m}_HHl*PT+t;D_F}V1z1X8Mn#XU54(u7u`F%q; zL?@8->Av)lIbhlE9&}1~qvgQvoVvR^w!CPL*vq!i@*j=sMyv24p$b>y?&6H+U+Gd% z<&9$<2gU(CN-pQT#)t#5`8KQh{Q;cozfbokbVh7hDq}xG^9W9*2wM`hT zsm{rF^!Y(LrVQe`342L&qjquzj}K<2B@yI*!YDZ>nS%?xSvR(~@OgW&*PdvuREy+& zwJ>fe3u5s%Kbk0cbE6*la)LXT>$tF>NjqHbu88S79Z@aMWVeVMyj@#^;xCWoy%t`D zrv@|M8u0NWODYJj$ln?~a#+4UtwLzrHj4gGyFZuFwyAby15XtZxG3@5ln@7z0@N0Q*nj6N_PB+D z>|ufz2Jy&lKQ7-a9kT1e7U$jgIkGzk-`B*__$DY>WsmNe6A<=cEj%?3A}YBKhbpBT z4C)M4HDHF3aD{Wau|)hgzv#;atAgn-8Rvz=VyJ&2mK%2UW`4Kc(%0RKRbJ6-J|lvE z`-O6ZaFT<52vgvpCy$)(!Mk5vCCBGXbJ0O_J~cs`;nvXJF%pOKg|mn%44r-#O>+L? zud*6JTg*tWnNm;W1xxpOn|8oSWC(>2T) z`VAX*w&GHCU8e4r^TQ!~o{;DKXIC#iydS{xCSkNwj^f7D7~UJ!i!-mq@Q7}-^cY1@ z_hhKduZ8vZ&X2BNytwcrctX5Vtt9(@-qn#St>@x>yc*QgOt5xUG!8YJkE4T1a7^a> zgRj298B5tqf79ZGw#GD`ZNvNGwV7Qe7`hdHeA+UED&os_ToOh9Dbc)iB}(Q85mcBG z#-}O4EbJ0M?|2_BAJ&r=#vXk0u{+HrPoI}!&o*E7q1h=DB#kvj?$H3;kC=hUYI%t2 zSc6v@kFa}I6P6xo!{bBrncmxiwI!lE5A)!pi&4)`dB_j{GSdx%Y*i*Kxc#HHwX>uC@RgFP!nN#tn<6^~Zzc z4D=gUh-tpZu}A$GntfNG>V9?caOg82!GZ(a9qA=ntH%mYj@0(!(t`m^A0NnpVFC2J zFFlPr#DAqt`k1-%>1k&gsypzcyA21g>%!rKOR*N~y{1+8jYa?twr>~hhVj*d1Qa@dKwqdj=9lPBk#@@B69A9lOp zCBDU;JTt+A7sTflX3~xK2HEkRpCuEF%;@TC$QAV+>93_DuQiHWgh{WxZa#8dgE1i| z5VsXaz`$z(Mp_rb!bf_Wq|4A{hV(<3sZptUN4C}K%zW9iG`LBZt&uz94hdi6M-S<# z^5DOHF09+yjWgwaTrtXuwCv1x1%`~Wl)SQDdm3-o;GSEme6?>d?Vjjz@UwE^B#eQH zFze&*N(X~MigcuBBdT{9yh1O+_Mz$pPDn&Pmyka$52I>ch3)afDTAlfg;G>_R(^dM$)TNVFyeESmw_)73 zeMt10iN$C7VE@d1cs(xxerGcAAhrarw;scuW_3u(l3uU#%3>(cVwJF-9oHN2<~+&! zXiNV0i#}fsl8jDVTc-4FO?4Y3=|pct^Q&J_##hMR_z0aJ-a*fd>$vtLk&6>!W!fb`>>7y+02vkLvy5Gd)Doc z8L7qOcLyFutYIqqY#$RxIk8%q z(!-jB-EUA&sK?_M_c34TD*Dx*76$5J6pydQ%VO!-8ng?we|F%2?{qf) z7%llBdF}|O+a~Z2=1Kk_^^5daN#ByCQXE*1@)ZE&gs?i#PMv zqA^SOX3OFk?lMe#(7}v&Va+jKEjY04N&MNf0x7K%#hX4BAND5Ux8(w1)2_kz!CPRz zxC*hg$HbR^8V66DK(zS*Ov)-ld5~zs9XDd%nH>20uEm0Jt5Ccy8y%WtW7FVlY_VR6 z`!`oY%OsuV*OQnjGY-9jy?LflGC%b)tHxLC_*j7(&P!0%G6@rZPlavlENmFJ6yJSv z(9>rN7G5vMwOM;G>{B^1Z*GB|>IUTfw-(nPuEaXcY&4bUyjrU)^!~aWIF|*Th-}3D zS&2s;tI=$Ruv8sWC1*U5BbyJQZcM0fB4l6f-Ii5d>QJAv1zjbBJ1%hsc05amq2>Z4 z^~}bPxO^-)RE(qNw?a*3lEMA*QMF(-j@hh$U&}0*jLpJZ+ZE_$Ag^)mN*J$P4FjEZ z7_cA*s`42sx&F5~rv|p5(}v4%ksR@d6N{niJP(?= z3s9|)1^W%_u`)z@@x~Y7=IH{&IOn2?%WAwmn~jBwS3>>lYTON7hbxzJ5VbTPhWj_6 z$g&9kdT+tE_1lmjnue9+*?e9kvP1cJ$s7)0Q(>i;sJOA&G3iOT^A`(52VEfAbL7Gm z=&8OO%{^D4VnQD3A8r<3*cNE|m7rBr0W#j^A?0r_HgN+?Gz!sRQGz3#N|D-8vbqr! zNYAgrmstmJ%C82wk^`DA*$HQt6kd5dS@egIOwH&^of)#WwG{5dBQ4sG5cZqFaol>d z1unxjU{b+GbT=)*im5yC)3^eYdsSeK`Yv>8y#u)}JFrkPC=FG6gmZKNGsYZ2;k1)@ zz5E>duf7Zwt?M`xc}ubZ=?vVMM*aUQ>Ab^w{{Qb!2}QKil9Uuu%3GAqQK?WWNl7Fl z3aJo^$_}YWA~O+X%P3OF-dkiP%AO_Jzx(I=`_px~JiVUt@wlIJZs%J(le)z6Q@2R= z3>nJ@w}&!%S3kNHTkuP*Dl^00BJA7+96f#%Q+Nm=%Zku5rVNMb&Oq(JIh40Li}k(A zan$k*t_`|?{e!OIuJk9=*FD7PtIsfMW)1Y-zeVMxPZ&I=4p$nbcUZKt;r-^)C3hBW ztYVlpSsG6^kK~FUUTo2_4^Ifg_-TSN9bOA-UO0VPb&~rHyM`#gJ9uFC5L-IQyXn;< z+)#Oh_}Ir7`>F~%{a&HL_ajV{z9GEtFI->Ig!RJ2S#n*8>IYh|r|7?a!J;AcOyKXt zIMD`Y@Se<;evTZ+s5t>VQ0>O(a^c3lHQ+|;mQ?y7oR@#Eu-v^C$a#xR(v8sYvJQ*( z)*~mm9;3_}5VrIu+Rv0eha6YFMbj7YQ;qlDxy;V#xsiGvwjB?WrJo*}B3ca(p6n zpU+`%m)U%xCbRc$Q+V*3T<5v~+BX@%>MJsvn$n4fn@JyRqbmPQRpD>(83vZLrbc=j zj-DvY3&~GjouJ0hei}4yE8lw8cAODnz^UEZ^ZKC7`YxTg>Z3W6UBd6XI5q?f8V!;%!{%CcitoS9|K^xi#qPLA88GKu$p zC-R120>6pgeb-?oJ?=zv*V+hof ziEMv=F7?afxN|`)ADoNgXZ>&{XooO&!4TS9aN#aLJBF>3+=Otg?^)}z*k6aul5`m> zd7Ib-J@%ch#{s2!+@WniC*Y@>M6Yer_`le zOp^;F`-=!ou9A6gu(CEAXX$c@zdm>9i%&Dln1T09II2#%ZRI`$bZ=d30iDcVM>%75@&+Eke5&1fFjf~o^mXk0Jt=}*>=NI-ZY{Jd$6luLpiBDR$34~Yzcz=N#WQ6s9L1})6PTG8#987~xhedio8t@^6QD?at>>_czk;>4XVIW_9s#GX zK<~^=Oc5sZtdlqJGxsX;La(Bz@CH6h{#IM-5zO{fVe_jR=&QYhQoql5Wcv-CW28Uj zgRHM4+ifl9*!uT$VZBQ)d&X$`e)Qq!#g6PVts@J1x1dSvQ#6}UhV3)=qhq_h@Y;3| z+MSCr_V5WbR-8aelViC4`7j>wFm(D9p`zmn#Emot5W1l=4w;NYR?nlb-Lr^a$!cjNTqtBj# z+p4n|cp-_y&dlZPr;_J8H<@$_;XF$(VJh}yyT`&o(D{VGpc2#;Z-DLeG(;^-g~!;{ z(2+lDH|FA7bk@9FgcDGuM_wCY=dNfVeo-$wclFs0=4A^YT6h7`+ge9#< z>8(w0KAMC4kx5*z&7$edC;pBdn~*Si_tmAdAJziRq^n4p9jq` z%ka)&4UT@xLGJScbeOdl>1Mk?Pw@^`u0@YetMD^D9o3a-IJ!7ZbeS|A%owv8K00eqJTQ@?hfQbPv`_}vda$OdT=_1d~g7Y{u_c+eCIRM4|`KXfnZe+@41ShY@@q=qI zE;I`oAz6sjTZ?Ck>+r-ift}-~(lk7PD{FhvucsO-TiwC|pS4KY8VhCBiRk-22Ae-D z#`^JV(O)=3aSKjjl+JBv9j(Fw^EWt?`v%v$RpYhJeGD+ZEcyCV&`K@F+Qh#`{I-RZ!5C$a-f)&gMNqR(6VMC9Y+pi;kJ&9t^0^~2|LjwC<))YkB6;x1kA?I z6CTtmWQxvY_Olcx_uNGbi#NDA>Ia+;H>JvR1<~Z1h(AWSt7l%r_uvz>|8fUUZ(YXr zf#oP`T7pUIj-Xnr5W^bfx!Sf5SB3Xw&}uYGSM}!YgVL!x`y}QnEknI`7~GS_;QWwS zOf5`-m*mxttvx2Z_uKdx{0@KQS}3+w=E3YX)Lf;;f=g}q*i?mMf}2U#l=vebd_{{T zuMrpZ6bb|GLFe%`*<)OUUH(~|6D@qmgIH?+H+Kv_THQC-#{tqh3 ze1Az%dfR>@rr;~yTzP{Nm#T4h!4ovbKSajyXqGH?rCmQ2`ehX&&~q+2e+hz8htcp@ z84tU=tDt^EeA3g;!=U90EI#)azRg?n`&Hqzxe4c2zY|9(T5$1m>0-6$M2oK-*wn+2 zKNslol;m!&1*r0D$CmUMBW(W5f8jK}5rGrya4aZ-HG0-;>HZXM%TwXiZxkB84a45S zld!Io$8-kv!$?g3imgs{^Jm4WNI@eWDk=k^SQv;fU&2 z{>Q3uwsb7$&Zl`@c-Kub$OXcg6Yh#f8<}IT(v~hl4IZ@erQ@Z~_|ocIn&5Wva;0xH#XIYt#B6a!Xn<*_Gb!VVyZ>m?2mcQIYYaYbWa|hF^ zRJ4XLZ^qy8;xEw^_P>za+h@td?r`R;)?luW1Gg9U;@Ai~K3Oa*aKj7eeY7iL4x7VH z=I(EfBth}gCL{G4DluR7tEN*-81bSTOJ!f!r4KpO!JU8ZdD1}Hm)VjJ z_5U?g)?!0B!qbmWO?{b`Ihbz_58}SL9&Gp7l}^3-F?dB^Mt2roi*Yah?YkaJ40NRf zw-W{__#Q@Fg939KJ}b0-)VW%S;Ld_%7nS-(4UoM{rDlliLX~WFvqPu{(3hR zHk&nyrcS}v4DnaZIF3L0PjEWEh_lOJy$HI-9f>{dvo#CNLfP z*Jqn5`ZtOf!GpRPD!yI9Zs(=V)G91cq+GGep?-`cQxgQQ{6f2P;c&XcBaKr;WVxDW_{dH z`bfXa?hQfwQ!;|3=_Bal6U4%RK+gW`PZiN8Lo9t5f6kNNEeCL7N`K)a{eQi^--YW= zXBvMw<^6I%moD?5;vj6pa>*y8yvHNqJ&kXz!wGWSvE{;gIMSP*uFl*h$Nik;%?q)j zg?0>N*R&x1*fE0sF(bIzJctub18G|*dh~}Oe0$5ACq{d6#1RkPXwjd;miJ>`kP|;G zz2o$!=7-Y??cQi=Jr@_sw&K3VdFVe9E?t?DcrUc*?JPWnDl5MHXixLZezbk>!Mv8< zZ285H4!HqbtQtf`$b7$K( zUb1H#!eM#C*h_0TS9g$eeN$$+PX8-WgN;;HGjCgLiCBwgn?yTx0Gl%|6 z5?*?|udno*`Sazv0OoEAl)QK#yBr!u#h9U7dfJ!!P7LOZ;6Xf)2`trJgw=j4IaqAcl&WUqn%TZG0KG2b| zvG#oT!;XpbWKONEjV6|6__$>VGI}pYszM%0@1MuZy4ScOtd|!#nhdNImVdZ-;zvr} z`6{Gm-i?Rl-BAYrXuZ^rcU- zVT|}ujmKE9c{few{EI>1iyn~6g@@H)*j%sz85z5fUUM1EcE3YviW0MbXfpA(=$gN~ zvTVkGw4f6gk8ow6A^58~jeK@8s~`gbSES3GcKq@lcDemQb=ggsxx+H%rpE0)X@ z{zZfEUOpRh{$l9{F4&2>mV@!N9;hEHY?XKO(DLX8EKom=@V1rce(nd#Z?~dvkRG3k zhj6I3HSe#p=P3D}?_T4=*_tvdp5n%+tFCN%M*5HIMB|os%$AnYiTtuVt;Sk%&~;Nf zh?cpoM32iBYV-3dPqy)TiQRfhICpp;nr`=o;p9jJ&rO9{Yw_PkT|nf?S{z@mz^%n< zto4vy4iyWwsj}gggYsKK`*OA1r@zHAJ3F~AdsX!2;W0fqEWuhj;JWbCaOu>rZ_j6D z?HIR6y2VzgQb^Z4wsIs7oK$D^uN}~|ls<%4-Wc3J1TNw^&MsXC%e=!FxbHTmHUEMm z^8SkJrbWv&#@yCu!RZz@Tv*?ek2~7abWJZtZna~A@XmKuTQU%4td|+vY+?CX{8Zzb zR2BZaDtUnNf9O5qJH`cs)6LC_d)*)5@Z%KpN*swAVFh)Hjl@Q+C8uxcTsF6y#MlU)Ol*b0~1U+&Qmyo)jB+6szwhzWiByN z;LWh_$ei>Z)#G2_KZVB#A0N$u;_DdJS%pLM3$gdjx+C$uHeiMsUTtvgBa%fh>^1IJamIj+qcj-&G6|cvJe^a4zZ4~-dOoD&;A{^PC z4P&)J1Pi~%BCQrdZ^UzVU38As>b$S3!zY(?=)FplBPOY`XY1zN5%345hM$n@@ErSO zKWeRc5no1^q59WRtdT75IJ>>rA^f<4nWJg_L)N%&TC?McGq7(=!Nb$xxZZpmED#I5 zcgvx1atoHqTw(X-8*r$tMSSyLaBQZ;3{_?7)JW&gie_wE`4eM3-r@7T#~5950}<(G zP!v~;#E$!+-ZUTfUvm*YdJE3xORlnN4nIx_=b73;96q=c4JLns#?w9MsJal-mrul+ zM-do5GYMX!S7L#%+j5+a!zSYfN(NP%}jyK@#!g{QG{sv+DEAgZF3YKe}L}Kt^ zq(9yTlXbZ;Y`GEPlh@+?%uFPS=C-iSD*T-)tjnfR^nVdZmCQa|^jDLMgh!>)V-qUJ z%*BbS2vpyV!MKBqu(IVEyxh4H-=`l(n~qoDVqJl@y29Kwd4SLnSCN!a3f1igu=?Q+ zOtRX7o{~vFAl~#Lmg%^^cLf?Jrog&nIbL;Gj^aq+Ki{0rl=EY#R3N^@#nOL4B|aE< z2JX#QL*c}1RO?9}(ua6Bq%VWTgLQDI$;Y!nMdIl>jmf*uA}--1G&dbWdgKmV^52L> z#cN=#y$ZQQQeo${9LfGmF@Iq)N){wzNm(+6981O{)g%@Rt8cYO1b1``U{h@;+BY-e ztvg@vHv1564NS#H#^dOKS;)Dz01D}8aCx&qcuM&Q&pm(z2Mh6~&0Z|}xfM?uvvA>D zI=*SHK~TT6Pw|tu>*$NcVgPmTy$Q$4kyp2 zLsdNmLFboAhIAR+7cQ6CX$owtR>16NDk2jl=kaGH4EL-;<(?##?3OM*=jkl)oXFQB z0$JO|g%3}eb3=A>zOT7~9d6t4?~1V0-DFl6zg+aV42;j-fFa?zSkP-bjz8G~!{2G^;reG2Y9u#PZk30Psrd-`yBn6n_M<_y5K|UNwxcFVGGz(ErgV_UY_rV>9-a@pgHoVhWye#RL@kL2IQ+tlW{I29uyiOyd@f>E$ ztmn|gGdM8g6ebm&f>rog^ijGDL)DwYthf+7(X@X)k@rLw>^#zoZP)gu?i>^9>8j8>@FON2uR!v;8_3hTg(F4xk^1ft zT3>vEuKgZE&8-q`K2#$5*;5R9`4Z3K-=V7H3&zEK$Liz1vGTgSuWJ;zeONQr{z~Fs zKj~a-m%w)(aco&Lon9NK@O{Ntp8hn9>msFJXLS#Lvlq^=cygLt{tiri2S?E^N;-eU z)`4{x(Xs)$hrS8hsUFLMzrm>Q512Xq!;nS=s){%0wU#nhl(gcfE^XLxps=&=tMi5E zIL6D9n070HdydA_1T%TIOEjM}72c=KD8|!=-nO7~x2}>QY{z27R&4N6WTJUfeh!mf zr2Wl!uU~V%No-CZo9497ZBEaqmQ)$kiXqF}@W@_u?pdHk1?dm@GEt9OaRxM0H{wUp zt!nlqal3z_c-Q98qiz=M&P?OWi<3Eb!B|d92;iU19!%NZi?{PS(e$9q;gs95PkAdE z3TOSkMjMvqx22Am8s96bane*Z+O^SO^E!Luk)Z}=d|lXy zv7*zBKQEfuoOWz5od@p=FKDQh3DuR_vHXiB zUH5CUpqDmBNEY$jP+e{qq{~4&r1wpH|98_2SU;;hgNB;0?xiWqyNlmVxLs4aiauG| zRdS@l$-k2%Yv4o%)z9?OE?QH+fVXJmE=8(sWpbxV3!_Vkpq-kHZLjriMB zkNvXRG4H+}mE88-Woh{z7q z+GW6BW%@i+X&|!)Blfsq#2IBqlEF0M+%Cpc2{mE8nwj)POYcQ{OO`ox6_%eBV}Et0 zZnDk)<(>5(2zOd^20O(Bwr?FTYqeP63`TRfb_BEMjOHlW?}Ylfarum1Ty0^=g~i5f zY-+%2W6^&58t`tqAv%^z6EO@U|7Y51R?!V_& z)Q+;|YWcmxZX|KejYKXfo+~_{IJUKoWy8=Yj#)U7dV!;OOZ>*y9*K`Q(UudNi~d_` zAkUyKXCBw#h2OedJ6Mm|`}F8PLXXqN>ha$SefB(O$iK6tZ^NV`UoSADy)d9RZm?i+ zbQg|O>P9O$u0ok)yjLW0bJw}Nk~5q0FHC2mOC)XAji>j>Aoh>;{jn$y@ znP$B0^b7r}zT?ulp9nPmhaLG%cvf_cNeBL*%KQ%$-u#2t5(TO!Dp6@@OWvze;RPq@ z5l?K(CDOwi;jT$d(SHZ_68}b#XzX)lvwHb7HY}e+cUc3xJ~MQdn{v*PvXG*kxXeA$WDuU@b^(oMl5@S zowtkdz5jaAmeWLEOvB@}On6jngxUUWDEgEKv&OCHpOKAlg$#84l!hAzWlh;S10$Da z;$&qOVn$@cYQRQJira!_(YctCyB)EaNz7PD=?-Z4Yf05P5ods>T}j0b&%Xwig6tAYzngsNAiV^JCo*H^866-ZVtK% z4TVkk^mGxH*Uk~&z7RPY4=mah}7YiqBkLZp$NEfaBz5Xi1 z%}>Rr)yr|heyQYHm*CFPB^W;~8E@j3VvFxGnIQ?6VZbzA^$6kL%dRw*?h=EH@3>!m z2>s1utz0B~BiAS#`J9CCm#K&^-;9%^4xn$0cKs{{qzFoBxgFo`7QD{R$=0Qw^8eQ z7G=$j!Tj2OSZvyfmti>=U$h=m-ee*1cqY8X+v|Bu_VS(P(C^Jkk(}70C_{LChdk-p6I!u4<%##0q+L4DK)S=3yU04s^MT z>i(;6zDGEo`Gml-F$yO{OIzx;1=}AMFGq(gH5TF zwFAr5rlakTAXJVYkK)w%xFOHw#a0Iq)$JONSG|GN5e2R?Y0Iv;dd$3TLT}-<9C>6( z183=<{VLsP8j}AgG~m$_)hF{Ka0fH)!60r55xPZa=VKjJ9?XPVW+O#yT*pE zgrA!@%Z?s8J$Tr{itW~0uhbz^aH&>|GJ+60|7$FQx*10>98z|$gS<~`8id6~bDE9=Vp-K8rn zRyr#uJF=p|k<+{#xZcv97yRvLYirGQ^8GY;(TP*ec3{>qS+5)EGS@(pt@pO&v?34w z=~{s>g&EjaH~{xe{4m*iI+XNQAW)zNoh_a82tTj|VjKk}HS{o^%}|AMSYni^U-rFd-76%4fpyAds6zC)fabS zyRx>_w3&qW$9mvsnhO$+PLiy?u<=v&qUyyp@$i2{)#K*W?J5kuw zik?3!;Pj;@_P_0nV=V@u{_Px0_uPQc1@bL_^%#HB|Da8iHvCy+z*{#vbH-p>syhk` zEYF4ZnFF{&Q8JOW!q@LAO!R7BT8!}#f81b3Cl6xU77tDyEBT{f{{CSz;*aV%BL}O^`&sO<^+}e`vUh*@kaBh?d#g zg?kzXFk+lH*Cz= n)` in the + # kernel). Catch it here so a wide matrix is a Julia error, not a + # process-killing C++ assert / abort. + m >= n || throw( + ArgumentError( + "svd only supports m >= n (got $(m)×$(n)); the backend does not factor wide matrices" + ), + ) k = min(m, n) S = real(T) # cuSolver requires full square buffers regardless of full_matrices diff --git a/src/ndarray/linalg.jl b/src/ndarray/linalg.jl index ae73a9f0f..879edafdd 100644 --- a/src/ndarray/linalg.jl +++ b/src/ndarray/linalg.jl @@ -132,6 +132,9 @@ Destructuring an `SVD` yields `(U, S, V)` — the adjoint of `F.Vt`, not `F.Vt` itself. `F.V` is a lazy `Adjoint` wrapper, so operating on it falls back to scalar indexing until `adjoint(::NDArray)` is implemented; prefer `F.Vt`. +The backend only factors tall or square matrices (`m >= n`). A wide input +throws `ArgumentError` rather than hitting the C++ `m >= n` assert. + Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. Integer and `Bool` inputs are converted to `Float64`. As everywhere else in the package, that conversion needs `@allowpromotion` only when it widens the diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 3f531b40d..f42955061 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -874,17 +874,63 @@ function Base.isapprox(arr::NDArray{T}, arr2::NDArray{T}; atol=0, rtol=0) where return compare(arr, arr2, atol, rtol) end +# HDF5 signature. A leftover empty/truncated file from a crashed write has no +# header, and opening it in a Legate HDF5 task aborts the runtime. +const _HDF5_MAGIC = UInt8[0x89, 0x48, 0x44, 0x46, 0x0d, 0x0a, 0x1a, 0x0a] + +function _hdf5_check_dataset(dataset::AbstractString) + return isempty(dataset) && throw(ArgumentError("HDF5 dataset name must be non-empty")) +end + +function _is_hdf5_file(path::AbstractString) + isfile(path) || return false + filesize(path) < length(_HDF5_MAGIC) && return false + return open(path, "r") do io + return read(io, length(_HDF5_MAGIC)) == _HDF5_MAGIC + end +end + +# A truncated leftover `.h5` is not a valid file and makes HDF5CombineVDS abort. +# Leave a real HDF5 file in place so Legate can truncate it. Do not touch +# `*_legate_vds`: that directory holds the payload of a VDS write, and we +# cannot tell a stale sidecar from one a later read still needs. +function _prepare_h5write(path::AbstractString, dataset::AbstractString) + _hdf5_check_dataset(dataset) + isdir(path) && throw(ArgumentError("h5write path must be a file, got directory $path")) + parent = dirname(path) + if !isempty(parent) && parent != "." && !isdir(parent) + throw(ArgumentError("h5write parent directory does not exist: $parent")) + end + if ispath(path) && !_is_hdf5_file(path) + rm(path; force=true) + end + return nothing +end + +function _prepare_h5read(path::AbstractString, dataset::AbstractString) + _hdf5_check_dataset(dataset) + isdir(path) && throw(ArgumentError("h5read path must be a file, got directory $path")) + isfile(path) || throw(ArgumentError("HDF5 file does not exist: $path")) + _is_hdf5_file(path) || throw(ArgumentError("not an HDF5 file: $path")) + return nothing +end + """ h5write(path::String, dataset::String, arr::NDArray) Write an `NDArray` directly to an HDF5 dataset without a host copy or dimension flip. +A leftover empty or truncated `.h5` from a crashed write is removed first. +A valid HDF5 file is left in place so Legate can overwrite it. The +`*_legate_vds` sidecar is not touched: it holds the payload of a VDS write. + # Arguments - `path`: Path to the HDF5 file. - `dataset`: Name of the dataset to write. - `arr`: The array to write. """ function h5write(path::String, dataset::String, arr::NDArray{T,N}) where {T,N} + _prepare_h5write(path, dataset) st_handle = get_store(arr) # NDArrays are row-major, so this writes straight through (no dim flip, no warning). la = Legate.LogicalArray{T,N}(st_handle, size(arr)) @@ -904,6 +950,7 @@ Read a dataset from an HDF5 file into an `NDArray`. - `layout`: On-disk memory order, either `:row` (default) or `:col`. """ function h5read(path::String, dataset::String; kwargs...) + _prepare_h5read(path, dataset) la = Legate.h5read(path, dataset; kwargs...) T = eltype(la) N = Int(Legate.dim(la)) diff --git a/test/array/linalg.jl b/test/array/linalg.jl index c5014032f..3a204c1bf 100644 --- a/test/array/linalg.jl +++ b/test/array/linalg.jl @@ -325,6 +325,11 @@ end end end +@testset "svd wide matrix (m < n) throws" begin + A = cuNumeric.NDArray(my_rand(Float32, 3, 5)) + @test_throws "m >= n" LinearAlgebra.svd(A) +end + @testset "svd thin output shapes (full=false)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) m, n = 6, 4 diff --git a/test/io/hdf5.jl b/test/io/hdf5.jl index a35cfb59b..704dff55c 100644 --- a/test/io/hdf5.jl +++ b/test/io/hdf5.jl @@ -53,4 +53,38 @@ end test_hdf5_roundtrip(T, shape) end end + + @testset "host-side path checks" begin + mktempdir() do dir + arr = cuNumeric.ones(Float32, 4, 3) + path = joinpath(dir, "values.h5") + + @test_throws ArgumentError cuNumeric.h5read(joinpath(dir, "missing.h5"), "values") + @test_throws ArgumentError cuNumeric.h5read(dir, "values") + @test_throws ArgumentError cuNumeric.h5write(dir, "values", arr) + @test_throws ArgumentError cuNumeric.h5write(path, "", arr) + @test_throws ArgumentError cuNumeric.h5read(path, "") + + stub = joinpath(dir, "stub.h5") + write(stub, "not hdf5") + @test_throws ArgumentError cuNumeric.h5read(stub, "values") + + # A leftover empty file is what aborted HDF5CombineVDS. + leftover = joinpath(dir, "leftover.h5") + write(leftover, UInt8[]) + cuNumeric.h5write(leftover, "values", arr) + cuNumeric.Legate.runtime_sync() + @test cuNumeric._is_hdf5_file(leftover) + @test size(cuNumeric.h5read(leftover, "values")) == size(arr) + cuNumeric.Legate.runtime_sync() + + # A second write to the same path must replace, not abort. + arr2 = cuNumeric.zeros(Float32, 4, 3) + cuNumeric.h5write(leftover, "values", arr2) + cuNumeric.Legate.runtime_sync() + @allowscalar @test safe_compare( + zeros(Float32, 4, 3), cuNumeric.h5read(leftover, "values"), 0, 0 + ) + end + end end From 6a544f0fa2ca44c3678d4ecdbf48bf5eaca94ef6 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Tue, 18 Aug 2026 02:16:44 -0500 Subject: [PATCH 13/49] New @accelerate macro explicit julia scoping semantics and CUDA.jl launching patches (#174) --- .buildkite/jll.pipeline.yml | 2 +- .buildkite/pipeline.yml | 2 +- .buildkite/run_developer_ci.sh | 2 +- .buildkite/upload_gpu_ci.sh | 23 +- .github/workflows/ci.yml | 39 ++- .github/workflows/container.yml | 4 +- .github/workflows/developer.yml | 57 +-- .github/workflows/docs-tags.yml | 1 + .github/workflows/docs.yml | 1 + .github/workflows/pkg_resolve.yml | 4 +- .github/workflows/version_check.yml | 1 + .gitignore | 3 + README.md | 89 ++--- benchmark/README.md | 19 +- benchmark/__plot_results.jl | 301 ---------------- benchmark/benchmarks.toml | 40 ++- benchmark/debug/grayscott_accelerate.jl | 66 ++++ benchmark/plot_results.jl | 284 +++++++++++++++ benchmark/run.jl | 4 +- benchmark/src/benchmarks/dmd.jl | 20 +- benchmark/src/benchmarks/grayscott.jl | 125 +++---- .../benchmarks/grayscott_accelerate_forms.jl | 95 +++++ benchmark/src/core.jl | 10 + docs/make.jl | 2 +- docs/src/api.md | 4 +- docs/src/api_cuda.md | 2 +- docs/src/api_unary.md | 2 +- docs/src/benchmarks/howto.md | 12 +- docs/src/benchmarks/results.md | 4 +- docs/src/configuration/hardware.md | 2 +- docs/src/debugging.md | 36 +- docs/src/developer_mode.md | 2 +- docs/src/examples/grayscott.md | 78 ++--- docs/src/examples/montecarlo.md | 43 +-- docs/src/index.md | 46 +-- docs/src/internals.md | 70 +--- docs/src/perf/kernel_fusion.md | 65 ++-- docs/src/perf/reduce_allocations.md | 93 ++++- examples/custom_cuda.jl | 37 +- examples/daxpy.jl | 2 +- examples/gray-scott.jl | 2 +- examples/gray-scott.py | 110 +++--- examples/integrate.jl | 18 +- examples/stencil.jl | 11 +- src/cuNumeric.jl | 6 +- src/cuda/cuda_ptx_task.jl | 157 +++++---- src/ndarray/broadcast.jl | 31 +- src/ndarray/broadcast_fusion.jl | 327 +++++++++++++++++- src/ndarray/detail/linalg.jl | 11 +- src/ndarray/detail/ndarray.jl | 53 ++- src/ndarray/ndarray.jl | 2 +- src/ndarray/promotion.jl | 20 ++ src/ndarray/unary.jl | 11 +- src/scoping/accelerate.jl | 257 ++++++++++++++ src/scoping/broadcast_lifetimes.jl | 10 +- src/scoping/lifetimes.jl | 4 +- src/scoping/scoping.jl | 136 +++----- test/Project.toml | 2 +- test/analysis/accelerate.jl | 190 ++++++++++ test/analysis/promotion.jl | 5 + test/analysis/type_stability.jl | 55 +++ test/array/broadcast_basic.jl | 27 ++ test/{defunct => cuda.jl}/fusion_compare.jl | 22 +- .../{defunct => cuda.jl}/fusion_compare_1d.jl | 22 +- test/cuda.jl/padding.jl | 224 ++++++++++++ test/{defunct => cuda.jl}/vecadd.jl | 31 +- test/gpu_only/broadcast_fusion.jl | 4 +- test/runtests.jl | 43 +-- test/workflows/grayscott.jl | 190 ++++------ 69 files changed, 2419 insertions(+), 1254 deletions(-) delete mode 100644 benchmark/__plot_results.jl create mode 100644 benchmark/debug/grayscott_accelerate.jl create mode 100644 benchmark/plot_results.jl create mode 100644 benchmark/src/benchmarks/grayscott_accelerate_forms.jl create mode 100644 src/scoping/accelerate.jl create mode 100644 test/analysis/accelerate.jl rename test/{defunct => cuda.jl}/fusion_compare.jl (87%) rename test/{defunct => cuda.jl}/fusion_compare_1d.jl (85%) create mode 100644 test/cuda.jl/padding.jl rename test/{defunct => cuda.jl}/vecadd.jl (81%) diff --git a/.buildkite/jll.pipeline.yml b/.buildkite/jll.pipeline.yml index 27656870d..7d591fb46 100644 --- a/.buildkite/jll.pipeline.yml +++ b/.buildkite/jll.pipeline.yml @@ -23,7 +23,7 @@ steps: fi julia --project -e 'using Pkg; Pkg.resolve(); Pkg.instantiate()' - JuliaCI/julia-test#v1: - test_args: "--quickfail --jobs=8 --verbose" + test_args: "--jobs=8 --verbose" - JuliaCI/julia-coverage#v1: dirs: - src diff --git a/.buildkite/pipeline.yml b/.buildkite/pipeline.yml index 8fab17e69..7fc8e4fef 100644 --- a/.buildkite/pipeline.yml +++ b/.buildkite/pipeline.yml @@ -10,6 +10,6 @@ steps: command: ".buildkite/upload_gpu_ci.sh" agents: queue: "cuda" - if: build.message !~ /\[skip tests\]/ + if: build.message !~ /\[skip ci\]/ if_changed: "{src/**,scripts/**,deps/build.jl,Project.toml,lib/CNPreferences/src/**,lib/cunumeric_jl_wrapper/**}" timeout_in_minutes: 5 diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index e4004be1f..4d130393c 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -86,5 +86,5 @@ cp LocalPreferences.toml test/LocalPreferences.toml julia --color=yes --project=. -e ' using Pkg - Pkg.test("cuNumeric"; test_args = ["--quickfail", "--jobs=8", "--verbose"]) + Pkg.test("cuNumeric"; test_args = ["--jobs=8", "--verbose"]) ' diff --git a/.buildkite/upload_gpu_ci.sh b/.buildkite/upload_gpu_ci.sh index 3e8546d1e..dc1fd3ec7 100755 --- a/.buildkite/upload_gpu_ci.sh +++ b/.buildkite/upload_gpu_ci.sh @@ -15,13 +15,24 @@ message="${BUILDKITE_MESSAGE:-}" run_jll=true run_developer=true -# Keep both suites for main and PRs into main. For non-main PRs, select the -# suite whose wrapper matches the code under test. +if [[ "$message" =~ \[skip[[:space:]]ci\] ]]; then + echo "Skipping all GPU CI because the build message requests it." + exit 0 +fi + +if [[ "$message" =~ \[skip[[:space:]]jll\] ]]; then + echo "Skipping JLL GPU CI because the build message contains [skip jll]." + run_jll=false +fi +if [[ "$message" =~ \[skip[[:space:]]dev\] ]]; then + echo "Skipping developer GPU CI because the build message contains [skip dev]." + run_developer=false +fi + +# Keep both suites for main and PRs into main. For non-main PRs and post-merge +# develop builds, select the suite whose wrapper matches the code under test. if [[ "$branch" != "main" && "$base_branch" != "main" ]]; then - if [[ "$message" =~ \[skip[[:space:]]jll\] ]]; then - echo "Skipping JLL GPU CI because the build message contains [skip jll]." - run_jll=false - elif [[ "$pull_request" != "false" && -n "$base_branch" ]]; then + if [[ ("$pull_request" != "false" && -n "$base_branch") || "$branch" == "develop" ]]; then base_ref="refs/remotes/origin/$WRAPPER_BASE_BRANCH" # The published wrapper JLL tracks main, so compare against main even # when the pull request targets develop. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 72d381fbb..a4eb983fc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,4 +1,4 @@ -name: CI CPU +name: CI on: workflow_dispatch: @@ -18,7 +18,10 @@ on: - 'scripts/**' - 'deps/build.jl' - 'Project.toml' + - 'lib/cunumeric_jl_wrapper/**' - 'lib/CNPreferences/src/**' + - '.github/workflows/ci.yml' + - '.github/workflows/developer.yml' tags: - 'v*' branches: @@ -29,13 +32,19 @@ on: - 'scripts/**' - 'deps/build.jl' - 'Project.toml' + - 'lib/cunumeric_jl_wrapper/**' - 'lib/CNPreferences/src/**' + - '.github/workflows/ci.yml' + - '.github/workflows/developer.yml' jobs: - pkg_resolve: + resolve: + name: Package resolution + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} uses: ./.github/workflows/pkg_resolve.yml - check_changes: - name: Check for wrapper changes + wrapper_changes: + name: Wrapper change detection + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} runs-on: ubuntu-latest outputs: wrapper_changed: ${{ steps.wrapper-changes.outputs.changed }} @@ -60,10 +69,22 @@ jobs: fi fi - test: - name: Julia ${{ matrix.julia }} - ${{ matrix.os }} - needs: [pkg_resolve, check_changes] - if: ${{ github.base_ref == 'main' || needs.check_changes.outputs.wrapper_changed != 'true' }} + developer_tests: + name: Developer wrapper tests + needs: [resolve, wrapper_changes] + if: ${{ !contains(toJSON(github.event), '[skip ci]') && !contains(toJSON(github.event), '[skip dev]') && (github.event_name != 'pull_request' || github.base_ref == 'main' || needs.wrapper_changes.outputs.wrapper_changed == 'true') }} + permissions: + contents: read + packages: write + attestations: write + id-token: write + actions: write + uses: ./.github/workflows/developer.yml + + jll_tests: + name: JLL wrapper tests - Julia ${{ matrix.julia }} - ${{ matrix.os }} + needs: [resolve, wrapper_changes] + if: ${{ !contains(toJSON(github.event), '[skip ci]') && !contains(toJSON(github.event), '[skip jll]') && (github.base_ref == 'main' || needs.wrapper_changes.outputs.wrapper_changed != 'true') }} runs-on: ${{ matrix.os }} strategy: fail-fast: false @@ -125,4 +146,4 @@ jobs: LEGATE_AUTO_CONFIG: "0" LEGATE_SKIP_RUNTIME: "true" LEGATE_CONFIG: "--cpus 1 --utility 1 --sysmem 500" - run: julia --project -e 'using Pkg; Pkg.test(test_args=["--quickfail", "--jobs=2", "--verbose"])' + run: julia --project -e 'using Pkg; Pkg.test(test_args=["--jobs=2", "--verbose"])' diff --git a/.github/workflows/container.yml b/.github/workflows/container.yml index 07a68c927..323a1e539 100644 --- a/.github/workflows/container.yml +++ b/.github/workflows/container.yml @@ -12,13 +12,13 @@ on: required: false default: false workflow_run: - workflows: ['CI CPU'] + workflows: ['CI'] types: [completed] branches: - main jobs: push_to_registry: - if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }} + if: ${{ !contains(toJSON(github.event), '[skip ci]') && (github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success') }} name: Container for ${{ matrix.platform }} - Julia ${{ matrix.julia }} - CUDA ${{ matrix.cuda }} permissions: contents: read diff --git a/.github/workflows/developer.yml b/.github/workflows/developer.yml index 93cd090ce..a1f9c16ce 100644 --- a/.github/workflows/developer.yml +++ b/.github/workflows/developer.yml @@ -1,55 +1,12 @@ -# Develeper CI test. This will build the workflow using Jlls and building wrappers from SRC -name: Develeper CI test +# Developer wrapper tests build the wrappers from source instead of using JLLs. +name: Developer Wrapper Tests on: - workflow_dispatch: - inputs: - tag: - description: 'Tag to build instead' - required: false - default: '' - mark_as_latest: - description: 'Mark as latest' - type: boolean - required: false - default: false - push: - paths: - - 'src/**' - - 'scripts/**' - - 'deps/build.jl' - - 'Project.toml' - - 'lib/cunumeric_jl_wrapper/src/**' - - 'lib/cunumeric_jl_wrapper/include/**' - - 'lib/CNPreferences/src/**' - - '.github/workflows/developer.yml' - tags: - - 'v*' - branches: - - main - pull_request: - paths: - - 'src/**' - - 'scripts/**' - - 'deps/build.jl' - - 'Project.toml' - - 'lib/cunumeric_jl_wrapper/src/**' - - 'lib/cunumeric_jl_wrapper/include/**' - - 'lib/CNPreferences/src/**' - - '.github/workflows/developer.yml' -jobs: - pkg_resolve: - uses: ./.github/workflows/pkg_resolve.yml + workflow_call: - docs: - name: Developer CI test - Julia ${{ matrix.julia }} - needs: pkg_resolve - permissions: - contents: read - packages: write - attestations: write - id-token: write - actions: write +jobs: + test: + name: Julia ${{ matrix.julia }} strategy: fail-fast: false matrix: @@ -157,4 +114,4 @@ jobs: cp LocalPreferences.toml test/LocalPreferences.toml - julia --color=yes --project=. -e 'using Pkg; Pkg.test("cuNumeric"; test_args=["--quickfail", "--jobs=2", "--verbose"])' + julia --color=yes --project=. -e 'using Pkg; Pkg.test("cuNumeric"; test_args=["--jobs=2", "--verbose"])' diff --git a/.github/workflows/docs-tags.yml b/.github/workflows/docs-tags.yml index ead34e481..5e982c79d 100644 --- a/.github/workflows/docs-tags.yml +++ b/.github/workflows/docs-tags.yml @@ -9,6 +9,7 @@ on: jobs: docs: name: Documentation + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} permissions: actions: write contents: write diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index 3317cea6a..84da51106 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -16,6 +16,7 @@ on: jobs: docs: name: Documentation + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} permissions: actions: write contents: write diff --git a/.github/workflows/pkg_resolve.yml b/.github/workflows/pkg_resolve.yml index 47c5302d1..6eb18b0bf 100644 --- a/.github/workflows/pkg_resolve.yml +++ b/.github/workflows/pkg_resolve.yml @@ -1,11 +1,11 @@ -name: Pkg Resolve +name: Package Resolution on: workflow_call: jobs: resolve: - name: Pkg.resolve + name: Resolve dependencies runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/version_check.yml b/.github/workflows/version_check.yml index 346bc7d6b..a738ada5d 100644 --- a/.github/workflows/version_check.yml +++ b/.github/workflows/version_check.yml @@ -8,6 +8,7 @@ on: jobs: version-check: name: Version Check + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 diff --git a/.gitignore b/.gitignore index d431bc365..cf02abbe8 100644 --- a/.gitignore +++ b/.gitignore @@ -16,6 +16,8 @@ logging logging/* debug debug/* +!benchmark/debug/ +!benchmark/debug/grayscott_accelerate.jl # example outputs (examples/data and the docs copy of gray-scott.gif are tracked) examples/*.h5 @@ -31,6 +33,7 @@ benchmark/results/* benchmark/plots** compile_wrapper.sh +__plot_results.jl *.tar.gz # generated by CMake diff --git a/README.md b/README.md index 39dd0896b..cad89042a 100644 --- a/README.md +++ b/README.md @@ -1,128 +1,89 @@

cuNumeric.jl - cuNumeric.jl -

-

- cuNumeric.jl - cuNumeric.jl + cuNumeric.jl

-[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) -[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) +[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) -cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. -cuNumeric.jl requires x86 Linux, an NVIDIA GPU, and Julia >= 1.10. If ARM support is of interest open an issue. cuNumeric.jl requires x86 Linux, an NVIDIA GPU, and Julia >= 1.10. If ARM support is of interest open an issue. ### Quick Start -cuNumeric.jl can be installed with the Julia package manager. Activate your preferred environment and then from the Julia REPL run: - - cuNumeric.jl can be installed with the Julia package manager. Activate your preferred environment and then from the Julia REPL run: ```julia using Pkg Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") -using Pkg -Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") ``` -The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). - -The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). +The first installation can take a while because it includes several large dependencies, such as the CUDA SDK. To use a local cupynumeric build, see [Build Modes](https://julialegate.github.io/cuNumeric.jl/dev/install). ```julia using cuNumeric -using cuNumeric cuNumeric.versioninfo() ``` > [!WARNING] > Starting more than one instance of cuNumeric.jl can lead to a hard-crash. The default hardware configuration reserves all available resources. -For more details, see [Hardware](./configuration/hardware.md). +For more details, see [Hardware](https://julialegate.github.io/cuNumeric.jl/dev/configuration/hardware). ### How `NDArray`s work The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it is a drop-in replacement. You can use the same constructors (i.e., `zeros`, `ones`, `rand`), broadcasting, slicing, and linear algebra. Under the hood a few details differ from Base, and knowing them can help you write fast code. -**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be partitioned over GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. Functions like `println` result in data being copied to the host and can also be slow. +**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be partitioned over GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. **Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communite with the Julia runtime. When you need a plain Julia number, call `unwrap` or `only`: +**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communicate with the Julia runtime. When you need a plain Julia number, call `unwrap`: ```julia s = sum(A) # NDArray{T,0} x = unwrap(s) # T, e.g. Float32 -x2 = only(s) -``` -s = sum(A) # NDArray{T,0} -x = unwrap(s) # T, e.g. Float32 ``` -**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. **The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. -For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). -For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). +For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/dev/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/dev/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/dev/perf/patterns_to_avoid). -### Kernel Fusion ### Kernel Fusion -Nested broadcast expressions fuse into a single kernel by default when on GPU. Prefer `@.` for multi-op elementwise code so every operator is dotted and the expression stays completely fused. Even just forgetting the `.` on unary negation (i.e., `y .= -a .+ b .* c`) will result in unfused code. Use the following pattern instead. Nested broadcast expressions fuse into a single kernel by default when on GPU. Prefer `@.` for multi-op elementwise code so every operator is dotted and the expression stays completely fused. Even just forgetting the `.` on unary negation (i.e., `y .= -a .+ b .* c`) will result in unfused code. Use the following pattern instead. -```julia -y .= @. -a + b * c ```julia y .= @. -a + b * c ``` -See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. - -See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. - -### Helping the Garbage Collector - -Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. -Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. - -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. +See [Kernel Fusion](https://julialegate.github.io/cuNumeric.jl/dev/perf/kernel_fusion) and [Debugging](https://julialegate.github.io/cuNumeric.jl/dev/debugging) for controls and diagnostics. -```julia -@analyze_lifetimes begin - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 -end -``` +### The `@accelerate` macro -### Performance at a glance +`@accelerate` fuses eligible GPU broadcasts within and across statements, then releases materialized temporary `NDArray`s after their last use on CPU or GPU. See [The `@accelerate` Macro](https://julialegate.github.io/cuNumeric.jl/dev/perf/reduce_allocations) for usage guidance. -A representative benchmark figure will go here (add something like `docs/src/images/benchmarks-overview.png` when ready). +### Benchmarks -Numbers, plots, and how to reproduce them live under [Benchmark Results](./benchmarks/results.md) and [How to Benchmark](./benchmarks/howto.md). +Results and reproduction instructions live under [Benchmark Results](https://julialegate.github.io/cuNumeric.jl/dev/benchmarks/results) and [How to Benchmark](https://julialegate.github.io/cuNumeric.jl/dev/benchmarks/howto). ### Try an example ```julia using cuNumeric -integrand = (x) -> @. exp(-x^2) +integrand(x) = @. exp(-x^2) + +@accelerate function monte_carlo(N, x_max) + Ω = 2 * x_max + raw_samples = cuNumeric.rand(N) + samples = @. Ω * raw_samples - x_max + return (Ω / N) * sum(integrand(samples)) +end N = 1_000_000 x_max = 10.0f0 -Ω = 2 * x_max - -samples = Ω .* cuNumeric.rand(N) -samples = samples .- x_max -estimate = (Ω / N) .* sum(integrand(samples)) +estimate = monte_carlo(N, x_max) println("Monte-Carlo Estimate: $(estimate)") ``` @@ -131,11 +92,3 @@ More worked examples (initialization, Gray-Scott, …) are in the documentation ### Known Limitations - There is no support for `Float16` or `ComplexF16` -- Arrays with 4 or more dimensions might have worse performance -- Maximum array dimension is 6 - - -### Known Deviations from Base Julia -- Reductions return 0D stores instead of scalars -- Slices return views -- `inv` does not throw `SingularException` for singular matrices diff --git a/benchmark/README.md b/benchmark/README.md index 205cd08ca..eb3b346eb 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -5,7 +5,7 @@ Benchmarks are declared in `benchmarks.toml`. `run.jl` parses it. ## Running ```bash -julia --project run.jl # runs whatever benchmarks.toml configures +julia --project=. run.jl # runs whatever benchmarks.toml configures ``` `run.jl` runs each (benchmark, backend) pair in its own process via @@ -35,7 +35,7 @@ n_warmup = 5 n_iter = 1000 n_trial = 5 -[[gemm]] # name registered in src/benchmarks.jl +[[gemm]] # name registered under src/benchmarks/ T = "Float32" # element type gpus = 1 cpus = 2 @@ -59,6 +59,10 @@ two axes: default `true`); it only affects cuNumeric, so comparison backends run once, not per variant. +Benchmark names are defined by the registered benchmark implementations. A +benchmark may expose baseline, optimized, backend-specific, or other variants; +the harness treats each name uniformly and records each result independently. + Each zipped field must be one of: - a scalar or single-element list (`cpus = 2` or `[2]`) -> broadcast to every config @@ -82,3 +86,14 @@ M = [150, 300, 600] # When `T = ["Float32", "Float64"]` and a length-2 `N`/`M` sweep you get all **4** combinations, not a paired `Float32 -> N[1], Float64 -> N[2]`. To pin a type to a specific size, use separate `[[name]]` blocks. + +## Plotting + +```bash +julia --project=benchmark benchmark/plot_results.jl +``` + +The plotter reads the result files in the selected results directory and writes +one weak-scaling figure per benchmark, plus aggregate fusion and no-fusion +figures when those result groups are present. Outputs are grouped under a +subdirectory named for the shared benchmark prefix. diff --git a/benchmark/__plot_results.jl b/benchmark/__plot_results.jl deleted file mode 100644 index fd1ed646a..000000000 --- a/benchmark/__plot_results.jl +++ /dev/null @@ -1,301 +0,0 @@ -#!/usr/bin/env julia -# Weak-scaling plots (1/2/4/8 GPUs) for the benchmark result CSVs. -# One figure per benchmark, three panels: throughput, time/step, parallel efficiency. -# -# CSV schema (see src/core.jl save_result): -# implementation,gpus,N,M,trial,time_ms,throughput,correctness -# `throughput` is the benchmark's `total_flops` divided by elapsed time. For -# Gray-Scott that unit is Gpoint-updates/s; other benchmarks report GFLOP/s. -# -# save_result appends and does NOT encode the code-path variant, so a cuNumeric CSV -# holds alternating runs: baseline, @accelerate, baseline, @accelerate, ... -# (a run boundary = the GPU count resetting downward). -# cuPyNumeric / CUDA.jl have no accelerated path -> a single block. -# -# Encoding: color = implementation; line style = code path -# solid = baseline, dashed = @accelerate. - -using Plots -using Statistics - -gr() - -function parse_args(args) - results_dir = "results" - out_dir = nothing - single_cunumeric_run = :baseline - hide_baseline = false - output_suffix = "" - - for arg in args - if startswith(arg, "--single-cu=") - value = Symbol(lowercase(last(split(arg, "="; limit=2)))) - value in (:baseline, :accelerated) || - error("--single-cu must be baseline or accelerated") - single_cunumeric_run = value - elseif arg == "--hide-baseline" - hide_baseline = true - elseif startswith(arg, "--out=") - out_dir = last(split(arg, "="; limit=2)) - elseif startswith(arg, "--suffix=") - output_suffix = last(split(arg, "="; limit=2)) - else - results_dir = arg - end - end - isempty(output_suffix) && hide_baseline && (output_suffix = "_no_baseline") - - results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) - if out_dir === nothing - out_dir = if basename(normpath(results_dir)) == "results" - joinpath(@__DIR__, "plots") - else - joinpath(@__DIR__, "plots", basename(normpath(results_dir))) - end - else - out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) - end - return (; results_dir, out_dir, single_cunumeric_run, hide_baseline, output_suffix) -end - -const CONFIG = parse_args(ARGS) -const RESULTS_DIR = CONFIG.results_dir -const OUT_DIR = CONFIG.out_dir -const SINGLE_CUNUMERIC_RUN = CONFIG.single_cunumeric_run -const HIDE_BASELINE = CONFIG.hide_baseline -const OUTPUT_SUFFIX = CONFIG.output_suffix - -# filekey, family label, color, marker, can_contain_accelerated_blocks -const FAMILIES = [ - ("cunumeric", "cuNumeric.jl (fused)", "#2a78d6", :circle, true), - ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond, true), - ("cupynumeric", "cuPyNumeric", "#eb6834", :rect, false), - ("CUDA.jl", "CUDA.jl", "#008300", :utriangle, false), -] - -const INK = "#0b0b0b" -const MUTED = "#898781" -const GRIDCOL = "#e1e0d9" -const IDEALCOL = "#c3c2b7" - -struct Row - gpus::Int - time_ms::Float64 - thr::Float64 -end - -# Parse a CSV into runs, split wherever the GPU count resets to a smaller value. -function load_runs(path) - rows = Row[] - for line in eachline(path) - isempty(strip(line)) && continue - f = split(line, ',') - push!(rows, Row(parse(Int, f[2]), parse(Float64, f[6]), parse(Float64, f[7]))) - end - isempty(rows) && return Vector{Row}[] - runs = [Row[]] - for (i, r) in enumerate(rows) - i > 1 && r.gpus < rows[i - 1].gpus && push!(runs, Row[]) - push!(runs[end], r) - end - return runs -end - -# Aggregate trials per GPU count -> sorted vector of (gpus, t, tsd, h, hsd). -function aggregate(rows) - by = Dict{Int,Vector{Row}}() - for r in rows - push!(get!(by, r.gpus, Row[]), r) - end - sd(x) = length(x) > 1 ? std(x) : 0.0 - return [ - (gpus=g, t=mean(getfield.(by[g], :time_ms)), tsd=sd(getfield.(by[g], :time_ms)), - h=mean(getfield.(by[g], :thr)), hsd=sd(getfield.(by[g], :thr))) - for g in sort(collect(keys(by))) - ] -end - -# Build the series (color+marker+linestyle+agg) present for one benchmark. -function series_for(bench) - series = [] # NamedTuple(label,color,marker,ls,agg) - for (key, fam, color, marker, splits) in FAMILIES - path = joinpath(RESULTS_DIR, "$(bench)_$(key).csv") - isfile(path) || continue - runs = load_runs(path) - isempty(runs) && continue - if splits && bench == "grayscott" - if length(runs) == 1 - label = - SINGLE_CUNUMERIC_RUN === :accelerated ? "$fam · accelerated" : - "$fam · baseline" - ls = SINGLE_CUNUMERIC_RUN === :accelerated ? :dash : :solid - push!( - series, (label=label, color=color, marker=marker, ls=ls, - agg=aggregate(runs[1])) - ) - else - # Repeated harness runs append alternating baseline/accelerated blocks. - baseline_rows = reduce(vcat, runs[1:2:end]) - accelerated_rows = reduce(vcat, runs[2:2:end]) - push!( - series, - (label="$fam · baseline", color=color, marker=marker, - ls=:solid, agg=aggregate(baseline_rows)), - ) - push!(series, - (label="$fam · accelerated", color=color, marker=marker, - ls=:dash, agg=aggregate(accelerated_rows))) - end - else - push!( - series, - (label=fam, color=color, marker=marker, ls=:solid, - agg=aggregate(reduce(vcat, runs))), - ) - end - end - return series -end - -function throughput_label(bench) - return bench == "grayscott" ? - "Throughput (Gpoint-updates/s)" : "Throughput (GFLOP/s)" -end - -function addline!(p, s, y; kw...) - return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, - lw=2.2, ls=s.ls, marker=s.marker, ms=6, msc=s.color, markerstrokewidth=0.8, - label=s.label, kw...) -end - -# One legend key: a short line sample (+ optional marker) with a text label. -function swatch!(p, x, y, color, ls, marker, label) - plot!(p, [x, x + 0.032], [y, y]; color=color, lw=2.6, ls=ls, label="") - marker !== nothing && scatter!(p, [x + 0.016], [y]; color=color, marker=marker, - ms=6, msc=color, markerstrokewidth=0.8, label="") - return annotate!(p, x + 0.045, y, text(label, 9, INK, :left)) -end - -# Grouped legend: color/marker = implementation, line style = code path. -function build_legend(series) - pl = plot(; framestyle=:none, legend=false, xlims=(0, 1), ylims=(0, 1), - grid=false, ticks=false) - # implementations present, in FAMILIES order, matched by color - present = [ - (fam, color, marker) for (key, fam, color, marker, _) in FAMILIES - if any(s.color == color for s in series) - ] - annotate!(pl, 0.015, 0.74, text("Implementation", 10, INK, :left)) - xs = range(0.18, 0.80; length=max(length(present), 1)) - for ((fam, color, marker), x) in zip(present, xs) - swatch!(pl, x, 0.74, color, :solid, marker, fam) - end - annotate!(pl, 0.015, 0.26, text("Line style", 10, INK, :left)) - has_baseline = any(endswith(s.label, "· baseline") for s in series) - has_accelerated = any(endswith(s.label, "· accelerated") for s in series) - if has_baseline && has_accelerated - swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") - swatch!(pl, 0.40, 0.26, MUTED, :dash, nothing, "@accelerate") - swatch!(pl, 0.70, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - elseif has_accelerated - swatch!(pl, 0.18, 0.26, MUTED, :dash, nothing, "@accelerate") - swatch!(pl, 0.52, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - elseif has_baseline - swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") - swatch!(pl, 0.48, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - else - swatch!(pl, 0.18, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - end - return pl -end - -function positive_ylim(vals; pad=0.12) - isempty(vals) && return (0, 1) - hi = maximum(vals) - hi > 0 || return (0, 1) - return (0, hi * (1 + pad)) -end - -function main() - mkpath(OUT_DIR) - files = filter(f -> endswith(f, ".csv"), readdir(RESULTS_DIR)) - benches = unique( - String[ - m.captures[1] for f in files for (key, _, _, _, _) in FAMILIES - for m in (match(Regex("^(.*)_" * replace(key, "." => "\\.") * "\\.csv\$"), f),) - if m !== nothing - ], - ) - - for bench in benches - series = series_for(bench) - if HIDE_BASELINE - series = filter(s -> !endswith(s.label, "· baseline"), series) - end - isempty(series) && continue - - common = (xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", - framestyle=:box, grid=true, gridcolor=GRIDCOL, gridalpha=1.0, - foreground_color_text=INK, tickfontcolor=MUTED, legend=false, - xlims=(0.85, 9.4)) - - # Panel 1: throughput (higher better) - throughput = [x.h for s in series for x in s.agg] - p1 = plot(; ylabel=throughput_label(bench), title="Throughput", - ylims=positive_ylim(throughput), common...) - for s in series - addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) - end - - # Panel 2: time per step (lower better; ideal = flat) - p2 = plot(; ylabel="Time / step (ms)", title="Time per step", common...) - for s in series - addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) - end - - # Panel 3: parallel efficiency = thr(p)/(p*thr(1)); ideal = 1.0 - efficiencies = Float64[] - for s in series - i1 = findfirst(x -> x.gpus == 1, s.agg) - i1 === nothing && continue - base = s.agg[i1].h - append!(efficiencies, [x.h/(x.gpus*base) for x in s.agg]) - end - p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", - ylims=positive_ylim(vcat(efficiencies, [1.0])), common...) - hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.4, label="") - for s in series - i1 = findfirst(x -> x.gpus == 1, s.agg) - i1 === nothing && continue - base = s.agg[i1].h - addline!(p3, s, [x.h/(x.gpus*base) for x in s.agg]) - end - - # grouped legend panel: color/marker = implementation, style = code path - pl = build_legend(series) - has_baseline = any(endswith(s.label, "· baseline") for s in series) - has_accelerated = any(endswith(s.label, "· accelerated") for s in series) - style_title = if has_baseline && has_accelerated - "solid = baseline · dashed = @accelerate" - elseif has_accelerated - "dashed = @accelerate" - elseif has_baseline - "solid = baseline" - else - "implementation comparison" - end - - fig = plot(p1, p2, p3, pl; layout=@layout([grid(1, 3); leg{0.16h}]), - size=(1400, 600), dpi=200, - plot_title=titlecase(bench) * " — weak scaling ($style_title)", - plot_titlefontsize=12, left_margin=6Plots.mm, - bottom_margin=6Plots.mm, top_margin=4Plots.mm, - background_color="#fcfcfb") - - out = joinpath(OUT_DIR, "$(bench)_weak_scaling$(OUTPUT_SUFFIX).png") - savefig(fig, out) - println("wrote $out") - end -end - -main() diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml index 1d8b3194f..747989673 100644 --- a/benchmark/benchmarks.toml +++ b/benchmark/benchmarks.toml @@ -36,16 +36,40 @@ T = "Float32" gpus = [1, 2, 4, 8] cpus = 16 fusion = [true, false] -N = [2000, 2832, 4000, 5656] -M = [2000, 2832, 4000, 5656] +N = [24000, 33944, 48000, 67888] +M = [24000, 33944, 48000, 67888] -[[grayscott_lifetimes]] +[[grayscott_function_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] cpus = 16 -fusion = false -N = [2000, 2832, 4000, 5656] -M = [2000, 2832, 4000, 5656] +fusion = [true, false] +N = [24000, 33944, 48000, 67888] +M = [24000, 33944, 48000, 67888] + +[[grayscott_begin_accelerated]] +T = "Float32" +gpus = [1, 2, 4, 8] +cpus = 16 +fusion = [true, false] +N = [24000, 33944, 48000, 67888] +M = [24000, 33944, 48000, 67888] + +[[grayscott_let_accelerated]] +T = "Float32" +gpus = [1, 2, 4, 8] +cpus = 16 +fusion = [true, false] +N = [24000, 33944, 48000, 67888] +M = [24000, 33944, 48000, 67888] + +[[grayscott_expression_accelerated]] +T = "Float32" +gpus = [1, 2, 4, 8] +cpus = 16 +fusion = [true, false] +N = [24000, 33944, 48000, 67888] +M = [24000, 33944, 48000, 67888] ################################# # DMD # @@ -72,11 +96,11 @@ fusion = [true, false] N = [50000, 100000, 200000, 400000] M = 512 -[[dmd_lifetimes]] +[[dmd_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] cpus = 16 -fusion = false +fusion = [true, false] N = [50000, 100000, 200000, 400000] M = 512 diff --git a/benchmark/debug/grayscott_accelerate.jl b/benchmark/debug/grayscott_accelerate.jl new file mode 100644 index 000000000..53a276bb0 --- /dev/null +++ b/benchmark/debug/grayscott_accelerate.jl @@ -0,0 +1,66 @@ +#!/usr/bin/env julia + +# Print @accelerate lifetime rewrites and the fused kernels launched by each +# Gray–Scott macro form. Run with a small grid, for example: +# julia --project=benchmark benchmark/debug/grayscott_accelerate.jl 64 + +using cuNumeric + +const BENCHMARK_SRC = joinpath(@__DIR__, "..", "src") +include(joinpath(BENCHMARK_SRC, "core.jl")) +include(joinpath(BENCHMARK_SRC, "benchmarks", "grayscott.jl")) +include(joinpath(BENCHMARK_SRC, "benchmarks", "grayscott_accelerate_forms.jl")) + +const N = length(ARGS) >= 1 ? parse(Int, ARGS[1]) : 64 +N >= 4 || error("N must be at least 4") + +const FORMS = ( + (:function, "function", GrayScottFunctionAccelerated), + (:begin, "begin", GrayScottBeginAccelerated), + (:let, "let", GrayScottLetAccelerated), + (:expression, "expression", GrayScottExpressionAccelerated), +) + +function lifetime_expansion(kind) + body = deepcopy(GRAYSCOTT_STEP_BODY) + input = if kind === :function + Expr(:function, Expr(:call, :debug_step, :u, :v, :u_new, :v_new, :args), body) + elseif kind === :let + Expr(:let, body) + else + body + end + return cuNumeric._accelerate_expand(input, @__MODULE__) +end + +function print_lifetimes(kind, label) + println("\n", "="^80, "\n", uppercase(label), " — lifetime analysis\n", "="^80) + if kind === :expression + # Expression form accelerates each assignment independently. + for statement in GRAYSCOTT_STEP_BODY.args + statement isa LineNumberNode && continue + statement isa Expr && statement.head === :(=) || continue + lhs, rhs = statement.args + println("\nRHS: ", lhs) + expansion = cuNumeric._accelerate_expand(rhs, @__MODULE__) + cuNumeric.print_lifetime_analysis(expansion) + end + else + cuNumeric.print_lifetime_analysis(lifetime_expansion(kind)) + end +end + +function run_form(label, type) + println("\n", "="^80, "\n", uppercase(label), " — runtime kernels\n", "="^80) + b = build_benchmark(type, Float32, N, N) + state = only(initialize(b; deterministic=true)) + return run!(b, state) +end + +println("Gray–Scott @accelerate debug; grid=$(N)x$(N)") +cuNumeric.BCAST_FUSION_DEBUG[] = true +println("BCAST_FUSION_DEBUG = ", cuNumeric.BCAST_FUSION_DEBUG[]) +for (kind, label, type) in FORMS + print_lifetimes(kind, label) + run_form(label, type) +end diff --git a/benchmark/plot_results.jl b/benchmark/plot_results.jl new file mode 100644 index 000000000..b53ac3618 --- /dev/null +++ b/benchmark/plot_results.jl @@ -0,0 +1,284 @@ +#!/usr/bin/env julia +# Generate weak-scaling plots from benchmark CSVs. +# Each figure shows throughput, time per step, and parallel efficiency. + +using Plots +using Statistics + +gr() + +function parse_args(args) + results_dir = "results" + out_dir = nothing + output_suffix = "" + + for arg in args + if startswith(arg, "--out=") + out_dir = last(split(arg, "="; limit=2)) + elseif startswith(arg, "--suffix=") + output_suffix = last(split(arg, "="; limit=2)) + else + results_dir = arg + end + end + results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) + if out_dir === nothing + out_dir = if basename(normpath(results_dir)) == "results" + joinpath(@__DIR__, "plots") + else + joinpath(@__DIR__, "plots", basename(normpath(results_dir))) + end + else + out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) + end + return (; results_dir, out_dir, output_suffix) +end + +const CONFIG = parse_args(ARGS) +const RESULTS_DIR = CONFIG.results_dir +const OUT_DIR = CONFIG.out_dir +const OUTPUT_SUFFIX = CONFIG.output_suffix + +# file key, display label, color, marker +const FAMILIES = [ + ("cunumeric", "cuNumeric.jl (fused)", "#2a78d6", :circle), + ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond), + ("cupynumeric", "cuPyNumeric", "#eb6834", :rect), + ("CUDA.jl", "CUDA.jl", "#008300", :utriangle), +] + +const INK = "#0b0b0b" +const MUTED = "#898781" +const GRIDCOL = "#e1e0d9" +const IDEALCOL = "#c3c2b7" +const GROUP_COLORS = ["#2a78d6", "#eb6834", "#008300", "#8b3fb0", "#c47f00", "#159a9c"] +const GROUP_MARKERS = [:circle, :diamond, :utriangle, :rect, :star5, :hexagon] + +struct Row + gpus::Int + time_ms::Float64 + thr::Float64 +end + +# Parse a CSV into runs, split wherever the GPU count resets to a smaller value. +function load_runs(path) + rows = Row[] + for line in eachline(path) + isempty(strip(line)) && continue + f = split(line, ',') + push!(rows, Row(parse(Int, f[2]), parse(Float64, f[6]), parse(Float64, f[7]))) + end + isempty(rows) && return Vector{Row}[] + runs = [Row[]] + for (i, r) in enumerate(rows) + i > 1 && r.gpus < rows[i - 1].gpus && push!(runs, Row[]) + push!(runs[end], r) + end + return runs +end + +# Aggregate trials per GPU count -> sorted vector of (gpus, t, tsd, h, hsd). +function aggregate(rows) + by = Dict{Int,Vector{Row}}() + for r in rows + push!(get!(by, r.gpus, Row[]), r) + end + sd(x) = length(x) > 1 ? std(x) : 0.0 + return [ + (gpus=g, t=mean(getfield.(by[g], :time_ms)), tsd=sd(getfield.(by[g], :time_ms)), + h=mean(getfield.(by[g], :thr)), hsd=sd(getfield.(by[g], :thr))) + for g in sort(collect(keys(by))) + ] +end + +_all_rows(runs) = reduce(vcat, runs; init=Row[]) + +function make_series(label, color, marker, ls, runs) + isempty(runs) && return nothing + return (label=label, color=color, marker=marker, ls=ls, + agg=aggregate(_all_rows(runs))) +end + +# Build the available implementation series for one benchmark. +function load_series(bench, specs; filename=(b, k) -> "$(b)_$(k).csv") + series = [] + for (key, label, color, marker) in specs + path = joinpath(RESULTS_DIR, filename(bench, key)) + isfile(path) || continue + runs = load_runs(path) + isempty(runs) && continue + + push!(series, make_series(label, color, marker, :solid, runs)) + end + return filter(!isnothing, series) +end + +series_for(bench) = load_series(bench, FAMILIES) + +function common_prefix(names) + words = split.(names, '_') + n = minimum(length, words) + i = 0 + while i < n && all(w -> w[i + 1] == words[1][i + 1], words) + i += 1 + end + return i == 0 ? String[] : words[1][1:i] +end + +function family_stem(name, prefix) + words = split(name, '_') + length(words) > length(prefix) || return name + return join(words[(length(prefix) + 1):end], "_") +end + +function grouped_series(benches, key) + series = [] + prefix = common_prefix(benches) + for (i, bench) in enumerate(benches) + path = joinpath(RESULTS_DIR, "$(bench)_$(key).csv") + isfile(path) || continue + runs = load_runs(path) + isempty(runs) && continue + color = GROUP_COLORS[mod1(i, length(GROUP_COLORS))] + marker = GROUP_MARKERS[mod1(i, length(GROUP_MARKERS))] + words = split(bench, '_') + length(words) > length(prefix) && (words = words[(length(prefix) + 1):end]) + label = titlecase(join(words, ' ')) + push!(series, make_series(label, color, marker, :solid, runs)) + end + return filter(!isnothing, series) +end + +function throughput_label() + return "Throughput" +end + +function addline!(p, s, y; kw...) + return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, + lw=2.2, ls=s.ls, marker=s.marker, ms=6, msc=s.color, markerstrokewidth=0.8, + label=s.label, kw...) +end + +# One legend key: a short line sample (+ optional marker) with a text label. +function swatch!(p, x, y, color, ls, marker, label) + plot!(p, [x, x + 0.032], [y, y]; color=color, lw=2.6, ls=ls, label="") + marker !== nothing && scatter!(p, [x + 0.016], [y]; color=color, marker=marker, + ms=6, msc=color, markerstrokewidth=0.8, label="") + return annotate!(p, x + 0.045, y, text(label, 9, INK, :left)) +end + +# Build a legend from the series currently being plotted. +function build_legend(series) + pl = plot(; framestyle=:none, legend=false, xlims=(0, 1), ylims=(0, 1), + grid=false, ticks=false) + present = [(s.label, s.color, s.marker) for s in series] + annotate!(pl, 0.015, 0.74, text("Series", 10, INK, :left)) + xs = range(0.18, 0.80; length=max(length(present), 1)) + for ((fam, color, marker), x) in zip(present, xs) + swatch!(pl, x, 0.74, color, :solid, marker, fam) + end + swatch!(pl, 0.18, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + return pl +end + +function positive_ylim(vals; pad=0.12) + isempty(vals) && return (0, 1) + hi = maximum(vals) + hi > 0 || return (0, 1) + return (0, hi * (1 + pad)) +end + +function weak_scaling_figure(bench, series, legend_panel; plot_title) + common = (xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", + framestyle=:box, grid=true, gridcolor=GRIDCOL, gridalpha=1.0, + foreground_color_text=INK, tickfontcolor=MUTED, legend=false, + xlims=(0.85, 9.4)) + + # Panel 1: throughput (higher better) + throughput = [x.h for s in series for x in s.agg] + p1 = plot(; ylabel=throughput_label(), title="Throughput", + ylims=positive_ylim(throughput), common...) + for s in series + addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) + end + + # Panel 2: time per step (lower better; ideal = flat) + p2 = plot(; ylabel="Time / step (ms)", title="Time per step", common...) + for s in series + addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) + end + + # Panel 3: parallel efficiency = thr(p)/(p*thr(1)); ideal = 1.0 + efficiencies = Float64[] + for s in series + i1 = findfirst(x -> x.gpus == 1, s.agg) + i1 === nothing && continue + base = s.agg[i1].h + append!(efficiencies, [x.h/(x.gpus*base) for x in s.agg]) + end + p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", + ylims=positive_ylim(vcat(efficiencies, [1.0])), common...) + hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.4, label="") + for s in series + i1 = findfirst(x -> x.gpus == 1, s.agg) + i1 === nothing && continue + base = s.agg[i1].h + addline!(p3, s, [x.h/(x.gpus*base) for x in s.agg]) + end + + return plot( + p1, p2, p3, legend_panel; + layout=@layout([grid(1, 3); leg{0.16h}]), + size=(1400, 600), dpi=200, plot_title, + plot_titlefontsize=12, left_margin=6Plots.mm, + bottom_margin=6Plots.mm, top_margin=4Plots.mm, + background_color="#fcfcfb", + ) +end + +function main() + mkpath(OUT_DIR) + files = filter(f -> endswith(f, ".csv"), readdir(RESULTS_DIR)) + benches = unique( + String[ + m.captures[1] for f in files for (key, _, _, _) in FAMILIES + for m in (match(Regex("^(.*)_" * replace(key, "." => "\\.") * "\\.csv\$"), f),) + if m !== nothing + ], + ) + family = common_prefix(benches) + benchmark_out = joinpath(OUT_DIR, isempty(family) ? "benchmarks" : join(family, "_")) + mkpath(benchmark_out) + + for bench in benches + series = series_for(bench) + isempty(series) && continue + + fig = weak_scaling_figure( + bench, series, build_legend(series); + plot_title=titlecase(bench) * " — weak scaling", + ) + + stem = family_stem(bench, family) + out = joinpath(benchmark_out, "$(stem)_weak_scaling$(OUTPUT_SUFFIX).png") + savefig(fig, out) + println("wrote $out") + end + + # Add aggregate views that compare all benchmark variants for each mode. + for (key, title, stem) in (("cunumeric", "Fusion enabled", "fusion"), + ("cunumeric_nofusion", "Fusion disabled", "no_fusion")) + series = grouped_series(benches, key) + isempty(series) && continue + fig = weak_scaling_figure( + stem, series, build_legend(series); + plot_title=title * " — weak scaling", + ) + out = joinpath(benchmark_out, "$(stem)_weak_scaling$(OUTPUT_SUFFIX).png") + savefig(fig, out) + println("wrote $out") + end + return nothing +end + +main() diff --git a/benchmark/run.jl b/benchmark/run.jl index cb748ccde..36c4c2661 100644 --- a/benchmark/run.jl +++ b/benchmark/run.jl @@ -23,8 +23,8 @@ const POSARGS = filter(a -> a ∉ VERBOSE_FLAGS, ARGS) banner(msg) = println("\n", "="^128, "\n", msg, "\n", "="^128) -# `_lifetimes` is a cuNumeric-only code-path variant (@analyze_lifetimes) -cunumeric_only(name) = endswith(name, "_lifetimes") +# `_accelerated` is a cuNumeric-only code-path variant (`@accelerate`). +cunumeric_only(name) = endswith(name, "_accelerated") const LAST_FUSION_TOGGLE = Ref{Union{Nothing,Bool}}(nothing) diff --git a/benchmark/src/benchmarks/dmd.jl b/benchmark/src/benchmarks/dmd.jl index 8099f34c5..7f5b41fea 100644 --- a/benchmark/src/benchmarks/dmd.jl +++ b/benchmark/src/benchmarks/dmd.jl @@ -13,13 +13,13 @@ Base.@kwdef struct DMDBaseline{T} <: AbstractDMD{T} M::Int end -Base.@kwdef struct DMDLifetimes{T} <: AbstractDMD{T} +Base.@kwdef struct DMDAccelerated{T} <: AbstractDMD{T} N::Int M::Int end name(::DMDBaseline) = "dmd_baseline" -name(::DMDLifetimes) = "dmd_lifetimes" +name(::DMDAccelerated) = "dmd_accelerated" dims(b::AbstractDMD) = (b.N, b.M) data(b::AbstractDMD{T}) where {T} = "DMD with T=$(T), N=$(b.N), M=$(b.M)" allowed_types(::Type{<:AbstractDMD}) = cuNumeric.SUPPORTED_FLOAT_TYPES @@ -65,8 +65,8 @@ _dmd_row(v) = v isa NDArray ? cuNumeric.reshape(v, (1, length(v))) : reshape(v, # svd / eigen return factorizations whose stores the lifetime rewriter cannot # see, so those stay outside the macro. The GEMM lift is wrapped. # -# Do not form Diagonal(1 ./ S) inside @analyze_lifetimes: the rewriter treats -# `1 ./ S` as a last-used temp and destroy!s it, while Diagonal still holds +# Do not form Diagonal(1 ./ S) inside @accelerate: the rewriter treats +# `1 ./ S` as a last-used temporary and frees it while Diagonal still holds # that same vector. Scale columns with a broadcast instead (same math). function _dmd_factors(X, r) n = size(X, 2) @@ -84,7 +84,11 @@ let body = quote (B, Ã) end @eval _dmd_project(::DMDBaseline, X, X2, U, Vt, S) = $body - @eval _dmd_project(::DMDLifetimes, X, X2, U, Vt, S) = @analyze_lifetimes $body + @eval @accelerate function _dmd_project( + ::DMDAccelerated, X, X2, U, Vt, S + ) + $body + end end function _dmd_compute!(b::AbstractDMD, X, r) @@ -108,9 +112,9 @@ function check_benchmark_correctness( Xh = rand(T, b.N, b.M) X = NDArray(Xh) r = _dmd_rank(b) - # Values, not lifetimes: compare against the baseline body on host and device. + # Values, not lifetimes: compare the selected device path against the host baseline. ref = DMDBaseline{T}(; N=b.N, M=b.M) - λ, _ = _dmd_compute!(ref, X, r) + λ, _ = _dmd_compute!(b, X, r) λh, _ = _dmd_compute!(ref, Xh, r) mag = sort(abs.(Array(λ)); rev=true) @@ -119,4 +123,4 @@ function check_benchmark_correctness( end register_benchmark("dmd_baseline", DMDBaseline) -register_benchmark("dmd_lifetimes", DMDLifetimes) +register_benchmark("dmd_accelerated", DMDAccelerated) diff --git a/benchmark/src/benchmarks/grayscott.jl b/benchmark/src/benchmarks/grayscott.jl index 3ba6e6398..2fee6146e 100644 --- a/benchmark/src/benchmarks/grayscott.jl +++ b/benchmark/src/benchmarks/grayscott.jl @@ -18,7 +18,7 @@ Base.@kwdef struct GrayScottBaseline{T} <: AbstractGrayScott{T} M::Int end -Base.@kwdef struct GrayScottLifetimes{T} <: AbstractGrayScott{T} +Base.@kwdef struct GrayScottAccelerated{T} <: AbstractGrayScott{T} N::Int M::Int end @@ -92,66 +92,69 @@ function check_benchmark_correctness( return (u_ok && v_ok) ? "pass" : "fail" end -# VARIANT DESCRIPTION -# baseline: as written -# lifetimes: step wrapped in @analyze_lifetimes -let body = quote - # currently we don't have NDArray^x working yet. every operator is dotted - # so each rhs fuses into a single broadcast kernel rather than shattering - # into bare +/-/* binary tasks. - F_u = ( - ( - .-u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) .+ args.f .* (1.0f0 .- u[2:(end - 1), 2:(end - 1)]) - ) - F_v = ( - ( - u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) .- (args.f + args.k) .* v[2:(end - 1), 2:(end - 1)] - ) - # 2-D Laplacian via slicing, excluding boundaries - u_lap = ( - ( - u[3:end, 2:(end - 1)] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ - u[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 .+ - ( - u[2:(end - 1), 3:end] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ - u[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - v_lap = ( - ( - v[3:end, 2:(end - 1)] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ - v[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 .+ - ( - v[2:(end - 1), 3:end] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ - v[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - - # Forward-Euler step for all interior points - u_new[2:(end - 1), 2:(end - 1)] = - ((args.c_u .* u_lap) .+ F_u) .* args.dt .+ u[2:(end - 1), 2:(end - 1)] - v_new[2:(end - 1), 2:(end - 1)] = - ((args.c_v .* v_lap) .+ F_v) .* args.dt .+ v[2:(end - 1), 2:(end - 1)] - - # Periodic boundary conditions - u_new[:, 1] = u[:, end - 1] - u_new[:, end] = u[:, 2] - u_new[1, :] = u[end - 1, :] - u_new[end, :] = u[2, :] - v_new[:, 1] = v[:, end - 1] - v_new[:, end] = v[:, 2] - v_new[1, :] = v[end - 1, :] - v_new[end, :] = v[2, :] - end +# Shared syntax tree keeps every Gray-Scott variant on the exact same workload. +const GRAYSCOTT_STEP_BODY = quote + # currently we don't have NDArray^x working yet. every operator is dotted + # so each rhs fuses into a single broadcast kernel rather than shattering + # into bare +/-/* binary tasks. + F_u = ( + ( + .-u[2:(end - 1), 2:(end - 1)] .* + (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) + ) .+ args.f .* (1.0f0 .- u[2:(end - 1), 2:(end - 1)]) + ) + F_v = ( + ( + u[2:(end - 1), 2:(end - 1)] .* + (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) + ) .- (args.f + args.k) .* v[2:(end - 1), 2:(end - 1)] + ) + # 2-D Laplacian via slicing, excluding boundaries + u_lap = ( + ( + u[3:end, 2:(end - 1)] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ + u[1:(end - 2), 2:(end - 1)] + ) ./ args.dx^2 .+ + ( + u[2:(end - 1), 3:end] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ + u[2:(end - 1), 1:(end - 2)] + ) ./ args.dx^2 + ) + v_lap = ( + ( + v[3:end, 2:(end - 1)] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ + v[1:(end - 2), 2:(end - 1)] + ) ./ args.dx^2 .+ + ( + v[2:(end - 1), 3:end] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ + v[2:(end - 1), 1:(end - 2)] + ) ./ args.dx^2 + ) + + # Forward-Euler step for all interior points + u_new[2:(end - 1), 2:(end - 1)] = + ((args.c_u .* u_lap) .+ F_u) .* args.dt .+ u[2:(end - 1), 2:(end - 1)] + v_new[2:(end - 1), 2:(end - 1)] = + ((args.c_v .* v_lap) .+ F_v) .* args.dt .+ v[2:(end - 1), 2:(end - 1)] + + # Periodic boundary conditions + u_new[:, 1] = u[:, end - 1] + u_new[:, end] = u[:, 2] + u_new[1, :] = u[end - 1, :] + u_new[end, :] = u[2, :] + v_new[:, 1] = v[:, end - 1] + v_new[:, end] = v[:, 2] + v_new[1, :] = v[end - 1, :] + v_new[end, :] = v[2, :] +end + +# Original baseline and recommended function-form benchmark. +let body = deepcopy(GRAYSCOTT_STEP_BODY) @eval _gs_step!(b::GrayScottBaseline, u, v, u_new, v_new, args::GSParams) = $body - @eval _gs_step!(b::GrayScottLifetimes, u, v, u_new, v_new, args::GSParams) = - @analyze_lifetimes $body + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottAccelerated, u, v, u_new, v_new, args::GSParams)), body + ) + @eval $definition end function run!(b::AbstractGrayScott, st::GrayScottState) @@ -163,4 +166,4 @@ function run!(b::AbstractGrayScott, st::GrayScottState) end register_benchmark("grayscott_baseline", GrayScottBaseline) -register_benchmark("grayscott_lifetimes", GrayScottLifetimes) +register_benchmark("grayscott_accelerated", GrayScottAccelerated) diff --git a/benchmark/src/benchmarks/grayscott_accelerate_forms.jl b/benchmark/src/benchmarks/grayscott_accelerate_forms.jl new file mode 100644 index 000000000..187dc299f --- /dev/null +++ b/benchmark/src/benchmarks/grayscott_accelerate_forms.jl @@ -0,0 +1,95 @@ +# Compare the four scope contracts of `@accelerate` on one shared Gray-Scott step. +# Each type has a distinct result name so benchmark runs produce separate CSVs. + +abstract type AbstractGrayScottAccelerateForm{T} <: AbstractGrayScott{T} end + +Base.@kwdef struct GrayScottFunctionAccelerated{T} <: + AbstractGrayScottAccelerateForm{T} + N::Int + M::Int +end + +Base.@kwdef struct GrayScottBeginAccelerated{T} <: AbstractGrayScottAccelerateForm{T} + N::Int + M::Int +end + +Base.@kwdef struct GrayScottLetAccelerated{T} <: AbstractGrayScottAccelerateForm{T} + N::Int + M::Int +end + +Base.@kwdef struct GrayScottExpressionAccelerated{T} <: + AbstractGrayScottAccelerateForm{T} + N::Int + M::Int +end + +name(::GrayScottFunctionAccelerated) = "grayscott_function_accelerated" +name(::GrayScottBeginAccelerated) = "grayscott_begin_accelerated" +name(::GrayScottLetAccelerated) = "grayscott_let_accelerated" +name(::GrayScottExpressionAccelerated) = "grayscott_expression_accelerated" + +# Function form is the reusable default: arguments and the return value survive, +# while non-returned locals may fuse across statements or die after their last use. +let body = deepcopy(GRAYSCOTT_STEP_BODY) + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottFunctionAccelerated, u, v, u_new, v_new, args::GSParams)), + body, + :function, + ) + @eval $definition +end + +# `begin` adds no scope. Every named local remains visible, so it measures the +# multi-output/materialized path rather than eliminating named intermediates. +let body = deepcopy(GRAYSCOTT_STEP_BODY) + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottBeginAccelerated, u, v, u_new, v_new, args::GSParams)), + body, + :begin, + ) + @eval $definition +end + +# `let` is a hard one-off scope. Only its result escapes, allowing aggressive +# inter-statement fusion and last-use cleanup for all other local temporaries. +let body = deepcopy(GRAYSCOTT_STEP_BODY) + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottLetAccelerated, u, v, u_new, v_new, args::GSParams)), + body, + :let, + ) + @eval $definition +end + +# Expression form has no multi-statement scope. Accelerating each RHS preserves +# fusion inside that expression but deliberately materializes statement results, +# isolating intra-expression fusion from the inter-statement rewrite cases above. +function accelerate_grayscott_rhs(body::Expr) + statements = Any[] + for statement in body.args + if statement isa LineNumberNode + push!(statements, statement) + elseif statement isa Expr && statement.head === :(=) + lhs, rhs = statement.args + push!(statements, :($lhs = @accelerate $rhs)) + else + error("Gray-Scott expression benchmark expected assignments; got $(repr(statement))") + end + end + return Expr(:block, statements...) +end + +let body = accelerate_grayscott_rhs(deepcopy(GRAYSCOTT_STEP_BODY)) + @eval function _gs_step!( + b::GrayScottExpressionAccelerated, u, v, u_new, v_new, args::GSParams + ) + $body + end +end + +register_benchmark("grayscott_function_accelerated", GrayScottFunctionAccelerated) +register_benchmark("grayscott_begin_accelerated", GrayScottBeginAccelerated) +register_benchmark("grayscott_let_accelerated", GrayScottLetAccelerated) +register_benchmark("grayscott_expression_accelerated", GrayScottExpressionAccelerated) diff --git a/benchmark/src/core.jl b/benchmark/src/core.jl index 526ce4eb1..7a0aacbc4 100644 --- a/benchmark/src/core.jl +++ b/benchmark/src/core.jl @@ -39,6 +39,16 @@ function total_flops end function initialize end function run! end +# Internal adapter for benchmark generators that share a quoted step body. +function _define_accelerated_definition(signature, body, form=:function) + if form === :function + return cuNumeric._accelerate_expand(Expr(:function, signature, body), @__MODULE__) + end + scoped = form === :begin ? Expr(:block, body.args...) : Expr(:let, body) + call = Expr(:macrocall, Symbol("@accelerate"), LineNumberNode(0), scoped) + return Expr(:function, signature, Expr(:block, Base.macroexpand(@__MODULE__, call))) +end + # Maps a benchmarks.toml table name to its benchmark type. Each benchmark file # registers itself via `register_benchmark`. const BENCHMARKS = Dict{String,Type}() diff --git a/docs/make.jl b/docs/make.jl index 628a272bc..5ce5328be 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -42,7 +42,7 @@ makedocs(; ], "Performance Tips" => [ "Kernel Fusion" => "perf/kernel_fusion.md", - "Reduce Allocations" => "perf/reduce_allocations.md", + "The @accelerate Macro" => "perf/reduce_allocations.md", "Patterns to Avoid" => "perf/patterns_to_avoid.md", ], "Configuration" => [ diff --git a/docs/src/api.md b/docs/src/api.md index a187a79e1..d8197626d 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -1,9 +1,9 @@ # NDArray Reference -Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). +Indexing, reshaping, reductions, comparisons, memory helpers, acceleration macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). ```@autodocs Modules = [cuNumeric] -Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl"] +Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl", "scoping/accelerate.jl"] Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!, :randn, :randn!, :randexp, :randexp!, :default_rng, :random, :random!)) ``` diff --git a/docs/src/api_cuda.md b/docs/src/api_cuda.md index 6e912cd8a..eedb9806c 100644 --- a/docs/src/api_cuda.md +++ b/docs/src/api_cuda.md @@ -58,7 +58,7 @@ allowscalar() do end ``` -See `examples/custom_cuda.jl` for a more complete example with multiple kernels. +See `examples/custom_cuda.jl` for a runnable two-kernel example. ## API Reference diff --git a/docs/src/api_unary.md b/docs/src/api_unary.md index 3ea0442b3..7582f4ce1 100644 --- a/docs/src/api_unary.md +++ b/docs/src/api_unary.md @@ -1,7 +1,7 @@ # Unary Operations >[!NOTE] -> Prefer `@.` for multi-op elementwise expressions so every operator is dotted (especially unary negation). This ensures broadcast operations are fused. See [Kernel Fusion](./perf/kernel_fusion.md). +> Prefer `@.` for multi-op elementwise expressions so every operator is dotted (especially unary negation). This makes eligible CUDA broadcasts fusion-friendly. See [Kernel Fusion](./perf/kernel_fusion.md). The following unary operations are supported and can be broadcast over `NDArray`: diff --git a/docs/src/benchmarks/howto.md b/docs/src/benchmarks/howto.md index 851f3bdd3..efce38239 100644 --- a/docs/src/benchmarks/howto.md +++ b/docs/src/benchmarks/howto.md @@ -47,7 +47,10 @@ n_correctness_iter = 5 - `cupynumeric` / `cuda`: optional comparison backends - `check_correctness`: one CPU-reference check per config (not per timed iter), recorded in the CSV -Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, `dmd_baseline`, `dmd_lifetimes`, `grayscott_baseline`, `grayscott_lifetimes`, …). Names must match what `src/benchmarks/*.jl` registers. +Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, +`dmd_baseline`, `dmd_accelerated`, `grayscott_baseline`, +`grayscott_function_accelerated`, …). Names must match what +`src/benchmarks/*.jl` registers. DMD's `N` is the number of spatial degrees of freedom (rows of the snapshot matrix), not a grid side length. The SVD is of the tall-skinny `N × (M-1)` matrix `X1`. Thin SVD plus the rank-`r` lift is `Θ(N)` when `M` and `r` are fixed, so weak scaling is `N ∝ P` (same idea as Monte Carlo, not GEMM's `N ∝ P^{1/3}`). The flop count is in `src/benchmarks/dmd.jl`. @@ -78,7 +81,12 @@ M = [150, 300, 600] That is 2 types × 3 sweep points = **6 runs**. -`fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, default `true`). Comparison backends ignore fusion and run once (on the fused pass), not per variant. Names ending in `_lifetimes` are cuNumeric-only code-path variants. +`fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, +default `true`). Comparison backends ignore fusion and run once (on the fused +pass), not per variant. Entries ending in `_accelerated` are cuNumeric-only. +The Gray-Scott function, `begin`, `let`, and expression entries compare the +four `@accelerate` scope contracts on the same step; `dmd_accelerated` applies +the recommended function form to the DMD projection. Gotcha: when `T = ["Float32", "Float64"]` and a length-2 `N`/`M` sweep you get all **4** combinations, not a paired `Float32 -> N[1]`. To pin a type to a size, use separate `[[name]]` blocks. diff --git a/docs/src/benchmarks/results.md b/docs/src/benchmarks/results.md index fd62aedcf..08cc9e3c2 100644 --- a/docs/src/benchmarks/results.md +++ b/docs/src/benchmarks/results.md @@ -1,6 +1,6 @@ # Benchmark Results -For JuliaCon2025 we benchmarks cuNumeric.jl on 8 A100 GPUs (single-node) and compared it to the Python library cuPyNumeric and other relevant benchmarks depending on the problem. All results shown are weak scaling. We hope to have multi-node benchmarks soon! +These historical JuliaCon 2025 results compare cuNumeric.jl with cuPyNumeric and problem-specific alternatives on one node with eight A100 GPUs. All plots show weak scaling. See [How to Benchmark](./howto.md) for the current harness and its baseline, `@accelerate`, fused, and unfused variants. ## SGEMM @@ -24,7 +24,7 @@ mul!(C, A, B) ## Monte-Carlo Integration -Monte-Carlo integration is embaressingly parallel and should scale perfectly. We do not know the exact number of operations in `exp` so the GFLOPs is off by a constant factor. +Monte-Carlo integration is embarrassingly parallel. Because the exact operation count of `exp` is implementation-dependent, the plotted operation rate is scaled by an approximate constant. Code Outline: ```julia diff --git a/docs/src/configuration/hardware.md b/docs/src/configuration/hardware.md index 0102ee183..4ddacc4c2 100644 --- a/docs/src/configuration/hardware.md +++ b/docs/src/configuration/hardware.md @@ -1,6 +1,6 @@ # Hardware Configuration -There is no programmatic way to set the hardware configuration used by CuPyNumeric (as of 26.01). By default, the hardware configuration is set automatically by Legate. This configuration can be manipulated through the following environment variables: +Legate chooses the hardware configuration automatically by default. Set these environment variables before starting Julia to override it: - `LEGATE_SHOW_CONFIG` : When set to 1, the Legate config is printed to stdout - `LEGATE_AUTO_CONFIG`: When set to 1, Legate will automatically choose the hardware configuration diff --git a/docs/src/debugging.md b/docs/src/debugging.md index 6b25941d5..3e868c907 100644 --- a/docs/src/debugging.md +++ b/docs/src/debugging.md @@ -6,7 +6,7 @@ Debug the layer that matches the problem: |---|---| | Which operations did Legate submit, and when did they run? | [Legate logs and profiles](#trace-legate-runtime-work) | | How were broadcasts fused? | [`BCAST_FUSION_DEBUG`](#inspect-fused-broadcasts-with-bcast_fusion_debug) | -| Where does `@analyze_lifetimes` free temporaries? | [`@show_lifetimes`](#inspect-lifetime-rewrites-with-show_lifetimes) | +| How does `@accelerate` rewrite code and free temporaries? | [`@show_lifetimes`](#inspect-lifetime-rewrites-with-show_lifetimes) | ## Trace Legate runtime work @@ -78,7 +78,7 @@ cuNumeric already supplies names for individual operations when task-scope namin When broadcast fusion is on, set `cuNumeric.BCAST_FUSION_DEBUG[] = true` to print inter-statement rewrites and each fused kernel's expression tree, arguments, and launch geometry. Inter-statement rewrites are reported when -`@analyze_lifetimes` expands, so enable the flag before defining or evaluating +`@accelerate` expands, so enable the flag before defining or evaluating the expression you want to inspect. Kernel details are reported at runtime. ```julia @@ -91,15 +91,16 @@ A = cuNumeric.ones(Float32, N, N) B = cuNumeric.ones(Float32, N, N) C = cuNumeric.zeros(Float32, N, N) -@analyze_lifetimes begin +@accelerate function combine!(C, A, B) product = A[2:end-1, 2:end-1] .* B[2:end-1, 2:end-1] C[2:end-1, 2:end-1] = product .+ 2.0f0 + return C end cuNumeric.BCAST_FUSION_DEBUG[] = false ``` -For example, a single-use producer inside `@analyze_lifetimes` is reported as: +For example, a single-use producer inside `@accelerate` is reported as: ```text ======================================== inter-broadcast fusion rewrite @@ -147,38 +148,23 @@ See [Kernel Fusion](./perf/kernel_fusion.md) for `@.` / fusion usage, and [Inter ## Inspect lifetime rewrites with `@show_lifetimes` -`@analyze_lifetimes` rewrites a block so temps are freed after their last use. `@show_lifetimes` prints the re-written code (without execution). It is pure AST work, so it works even without a GPU. +`@accelerate` rewrites straight-line code so eligible broadcasts combine and non-returned temporaries are freed after their final use. `@show_lifetimes` prints the exact expansion without executing it, so it works without a GPU. ```julia using cuNumeric -@show_lifetimes begin +@show_lifetimes function update!(C, A, B) result = A[1:end, :] .+ B[1:end, :] C .= result .* 2.0 + return C end ``` -Example output when broadcast fusion is enabled (fusion-aware analysis): - -```text -@analyze_lifetimes expansion (fusion-aware analysis) ------------------------------------------------------------- - 1 tmp1 = A[1:end, :] - 2 tmp2 = B[1:end, :] - 3 tmp3 = tmp1 .+ tmp2 - ✗ free tmp1 - ✗ free tmp2 - 4 result = tmp3 - 5 res3 = (C .= result .* 2.0) - ✗ free tmp3 - 6 res3 ------------------------------------------------------------- -``` - How to read it: -- Numbered lines are the rewritten statements. +- The header identifies the exact function, `let`, block, or expression form expanded. +- Numbered lines are rewritten statements. - Red `✗ free tmpN` lines are the inserted `maybe_insert_delete` calls. -- With fusion enabled, dotted intermediates stay as broadcast expressions instead of being treated as many separate allocations. With fusion disabled, the header says `plain analysis` and more call sites are hoisted. +- With fusion enabled, dotted intermediates stay as broadcast expressions instead of being treated as separate allocations. With fusion disabled, the header says `plain` and more call sites are hoisted. Use this when a hot loop still looks allocation-heavy, or when you want to confirm that a value is freed before it escapes the block. diff --git a/docs/src/developer_mode.md b/docs/src/developer_mode.md index 5cac1631d..913977942 100644 --- a/docs/src/developer_mode.md +++ b/docs/src/developer_mode.md @@ -69,4 +69,4 @@ Restart Julia. You do not need `Pkg.build` for pure JLL mode (the build script e - [Build Modes](./install.md): JLL, developer, and conda providers - [CNPreferences](./api_preferences.md): preference defaults and function reference - [Debugging](./debugging.md): fusion and lifetime printers while developing -- [Internals](./internals.md): how fusion and `@analyze_lifetimes` work +- [Internals](./internals.md): how fusion and `@accelerate` work diff --git a/docs/src/examples/grayscott.md b/docs/src/examples/grayscott.md index 59ef00463..7c230143b 100644 --- a/docs/src/examples/grayscott.md +++ b/docs/src/examples/grayscott.md @@ -1,59 +1,27 @@ # Gray-Scott Reaction Diffusion -```julia -# found in examples/gray-scott.jl -using cuNumeric -using Plots - -# Flattened u snapshots land here for examples/dmd.jl to analyze. -const SNAPSHOT_FILE = "gray-scott.h5" - -struct Params{T} - dx::T - dt::T - c_u::T - c_v::T - f::T - k::T - - function Params(dx=1.0f0, c_u=1.0f0, c_v=0.3f0, f=0.03f0, k=0.06f0) - new{Float32}(dx, dx/5, c_u, c_v, f, k) - end -end - -function bc!(u_new, v_new, u, v) - u_new[:,1] = u[:,end-1] - u_new[:,end] = u[:,2] - u_new[1,:] = u[end-1,:] - u_new[end,:] = u[2,:] - v_new[:,1] = v[:,end-1] - v_new[:,end] = v[:,2] - v_new[1,:] = v[end-1,:] - v_new[end,:] = v[2,:] -end - -function step!(u, v, u_new, v_new, args::Params) - @analyze_lifetimes begin - # Prefer @. so every op is dotted and the tree can fuse - F_u = @. -u[2:end-1, 2:end-1] * (v[2:end-1, 2:end-1]^2) + - args.f * (1.0f0 - u[2:end-1, 2:end-1]) - F_v = @. u[2:end-1, 2:end-1] * (v[2:end-1, 2:end-1]^2) - - (args.f + args.k) * v[2:end-1, 2:end-1] - - u_lap = @. ( - (u[3:end, 2:end-1] - 2 * u[2:end-1, 2:end-1] + u[1:end-2, 2:end-1]) / args.dx^2 + - (u[2:end-1, 3:end] - 2 * u[2:end-1, 2:end-1] + u[2:end-1, 1:end-2]) / args.dx^2 - ) - v_lap = @. ( - (v[3:end, 2:end-1] - 2 * v[2:end-1, 2:end-1] + v[1:end-2, 2:end-1]) / args.dx^2 + - (v[2:end-1, 3:end] - 2 * v[2:end-1, 2:end-1] + v[2:end-1, 1:end-2]) / args.dx^2 - ) - - u_new[2:end-1, 2:end-1] = @. (args.c_u * u_lap + F_u) * args.dt + u[2:end-1, 2:end-1] - v_new[2:end-1, 2:end-1] = @. (args.c_v * v_lap + F_v) * args.dt + v[2:end-1, 2:end-1] - end +The runnable example in `examples/gray-scott.jl` evolves two chemical fields with periodic boundaries. Its update is a straight-line function, so the recommended function form of `@accelerate` can release reaction and Laplacian temporaries after their final use and fuse eligible CUDA broadcasts. +```julia +@accelerate function step!(u, v, u_new, v_new, args::Params) + F_u = @. -u[2:end-1, 2:end-1] * v[2:end-1, 2:end-1]^2 + + args.f * (1.0f0 - u[2:end-1, 2:end-1]) + F_v = @. u[2:end-1, 2:end-1] * v[2:end-1, 2:end-1]^2 - + (args.f + args.k) * v[2:end-1, 2:end-1] + + u_lap = @. ( + (u[3:end, 2:end-1] - 2u[2:end-1, 2:end-1] + u[1:end-2, 2:end-1]) / args.dx^2 + + (u[2:end-1, 3:end] - 2u[2:end-1, 2:end-1] + u[2:end-1, 1:end-2]) / args.dx^2 + ) + v_lap = @. ( + (v[3:end, 2:end-1] - 2v[2:end-1, 2:end-1] + v[1:end-2, 2:end-1]) / args.dx^2 + + (v[2:end-1, 3:end] - 2v[2:end-1, 2:end-1] + v[2:end-1, 1:end-2]) / args.dx^2 + ) + + u_new[2:end-1, 2:end-1] = @. (args.c_u * u_lap + F_u) * args.dt + u[2:end-1, 2:end-1] + v_new[2:end-1, 2:end-1] = @. (args.c_v * v_lap + F_v) * args.dt + v[2:end-1, 2:end-1] bc!(u_new, v_new, u, v) + return nothing end function gray_scott() @@ -103,11 +71,9 @@ function gray_scott() cuNumeric.Legate.runtime_sync() return u, v - -end - -u, v = gray_scott() + end ``` + ![Simulation Output](../gray-scott.gif) The snapshots written to `gray-scott.h5` are the input to diff --git a/docs/src/examples/montecarlo.md b/docs/src/examples/montecarlo.md index 5a66572db..eb45188a0 100644 --- a/docs/src/examples/montecarlo.md +++ b/docs/src/examples/montecarlo.md @@ -1,40 +1,29 @@ # Monte-Carlo Integration -Most integrals can be estimated with a basic Monte-Carlo estimator: +For uniformly sampled points `x_i` in a domain of volume `\Omega`, a basic Monte-Carlo estimator is ```math -\hat{I}_N = \frac{\Omega}{N}\sum_{i=1}^Nf(x_i) +\hat{I}_N = \frac{\Omega}{N}\sum_{i=1}^N f(x_i). ``` -where `N` is the number of samples, ``\Omega`` is the volume of the domain and ``x_i`` are sampled indpendently and uniformly at random from the domain. This estimator is guranteed to converge (subject to some minor constraints) at a rate independent of the dimension and is embaressingly parallel to compute! -In the example below, we estimate the integral: -```math -I = \int_{-\infty}^{\infty}e^{-x^2}. -``` +This example estimates `\int_{-\infty}^{\infty} e^{-x^2}\,dx` by sampling the finite interval `[-10, 10]`. `@accelerate` frees the non-returned sample arrays after their final use; CUDA may also fuse eligible broadcasts. -Since we cannot uniformly sample form negative to positive infinity, we truncate the domain between -5 and 5. This is ok since the integrand exponentially decays and we won't be off by much in the end. ```julia -# found in examples/integrate.jl +# examples/integrate.jl using cuNumeric -# Note that we do not yet support broadcasting -# custom functions over NDArray, so the broadcasting MUST -# be done inside the function -integrand = (x) -> @. exp(-x^2) - -N = 1_000_000 +integrand(x) = @. exp(-x^2) -x_max = 10.0f0 -domain = [-x_max, x_max] -Ω = domain[2] - domain[1] +@accelerate function monte_carlo(N, x_max) + Ω = 2 * x_max + raw_samples = cuNumeric.rand(N) + samples = @. Ω * raw_samples - x_max + return (Ω / N) * sum(integrand(samples)) +end -samples = Ω * cuNumeric.rand(N) -samples = @. samples - x_max - -# Reductions return 0D NDArrays instead -# of a scalar to avoid blocking runtime -estimate = (Ω / N) * sum(integrand(samples)) - -println("Monte-Carlo Estimate: $(estimate)") -println("Analytical: $(sqrt(pi))") +estimate = monte_carlo(1_000_000, 10.0f0) +println("Monte-Carlo estimate: $(estimate)") +println("Analytical value: $(sqrt(pi))") ``` + +The result is a 0-dimensional `NDArray`, which keeps the reduction asynchronous. Use `unwrap(estimate)` only when a Julia scalar is required. diff --git a/docs/src/index.md b/docs/src/index.md index fe626eb6a..bb266637d 100644 --- a/docs/src/index.md +++ b/docs/src/index.md @@ -5,7 +5,7 @@ ``` -[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) +[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. @@ -20,7 +20,7 @@ using Pkg Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") ``` -The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). +The first installation can take a while because it includes several large dependencies, such as the CUDA SDK. To use a local cupynumeric build, see [Build Modes](https://julialegate.github.io/cuNumeric.jl/dev/install). ```julia using cuNumeric @@ -30,7 +30,7 @@ cuNumeric.versioninfo() > [!WARNING] > Starting more than one instance of cuNumeric.jl can lead to a hard-crash. The default hardware configuration reserves all available resources. -For more details, see [Hardware](./configuration/hardware.md). +For more details, see [Hardware](https://julialegate.github.io/cuNumeric.jl/dev/configuration/hardware). ### How `NDArray`s work @@ -40,7 +40,7 @@ The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it **Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communite with the Julia runtime. When you need a plain Julia number, call `unwrap`: +**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communicate with the Julia runtime. When you need a plain Julia number, call `unwrap`: ```julia s = sum(A) # NDArray{T,0} @@ -49,7 +49,7 @@ x = unwrap(s) # T, e.g. Float32 **The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. -For API details see [Initialization](./api_initialization.md), [Random](./api_random.md), and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). +For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/dev/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/dev/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/dev/perf/patterns_to_avoid). ### Kernel Fusion @@ -59,41 +59,33 @@ Nested broadcast expressions fuse into a single kernel by default when on GPU. P y .= @. -a + b * c ``` -See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. +See [Kernel Fusion](https://julialegate.github.io/cuNumeric.jl/dev/perf/kernel_fusion) and [Debugging](https://julialegate.github.io/cuNumeric.jl/dev/debugging) for controls and diagnostics. -### Helping the Garbage Collector +### The `@accelerate` macro -Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. +`@accelerate` fuses eligible GPU broadcasts within and across statements, then releases materialized temporary `NDArray`s after their last use on CPU or GPU. See [The `@accelerate` Macro](https://julialegate.github.io/cuNumeric.jl/dev/perf/reduce_allocations) for usage guidance. -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. +### Benchmarks -```julia -@analyze_lifetimes begin - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 -end -``` - -### Performance at a glance - -A representative benchmark figure will go here (add something like `docs/src/images/benchmarks-overview.png` when ready). - -Numbers, plots, and how to reproduce them live under [Benchmark Results](./benchmarks/results.md) and [How to Benchmark](./benchmarks/howto.md). +Results and reproduction instructions live under [Benchmark Results](https://julialegate.github.io/cuNumeric.jl/dev/benchmarks/results) and [How to Benchmark](https://julialegate.github.io/cuNumeric.jl/dev/benchmarks/howto). ### Try an example ```julia using cuNumeric -integrand = (x) -> @. exp(-x^2) +integrand(x) = @. exp(-x^2) + +@accelerate function monte_carlo(N, x_max) + Ω = 2 * x_max + raw_samples = cuNumeric.rand(N) + samples = @. Ω * raw_samples - x_max + return (Ω / N) * sum(integrand(samples)) +end N = 1_000_000 x_max = 10.0f0 -Ω = 2 * x_max - -samples = Ω .* cuNumeric.rand(N) -samples = samples .- x_max -estimate = (Ω / N) .* sum(integrand(samples)) +estimate = monte_carlo(N, x_max) println("Monte-Carlo Estimate: $(estimate)") ``` diff --git a/docs/src/internals.md b/docs/src/internals.md index 1afa20c03..652433251 100644 --- a/docs/src/internals.md +++ b/docs/src/internals.md @@ -1,72 +1,30 @@ # Internals -This page describes the implementation details of kernel fusion, manual memory management via `@analyze_lifetimes` and automatic memory management via GC heuristics. For docs on how to use these features, see [Kernel Fusion](./perf/kernel_fusion.md) and [Reduce Allocations](./perf/reduce_allocations.md). +This page summarizes the machinery behind [`@accelerate`](./perf/reduce_allocations.md), [kernel fusion](./perf/kernel_fusion.md), and allocation-driven garbage collection. ## Broadcast fusion -We compile nested Julia broadcast expressions on `NDArray` to a single CUDA kernel instead of launching one kernel per operation. +Julia first builds a nested `Broadcasted` tree for a dotted expression. When CUDA fusion is enabled and the tree is eligible, cuNumeric flattens it, generates a CUDA.jl kernel, and launches it through Legate. Otherwise `unravel_broadcast_tree` evaluates the operations individually. CPU execution always uses the unfused path. -### Pipeline +`@accelerate` adds an inter-statement syntax pass. It can merge a single-use broadcast producer into its consumer when doing so preserves mutation and aliasing semantics. The soft-scope `begin` form instead keeps named values materialized and may lower an eligible chain to a multi-output CUDA kernel. -- **Expression:** You write a dotted or `@.` expression such as `y .= @. a * b + c`. -- **Broadcast tree:** Julia builds the `Broadcasted` tree. -- **Fused path:** Flatten the tree, build a kernel with CUDA.jl, and launch it with Legate. -- **Unfused path:** `unravel_broadcast_tree` recursively unravels the tree and executes each operation one at a time. +Relevant source: `src/ndarray/broadcast_fusion.jl` and `src/scoping/`. -## Lifetimes and GC +## Lifetime analysis -Julia's GC sees an `NDArray` as a small handle. The actual data is owned by the Legate runtime. This means that to Julia's GC `NDArrays` do not create memory pressure and GC is never executed. To avoid out-of-memory errors we created the `@analyze_lifetimes` macro so users can manually manage the lifetimes of a code block and also manually track device memory to automatically invoke GC. +An `NDArray` is a small Julia handle to storage owned by Legate, so Julia's heap pressure understates the size of live array data. `@accelerate` therefore performs static last-use analysis: -### Eager last-use freeing with `@analyze_lifetimes` +1. Expand nested `@.` macros and reject non-straight-line code. +2. Apply conservative inter-statement fusion where the selected macro form permits it. +3. Hoist materialized temporary values and find their final uses. +4. Insert `maybe_insert_delete` calls after those uses while protecting caller-owned arguments, named soft-scope bindings, and returned values. -`@analyze_lifetimes` rewrites a block at macro-expansion time: +With fusion enabled, nested dotted nodes remain lazy and are not counted as separate array allocations. With fusion disabled, allocating calls are analyzed individually. `@show_lifetimes` prints the exact expansion without running it. -- Hoist temporary allocations into named temps. -- Find each temp's static last use. -- Insert calls to free temporary `NDArrays` after last-use. +The analysis is statement-linear rather than a control-flow graph pass. Apply it to straight-line function or loop bodies, not to control flow itself. -Under broadcast fusion, intermediate dotted nodes are **not** real `NDArray` allocations. The macro switches to a fusion-aware hoist that keeps dotted trees lazy and only treats slices, broadcast roots, and non-broadcast calls as real allocations. When fusion is off, every call (including dotted ops) is treated as a real allocation. +## Allocation-driven GC -```julia -@analyze_lifetimes begin - result = A[1:end, :] .+ B[1:end, :] - C .= result .* 2 -end -``` - -Use `@show_lifetimes` to print the rewritten block and the free sites without running the code. That is pure AST work and works without a GPU. - -The implementation uses separate passes for inter-statement broadcast fusion, -allocation hoisting, and finalizer insertion. The top-level scoping pass selects -the appropriate lifetime analysis based on whether broadcast fusion is enabled. - -- **Inter-statement broadcast fusion** (`rewrite_scope`): - Merges single-use broadcast statements into their consumer, e.g.: - `p = A .* B; C .= p .+ 1` → `C .= A .* B .+ 1` - This pass operates only on syntax and does not depend on cuNumeric types. - -- **Lifetime analysis**: - - With fusion enabled (`rewrite_broadcast_lifetimes`): keeps dotted trees lazy - and hoists only materialized values (slices, non-broadcast calls, etc.) - - With fusion disabled (`rewrite_eager_lifetimes`): hoists all allocating calls - Both passes rename temps and insert free calls after static last uses. - -- **Finalizer insertion** (`insert_finalizers`): - Traverses the AST generated by either lifetime pass and inserts `delete` calls - at the computed last-use sites. It preserves the final return value of the - block by not freeing it. - -Limits to keep in mind: - -- Analysis is statement-linear. It is not a full control-flow graph pass. Wrap hot loop bodies, not entire programs. -- Some paths free eagerly outside the macro (for example LHS slice views created during indexed assignment). - -Relevant source: `src/scoping/`. - -### Allocation-driven GC heuristics - -Every `NDArray` registers its byte size on construct and free. When predicted live bytes cross soft (~80%) or hard (~90%) fractions of available memory, and enough new growth has accumulated since the last collection, cuNumeric.jl triggers Julia `GC.gc`. - -`@analyze_lifetimes` reduces peak live temps. The heuristics catch cases the macro cannot see. +Each `NDArray` reports its byte size when constructed and freed. When predicted live bytes cross soft and hard fractions of available memory—and enough new growth has accumulated—cuNumeric asks Julia to collect garbage. Eager last-use freeing reduces peak live storage; the GC heuristic covers allocations the static pass cannot prove dead. Relevant source: `src/memory.jl`. diff --git a/docs/src/perf/kernel_fusion.md b/docs/src/perf/kernel_fusion.md index 04c437742..cb715fadf 100644 --- a/docs/src/perf/kernel_fusion.md +++ b/docs/src/perf/kernel_fusion.md @@ -1,61 +1,47 @@ # Kernel Fusion -On CUDA, nested broadcast expressions are fused into a single PTX kernel when fusion is enabled (the default). There is no separate `@fuse` macro. You write ordinary Julia broadcast code, and cuNumeric compiles eligible trees into one kernel instead of launching one op at a time. +When CUDA is available, cuNumeric can compile an eligible nested broadcast tree into one PTX kernel instead of launching one operation at a time. CPU execution follows the normal unfused path. -Prefer Julia's `@.` macro for multi-op elementwise expressions. Placing `.` on every operator by hand is easy to get wrong: missing a dot on unary negation or addition silently changes the meaning, and it can also break fusion by splitting work into the wrong ops. +Prefer Julia's `@.` macro for multi-operation elementwise expressions so every operator is dotted: ```julia -# Easy to miss the dot on negation +# A missing dot on unary negation changes the expression and can prevent fusion. y .= .-a .+ b .* c -# Prefer: @. dots every operator, which is clearer and fusion-friendly +# Prefer this form. y .= @. -a + b * c ``` -## Avoid preallocated intermediate broadcast buffers +Fusion requires array leaves with the same shape and at least `FUSE_BROADCAST_MIN_OPS` broadcast operations (default: 2). Shape-mismatched broadcasts such as `matrix .+ vector` use the unfused path. -Preallocation is useful for a final output or a buffer that must persist across -iterations. It can be counterproductive for a single-use intermediate inside -`@analyze_lifetimes`, however. An in-place `.=` assignment is an observable -mutation, so inter-statement broadcast fusion treats it as a kernel boundary: +## Fuse across statements with `@accelerate` -```julia -tmp = cuNumeric.zeros(Float32, N, N) -result = cuNumeric.zeros(Float32, N, N) +An ordinary assignment lets `@accelerate` substitute a single-use producer into its consumer: -@analyze_lifetimes begin - tmp .= @. A + B +```julia +@accelerate function update!(result, A, B, C) + tmp = @. A + B result .= @. tmp * C + 1.0f0 + return result end ``` -This materializes `tmp` before the second expression and requires separate -kernel launches. Instead, use an ordinary assignment for a single-use -intermediate and keep `.=` for the final destination: +This can become the equivalent of `result .= @. (A + B) * C + 1.0f0`. The rewrite is conservative: the producer must have one use, no intervening statement may invalidate its inputs, and the normal fusion requirements still apply. -```julia -result = cuNumeric.zeros(Float32, N, N) +Do not preallocate a single-use intermediate with `.=` merely to avoid allocation: -@analyze_lifetimes begin - tmp = @. A + B - result .= @. tmp * C + 1.0f0 -end +```julia +tmp .= @. A + B +result .= @. tmp * C + 1.0f0 ``` -The inter-statement pass can substitute `tmp` into its only consumer, producing -the equivalent of `result .= @. (A + B) * C + 1.0f0`. The intermediate is never -materialized, so the full expression can run as one fused kernel. +The mutation of `tmp` is observable, so it remains a kernel boundary. Keep preallocation when the intermediate is reused or its mutation must be visible. -This rewrite is intentionally conservative: the intermediate must have one -use, no intervening statement may invalidate its inputs, and all normal fusion -requirements still apply. Keep preallocation when an intermediate is reused, -must preserve mutation semantics, or cannot be fused. Use -[`BCAST_FUSION_DEBUG`](../debugging.md#inspect-fused-broadcasts-with-bcast_fusion_debug) -to confirm whether the rewrite occurred. +The `begin` form has different ownership semantics: named bindings remain live. On CUDA, an eligible same-shape chain can still be emitted as one multi-output kernel that materializes each binding. See [Accelerate Array Code](./reduce_allocations.md) for all macro forms. -Fusion applies when CUDA is available, the array leaves share the same shape, and the expression has at least `FUSE_BROADCAST_MIN_OPS` ops (default 2). Otherwise cuNumeric falls back to evaluating one op at a time. Shape-mismatched broadcasts such as `matrix .+ vector` use the unfused path. +## Configure and inspect fusion -Toggle fusion through `CNPreferences` (restart Julia after changing these): +Set preferences in one Julia process, then restart Julia: ```julia using CNPreferences @@ -63,14 +49,7 @@ using CNPreferences CNPreferences.enable_broadcast_fusion!() # default CNPreferences.disable_broadcast_fusion!() CNPreferences.set_broadcast_fusion_min_ops!(2) # default -CNPreferences.set_broadcast_fusion_min_ops!(1) # also fuse single-ops +CNPreferences.set_broadcast_fusion_min_ops!(1) # include single-op broadcasts ``` -What `set_broadcast_fusion_min_ops!` controls: - -- **`2` (default):** only trees with two or more ops fuse. Example: `y .= @. a * b + c` can fuse; `y .= cos.(x)` does not. Keeping single-ops on the unfused C-API path avoids PTX compile overhead when there is little to gain. -- **`1`:** every eligible broadcast can fuse, including unary / single-op forms. Prefer this when you want uniform fused behavior (for example in tests) rather than for typical apps. - -The threshold counts `Broadcasted` nodes in the expression tree. Set it through `CNPreferences`, then restart Julia. See [CNPreferences](../api_preferences.md). - -To inspect a fused launch or a lifetime rewrite, see [Debugging](../debugging.md). For the implementation pipeline, see [Internals](../internals.md). +The default threshold avoids PTX compilation overhead when there is little work to combine. See [CNPreferences](../api_preferences.md) for preference details and [`BCAST_FUSION_DEBUG`](../debugging.md#inspect-fused-broadcasts-with-bcast_fusion_debug) to confirm which path ran. diff --git a/docs/src/perf/reduce_allocations.md b/docs/src/perf/reduce_allocations.md index 37b38ca50..f36ab803b 100644 --- a/docs/src/perf/reduce_allocations.md +++ b/docs/src/perf/reduce_allocations.md @@ -1,31 +1,88 @@ -# Reduce Allocations +# The `@accelerate` Macro -Every intermediate `NDArray` (from a slice, broadcast, or function call) allocates a fresh buffer and waits for the Julia GC to free it. Because the GC runs on memory pressure, many dead buffers accumulate and pressure cuNumeric's allocator. +`@accelerate` optimizes *straight-line* array code: a fixed sequence of statements +with no branches, loops, jumps, `try`, or nested functions. Ordinary calls are +opaque boundaries; general control flow is not rewritten. -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager `maybe_insert_delete` calls immediately after each temporary's final use. Freed buffers can then be reused by later same-sized allocations instead of waiting on GC. +Within that restricted body, it performs: -When broadcast fusion is on, intermediate dotted nodes in a broadcast tree are not real `NDArray` allocations. The macro accounts for that automatically. +```@raw html +
    +
  1. Fusion within a broadcast expression. On CUDA, an eligible dotted expression such as @. A + B * C can run as one kernel. CPU execution uses the normal unfused path.
  2. +
  3. Fusion across broadcast statements. A single-use broadcast result can be substituted into its consumer, producing fewer GPU kernel launches.
  4. +
  5. Temporary lifetime analysis. After rewriting the code, the macro releases materialized, non-returned NDArrays after their final use on CPU or GPU.
  6. +
+``` + +These jobs must happen together: an intermediate that fuses into its consumer is never allocated, while an intermediate that cannot fuse is materialized and then released after its last use. + +## Use the function form by default + +Annotate a reusable straight-line function: + +```julia +@accelerate function update!(C, A, B) + combined = @. A + B + C .= @. 2.0f0 * combined + return C +end +``` + +On an eligible GPU path, `combined` can be folded into the second broadcast so the chain runs as one kernel. On CPU, or when fusion is ineligible, `combined` is materialized and released after the update. Function arguments belong to the caller and are never released by `@accelerate`; returned values also remain valid. + +## Choose a form + +The forms differ in which values must remain available, which determines how aggressively the macro may fuse or release intermediates. + +| Form | Use it when | Fusion and lifetime behavior | +| :--- | :--- | :--- | +| `@accelerate function ... end` | Defining reusable array code. This is the recommended default. | Arguments and returned values are protected. Non-returned locals may fuse into consumers or be released after their last use. | +| `@accelerate begin ... end` | Named results must remain in the current scope. | `begin` creates no new Julia scope, so every named binding is protected. An eligible same-shape CUDA chain may still use one multi-output kernel, but each named result is materialized. | +| `@accelerate let ... end` | Writing a one-off multi-statement calculation when only its result is needed. | `let` creates a local scope. Only the result escapes; other locals may fuse away or be released after their last use. | +| `@accelerate expr` | Evaluating one expression without named intermediates. | The result is materialized and returned. Eligible operations fuse within the expression, and transient temporaries are released. | + +For example, nested scope lets a private intermediate feed a value that is also +used by the outer block: ```julia -T = Float32 -A = cuNumeric.ones(T, (N, N)) -B = cuNumeric.ones(T, (N, N)) -C = cuNumeric.zeros(T, (N, N)) - -@analyze_lifetimes begin - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 +@accelerate begin + shifted = let + product = @. A * B + @. product + 1 + end + x = @. shifted * C end + +consume(shifted, x) ``` -**Benchmark** (Gray-Scott reaction-diffusion, 512×512, 10 000 steps): +Choose `let` when only the final result should escape: +```julia +result = @accelerate let + product = @. A * B + @. product + 1 +end ``` - user system elapsed CPU max RSS -without 106.50 s 23.87 s 58.66 s 222% 3786 MB -with 61.74 s 13.66 s 27.84 s 270% 2999 MB + +For a single unnamed expression, use: + +```julia +result = @accelerate (@. A + B * C) ``` -~2× wall-clock speedup and ~800 MB lower peak memory with no algorithmic changes. +## Writing an accelerated body + +- Apply `@.` to each elementwise right-hand side. Applying it to the entire body would change `x = ...` into `x .= ...` and `f(...)` into `f.(...)`. +- Use ordinary `=` for a disposable intermediate. This allows a single-use producer to fuse into its consumer. +- Use `.=` when the mutation must be visible. The destination write is preserved, although an eligible producer may fuse into it. +- Keep control flow outside the accelerated body. Loops, conditionals, `try`, short-circuit operators, and nested functions are rejected. +- Ordinary function calls run in program order and form rewrite boundaries. Annotate the called function separately if its body should also be accelerated. + +```julia +for _ in 1:nsteps + update!(C, A, B) +end +``` -Use `@show_lifetimes` to print the rewrite without running it ([Debugging](../debugging.md)). For how the rewriter and GC heuristics work, see [Internals](../internals.md). +See [Kernel Fusion](./kernel_fusion.md) for CUDA fusion requirements and [`@show_lifetimes`](../debugging.md#inspect-lifetime-rewrites-with-show_lifetimes) to inspect the exact rewrite without executing it. diff --git a/examples/custom_cuda.jl b/examples/custom_cuda.jl index 6092642bb..c968b29ae 100644 --- a/examples/custom_cuda.jl +++ b/examples/custom_cuda.jl @@ -3,6 +3,8 @@ using cuNumeric using CUDA import CUDA: i32 +cuNumeric.Experimental(true) + function kernel_add(a, b, c, N) i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x if i <= N @@ -19,25 +21,26 @@ function kernel_sin(a, b, N) return nothing end -N = 1024 -threads = 256 -blocks = cld(N, threads) +function run_custom_cuda(N=1024) + threads = 256 + blocks = cld(N, threads) + a = cuNumeric.fill(1.0f0, N) + b = cuNumeric.fill(2.0f0, N) + c = cuNumeric.zeros(Float32, N) + n_scalar = UInt32(N) -a = cuNumeric.fill(1.0f0, N) -b = cuNumeric.fill(2.0f0, N) -c = cuNumeric.ones(Float32, N) + add_task = cuNumeric.@cuda_task kernel_add(a, b, c, n_scalar) + cuNumeric.@launch task=add_task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=n_scalar -# task = cuNumeric.@cuda_task kernel_add(a, b, c, UInt32(1)) -# cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=UInt32(N) -# allowscalar() do -# c_cpu = c[:] -# println("Result of c after kenel launch: ", c_cpu[1]) -# end + sin_task = cuNumeric.@cuda_task kernel_sin(c, b, n_scalar) + cuNumeric.@launch task=sin_task threads=threads blocks=blocks inputs=c outputs=b scalars=n_scalar -task = cuNumeric.@cuda_task kernel_sin(a, b, UInt32(1)) -cuNumeric.@launch task=task threads=threads blocks=blocks inputs=a outputs=b scalars=UInt32(N) + allowscalar() do + return println("sin(1 + 2) = ", b[1]) + end + return b +end -allowscalar() do - b_cpu = b[:] - println("Result of b after kenel launch: ", b_cpu[1]) +if abspath(PROGRAM_FILE) == @__FILE__ + run_custom_cuda() end diff --git a/examples/daxpy.jl b/examples/daxpy.jl index f7983fff0..7ecb82886 100644 --- a/examples/daxpy.jl +++ b/examples/daxpy.jl @@ -6,6 +6,6 @@ arr = cuNumeric.rand(20) α = 1.32f0 b = 2.0f0 -arr2 = @. α * arr + b +arr2 = @accelerate @. α * arr + b println(arr2) diff --git a/examples/gray-scott.jl b/examples/gray-scott.jl index a52617a35..9d2a757e3 100644 --- a/examples/gray-scott.jl +++ b/examples/gray-scott.jl @@ -28,7 +28,7 @@ function bc!(u_new, v_new, u, v) return v_new[end, :] = v[2, :] end -function step!(u, v, u_new, v_new, args::Params) +@accelerate function step!(u, v, u_new, v_new, args::Params) # calculate F_u and F_v functions F_u = ( (-u[2:(end - 1), 2:(end - 1)] .* (v[2:(end - 1), 2:(end - 1)] .^ 2)) .+ diff --git a/examples/gray-scott.py b/examples/gray-scott.py index dce4e6cab..5eab0b8c8 100644 --- a/examples/gray-scott.py +++ b/examples/gray-scott.py @@ -1,82 +1,54 @@ -# python equivalent of gray-scott.jl to test the GC problem +"""cuPyNumeric equivalent of examples/gray-scott.jl.""" import cupynumeric as np -# import matplotlib.animation as animation -# from IPython.display import HTML -# import matplotlib.pyplot as plt - -def greyScottSys(u, v, dx, dt, c_u, c_v, f, k): - # u,v are arrays - # dx,dt are space and time steps - # c_u, c_v, f, k are constant paramaters - - #create new u array +def step(u, v, dx, dt, c_u, c_v, feed, kill): u_new = np.zeros_like(u) v_new = np.zeros_like(v) - #calculate F_u and F_v functions - F_u = (-u[1:-1,1:-1]*(v[1:-1,1:-1]**2)) + f*(1-u[1:-1,1:-1]) - F_v = (u[1:-1,1:-1]*(v[1:-1,1:-1]**2)) - (f+k)*v[1:-1,1:-1] - - # 2-D Laplacian of f using array slicing, excluding boundaries - # For an N x N array f, f_lap is the N-1 x N-1 array in the "middle" - u_lap = (u[2:,1:-1] - 2*u[1:-1,1:-1] + u[:-2,1:-1]) / dx**2\ - + (u[1:-1,2:] - 2*u[1:-1,1:-1] + u[1:-1,:-2]) / dx**2 - v_lap = (v[2:,1:-1] - 2*v[1:-1,1:-1] + v[:-2,1:-1]) / dx**2\ - + (v[1:-1,2:] - 2*v[1:-1,1:-1] + v[1:-1,:-2]) / dx**2 - - # Forward-Euler time step for all points except the boundaries - u_new[1:-1,1:-1] = ((c_u * u_lap) + F_u)*dt + u[1:-1,1:-1] - v_new[1:-1,1:-1] = ((c_v * v_lap) + F_v)*dt + v[1:-1,1:-1] - - # Apply periodic boundary conditions - u_new[:,0] = u[:,-2] - u_new[:,-1] = u[:,1] - u_new[0,:] = u[-2,:] - u_new[-1,:] = u[1,:] - v_new[:,0] = v[:,-2] - v_new[:,-1] = v[:,1] - v_new[0,:] = v[-2,:] - v_new[-1,:] = v[1,:] - + u_mid = u[1:-1, 1:-1] + v_mid = v[1:-1, 1:-1] + reaction = u_mid * v_mid**2 + f_u = -reaction + feed * (1 - u_mid) + f_v = reaction - (feed + kill) * v_mid + + u_lap = ( + u[2:, 1:-1] - 2 * u_mid + u[:-2, 1:-1] + + u[1:-1, 2:] - 2 * u_mid + u[1:-1, :-2] + ) / dx**2 + v_lap = ( + v[2:, 1:-1] - 2 * v_mid + v[:-2, 1:-1] + + v[1:-1, 2:] - 2 * v_mid + v[1:-1, :-2] + ) / dx**2 + + u_new[1:-1, 1:-1] = (c_u * u_lap + f_u) * dt + u_mid + v_new[1:-1, 1:-1] = (c_v * v_lap + f_v) * dt + v_mid + + u_new[:, 0] = u[:, -2] + u_new[:, -1] = u[:, 1] + u_new[0, :] = u[-2, :] + u_new[-1, :] = u[1, :] + v_new[:, 0] = v[:, -2] + v_new[:, -1] = v[:, 1] + v_new[0, :] = v[-2, :] + v_new[-1, :] = v[1, :] return u_new, v_new +def gray_scott(n=4000, n_steps=100): + dx = 1.0 + dt = dx / 5 + u = np.ones((n, n)) + v = np.zeros((n, n)) + seed = min(150, n) + u[:seed, :seed] = np.random.rand(seed, seed) + v[:seed, :seed] = np.random.rand(seed, seed) -# initial conditions and discretizaiton -dx = 1 -dt = dx/5 -u = np.ones((4000,4000)) -v = np.zeros((4000,4000)) -u[:150,:150] = np.random.rand(150,150) -v[:150,:150] = np.random.rand(150,150) - - -# fig = plt.figure() - -c_u = 1 -c_v = 0.3 -f = 0.03 -k = 0.06 - -# t_final = 1000 - -# ims = [] -n_steps = 100 # number of steps to take -frame_interval = 200 # steps to take between making plots - -# build a list of images -for n in range(n_steps) : - - ## This may need to be changed. - u,v = greyScottSys(u, v, dx, dt, c_u, c_v, f, k) + for _ in range(n_steps): + u, v = step(u, v, dx, dt, 1.0, 0.3, 0.03, 0.06) + return u, v - # ## Store frames when n is a multiple of frame_interval - # if n%frame_interval == 0: - # im = plt.imshow(u, vmin=0, vmax=1) # Show a plot of u. - # ims.append([im]) # append single image to the list of images -# anim = animation.ArtistAnimation(fig, ims, interval=100, repeat=False) -# HTML(anim.to_jshtml()) +if __name__ == "__main__": + gray_scott() diff --git a/examples/integrate.jl b/examples/integrate.jl index f27dae031..8d4e9653d 100644 --- a/examples/integrate.jl +++ b/examples/integrate.jl @@ -1,9 +1,4 @@ -using cuNumeric - -# Note that we do not yet support broadcasting -# custom functions over NDArray, so the broadcasting MUST -# be done inside the function -integrand = (x) -> @. exp(-x^2) +integrand = (x) -> exp(-x^2) N = 1_000_000 @@ -11,12 +6,13 @@ x_max = 10.0f0 domain = [-x_max, x_max] Ω = domain[2] - domain[1] -samples = Ω * cuNumeric.rand(N) -samples = @. samples - x_max +estimate = @accelerate begin + samples = @. Ω * cuNumeric.rand(N) - x_max -# Reductions return 0D NDArrays instead -# of a scalar to avoid blocking runtime -estimate = (Ω / N) * sum(integrand(samples)) + # Reductions return 0D NDArrays instead + # of a scalar to avoid blocking runtime + return (Ω / N) * sum(integrand.(samples)) +end println("Monte-Carlo Estimate: $(estimate)") println("Analytical: $(sqrt(pi))") diff --git a/examples/stencil.jl b/examples/stencil.jl index b8d8e3586..ee8c3e6a9 100644 --- a/examples/stencil.jl +++ b/examples/stencil.jl @@ -1,4 +1,3 @@ - using cuNumeric: cuNumeric function initialize(N) @@ -13,7 +12,6 @@ end function run_stencil(N, I, warmup) grid = initialize(N) - println("Running Jacobi stencil...") center = grid[2:(N + 1), 2:(N + 1)] @@ -22,11 +20,14 @@ function run_stencil(N, I, warmup) west = grid[2:(N + 1), 1:N] south = grid[3:(N + 2), 2:(N + 1)] - for i in 1:(I + warmup) + for _ in 1:(I + warmup) average = center .+ north .+ east .+ west .+ south work = 0.2 .* average - center = work + center .= work end + return grid end -run_stencil(1000, 100, 5) +if abspath(PROGRAM_FILE) == @__FILE__ + run_stencil(1000, 100, 5) +end diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 468904acc..bd080578f 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -59,6 +59,10 @@ if !HAS_CUDA @warn "We couldn't find a CUDA-enabled GPU. If you have an NVIDIA GPU something might be wrong." end +# `HAS_CUDA` describes the machine. A CPU-only Legate configuration on a GPU +# machine must still avoid registering or launching GPU tasks. +@inline _has_gpu_target() = HAS_CUDA && Int(Legate.num_gpus()) > 0 + const DEFAULT_FLOAT = Float32 const DEFAULT_INT = Int32 @@ -263,8 +267,6 @@ function __init__() _is_precompiling() && return nothing - _register_scoping_error_hint!() - # Cannot set LEGATE_CONFIG on CI machines used # to register packages. So we will just skip starting # legate/cunumeric when using registry CI machines. diff --git a/src/cuda/cuda_ptx_task.jl b/src/cuda/cuda_ptx_task.jl index 87962b7fa..44213a47a 100644 --- a/src/cuda/cuda_ptx_task.jl +++ b/src/cuda/cuda_ptx_task.jl @@ -2,7 +2,11 @@ export @cuda_task, @launch, CUDATask struct CUDATask func::String - argtypes::NTuple{N,Type} where {N} #! THIS IS TYPE UNSTABLE + argtypes::Vector{DataType} + + function CUDATask(func, argtypes) + return new(convert(String, func), collect(DataType, argtypes)) + end end #! JUST PASS TYPES HERE INSTEAD OF CALLING typeof() @@ -16,57 +20,59 @@ function to_stdvec(::Type{T}, vec) where {T} return stdvec end -function add_padding(arr::NDArray, dims::Dims{N}; copy=false) where {N} - old_size = size(arr) - - @assert all(dims .>= old_size) "newdims must be ≥ current dims elementwise" - new = zeros(eltype(arr), dims) +@inline _launch_shape(arr::NDArray) = _launch_shape(arr, _padding(arr)) +@inline _launch_shape(arr::NDArray, ::Nothing) = size(arr) +@inline _launch_shape(::NDArray, padding::PaddedStorage) = padding.shape + +@inline _physical_array(arr::NDArray, ::Nothing) = arr +@inline _physical_array(::NDArray, padding::PaddedStorage) = padding.backing + +function _ensure_launch_padding!(arr::NDArray{T,N}, target_shape; copy=false) where {T,N} + padding = _padding(arr) + !isnothing(padding) && padding.shape == target_shape && return arr + isnothing(padding) && size(arr) == target_shape && return arr + @assert all(target_shape .>= size(arr)) "cannot pad $(size(arr)) to $target_shape" + + padded = zeros(T, target_shape) + slices = ntuple(d -> (0, size(arr, d)), N) + logical = nda_get_slice(padded, slice_array(slices...)) + aliases_parent = !isnothing(arr.parent) + copy && !aliases_parent && copyto!(logical, arr) + storage = PaddedStorage{T,N}( + padded, + aliases_parent ? logical : nothing, + target_shape, + ) - if copy # due to being an input. we don't need to copy outputs - slices = ntuple(d -> (0, Int(old_size[d])), length(old_size)) - s = nda_get_slice(new, slice_array(slices...)) - copyto!(s, arr) - destroy!(s) + if aliases_parent + old_padding = _padding(arr) + arr.padding = storage + !isnothing(old_padding) && _destroy_padded_storage!(old_padding) + else + destroy!(arr) + arr.ptr = logical.ptr + arr.nbytes = logical.nbytes + arr.padding = storage + logical.ptr = Ptr{Cvoid}(0) + logical.nbytes = 0 end - - nda_destroy_array(arr.ptr) - register_free!(arr.nbytes) - - # update pointer & update metadata - arr.ptr = new.ptr - arr.nbytes = new.nbytes - arr.padding = old_size # remember the prior (before the padding) - - # julia GC will call finalizer, but we manually cleaned it - new.ptr = Ptr{Cvoid}(0) - new.nbytes = 0 - return new.padding = nothing -end - -function add_padding(arr::NDArray, i::Int64; copy=false) - return add_padding(arr, (i,); copy=copy) + return arr end -function check_sz!(arr, maxshape; copy=false) - sz = cuNumeric.size(arr) - if maxshape != nothing - # currently require all ndarray inputs to be equal - alligned_equal_size = sz == maxshape - if !alligned_equal_size - cuNumeric.add_padding(arr, maxshape; copy=copy) - new_size = padded_shape(arr) - @warn "[Padding Added] $sz output is now $new_size" - end +function _sync_to_launch_padding!(arr::NDArray) + padding = _padding(arr) + if !isnothing(padding) && !isnothing(padding.staging) + copyto!(padding.staging, arr) end + return nothing end -function check_sz(arr, maxshape) - sz = cuNumeric.size(arr) - if maxshape != nothing - # currently require all ndarray inputs to be equal - alligned_equal_size = sz == maxshape - @assert alligned_equal_size +function _sync_from_launch_padding!(arr::NDArray) + padding = _padding(arr) + if !isnothing(padding) && !isnothing(padding.staging) + copyto!(arr, padding.staging) end + return nothing end # `get_store` returns a Julia-owned `LogicalArrayImplAllocated` that shares the @@ -74,42 +80,32 @@ end # array into the task; if we leave the temporary alive until GC, store refcounts # stay elevated and framebuffer reclaim stalls (fusion 1-GPU OOM under load). # Finalize the temporary immediately after the copy into the task. -function _add_task_array!(add_to, task, arr::NDArray) - st = cuNumeric.get_store(arr) - var = add_to(task, st) - finalize(st) - return var +function _add_task_array!(add_to, task, arr::NDArray; physical=false) + task_arr = physical ? _physical_array(arr, _padding(arr)) : arr + st = cuNumeric.get_store(task_arr) + try + return add_to(task, st) + finally + finalize(st) + end end function Launch(kernel::CUDATask, inputs::Tuple{Vararg{NDArray}}, outputs::Tuple{Vararg{NDArray}}, scalars::Tuple{Vararg{Any}}; - blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing, validate_shapes=true) - max_shape = if validate_shapes - # Generic PTX tasks retain the existing padding/shape behavior. - ndarrays = vcat(inputs..., outputs...) # returns (nbytes, position) - mx = findmax(arr -> arr.nbytes, ndarrays) # first elem nbytes - shape = size(ndarrays[mx[2]]) # second elem max position - @assert !isnothing(shape) - shape - else - # Fused linear broadcast verifies shapes match - nothing - end - + blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing) rt = Legate.get_runtime() lib = cuNumeric.get_lib() task = Legate.create_auto_task(rt, lib, taskid) + physical = taskid == cuNumeric.RUN_PTX input_vars = Vector{Legate.Variable}() for arr in inputs - validate_shapes && check_sz!(arr, max_shape; copy=true) - push!(input_vars, _add_task_array!(Legate.add_input, task, arr)) + push!(input_vars, _add_task_array!(Legate.add_input, task, arr; physical)) end output_vars = Vector{Legate.Variable}() for arr in outputs - validate_shapes && check_sz!(arr, max_shape; copy=false) - push!(output_vars, _add_task_array!(Legate.add_output, task, arr)) + push!(output_vars, _add_task_array!(Legate.add_output, task, arr; physical)) end # Reserved scalars: kernel_name (0), blocks (1,2,3), threads (4,5,6) @@ -136,15 +132,34 @@ function Launch(kernel::CUDATask, inputs::Tuple{Vararg{NDArray}}, end function launch(kernel::CUDATask, inputs, outputs, scalars; - blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing, validate_shapes=true) - return Launch(kernel, - isa(inputs, Tuple) ? inputs : (inputs,), - isa(outputs, Tuple) ? outputs : (outputs,), + blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing) + input_tuple = isa(inputs, Tuple) ? inputs : (inputs,) + output_tuple = isa(outputs, Tuple) ? outputs : (outputs,) + + # Custom tasks require equal physical shapes. Keep the padded backing so + # repeated launches do not allocate or copy again. + if taskid == cuNumeric.RUN_PTX + arrays = (input_tuple..., output_tuple...) + if !isempty(arrays) + rank = ndims(first(arrays)) + @assert all(ndims(arr) == rank for arr in arrays) "custom task arrays must have equal ranks" + max_shape = ntuple(d -> maximum(_launch_shape(arr)[d] for arr in arrays), rank) + foreach(arr -> _ensure_launch_padding!(arr, max_shape; copy=true), input_tuple) + foreach(arr -> _ensure_launch_padding!(arr, max_shape), output_tuple) + foreach(_sync_to_launch_padding!, input_tuple) + end + end + + result = Launch(kernel, + input_tuple, + output_tuple, isa(scalars, Tuple) ? scalars : (scalars,); blocks=isa(blocks, Tuple) ? blocks : (blocks,), threads=isa(threads, Tuple) ? threads : (threads,), - taskid=taskid, ctx=ctx, validate_shapes=validate_shapes, + taskid=taskid, ctx=ctx, ) + taskid == cuNumeric.RUN_PTX && foreach(_sync_from_launch_padding!, output_tuple) + return result end function ptx_task(ptx::String, kernel_name) diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 8df0b0ee2..6ac971e4c 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -123,11 +123,9 @@ __materialize(x::Base.RefValue{Val{V}}) where {V} = NDArray(V) # Use binary_op P # Catch unknown things... __materialize(x) = error("Unrecognized leaf in broadcast expression: $(x)") -# Scalar-only nested broadcasts (e.g. `s1 .* s2 .+ A`): the inner -# `Broadcasted(*, (s1, s2))` keeps DefaultArrayStyle{0}, not NDArrayStyle. -# Fold to a Number so the parent unravel sees a scalar leaf. +# Use Base for scalar-only broadcasts, including `literal_pow` wrappers. @inline function __materialize(bc::Broadcasted{<:DefaultArrayStyle{0}}) - return bc.f((__materialize.(bc.args))...) + return Base.materialize(bc) end function __materialize(bc::Broadcasted{<:NDArrayStyle}) @@ -135,6 +133,22 @@ function __materialize(bc::Broadcasted{<:NDArrayStyle}) return unravel_broadcast_tree(bc) end +# The C API is binary, so evaluate flattened `+` and `*` chains pairwise. +function _unravel_flattened_associative(f, args::Tuple) + acc = first(args) + owns_acc = false + for arg in Base.tail(args) + next = try + __materialize(Base.broadcasted(f, acc, arg)) + finally + owns_acc && acc isa NDArray && destroy!(acc) + end + acc = next + owns_acc = acc isa NDArray + end + return acc +end + # Destroy promote copies and non-leaf materialized NDArrays (nested results / Val{V}). @inline function _destroy_unfused_arg_temps!(orig, materialized, promoted) if promoted isa NDArray && promoted !== materialized @@ -148,6 +162,9 @@ end # Un-fused implementation of broadcast tree function unravel_broadcast_tree(bc::Broadcasted) + if length(bc.args) > 2 && _is_flattened_associative(bc.f) + return _unravel_flattened_associative(bc.f, bc.args) + end # Recursively materialize/unravel any nested broadcasts # until we reach a Broadcasted expression with only @@ -234,13 +251,11 @@ end ) end - # Fused writes `dest` in place (no post-fuse `nda_move`); promotion is - # checked pre-launch in `fuse_broadcast_tree!`. CPU vs GPU is compile-time - # via `@static if FUSE_BROADCAST_EXPRS && HAS_CUDA`. + # Require an active GPU target so `--gpus 0` stays on the unfused path. # Fusion requires same-shaped NDArray leaves; otherwise fall back. # Single-op exprs (length < `FUSE_BROADCAST_MIN_OPS`) stay unfused by default. @static if FUSE_BROADCAST_EXPRS && HAS_CUDA - if _should_attempt_broadcast_fusion(dest, bc) + if _has_gpu_target() && _should_attempt_broadcast_fusion(dest, bc) return fuse_broadcast_tree!(dest, bc) else return _copyto_unfused!(dest, unravel_broadcast_tree(bc)) diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 728deec17..64149a9f7 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -619,7 +619,11 @@ end bc::Base.Broadcast.Broadcasted{S,Ax,F,Args} ) where {S,Ax,F,Args} eltypes = _fused_checked_eltypes(bc.args) - T_OUT = __checked_promote_op(bc.f, eltypes) + T_OUT = if length(bc.args) > 2 && _is_flattened_associative(bc.f) + _checked_promote_associative(bc.f, eltypes.parameters...) + else + __checked_promote_op(bc.f, eltypes) + end __my_promote_type(eltypes.parameters...) return T_OUT end @@ -748,10 +752,329 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas threads=fkm.threads, taskid=cuNumeric.RUN_PTX_BROADCAST, ctx=fkm.ctx, - validate_shapes=false, ) end # Fused kernel already wrote `dest` in place; promotion was checked pre-launch. return dest end + +# ============================================================================ +# Multi-output fused broadcast: materialize named intermediates in one launch. +# Segments (dependency order, root last) are flattened independently; a +# `MatRef{K}` leaf reads the K-th segment's per-element local. One kernel +# computes each segment into a local, stores it to that segment's output buffer, +# and chains locals into parents (segmented flatten + chained-local multi-store). +# ============================================================================ + +# Opaque scalar-like leaf that survives `Base.Broadcast.flatten` (never descended +# into, never wrapped in a Ref). +struct MatRef{K} end +MatRef(k::Int) = MatRef{k}() +Base.broadcastable(m::MatRef) = m +Base.Broadcast.BroadcastStyle(::Type{<:MatRef}) = Base.Broadcast.DefaultArrayStyle{0}() +Base.axes(::MatRef) = () +Base.ndims(::Type{<:MatRef}) = 0 + +# Third arg-plan variant (alongside Runtime/Static): read the K-th chained local. +struct LocalBroadcastArg{K} end + +Base.@propagate_inbounds @inline _materialize_ml_arg( + ::RuntimeBroadcastArg{J}, rt, sa, locals, I +) where {J} = _gpu_broadcast_getindex(getfield(rt, J), I) +Base.@propagate_inbounds @inline _materialize_ml_arg( + ::StaticBroadcastArg{J}, rt, sa, locals, I +) where {J} = getfield(sa, J) +Base.@propagate_inbounds @inline _materialize_ml_arg( + ::LocalBroadcastArg{K}, rt, sa, locals, I +) where {K} = getfield(locals, K) + +Base.@propagate_inbounds @inline _materialize_ml_args(::Tuple{}, rt, sa, locals, I) = () +Base.@propagate_inbounds @inline function _materialize_ml_args(plan::Tuple, rt, sa, locals, I) + return ( + @inbounds(_materialize_ml_arg(getfield(plan, 1), rt, sa, locals, I)), + @inbounds(_materialize_ml_args(Base.tail(plan), rt, sa, locals, I))..., + ) +end + +# Device-side: run each segment in order, store to its output, chain the local. +# Generate straight-line code because recursive tuple traversal eventually hits +# Julia's inference limit and leaves a dynamic call in GPU kernels on Julia 1.10. +Base.@propagate_inbounds @inline @generated function _run_segments( + segs::S, outs, rt, sa, locals::L, I +) where {S<:Tuple,L<:Tuple} + body = Expr(:block) + local_values = Any[:(getfield(locals, $k)) for k in 1:fieldcount(L)] + + for k in 1:fieldcount(S) + seg = gensym(:seg) + vals = gensym(:vals) + value = gensym(:value) + local_tuple = Expr(:tuple, local_values...) + push!( + body.args, + quote + $seg = getfield(segs, $k) + $vals = _materialize_ml_args( + getfield($seg, 2), rt, sa, $local_tuple, I + ) + $value = Base.Broadcast._broadcast_getindex_evalf( + getfield($seg, 1), $vals... + ) + @inbounds getfield(outs, $k)[I] = $value + end, + ) + push!(local_values, value) + end + + push!(body.args, :(nothing)) + return body +end + +# Dimension-dispatched (mirrors the single-output linear/cartesian kernels). +# `args` = (outputs[1:NOUT]..., runtime_args...); bounds from the first output. +function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val{2}) where {NOUT} + @kernel unsafe_indices = true function broadcast_kernel_multi_2d(args...) + I = _broadcast_cartesian_work_id() + dest = getfield(args, 1) + @inbounds if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + end + end + return broadcast_kernel_multi_2d +end + +function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val{3}) where {NOUT} + @kernel unsafe_indices = true function broadcast_kernel_multi_3d(args...) + I = _broadcast_cartesian_work_id_3d() + dest = getfield(args, 1) + @inbounds if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) && I[3] <= size(dest, 3) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + end + end + return broadcast_kernel_multi_3d +end + +# 1-D and any other rank: linear indexing (matches the single-output default). +function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val) where {NOUT} + @kernel unsafe_indices = true function broadcast_kernel_multi_linear(args...) + I = _broadcast_linear_work_id() + @inbounds if I <= length(getfield(args, 1)) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + end + end + return broadcast_kernel_multi_linear +end + +# Flatten a segment and classify its leaves, deduping NDArrays into shared +# `runtime_args` and static leaves into shared `static_args`. +function _split_segment!(seg_bc, runtime_args, static_args, ndarray_idx) + flat = Base.Broadcast.flatten(seg_bc) + plan = Any[] + for leaf in flat.args + if leaf isa MatRef + push!(plan, LocalBroadcastArg{_matref_k(leaf)}()) + elseif leaf isa Base.RefValue + v = leaf[] + if v isa Number + push!(runtime_args, v) + push!(plan, RuntimeBroadcastArg{length(runtime_args)}()) + else + _push_static_arg!(static_args, plan, v) + end + elseif leaf isa NDArray || leaf isa Base.Broadcast.Extruded + nda = get_ndarray(leaf) + j = get!(() -> (push!(runtime_args, leaf); length(runtime_args)), + ndarray_idx, objectid(nda)) + push!(plan, RuntimeBroadcastArg{j}()) + elseif leaf isa Number + push!(runtime_args, leaf) + push!(plan, RuntimeBroadcastArg{length(runtime_args)}()) + elseif isbits(leaf) + _push_static_arg!(static_args, plan, leaf) + else + throw(ArgumentError("multi-output fusion: cannot lower leaf $(typeof(leaf))")) + end + end + return (flat.f, tuple(plan...)) +end + +_matref_k(::MatRef{K}) where {K} = K + +const _MULTI_PTX_CACHE = Dict{Any,Any}() +const _MULTI_PTX_CACHE_LOCK = ReentrantLock() + +# Compile + register the multi-output kernel -> (ctx, threads, CUDATask). Cached +# by (closure type, arg types) so a repeated fusion signature compiles once. +function get_multi_cuda_task(obj, out_arrs, runtime_args) + arg_types = (map_cuda_type.(typeof.(out_arrs))..., map_cuda_type.(typeof.(runtime_args))...) + key = (typeof(obj), arg_types) + lock(_MULTI_PTX_CACHE_LOCK) do + return get!(_MULTI_PTX_CACHE, key) do + ptx, threads, ctx = get_ptx(obj, arg_types...) + threads == 0 && return (ctx, 0, nothing) + orig = extract_kernel_name(ptx) + uname = orig * "_" * string(hash(ptx); base=16) + ptx = replace(ptx, orig => uname) + ptx_task(ptx, uname) + return (ctx, threads, CUDATask(uname, arg_types)) + end + end +end + +# First NDArray leaf across all segments; used as an allocation template. +function _first_ndarray(seg_bcs::Tuple) + for seg_bc in seg_bcs + for leaf in Base.Broadcast.flatten(seg_bc).args + leaf isa NDArray && return leaf + leaf isa Base.Broadcast.Extruded && return get_ndarray(leaf) + end + end + return throw(ArgumentError("multi-output fusion: no NDArray leaf to size buffers from")) +end + +# Result eltype of a segment, resolving `MatRef{k}` to `eltype(bufs[k])` (earlier +# segments already allocated). Lets each intermediate use its own eltype. +function _segment_eltype(flat, bufs) + ets = map(flat.args) do leaf + if leaf isa MatRef + eltype(bufs[_matref_k(leaf)]) + elseif leaf isa NDArray + eltype(leaf) + elseif leaf isa Base.Broadcast.Extruded + eltype(leaf.x) + else + typeof(leaf) + end + end + T = Base.promote_op(flat.f, ets...) + # Fall back to promoting the leaf eltypes when promote_op can't infer. + return isconcretetype(T) ? T : promote_type(ets...) +end + +# Render a segment's broadcast tree, showing `MatRef{k}` leaves as `seg{k}`. +function _bcast_multi_tree_str(bc) + return _bcast_tree_str(bc) do x + x isa MatRef && return "seg{$(_matref_k(x))}" + x isa NDArray && return "NDArray" + x isa Base.Broadcast.Extruded && return "NDArray" + x isa Number && return repr(x) + x isa Base.RefValue && return string("^", repr(x[])) + return string("<", typeof(x), ">") + end +end + +# Fused multi-output introspection (mirrors `_describe_fused_broadcast`). Enable +# with `cuNumeric.BCAST_FUSION_DEBUG[] = true`. +function _describe_fused_multi( + out_arrs, seg_bcs, input_ndarrays, actual_scalars, argmap, threads, ndrange +) + io = IOBuffer() + field(k, v) = println(io, " ", rpad(k, 8), v) + NOUT = length(out_arrs) + println(io, "\n", "="^40, " fused multi-output broadcast ($NOUT outputs)") + println(io, " segments (each materialized to its own output):") + for (i, seg) in enumerate(seg_bcs) + role = i == NOUT ? "root" : "seg{$i}" + println( + io, " ", rpad(role, 7), _ndarray_debug_summary(out_arrs[i]), + " <- ", _bcast_multi_tree_str(seg), + ) + end + field("inputs", "input{N} ($(length(input_ndarrays)) unique)") + for (i, nd) in enumerate(input_ndarrays) + println(io, " ", rpad(string(i - 1), 4), _ndarray_debug_summary(nd)) + end + isempty(actual_scalars) || field("scalars", join(repr.(actual_scalars), ", ")) + indexing = ndims(out_arrs[1]) in (2, 3) ? "cartesian" : "linear" + field( + "launch", + "host thread budget=$threads, indexing=$indexing, num_outputs=$NOUT, " * + "blocks=device(local tile), global_ndrange=$ndrange", + ) + field("arg_map", string(argmap)) + print(String(take!(io))) + return nothing +end + +# Launch one kernel writing each segment into preallocated `out_arrs[i]` +# (dependency order; `out_arrs[end]` is the root). +function _fused_multi_launch!(out_arrs::Tuple, seg_bcs::Tuple) + NOUT = length(seg_bcs) + runtime_args = Any[] + static_args = Any[] + ndarray_idx = Dict{UInt,Int}() + segs = Any[] + for seg_bc in seg_bcs + push!(segs, _split_segment!(seg_bc, runtime_args, static_args, ndarray_idx)) + end + segs = tuple(segs...) + static_args = tuple(static_args...) + + kernel = make_multi_output_kernel(segs, Val(NOUT), static_args, Val(ndims(out_arrs[1]))) + bck = kernel(CUDACore.CUDAKernels.CUDABackend()) + ctx, threads, task = get_multi_cuda_task(bck, out_arrs, tuple(runtime_args...)) + isnothing(task) && return out_arrs + + # arg_map: kernel args in order (outputs..., runtime_args...). + argmap = Int32[Int32(i) for i in 0:(NOUT - 1)] + input_ndarrays = NDArray[] + ndinput_idx = Dict{UInt,Int}() + actual_scalars = Any[] + for arg in runtime_args + if stores_cudevicearray(map_cuda_type(typeof(arg))) + nda = get_ndarray(arg) + j = get!(() -> (push!(input_ndarrays, nda); length(input_ndarrays) - 1), + ndinput_idx, objectid(nda)) + push!(argmap, Int32(NOUT + j)) + else + push!(argmap, Int32(-1 - length(actual_scalars))) + push!(actual_scalars, arg) + end + end + + if BCAST_FUSION_DEBUG[] + ndrange = ndims(out_arrs[1]) > 0 ? size(out_arrs[1]) : (1,) + _describe_fused_multi( + out_arrs, seg_bcs, input_ndarrays, actual_scalars, argmap, threads, ndrange + ) + end + + launch( + task, tuple(input_ndarrays...), out_arrs, + (Int32(length(argmap)), argmap..., actual_scalars...); + blocks=1, threads=threads, taskid=cuNumeric.RUN_PTX_BROADCAST, ctx=ctx, + ) + return out_arrs +end + +# Allocate a typed tuple so callers retain each segment's concrete NDArray type. +function _alloc_segment_buffers(template::NDArray, seg_bcs::Tuple, dims) + return _alloc_segment_buffers(template, seg_bcs, dims, ()) +end + +@inline _alloc_segment_buffers(template, ::Tuple{}, dims, bufs::Tuple) = bufs + +@inline function _alloc_segment_buffers(template, seg_bcs::Tuple, dims, bufs::Tuple) + flat = Base.Broadcast.flatten(first(seg_bcs)) + buf = similar(template, _segment_eltype(flat, bufs), dims) + return _alloc_segment_buffers(template, Base.tail(seg_bcs), dims, (bufs..., buf)) +end + +# `seg_bcs[1:end-1]` are materialized producers (dependency order); `seg_bcs[end]` +# writes `dest`. Producer buffers are allocated (returned so callers bind names). +function copyto_fused_multi!(dest::NDArray, seg_bcs::Tuple) + bufs = _alloc_segment_buffers(dest, seg_bcs[1:(end - 1)], size(dest)) + outs = (bufs..., dest) + _fused_multi_launch!(outs, seg_bcs) + return outs +end + +# Every segment gets a fresh buffer (all named results stay live). Returns the +# buffers in segment order so callers can bind each user name. +function copyto_fused_multi_alloc!(seg_bcs::Tuple) + tmpl = _first_ndarray(seg_bcs) + outs = _alloc_segment_buffers(tmpl, seg_bcs, size(tmpl)) + _fused_multi_launch!(outs, seg_bcs) + return outs +end diff --git a/src/ndarray/detail/linalg.jl b/src/ndarray/detail/linalg.jl index a15cf9e1a..6722495c1 100644 --- a/src/ndarray/detail/linalg.jl +++ b/src/ndarray/detail/linalg.jl @@ -263,15 +263,12 @@ function _svd(a::NDArray{T,2}, full_matrices::Bool) where {T} ) k = min(m, n) S = real(T) - # cuSolver requires full square buffers regardless of full_matrices - u_buf = cuNumeric.zeros(T, m, m) + + u_buf = full_matrices ? cuNumeric.zeros(T, m, m) : cuNumeric.zeros(T, m, k) s = cuNumeric.zeros(S, k) - vh_buf = cuNumeric.zeros(T, n, n) + vh_buf = full_matrices ? cuNumeric.zeros(T, n, n) : cuNumeric.zeros(T, k, n) svd_single(a, u_buf, s, vh_buf) - # Backend factors are logically ordered; only thin strided views need materialization. - u = full_matrices ? u_buf : copy(u_buf[:, 1:k]) - vh = full_matrices ? vh_buf : copy(vh_buf[1:k, :]) - return u, s, vh + return u_buf, s, vh_buf end # svd runs on float/complex only — no integer backend diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 8b7e67671..a8b6f8118 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -43,21 +43,24 @@ get_n_dim(ptr::NDArray_t) = Int(ccall((:nda_array_dim, libnda), Int32, (NDArray_ abstract type AbstractNDArray{T<:SUPPORTED_TYPES,N} <: AbstractArray{T,N} end +# Runtime padding uses an abstract field to break the recursive storage definition. +abstract type AbstractPaddedStorage{T,N} end + @doc""" The NDArray type represents a multi-dimensional array in cuNumeric. It is a wrapper around a Legate array and provides various methods for array manipulation and operations. Finalizer calls `nda_destroy_array` to clean up the underlying Legate array when the NDArray is garbage collected. """ -mutable struct NDArray{T,N,PADDED,P} <: AbstractNDArray{T,N} +mutable struct NDArray{T,N,P} <: AbstractNDArray{T,N} ptr::NDArray_t nbytes::Int64 - padding::Union{Nothing,NTuple{N,Int}} + padding::Union{Nothing,AbstractPaddedStorage{T,N}} parent::P function NDArray(ptr::NDArray_t, ::Type{T}, ::Val{N}) where {T,N} nbytes = cuNumeric.nda_nbytes(ptr) cuNumeric.register_alloc!(nbytes) - handle = new{T,N,false,Nothing}(ptr, nbytes, nothing, nothing) + handle = new{T,N,Nothing}(ptr, nbytes, nothing, nothing) finalizer(_finalize_ndarray!, handle) return handle end @@ -66,26 +69,53 @@ mutable struct NDArray{T,N,PADDED,P} <: AbstractNDArray{T,N} function NDArray(ptr::NDArray_t, ::Type{T}, ::Val{N}, parent::P) where {T,N,P} nbytes = cuNumeric.nda_nbytes(ptr) cuNumeric.register_alloc!(nbytes) - handle = new{T,N,false,P}(ptr, nbytes, nothing, parent) + handle = new{T,N,P}(ptr, nbytes, nothing, parent) finalizer(_finalize_ndarray!, handle) return handle end end +struct PaddedStorage{T,N} <: AbstractPaddedStorage{T,N} + backing::NDArray{T,N,Nothing} + staging::Union{Nothing,NDArray{T,N,NDArray{T,N,Nothing}}} + shape::NTuple{N,Int} +end + +# Narrow the abstract field to its concrete storage type. +@inline _padding(arr::NDArray{T,N}) where {T,N} = + arr.padding::Union{Nothing,PaddedStorage{T,N}} + +function _finalize_padded_storage!(storage::PaddedStorage) + !isnothing(storage.staging) && finalize(storage.staging) + finalize(storage.backing) + return nothing +end + +function _destroy_padded_storage!(storage::PaddedStorage) + !isnothing(storage.staging) && destroy!(storage.staging) + destroy!(storage.backing) + return nothing +end + # May run off the launch thread, so defer the Legate free to drain_pending_frees!. # Accounting is atomic and safe to do here immediately. function _finalize_ndarray!(arr::NDArray) ptr = arr.ptr - ptr == C_NULL && return nothing arr.ptr = Ptr{Cvoid}(0) nbytes = arr.nbytes arr.nbytes = 0 - nbytes > 0 && register_free!(nbytes) - _enqueue_free!(ptr) + padding = _padding(arr) + arr.padding = nothing + + if ptr != C_NULL + nbytes > 0 && register_free!(nbytes) + _enqueue_free!(ptr) + end + !isnothing(padding) && _finalize_padded_storage!(padding) return nothing end -@inline _is_ndarray_slice(arr::NDArray) = arr.parent isa NDArray +@inline _is_ndarray_slice(arr::NDArray) = arr.parent isa NDArray || !isnothing(_padding(arr)) """ destroy!(arr::NDArray) @@ -102,6 +132,9 @@ function destroy!(arr::NDArray) arr.nbytes = 0 nbytes > 0 && register_free!(nbytes) end + padding = _padding(arr) + arr.padding = nothing + !isnothing(padding) && _destroy_padded_storage!(padding) return arr end @@ -591,9 +624,7 @@ end Return the size of the given `NDArray`. """ -shape(arr::NDArray{<:Any,N,true}) where {N} = arr.padding - -function shape(arr::NDArray{<:Any,N,false}) where {N} +function shape(arr::NDArray{<:Any,N}) where {N} shp = cuNumeric.nda_array_shape(arr) return ntuple(i -> Int(shp[i]), Val(N)) end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index f42955061..437d69d5c 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -399,7 +399,7 @@ function _setindex!( end #### START OF SLICING #### -# LHS slices from `nda_get_slice` are invisible to `@analyze_lifetimes`; destroy +# LHS slices from `nda_get_slice` are invisible to `@accelerate`; destroy # the view handle after submitting the assign so they cannot pile up under Julia # GC (which sees each NDArray as ~pointer-sized). function _setindex_slice!(lhs::NDArray, rhs::NDArray, slices) diff --git a/src/ndarray/promotion.jl b/src/ndarray/promotion.jl index 20e5ab607..cb1034098 100644 --- a/src/ndarray/promotion.jl +++ b/src/ndarray/promotion.jl @@ -30,9 +30,19 @@ unchecked_promote_scalar(x, ::Type) = x unchecked_promote_arr(::Base.RefValue{typeof(^)}, ::Type{T}) where {T} = typeof(Base.:(^)) unchecked_promote_arr(::Base.RefValue{Val{V}}, ::Type{T}) where {T,V} = Val{V} +@inline _is_flattened_associative(f) = f === (+) || f === (*) + __checked_promote_op(op, ::Type{Tuple{A}}) where {A} = __checked_promote_op(op, A) __checked_promote_op(op, ::Type{Tuple{A,B}}) where {A,B} = __checked_promote_op(op, A, B) +# Julia flattens dotted `+` and `*` chains into n-ary Broadcasted nodes. Fold +# their input types pairwise, matching both the binary C API and fused path. +@inline function __checked_promote_op( + op::Union{typeof(+),typeof(*)}, ::Type{Args} +) where {Args<:Tuple{Any,Any,Any,Vararg{Any}}} + return _checked_promote_associative(op, Args.parameters...) +end + # Path for literal powers @inline function __checked_promote_op( f::typeof(Base.literal_pow), a::Type{Tuple{_,ARR_TYPE,Val{POWER}}} @@ -77,6 +87,16 @@ end return T end +@inline _checked_promote_associative(op, ::Type{A}, ::Type{B}) where {A,B} = + __checked_promote_op(op, A, B) + +@inline function _checked_promote_associative( + op, ::Type{A}, ::Type{B}, ::Type{C}, rest::Type... +) where {A,B,C} + T = __checked_promote_op(op, A, B) + return _checked_promote_associative(op, T, C, rest...) +end + # For literal powers which are often Int64, do not check for promotion to double # The result of promote_op with a literal integer power is always the base type # Base.promote_op(^, Float32, Int64) == Float32 diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index ef234d694..d27a78951 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -48,6 +48,13 @@ global const unary_op_map_no_args = Dict{Function,UnaryOpCode}( ### SPECIAL CASES ### +# `dest .= src` lowers to `identity.(src)`. Treat identity like the native +# unary operation it is so ordinary Julia broadcast assignment works for +# NDArrays, including writable slices. +@inline function __broadcast(::typeof(identity), out::NDArray, input::NDArray) + return nda_unary_op!(out, cuNumeric.COPY, input) +end + # Needed to support != Base.:(!)(input::NDArray{Bool,0}) = nda_unary_op!(similar(input), cuNumeric.LOGICAL_NOT, input) Base.:(!)(input::NDArray{Bool,1}) = nda_unary_op!(similar(input), cuNumeric.LOGICAL_NOT, input) @@ -92,7 +99,7 @@ end # Fallbacks for Real types @inline function __broadcast(f::typeof(Base.real), out::NDArray, input::NDArray{<:Real}) # real(real_array) is just the array - return nda_unary_op!(out, cuNumeric.IDENTITY, input) + return nda_unary_op!(out, cuNumeric.COPY, input) end @inline function __broadcast(f::typeof(Base.imag), out::NDArray, input::NDArray{<:Real}) # imag(real_array) is all zeros @@ -100,7 +107,7 @@ end end @inline function __broadcast(f::typeof(Base.conj), out::NDArray, input::NDArray{<:Real}) # conj(real_array) is just the array - return nda_unary_op!(out, cuNumeric.IDENTITY, input) + return nda_unary_op!(out, cuNumeric.COPY, input) end function Base.:(-)(input::NDArray{Bool}) diff --git a/src/scoping/accelerate.jl b/src/scoping/accelerate.jl new file mode 100644 index 000000000..3b0268c69 --- /dev/null +++ b/src/scoping/accelerate.jl @@ -0,0 +1,257 @@ +using MacroTools: MacroTools + +# Rejected everywhere: control flow makes last-use freeing unsound (a temp freed +# after its textual last use could be revived on another path). +const _CONTROL_FLOW_HEADS = (:if, :elseif, :for, :while, :try, :do, :break, :continue, :&&, :||) + +_is_function_lhs(::Any) = false +function _is_function_lhs(lhs::Expr) + lhs.head === :call && return true + lhs.head === :where && return _is_function_lhs(first(lhs.args)) + return false +end + +function _reject_nonstraightline(body) + MacroTools.postwalk(body) do node + node isa Expr || return node + if node.head === :function || node.head === :-> || + (node.head === :(=) && _is_function_lhs(first(node.args))) + error("@accelerate: nested/anonymous function definitions are not supported") + end + if node.head in _CONTROL_FLOW_HEADS + error( + "@accelerate: control flow (`$(node.head)`) is not supported; " * + "only straight-line code can be accelerated", + ) + end + return node + end + return nothing +end + +# Function args are caller-owned: protected roots, never freed or fused away. +function _argument_symbols(def) + names = Set{Symbol}() + for arg in Iterators.flatten((get(def, :args, Any[]), get(def, :kwargs, Any[]))) + name, _, _, _ = MacroTools.splitarg(arg) + name isa Symbol && push!(names, name) + end + return names +end + +# Function form: validate, expand `@.`, normalize trailing `return`, run the +# lifetime/fusion passes protecting `protected_roots` (the args). +function _accelerate_rewrite(body, caller::Module, protected_roots::Set{Symbol}) + _reject_nonstraightline(body) + body = _normalize_return(_expand_dot_macros(body, caller)) + on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing + return process_ndarray_scope(body; on_rewrite, protected_roots) +end + +# `begin`/expr form: 1:1 Julia scope (no `let`) — named bindings stay live. +# On GPU, same-shape chains fuse into one multi-output launch; otherwise only +# anonymous temporaries (slices) are freed. +function _accelerate_block_soft(block, caller::Module) + _reject_nonstraightline(block) + nb = _normalize_return(_expand_dot_macros(block, caller)) + on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing + fallback = process_ndarray_scope( + nb; on_rewrite, protected_roots=_assigned_symbols(nb) + ) + @static if FUSE_BROADCAST_EXPRS && HAS_CUDA + fused = _try_fuse_block_multi(nb) + if !isnothing(fused) + return quote + if cuNumeric._has_gpu_target() + $fused + else + $fallback + end + end + end + end + # Protect named bindings; free only anonymous temps. + return fallback +end + +# Flatten a `let` node's bindings + body into one statement block. +function _let_body(letexpr::Expr) + stmts = Any[] + for part in letexpr.args + if part isa Expr && part.head === :block + append!(stmts, part.args) + elseif part isa Expr && part.head === :(=) + push!(stmts, part) + elseif !isnothing(part) + push!(stmts, part) + end + end + return Expr(:block, stmts...) +end + +# `let` form: hard scope. Full analysis — combine single-use producers, free +# every non-returned temp — re-wrapped in a `let` so only the result escapes. +function _accelerate_block_hard(letexpr, caller::Module) + body = _let_body(letexpr) + _reject_nonstraightline(body) + nb = _normalize_return(_expand_dot_macros(body, caller)) + on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing + rewritten = process_ndarray_scope(nb; on_rewrite, protected_roots=Set{Symbol}()) + bindings = union(_assigned_symbols(nb), _assigned_symbols(rewritten)) + return _lexical_scope(rewritten, bindings) +end + +# Drop the leading dot: `.+` -> `+`, `.^` -> `^`. +_undot(op::Symbol) = Symbol(chop(string(op); head=1, tail=0)) + +# Dotted RHS -> lazy `Base.broadcasted(...)`: chain vars become `MatRef{k}`, +# slice leaves are hoisted into `hoisted` (temp => slice) to free post-launch. +# `nothing` when not lowerable (caller falls back). +function _to_broadcasted(expr, idx::AbstractDict{Symbol,Int}, hoisted::Vector) + if expr isa Symbol + haskey(idx, expr) && return :(cuNumeric.MatRef($(idx[expr]))) + return expr + end + expr isa Expr || return expr + if expr.head === :ref + # Slice temp: bail if it indexes a chain var, else hoist to free later. + any(s -> haskey(idx, s), walk_symbols(expr)) && return nothing + tmp = gensym(:slice) + push!(hoisted, tmp => expr) + return tmp + end + if expr.head === :call && expr.args[1] isa Symbol && _is_broadcast_op(expr.args[1]) + cargs = map(a -> _to_broadcasted(a, idx, hoisted), expr.args[2:end]) + any(isnothing, cargs) && return nothing + return Expr(:call, :(Base.broadcasted), _undot(expr.args[1]), cargs...) + end + if expr.head === :. && length(expr.args) == 2 && + expr.args[2] isa Expr && expr.args[2].head === :tuple + cargs = map(a -> _to_broadcasted(a, idx, hoisted), expr.args[2].args) + any(isnothing, cargs) && return nothing + return Expr(:call, :(Base.broadcasted), expr.args[1], cargs...) + end + # Non-dotted scalar leaf; unsafe if it reads a chain var as a scalar. + any(s -> haskey(idx, s), walk_symbols(expr)) && return nothing + return expr +end + +function _is_top_broadcast(rhs) + return ( + rhs isa Expr && rhs.head === :call && rhs.args[1] isa Symbol && + _is_broadcast_op(rhs.args[1]) + ) || + ( + rhs isa Expr && rhs.head === :. && length(rhs.args) == 2 && + rhs.args[2] isa Expr && rhs.args[2].head === :tuple + ) +end + +# SSA chain of `sym = ` (+ optional trailing return) -> one +# multi-output launch materializing each result. `nothing` -> caller falls back. +function _try_fuse_block_multi(block) + stmts = _scope_statements(block) + isnothing(stmts) && return nothing + stmts = filter(s -> !(s isa LineNumberNode), stmts) + isempty(stmts) && return nothing + + assigns = stmts + ret = nothing + if isnothing(_assignment(last(stmts))) + ret = last(stmts) + assigns = stmts[1:(end - 1)] + end + length(assigns) >= 2 || return nothing + + syms = Symbol[] + idx = Dict{Symbol,Int}() + seg_exprs = Any[] + hoisted = Pair{Symbol,Any}[] # slice temp => slice expr + for stmt in assigns + a = _assignment(stmt) + isnothing(a) && return nothing + a.lhs isa Symbol || return nothing # no indexed-assign in this path + a.lhs in syms && return nothing # SSA: no reassignment + _is_top_broadcast(a.rhs) || return nothing # must be a real broadcast + seg = _to_broadcasted(a.rhs, idx, hoisted) + isnothing(seg) && return nothing + push!(seg_exprs, seg) + push!(syms, a.lhs) + idx[a.lhs] = length(syms) + end + outs = gensym(:outs) + slice_binds = [:($t = $e) for (t, e) in hoisted] + slice_frees = [:(cuNumeric.maybe_insert_delete($t)) for (t, _) in hoisted] + binds = [:($(syms[i]) = $outs[$i]) for i in eachindex(syms)] + value = isnothing(ret) ? last(syms) : ret + return quote + $(slice_binds...) # materialize slice views + $outs = cuNumeric.copyto_fused_multi_alloc!(($(seg_exprs...),)) + $(slice_frees...) # free them after the launch + $(binds...) + $value + end +end + +# AST `@accelerate` emits (pre-`esc`); shared with `@show_lifetimes`. Dispatch: +# function def / `let` (hard scope) / `begin`-expr (soft, 1:1 Julia scope). +function _accelerate_expand(input, caller::Module) + if MacroTools.isdef(input) + def = MacroTools.splitdef(input) + def[:body] = _accelerate_rewrite(def[:body], caller, _argument_symbols(def)) + return MacroTools.combinedef(def) + elseif input isa Expr && input.head === :let + return _accelerate_block_hard(input, caller) + end + return _accelerate_block_soft(input, caller) +end + +@doc""" + @accelerate function f(args...) ... end + @accelerate begin ... end + @accelerate let ... end + @accelerate expr + +Optimize straight-line array code by coordinating CUDA broadcast fusion within +expressions, fusion across broadcast statements, and scope-aware cleanup of +materialized temporaries. Control flow and nested/anonymous functions are +rejected. Four forms determine which values must remain valid: + + * **function** (preferred): arguments and returned values are protected; + non-returned locals may fuse into consumers or be freed after their last use. + * **`begin`**: creates no new Julia scope, so every named binding stays live; + eligible GPU chains may use one multi-output kernel that materializes them. + * **`let`**: creates a local scope; only the result escapes, so other locals may + fuse away or be freed after their last use. + * **expression**: materializes and returns one expression; eligible operations + fuse within it and transient temporaries are released. + +```julia +@accelerate function step(u, v) # c may fuse away; the result is returned + c = u .* v + return c .^ 2 +end +a, b = @accelerate begin # a and b both stay live, one GPU launch + a = x .* y + b = a .+ 1 + (a, b) +end +result = @accelerate (x .+ y .* z) +``` +""" +macro accelerate(input) + return esc(_accelerate_expand(input, __module__)) +end + +@doc""" + @show_lifetimes function f(args...) ... end + @show_lifetimes begin ... end + @show_lifetimes let ... end + +Print the exact expansion [`@accelerate`](@ref) produces for the same input +(all forms), without running it; inserted frees are highlighted. Pure AST work. +""" +macro show_lifetimes(input) + expansion = _accelerate_expand(input, __module__) + return :(print_lifetime_analysis($(QuoteNode(expansion)))) +end diff --git a/src/scoping/broadcast_lifetimes.jl b/src/scoping/broadcast_lifetimes.jl index 47bb6b301..f1228f58c 100644 --- a/src/scoping/broadcast_lifetimes.jl +++ b/src/scoping/broadcast_lifetimes.jl @@ -115,9 +115,11 @@ function rewrite_broadcast_lifetimes(scope) return _prepend_statements(rewritten, temps), assigned_vars end -function process_broadcast_lifetime_scope(scope; on_rewrite=nothing) - # Returned producers must stay materialized, so exempt them from fusion. - protected = _returned_symbols(scope) +function process_broadcast_lifetime_scope( + scope; on_rewrite=nothing, protected_roots=Set{Symbol}() +) + # Returned producers and caller-owned roots stay materialized: exempt from fusion. + protected = union(_returned_symbols(scope), protected_roots) scope = InterBroadcastFusion.rewrite_scope(scope; on_rewrite, protected) - return _process_lifetime_scope(scope, rewrite_broadcast_lifetimes) + return _process_lifetime_scope(scope, rewrite_broadcast_lifetimes; protected_roots) end diff --git a/src/scoping/lifetimes.jl b/src/scoping/lifetimes.jl index b5b88469c..7f3d97db4 100644 --- a/src/scoping/lifetimes.jl +++ b/src/scoping/lifetimes.jl @@ -70,6 +70,6 @@ function rewrite_eager_lifetimes(scope) return _prepend_statements(rewritten, temps), assigned_vars end -function process_lifetime_scope(scope) - return _process_lifetime_scope(scope, rewrite_eager_lifetimes) +function process_lifetime_scope(scope; protected_roots=Set{Symbol}()) + return _process_lifetime_scope(scope, rewrite_eager_lifetimes; protected_roots) end diff --git a/src/scoping/scoping.jl b/src/scoping/scoping.jl index 676708b30..dec17913f 100644 --- a/src/scoping/scoping.jl +++ b/src/scoping/scoping.jl @@ -1,4 +1,4 @@ -export @analyze_lifetimes, @show_lifetimes +export @accelerate, @show_lifetimes # Include generic syntax layers before the cuNumeric-specific lifetime passes. include("util.jl") @@ -44,9 +44,7 @@ function _normalize_return(block) for (i, stmt) in enumerate(stmts) stmt isa Expr && stmt.head === :return || continue i == length(stmts) || throw( - ArgumentError( - "@analyze_lifetimes: `return` is only allowed as the block's final statement" - ), + ArgumentError("`return` is only allowed as the final statement") ) value = isempty(stmt.args) ? :nothing : only(stmt.args) return Expr(block.head, stmts[1:(end - 1)]..., value) @@ -54,47 +52,6 @@ function _normalize_return(block) return block end -@doc""" - @analyze_lifetimes expr - -Wraps a block of code so that all temporary `NDArray` allocations -(e.g. from slicing or function calls) are tracked and safely freed -at the end of the block. Ensures proper cleanup of GPU memory by -inserting `maybe_insert_delete` calls automatically. - -Assignments created inside the macro are scoped to its lexical region. Existing -arrays can still be mutated in place, and the final value of the block is -returned, but internal bindings do not leak into the surrounding scope. - -The block's final statement determines what leaves the region. Any binding it -returns (a bare name or the elements of a returned tuple) is both protected from -the automatic free and, under fusion, kept materialized rather than inlined into -its consumer, so a real `NDArray` escapes rather than a lazy broadcast tree: - - x, y = @analyze_lifetimes begin - x = e1 .+ e2 - c = x .* e1 # not returned, single-use -> fused into y - y = c .^ 2 - (x, y) # returned -> x and y stay materialized - end - -A trailing `return expr` is accepted as an explicit spelling of the final -statement (`return (x, y)` above); a `return` anywhere else is an error. - -When broadcast fusion is enabled (`FUSE_BROADCAST_EXPRS`), dotted operators -(`.+`, `.*`, etc.) form a lazy `Base.Broadcast.Broadcasted` tree compiled into -a single PTX kernel; intermediate nodes are not real `NDArray` allocations and -are not individually hoisted. The macro automatically selects the -broadcast-aware analysis in that case and the plain analysis otherwise. -""" -macro analyze_lifetimes(block) - block = _normalize_return(_expand_dot_macros(block, __module__)) - on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing - rewritten = process_ndarray_scope(block; on_rewrite) - bindings = union(_assigned_symbols(block), _assigned_symbols(rewritten)) - return esc(_lexical_scope(rewritten, bindings)) -end - const counter = Ref(0) function maybe_insert_delete(var::NDArray) @@ -103,10 +60,8 @@ end maybe_insert_delete(x) = x -# `@analyze_lifetimes` is an ownership region, analogous to a C++ `{ ... }` -# block. Bind every source and generated assignment explicitly so it cannot -# accidentally reuse or leak a caller local with the same name. Indexed and -# broadcast assignments are mutations, not new bindings, and remain visible. +# Symbols bound by an assignment anywhere in `expr` (let-form locals; soft-form +# protected bindings). function _assigned_symbols(expr) assigned = Set{Symbol}() @@ -124,9 +79,7 @@ function _assigned_symbols(expr) function visit(node) node isa Expr || return nothing assignment = _assignment(node) - if !isnothing(assignment) - collect_binding(assignment.lhs) - end + isnothing(assignment) || collect_binding(assignment.lhs) foreach(visit, node.args) return nothing end @@ -140,20 +93,6 @@ function _lexical_scope(body, bindings::Set{Symbol}) return Expr(:let, Expr(:block, ordered...), body) end -function _register_scoping_error_hint!() - isdefined(Base.Experimental, :register_error_hint) || return nothing - Base.Experimental.register_error_hint(UndefVarError) do io, exc - return print( - io, - "\nHint: bindings assigned inside `@analyze_lifetimes` are local to its " * - "block. If `", - exc.var, - "` was created there, return it from the block to use it afterward.", - ) - end - return nothing -end - function _hoist_temporary(expr, assigned_vars) counter[] += 1 temporary = Symbol(:tmp, counter[]) @@ -202,7 +141,9 @@ end insert_finalizers(stmts::Vector) Insert `cuNumeric.maybe_insert_delete(var)` after the last use of each temporary variable. """ -function insert_finalizers(exprs::Vector, assigned_vars::Set{Symbol}) +function insert_finalizers( + exprs::Vector, assigned_vars::Set{Symbol}; protected_roots::Set{Symbol}=Set{Symbol}() +) last_use = Dict{Symbol,Int}() alias_map = Dict{Symbol,Symbol}() @@ -261,7 +202,8 @@ function insert_finalizers(exprs::Vector, assigned_vars::Set{Symbol}) # return `nothing` rather than leak it or hand back a dangling handle. terminal_indexed = n > 0 && is_indexed_assign(stmts[n]) - protected = Set{Symbol}() + # Roots (function args) are protected regardless of the terminal statement. + protected = Set{Symbol}(canon(root) for root in protected_roots) if n > 0 && !terminal_indexed for result in _result_symbols(stmts[n]) push!(protected, canon(result)) @@ -315,16 +257,20 @@ end insert_finalizers(block::Expr) Apply finalizer insertion to a `begin ... end` or `:block` expression. """ -function insert_finalizers(block::Expr, assigned_vars::Set{Symbol}) +function insert_finalizers( + block::Expr, assigned_vars::Set{Symbol}; protected_roots::Set{Symbol}=Set{Symbol}() +) stmts = _scope_statements(block) isnothing(stmts) && error("Expected a begin/block expression") - return Expr(:block, insert_finalizers(stmts, assigned_vars)...) + return Expr(:block, insert_finalizers(stmts, assigned_vars; protected_roots)...) end -function _process_lifetime_scope(scope, rewrite_lifetimes) +function _process_lifetime_scope( + scope, rewrite_lifetimes; protected_roots::Set{Symbol}=Set{Symbol}() +) try rewritten, assigned_vars = rewrite_lifetimes(scope) - return insert_finalizers(rewritten, assigned_vars) + return insert_finalizers(rewritten, assigned_vars; protected_roots) finally counter[] = 0 end @@ -335,13 +281,15 @@ end include("lifetimes.jl") include("broadcast_lifetimes.jl") -function process_ndarray_scope(scope; on_rewrite=nothing) +function process_ndarray_scope( + scope; on_rewrite=nothing, protected_roots::Set{Symbol}=Set{Symbol}() +) # Broadcast expressions stay lazy only when fusion is enabled; otherwise # every call is analyzed as an eager allocation. @static if FUSE_BROADCAST_EXPRS - return process_broadcast_lifetime_scope(scope; on_rewrite) + return process_broadcast_lifetime_scope(scope; on_rewrite, protected_roots) end - return process_lifetime_scope(scope) + return process_lifetime_scope(scope; protected_roots) end # Return the deleted value for a generated finalizer call. @@ -354,12 +302,27 @@ function _delete_argument(expr) return only(call.args) end -function print_lifetime_analysis(block; io::IO=stdout) +# Header + body statements per form, so the printout mirrors the real expansion. +function _analysis_parts(ex) + ex = _strip_lines(ex) + if ex isa Expr && ex.head === :function + return "function " * string(first(ex.args)), _flatten_statements(ex.args[2]) + elseif ex isa Expr && ex.head === :let + binds = _strip_lines(first(ex.args)) + bindstr = binds isa Expr ? join(binds.args, ", ") : string(binds) + return "let " * bindstr, _flatten_statements(ex.args[2]) + end + return nothing, _flatten_statements(ex) +end + +# Pretty-print `expansion` (the exact `@accelerate` output), highlighting frees. +function print_lifetime_analysis(expansion; io::IO=stdout) rule = "-"^60 - stmts = _flatten_statements(process_ndarray_scope(block)) + header, stmts = _analysis_parts(expansion) mode = FUSE_BROADCAST_EXPRS ? "fusion-aware" : "plain" - println(io, "@analyze_lifetimes expansion ($mode analysis)\n", rule) + println(io, "@accelerate expansion ($mode)\n", rule) + isnothing(header) || println(io, header) n = 0 for s in stmts @@ -368,24 +331,13 @@ function print_lifetime_analysis(block; io::IO=stdout) printstyled(io, lpad("✗ free ", 11), deleted, "\n"; color=:red) else n += 1 - println(io, lpad(n, 4), " ", s) + println(io, lpad(n, 4), " ", _strip_lines(s)) end end + isnothing(header) || println(io, "end") println(io, rule) return nothing end -@doc""" - @show_lifetimes expr - -Print the lifetime-analysis rewrite of `expr` — the same transformation -[`@analyze_lifetimes`](@ref) applies — without running it. Every statement is -shown in source order and each inserted `maybe_insert_delete` is highlighted so -you can see exactly where each temporary is freed. Pure AST work, so it runs on -CPU-only checkouts. -""" -macro show_lifetimes(block) - block = _normalize_return(_expand_dot_macros(block, __module__)) - return :(print_lifetime_analysis($(QuoteNode(block)))) -end +include("accelerate.jl") diff --git a/test/Project.toml b/test/Project.toml index 73f4b047f..6e9f7cd1a 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -1,7 +1,7 @@ [deps] CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" -CUDA_Driver_jll = "4ee394cb-3365-5eb0-8335-949819d2adfc" +CUDACore = "bd0ed864-bdfe-4181-a5ed-ce625a5fdea2" InteractiveUtils = "b77e0a4c-d291-57a0-90e8-8db25a27a240" LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc" diff --git a/test/analysis/accelerate.jl b/test/analysis/accelerate.jl new file mode 100644 index 000000000..28eee8897 --- /dev/null +++ b/test/analysis/accelerate.jl @@ -0,0 +1,190 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +# Coverage of the four `@accelerate` forms and their scope contracts: +# @accelerate function f(...) ... end -> function scope, frees non-returned +# @accelerate begin ... end -> 1:1 Julia scope, bindings stay alive +# @accelerate let ... end -> hard scope, combine + free non-returned +# @accelerate expr -> materialized result, temps freed + +using InteractiveUtils: code_typed + +@testset "@accelerate — four forms" begin + T = Float32 + N = 64 + _nd(v) = @allowscalar NDArray(v) + approx(x, ref) = isapprox(Array(x), ref; rtol=1.0f-4) + ja = my_rand(T, N) + jb = my_rand(T, N) + + @testset "1. function form" begin + @accelerate function _acc_fsq(a, b) + c = a .* b + return c .^ 2 + end + a = _nd(ja) + b = _nd(jb) + @test approx(_acc_fsq(a, b), (ja .* jb) .^ 2) + # Arguments are caller-owned: a second call on the same inputs still works. + @test approx(_acc_fsq(a, b), (ja .* jb) .^ 2) + end + + @testset "3. let form (hard scope)" begin + function _acc_let(a, b) + s = @accelerate let + r = a .+ b + s = r .* T(2) + s + end + return s, @isdefined(r) + end + s, r_leaked = _acc_let(_nd(ja), _nd(jb)) + @test approx(s, (ja .+ jb) .* T(2)) + @test r_leaked == false # `r` must not escape the let scope + end + + @testset "4. expr form" begin + a = _nd(ja) + b = _nd(jb) + res = @accelerate (a .+ b) .^ 2 + @test res isa NDArray + @test approx(res, (ja .+ jb) .^ 2) + end + + @testset "2. begin form (bindings stay alive)" begin + function _acc_begin(a, b) + q = @accelerate begin + p = a .* b + q = p .+ one(T) + q + end + return p, q # both must be defined in this scope + end + p, q = _acc_begin(_nd(ja), _nd(jb)) + @test approx(p, ja .* jb) + @test approx(q, (ja .* jb) .+ one(T)) + + # A nested `let` keeps its intermediate private while the outer block + # can consume and return the value it produces. + a = _nd(ja) + b = _nd(jb) + one_t = one(T) + shifted, x = @accelerate begin + shifted = let + product = @. a * b + @. product + one_t + end + x = @. shifted * 2 + (shifted, x) + end + @test approx(shifted, (ja .* jb) .+ one(T)) + @test approx(x, ((ja .* jb) .+ one(T)) .* 2) + end + + @testset "expansion contracts (white-box)" begin + expand(ex) = cuNumeric._accelerate_expand(ex, @__MODULE__) + hasfree(ex) = occursin("maybe_insert_delete", string(expand(ex))) + + # Scope shape per form. + @test expand(:(function f(a) + ;c = a .* a; + c .^ 2; + end)).head === :function + @test expand(:( + begin + C .= a[2:end] .+ b[2:end] + end + )).head === :block + @test expand(:( + let + r = a .+ b; + r .* 2 + end + )).head === :let + + # Slices are freed in every non-`let` form (uniform cleanup). + @test hasfree(:(function f(a) + ;s = a[2:end]; + s .+ 1; + end)) + @test hasfree(:( + begin + C .= a[2:end] .+ b[2:end] + end + )) + + if cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric.HAS_CUDA + # A same-shape chain fuses into one multi-output launch and still + # frees the hoisted slice temporaries. + mo = string(expand(:( + begin + p = a[2:end] .* b[2:end] + q = p .+ 1 + q + end + ))) + @test occursin("copyto_fused_multi_alloc!", mo) + @test occursin("maybe_insert_delete", mo) + else + # Multi-output fusion is GPU-only; CPU expansion must use the + # ordinary broadcast path even when fusion is enabled in preferences. + cpu = string(expand(:( + begin + p = a .* b + q = p .+ 1 + q + end + ))) + @test !occursin("copyto_fused_multi_alloc!", cpu) + end + end + + @testset "multi-output segment runner is fully unrolled" begin + # GPU compilation requires every chained segment call to be statically + # dispatched. This three-segment shape crossed Julia 1.10's recursive + # inference limit when `_run_segments` recursed over `Base.tail`. + segs = ( + (+, (cuNumeric.RuntimeBroadcastArg{1}(), cuNumeric.RuntimeBroadcastArg{2}())), + (*, (cuNumeric.LocalBroadcastArg{1}(), cuNumeric.RuntimeBroadcastArg{1}())), + (^, (cuNumeric.LocalBroadcastArg{2}(), cuNumeric.RuntimeBroadcastArg{3}())), + ) + outs = ntuple(_ -> zeros(T, 2, 2), 3) + runtime_args = (ones(T, 2, 2), ones(T, 2, 2), 2) + + @test @inferred( + cuNumeric._run_segments( + segs, outs, runtime_args, (), (), CartesianIndex(1, 1) + ) + ) === nothing + @test getindex.(outs, Ref(CartesianIndex(1, 1))) == (T(2), T(2), T(4)) + + argtypes = ( + typeof(segs), + typeof(outs), + typeof(runtime_args), + Tuple{}, + Tuple{}, + CartesianIndex{2}, + ) + typed = only(code_typed(cuNumeric._run_segments, argtypes; optimize=true)).first + @test !occursin( + "_run_segments", sprint(show, MIME("text/plain"), typed) + ) + end +end diff --git a/test/analysis/promotion.jl b/test/analysis/promotion.jl index 7b1fb39a7..422e5766a 100644 --- a/test/analysis/promotion.jl +++ b/test/analysis/promotion.jl @@ -19,3 +19,8 @@ @test safe_compare(r1, r2, atol(Float64), rtol(Float64)) end end + +@testset "Flattened associative broadcast promotion" begin + @test @inferred(cuNumeric.__checked_promote_op(+, NTuple{5,Float64})) === Float64 + @test @inferred(cuNumeric.__checked_promote_op(*, NTuple{4,Int32})) === Int32 +end diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index 9d0240749..ce81c1cb0 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -17,6 +17,32 @@ * Ethan Meitz =# +@accelerate function _type_stable_accelerate_function(a, b) + intermediate = @. a + b + return @. intermediate * 2.0f0 +end + +function _type_stable_accelerate_begin(a, b) + return @accelerate begin + intermediate = @. a + b + result = @. intermediate * 2.0f0 + (intermediate, result) + end +end + +function _type_stable_accelerate_let(a, b) + return @accelerate let + intermediate = @. a + b + @. intermediate * 2.0f0 + end +end + +function _type_stable_accelerate_expr(a, b) + return @accelerate (@. (a + b) * 2.0f0) +end + +_type_stable_cuda_argtypes(task::cuNumeric.CUDATask) = task.argtypes + @testset verbose = true "core" begin a = cuNumeric.zeros(5) b = cuNumeric.zeros(Float64, 3, 4) @@ -66,6 +92,18 @@ end @test @inferred(cuNumeric.NDArray(rand(Float32, 3, 3))) !== nothing end +@testset verbose = true "custom CUDA metadata" begin + task = cuNumeric.CUDATask("kernel", (Float32, Int32)) + @test isconcretetype(typeof(task)) + @test all(isconcretetype, fieldtypes(typeof(task))) + @test @inferred(_type_stable_cuda_argtypes(task)) == DataType[Float32, Int32] + + storage_type = cuNumeric.PaddedStorage{Float32,1} + @test all(isconcretetype, Base.uniontypes(fieldtype(storage_type, :backing))) + @test all(isconcretetype, Base.uniontypes(fieldtype(storage_type, :staging))) + @test all(isconcretetype, Base.uniontypes(fieldtype(storage_type, :shape))) +end + @testset verbose = true "conversion" begin # cast to array, as_type a = cuNumeric.zeros(Float64, 5, 5) @@ -104,6 +142,23 @@ end @test @inferred(((a .* b) .+ a) .* 2.0f0) !== nothing end +@testset verbose = true "@accelerate forms" begin + a = cuNumeric.ones(Float32, 3, 3) + b = cuNumeric.ones(Float32, 3, 3) + + function_result = @inferred _type_stable_accelerate_function(a, b) + @test function_result isa NDArray{Float32,2} + + begin_result = @inferred _type_stable_accelerate_begin(a, b) + @test begin_result isa Tuple{NDArray{Float32,2},NDArray{Float32,2}} + + let_result = @inferred _type_stable_accelerate_let(a, b) + @test let_result isa NDArray{Float32,2} + + expr_result = @inferred _type_stable_accelerate_expr(a, b) + @test expr_result isa NDArray{Float32,2} +end + @testset verbose = true "solve" begin # native float/complex, 2D and 1D rhs @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) diff --git a/test/array/broadcast_basic.jl b/test/array/broadcast_basic.jl index e9eb95c04..16b7f710b 100644 --- a/test/array/broadcast_basic.jl +++ b/test/array/broadcast_basic.jl @@ -68,6 +68,18 @@ end result_cpu = zeros(dims) @test result == result_cpu + # Plain broadcast assignment lowers to identity.(source). It must copy + # into both dense NDArrays and writable views. + result .= arrA + @test result == arrA_cpu + + parent = cuNumeric.zeros(Float64, N + 2, N + 2) + center = parent[2:(N + 1), 2:(N + 1)] + center .= arrA + expected_parent = zeros(N + 2, N + 2) + expected_parent[2:(N + 1), 2:(N + 1)] .= arrA_cpu + @test parent == expected_parent + # where the real testing starts arrA = 13.74 .- arrA arrA_cpu = 13.74 .- arrA_cpu @@ -125,6 +137,21 @@ end result_cpu = arrA_cpu .* arrB_cpu @test result == result_cpu + # `@.` lowers associative chains to n-ary Broadcasted nodes. Both the + # fused GPU path and pairwise unfused CPU path must accept them. + result = @. arrA + arrB + arrA + arrB + arrA + result_cpu = @. arrA_cpu + arrB_cpu + arrA_cpu + arrB_cpu + arrA_cpu + @test result == result_cpu + + result = @. 0.5 * arrA * arrB * arrA + result_cpu = @. 0.5 * arrA_cpu * arrB_cpu * arrA_cpu + @test result == result_cpu + + dx = 0.1 + result = @. arrA / dx^2 + result_cpu = @. arrA_cpu / dx^2 + @test result == result_cpu + operator(arrA, arrB) operator(arrA_cpu, arrB_cpu) @test arrA == arrA_cpu diff --git a/test/defunct/fusion_compare.jl b/test/cuda.jl/fusion_compare.jl similarity index 87% rename from test/defunct/fusion_compare.jl rename to test/cuda.jl/fusion_compare.jl index dd4934fca..be0d5090e 100644 --- a/test/defunct/fusion_compare.jl +++ b/test/cuda.jl/fusion_compare.jl @@ -1,3 +1,9 @@ +using CUDA: CUDA, @cuda +using CUDACore: blockDim, blockIdx, threadIdx +import CUDACore: i32 + +cuNumeric.Experimental(true) + function unfused_cunumeric(u, v, f, k) F_u = ( ( @@ -101,8 +107,8 @@ function run_unfused_baseline(N, u, v) end function fusion_test(; N=1024, atol=1.0f-6, rtol=1.0f-6) - u = cuNumeric.as_type(cuNumeric.random(Float32, (N, N)), Float32) - v = cuNumeric.as_type(cuNumeric.random(Float32, (N, N)), Float32) + u = cuNumeric.rand(Float32, (N, N)) + v = cuNumeric.rand(Float32, (N, N)) # using CUDA u_base = CUDA.rand(Float32, (N, N)) @@ -117,8 +123,14 @@ function fusion_test(; N=1024, atol=1.0f-6, rtol=1.0f-6) Fu_fused, Fv_fused = run_fused_cunumeric(N, u, v) Fu_unfused, Fv_unfused = run_unfused_cunumeric(N, u, v) - @test isapprox(Fu_fused, Fu_unfused; atol=atol, rtol=rtol) - @test isapprox(Fv_fused, Fv_unfused; atol=atol, rtol=rtol) + @test isapprox(Array(Fu_fused), Array(Fu_unfused); atol=atol, rtol=rtol) + @test isapprox(Array(Fv_fused), Array(Fv_unfused); atol=atol, rtol=rtol) end -fusion_test() +try + @testset "2D fusion comparison" begin + fusion_test() + end +finally + cuNumeric.Experimental(false) +end diff --git a/test/defunct/fusion_compare_1d.jl b/test/cuda.jl/fusion_compare_1d.jl similarity index 85% rename from test/defunct/fusion_compare_1d.jl rename to test/cuda.jl/fusion_compare_1d.jl index 8163a5db1..8efab2989 100644 --- a/test/defunct/fusion_compare_1d.jl +++ b/test/cuda.jl/fusion_compare_1d.jl @@ -1,4 +1,10 @@ +using CUDA: CUDA, @cuda +using CUDACore: blockDim, blockIdx, threadIdx +import CUDACore: i32 + +cuNumeric.Experimental(true) + function unfused_cunumeric(u, v, f, k) F_u = ( ( @@ -102,8 +108,8 @@ function run_unfused_baseline(N, u, v) end function fusion_test(; N=1024*1024, atol=1.0f-6, rtol=1.0f-6) - u = cuNumeric.as_type(cuNumeric.rand(NDArray, N), Float32) - v = cuNumeric.as_type(cuNumeric.rand(NDArray, N), Float32) + u = cuNumeric.rand(Float32, N) + v = cuNumeric.rand(Float32, N) # using CUDA u_base = CUDA.rand(Float32, N) @@ -116,8 +122,14 @@ function fusion_test(; N=1024*1024, atol=1.0f-6, rtol=1.0f-6) # using cuNumeric Fu_fused, Fv_fused = run_fused_cunumeric(N, u, v) Fu_unfused, Fv_unfused = run_unfused_cunumeric(N, u, v) - @test isapprox(Fu_fused, Fu_unfused; atol=atol, rtol=rtol) - @test isapprox(Fv_fused, Fv_unfused; atol=atol, rtol=rtol) + @test isapprox(Array(Fu_fused), Array(Fu_unfused); atol=atol, rtol=rtol) + @test isapprox(Array(Fv_fused), Array(Fv_unfused); atol=atol, rtol=rtol) end -fusion_test() +try + @testset "1D fusion comparison" begin + fusion_test() + end +finally + cuNumeric.Experimental(false) +end diff --git a/test/cuda.jl/padding.jl b/test/cuda.jl/padding.jl new file mode 100644 index 000000000..f38e875a8 --- /dev/null +++ b/test/cuda.jl/padding.jl @@ -0,0 +1,224 @@ +#= Copyright 2025 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +#= Purpose of test: cuda + -- Validate custom-kernel padding, synchronization, and lifetime management +=# + +using CUDACore: blockDim, blockIdx, threadIdx +import CUDACore: i32 + +cuNumeric.Experimental(true) + +function padding_add(a, b, c, N) + i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x + if i <= N + @inbounds c[i] = a[i] + b[i] + end + return nothing +end + +function padding_mul(a, c, b, N) + i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x + if i <= N + @inbounds b[i] = a[i] * c[i] + end + return nothing +end + +function cuda_padding_lifetime() + N = 1_000_000 + M = N - 2 + threads = 256 + blocks = cld(M, threads) + initial_bytes = cuNumeric.current_device_bytes[] + + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + c = cuNumeric.zeros(Float32, M) + task = cuNumeric.@cuda_task padding_add(a, b, c, UInt32(M)) + unpadded_bytes = cuNumeric.current_device_bytes[] + + try + @test @inferred( + cuNumeric.launch( + task, (a, b), c, UInt32(M); threads=threads, blocks=blocks + ) + ) === nothing + padded_bytes = cuNumeric.current_device_bytes[] + + @test @inferred(cuNumeric._launch_shape(c)) == (N,) + @test @inferred(cuNumeric._sync_to_launch_padding!(c)) === nothing + @test @inferred(cuNumeric._sync_from_launch_padding!(c)) === nothing + + accounting_ok = true + for _ in 1:15 + cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=UInt32( + M + ) + accounting_ok &= cuNumeric.current_device_bytes[] == padded_bytes + end + + @test padded_bytes > unpadded_bytes + @test accounting_ok + @test size(c) == (M,) + @test all(Array(c) .== 2.0f0) + finally + cuNumeric.destroy!(a) + cuNumeric.destroy!(b) + cuNumeric.destroy!(c) + end + @test cuNumeric.current_device_bytes[] == initial_bytes +end + +function cuda_padding_api_interop() + N = 4096 + M = N - 2 + threads = 256 + blocks = cld(M, threads) + initial_bytes = cuNumeric.current_device_bytes[] + + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + c = cuNumeric.zeros(Float32, M) + library_result = nothing + + try + task = cuNumeric.@cuda_task padding_add(a, b, c, UInt32(M)) + cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=UInt32( + M + ) + + # The broadcast writes through c's logical view into its padded backing. + c .= c .* 2.0f0 .+ 0.0f0 + @test all(Array(c) .== 4.0f0) + + # Non-broadcasted operators also consume the logical shape. + library_result = c + c + @test size(library_result) == (M,) + @test all(Array(library_result) .== 8.0f0) + + # A later custom launch sees the values written by the regular API. + task = cuNumeric.@cuda_task padding_mul(a, c, b, UInt32(M)) + cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, c) outputs=b scalars=UInt32( + M + ) + result = Array(b) + @test all(result[1:M] .== 4.0f0) + @test all(result[(M + 1):N] .== 1.0f0) + finally + !isnothing(library_result) && cuNumeric.destroy!(library_result) + cuNumeric.destroy!(a) + cuNumeric.destroy!(b) + cuNumeric.destroy!(c) + end + @test cuNumeric.current_device_bytes[] == initial_bytes +end + +function cuda_padding_slice_output() + N = 4096 + M = N - 2 + threads = 256 + blocks = cld(M, threads) + initial_bytes = cuNumeric.current_device_bytes[] + + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + parent = cuNumeric.zeros(Float32, N) + output = parent[1:M] + task = cuNumeric.@cuda_task padding_add(a, b, output, UInt32(M)) + unpadded_bytes = cuNumeric.current_device_bytes[] + + try + @test @inferred( + cuNumeric.launch( + task, (a, b), output, UInt32(M); threads=threads, blocks=blocks + ) + ) === nothing + padded_bytes = cuNumeric.current_device_bytes[] + @test @inferred(cuNumeric._launch_shape(output)) == (N,) + @test @inferred(cuNumeric._sync_from_launch_padding!(output)) === nothing + + # Mutate the logical parent view before reusing it as a custom-kernel input. + output .= output .* 1.0f0 .+ 1.0f0 + library_result = output + output + @test all(Array(library_result) .== 6.0f0) + cuNumeric.destroy!(library_result) + + task = cuNumeric.@cuda_task padding_mul(a, output, b, UInt32(M)) + @test @inferred( + cuNumeric.launch( + task, (a, output), b, UInt32(M); threads=threads, blocks=blocks + ) + ) === nothing + values = Array(parent) + product = Array(b) + + @test padded_bytes > unpadded_bytes + @test cuNumeric.current_device_bytes[] == padded_bytes + @test all(values[1:M] .== 3.0f0) + @test all(product[1:M] .== 3.0f0) + @test values[end] == 0.0f0 + finally + cuNumeric.destroy!(output) + cuNumeric.destroy!(parent) + cuNumeric.destroy!(a) + cuNumeric.destroy!(b) + end + @test cuNumeric.current_device_bytes[] == initial_bytes +end + +Base.@noinline function drop_padded_arrays() + N = 4096 + M = N - 2 + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + c = cuNumeric.zeros(Float32, M) + task = cuNumeric.@cuda_task padding_add(a, b, c, UInt32(M)) + cuNumeric.@launch task=task threads=256 blocks=cld(M, 256) inputs=(a, b) outputs=c scalars=UInt32( + M + ) + return nothing +end + +function cuda_padding_finalizer() + GC.gc(true) + cuNumeric.drain_pending_frees!() + baseline = cuNumeric.current_device_bytes[] + + drop_padded_arrays() + allocated = cuNumeric.current_device_bytes[] + GC.gc(true) + GC.gc(true) + cuNumeric.drain_pending_frees!() + + @test allocated > baseline + @test cuNumeric.current_device_bytes[] == baseline +end + +try + @testset "Custom CUDA padding" begin + cuda_padding_lifetime() + cuda_padding_api_interop() + cuda_padding_slice_output() + cuda_padding_finalizer() + end +finally + cuNumeric.Experimental(false) +end diff --git a/test/defunct/vecadd.jl b/test/cuda.jl/vecadd.jl similarity index 81% rename from test/defunct/vecadd.jl rename to test/cuda.jl/vecadd.jl index ea01e7ab3..e2b5de552 100644 --- a/test/defunct/vecadd.jl +++ b/test/cuda.jl/vecadd.jl @@ -21,6 +21,11 @@ -- Register various custom kernels using CUDA.jl =# +using CUDACore: blockDim, blockIdx, threadIdx +import CUDACore: i32 + +cuNumeric.Experimental(true) + function kernel_add(a, b, c, N) i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x if i <= N @@ -29,9 +34,8 @@ function kernel_add(a, b, c, N) return nothing end -# testing a second kernel -# on purpose switching inputs and outputs -function kernel_mul(a, b, c, N) +# Test a second kernel with `c` as an input and `b` as the output. +function kernel_mul(a, c, b, N) i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x if i <= N @inbounds b[i] = a[i] * c[i] @@ -70,18 +74,16 @@ function cuda_binaryop(max_diff) N ) - @test @allowscalar cuNumeric.compare(c, c_cpu, atol(Float32), rtol(Float32)) + @test @allowscalar cuNumeric.compare(c, c_cpu, max_diff, max_diff) - for i in 1:N - @allowscalar b[i] = a[i] * c[i] - end + b_cpu .= a_cpu .* c_cpu - task = cuNumeric.@cuda_task kernel_mul(a, b, c, UInt32(1)) + task = cuNumeric.@cuda_task kernel_mul(a, c, b, UInt32(1)) cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, c) outputs=b scalars=UInt32( N ) - @test @allowscalar cuNumeric.compare(b, b_cpu, atol(Float32), rtol(Float32)) + @test @allowscalar cuNumeric.compare(b, b_cpu, max_diff, max_diff) end function kernel_sin(a, b, N) @@ -119,5 +121,14 @@ function cuda_unaryop(max_diff) # TODO explore getting inplace ops working. cuNumeric.@launch task=task threads=threads blocks=blocks inputs=a outputs=b scalars=UInt32(N) - @test @allowscalar cuNumeric.compare(b, b_cpu, atol(Float32), rtol(Float32)) + @test @allowscalar cuNumeric.compare(b, b_cpu, max_diff, max_diff) +end + +try + @testset "Custom CUDA kernels" begin + cuda_binaryop(1.0f-5) + cuda_unaryop(1.0f-5) + end +finally + cuNumeric.Experimental(false) end diff --git a/test/gpu_only/broadcast_fusion.jl b/test/gpu_only/broadcast_fusion.jl index 5c92911c8..8ab3afb3d 100644 --- a/test/gpu_only/broadcast_fusion.jl +++ b/test/gpu_only/broadcast_fusion.jl @@ -203,7 +203,7 @@ _broadcast_fusion_user_add(x, y) = x + y @testset "z .= scalar * f.(A, B)" begin expected = T(2.0) .* (julia_a .+ julia_b) z = cuNumeric.zeros(T, (N,)) - @analyze_lifetimes begin + @accelerate begin z .= T(2.0) .* _broadcast_fusion_user_add.(a, b) end @allowscalar @test safe_compare(expected, z, atol, rtol) @@ -566,7 +566,7 @@ end ja = reshape(T.(1:(N * N)), N, N) a = @allowscalar NDArray(ja) out = cuNumeric.zeros(T, (N + 2, N + 2)) - @analyze_lifetimes begin + @accelerate begin producer = a .* s1 out[2:(end - 1), 2:(end - 1)] = producer .+ s2 end diff --git a/test/runtests.jl b/test/runtests.jl index da06059c3..a7ca7bbb7 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -1,28 +1,18 @@ using cuNumeric -using CUDA: CUDA +using CUDACore: CUDACore using ParallelTestRunner using Pkg using InteractiveUtils: versioninfo -run_gpu_tests = CUDA.functional() +run_gpu_tests = CUDACore.functional() @info "Julia information:\n" * sprint(io -> versioninfo(io)) -run_gpu_tests && @info "CUDA information:\n" * sprint(io -> CUDA.versioninfo(io)) @info "cuNumeric information:\n" * sprint(io -> cuNumeric.versioninfo(io)) # Forcibly precompile the current environment in parallel: Pkg sometimes ignores # dependencies pointed through via `[sources]` Pkg.precompile() -cuda_init = if run_gpu_tests - quote - using CUDA - import CUDA: i32 - end -else - :() -end - const init_code = quote using LinearAlgebra using Random @@ -32,8 +22,6 @@ const init_code = quote ENV["LEGATE_SKIP_RUNTIME"] = "false" using cuNumeric - $cuda_init - include("util.jl") end @@ -44,16 +32,23 @@ delete!(testsuite, "util") delete!(testsuite, "array/unary/tests") delete!(testsuite, "array/binary/tests") -if !run_gpu_tests - @warn "CUDA GPU not available, skipping GPU-only tests" - filter!(test -> !startswith(first(test), "gpu_only/"), testsuite) -end +test_args = parse_args(ARGS) +if filter_tests!(testsuite, test_args) + if !run_gpu_tests + @warn "CUDA GPU not available, skipping GPU-only tests" + filter!( + test -> + !startswith(first(test), "gpu_only/") && + !startswith(first(test), "cuda.jl/"), + testsuite, + ) + end -if !run_gpu_tests || !cuNumeric.FUSE_BROADCAST_EXPRS - @warn "Broadcast fusion is disabled, skipping fusion tests" - filter!(test -> !startswith(first(test), "gpu_only/broadcast_fusion"), testsuite) + if !run_gpu_tests || !cuNumeric.FUSE_BROADCAST_EXPRS + @warn "Broadcast fusion is disabled, skipping fusion tests" + filter!(test -> !startswith(first(test), "gpu_only/broadcast_fusion"), testsuite) + end end -filter!(test -> !startswith(first(test), "defunct/"), testsuite) - -runtests(cuNumeric, ARGS; testsuite, init_code) +cuda_tests = filter(test -> startswith(test, "cuda.jl/"), collect(keys(testsuite))) +runtests(cuNumeric, test_args; testsuite, init_code, serial=cuda_tests) diff --git a/test/workflows/grayscott.jl b/test/workflows/grayscott.jl index e81c364f6..7be2662ba 100644 --- a/test/workflows/grayscott.jl +++ b/test/workflows/grayscott.jl @@ -33,64 +33,62 @@ struct ParamsGS{T<:AbstractFloat} end end -function step(u, v, u_new, v_new, args::ParamsGS) - @analyze_lifetimes begin - # calculate F_u and F_v functions - # currently we don't have NDArray^x working yet. - F_u = ( - ( - -u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) + args.f*(1 .- u[2:(end - 1), 2:(end - 1)]) - ) - F_v = ( - ( - u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) - (args.f+args.k)*v[2:(end - 1), 2:(end - 1)] - ) - # 2-D Laplacian of f using array slicing, excluding boundaries - # For an N x N array f, f_lap is the Nend x Nend array in the "middle" - u_lap = ( - ( - u[3:end, 2:(end - 1)] - 2*u[2:(end - 1), 2:(end - 1)] + - u[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 + - ( - u[2:(end - 1), 3:end] - 2*u[2:(end - 1), 2:(end - 1)] + - u[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - v_lap = ( - ( - v[3:end, 2:(end - 1)] - 2*v[2:(end - 1), 2:(end - 1)] + - v[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 + - ( - v[2:(end - 1), 3:end] - 2*v[2:(end - 1), 2:(end - 1)] + - v[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) +@accelerate function step(u, v, u_new, v_new, args::ParamsGS) + # calculate F_u and F_v functions + # currently we don't have NDArray^x working yet. + F_u = ( + ( + -u[2:(end - 1), 2:(end - 1)] .* + (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) + ) + args.f*(1 .- u[2:(end - 1), 2:(end - 1)]) + ) + F_v = ( + ( + u[2:(end - 1), 2:(end - 1)] .* + (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) + ) - (args.f+args.k)*v[2:(end - 1), 2:(end - 1)] + ) + # 2-D Laplacian of f using array slicing, excluding boundaries + # For an N x N array f, f_lap is the Nend x Nend array in the "middle" + u_lap = ( + ( + u[3:end, 2:(end - 1)] - 2*u[2:(end - 1), 2:(end - 1)] + + u[1:(end - 2), 2:(end - 1)] + ) ./ args.dx^2 + + ( + u[2:(end - 1), 3:end] - 2*u[2:(end - 1), 2:(end - 1)] + + u[2:(end - 1), 1:(end - 2)] + ) ./ args.dx^2 + ) + v_lap = ( + ( + v[3:end, 2:(end - 1)] - 2*v[2:(end - 1), 2:(end - 1)] + + v[1:(end - 2), 2:(end - 1)] + ) ./ args.dx^2 + + ( + v[2:(end - 1), 3:end] - 2*v[2:(end - 1), 2:(end - 1)] + + v[2:(end - 1), 1:(end - 2)] + ) ./ args.dx^2 + ) - # # Forward-Euler time step for all points except the boundaries - u_new[2:(end - 1), 2:(end - 1)] = - ((args.c_u * u_lap) + F_u) * args.dt + u[2:(end - 1), 2:(end - 1)] - v_new[2:(end - 1), 2:(end - 1)] = - ((args.c_v * v_lap) + F_v) * args.dt + v[2:(end - 1), 2:(end - 1)] - - # Apply periodic boundary conditions - u_new[:, 1] = u[:, end - 1] - u_new[:, end] = u[:, 2] - u_new[1, :] = u[end - 1, :] - u_new[end, :] = u[2, :] - v_new[:, 1] = v[:, end - 1] - v_new[:, end] = v[:, 2] - v_new[1, :] = v[end - 1, :] - v_new[end, :] = v[2, :] - end + # # Forward-Euler time step for all points except the boundaries + u_new[2:(end - 1), 2:(end - 1)] = + ((args.c_u * u_lap) + F_u) * args.dt + u[2:(end - 1), 2:(end - 1)] + v_new[2:(end - 1), 2:(end - 1)] = + ((args.c_v * v_lap) + F_v) * args.dt + v[2:(end - 1), 2:(end - 1)] + + # Apply periodic boundary conditions + u_new[:, 1] = u[:, end - 1] + u_new[:, end] = u[:, 2] + u_new[1, :] = u[end - 1, :] + u_new[end, :] = u[2, :] + v_new[:, 1] = v[:, end - 1] + v_new[:, end] = v[:, 2] + v_new[1, :] = v[end - 1, :] + v_new[end, :] = v[2, :] end -# same as above but without @analyze_lifetimes macro +# same as above but without the @accelerate macro function step_base(u, v, u_new, v_new, args::ParamsGS) # calculate F_u and F_v functions # currently we don't have NDArray^x working yet. @@ -239,8 +237,8 @@ function run_slice_test(op, op_scoped, FT, N; f=0.04, k=0.06, dx=1.0) return base, scoped end -binary_scope(op) = (a, b, out) -> @analyze_lifetimes out[:, :] = op(a, b) -slice_scope(op) = (u, v, out, args) -> @analyze_lifetimes out[:, :] = op(u, v, args) +binary_scope(op) = (a, b, out) -> @accelerate out[:, :] = op(a, b) +slice_scope(op) = (u, v, out, args) -> @accelerate out[:, :] = op(u, v, args) const OPS = Dict( :add => (+), @@ -436,68 +434,30 @@ function test_scoping_rewrite_pipeline() @test isempty(freed) end - @testset "Lexical lifetime scope" begin - function hidden_binding() - @analyze_lifetimes begin - internal_result = 41 - nothing - end - return internal_result - end - - function shadowed_binding() - internal_result = :outer - @analyze_lifetimes begin - internal_result = :inner - nothing - end - return internal_result - end - - function hidden_destructured_bindings() - @analyze_lifetimes begin - internal_first, internal_second = (1, 2) + @testset "Block form keeps Julia scope" begin + # Block form adds no scope: bindings stay live in the enclosing scope (1:1 Julia). + function visible_binding() + @accelerate begin + internal_result = 42 nothing end - return internal_first, internal_second - end - - function unrelated_undefined_binding() - return unrelated_result - end - - function rendered_error(f) - try - f() - catch exc - return sprint(io -> showerror(io, exc, catch_backtrace())) - end - return "" + return internal_result # would be UndefVar under a `let` end + @test visible_binding() == 42 output = [0] - returned = @analyze_lifetimes begin + returned = @accelerate begin internal_result = 42 output[1] = internal_result internal_result end - - @test_throws UndefVarError hidden_binding() - @test_throws UndefVarError hidden_destructured_bindings() - @test occursin( - "If `internal_result` was created there", rendered_error(hidden_binding) - ) - @test occursin( - "If `unrelated_result` was created there", - rendered_error(unrelated_undefined_binding), - ) - @test shadowed_binding() == :outer @test output == [42] @test returned == 42 if cuNumeric.FUSE_BROADCAST_EXPRS - function hidden_fused_binding(a, b, destination) - @analyze_lifetimes begin + # A fused intermediate is materialized and also stays live. + function fused_binding(a, b, destination) + @accelerate begin fused_result = a .* b destination .= fused_result .+ 1 end @@ -505,9 +465,7 @@ function test_scoping_rewrite_pipeline() end destination = zeros(Int, 2) - @test_throws UndefVarError hidden_fused_binding( - [2, 3], [4, 5], destination - ) + @test fused_binding([2, 3], [4, 5], destination) == [8, 15] @test destination == [9, 16] end end @@ -519,7 +477,7 @@ function test_scoping_regressions(T, N) C = cuNumeric.zeros(T, (N, N)) @testset "In-place assignment" begin - @analyze_lifetimes begin + @accelerate begin result = A[1:end, :] .+ B[1:end, :] C .= result .* T(2.0) end @@ -529,7 +487,7 @@ function test_scoping_regressions(T, N) @testset "Macro as RHS" begin # Test values: (1+1)^2 = 4 - res = @analyze_lifetimes (A .+ B) .^ 2 + res = @accelerate (A .+ B) .^ 2 @test res isa cuNumeric.NDArray @test all(Array(res) .== T(4.0)) end @@ -537,7 +495,7 @@ function test_scoping_regressions(T, N) @testset "Returned bindings stay materialized" begin # A returned producer must come back as a materialized NDArray, not a # lazy broadcast tree; `c` stays a private intermediate that fuses away. - x, y = @analyze_lifetimes begin + x, y = @accelerate begin x = A .+ B c = x .* A y = c .^ 2 @@ -551,14 +509,14 @@ function test_scoping_regressions(T, N) @testset "Return forms yield materialized bindings" begin # `x = y` alias, tuple, and trailing `return` all return real NDArrays. - aliased = @analyze_lifetimes begin + aliased = @accelerate begin y = A .+ B x = y end @test aliased isa cuNumeric.NDArray @test all(Array(aliased) .== T(2)) - rx, ry = @analyze_lifetimes begin + rx, ry = @accelerate begin rx = A .+ B ry = rx .^ 2 return (rx, ry) @@ -570,7 +528,7 @@ function test_scoping_regressions(T, N) if cuNumeric.FUSE_BROADCAST_EXPRS @testset "Indexed fused assignment writes through NDArray slices" begin out = cuNumeric.zeros(T, (N + 2, N + 2)) - @analyze_lifetimes begin + @accelerate begin producer = A .* T(2) out[2:(end - 1), 2:(end - 1)] = producer .+ T(1) end @@ -582,7 +540,7 @@ function test_scoping_regressions(T, N) @testset "Nested @. macros fuse before lifetime analysis" begin multiplier = cuNumeric.ones(T, (N, N)) result = cuNumeric.zeros(T, (N, N)) - @analyze_lifetimes begin + @accelerate begin tmp = @. A + B result .= @. tmp * multiplier + T(1.0) end From b79a17fbc560d5056e69297782537db612160a8f Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Tue, 18 Aug 2026 18:28:10 -0400 Subject: [PATCH 14/49] Add GPU fft/ifft/fft!/ifft! and batched FFT on NDArray (#185) * Add GPU fft/ifft/fft!/ifft! and batched FFT on NDArray --- Project.toml | 2 + benchmark/benchmarks.toml | 18 ++ benchmark/src/benchmarks/poisson_fft.jl | 84 ++++++++ benchmark/src_py/benchmarks/poisson_fft.py | 56 +++++ docs/make.jl | 2 + docs/src/api.md | 2 +- docs/src/benchmarks/howto.md | 7 +- docs/src/examples/poisson_fft.md | 63 ++++++ docs/src/fft.md | 58 +++++ docs/src/index.md | 2 +- examples/poisson_fft.jl | 84 ++++++++ lib/cunumeric_jl_wrapper/include/types.h | 3 + lib/cunumeric_jl_wrapper/src/types.cpp | 12 ++ lib/cunumeric_jl_wrapper/src/wrapper.cpp | 1 + src/cuNumeric.jl | 4 + src/ndarray/detail/fft.jl | 239 +++++++++++++++++++++ src/ndarray/fft.jl | 188 ++++++++++++++++ test/Project.toml | 2 +- test/analysis/type_stability.jl | 11 + test/cuda.jl/fusion_compare.jl | 5 +- test/cuda.jl/fusion_compare_1d.jl | 5 +- test/cuda.jl/padding.jl | 4 +- test/cuda.jl/vecadd.jl | 4 +- test/gpu_only/fft.jl | 166 ++++++++++++++ test/runtests.jl | 5 +- 25 files changed, 1008 insertions(+), 19 deletions(-) create mode 100644 benchmark/src/benchmarks/poisson_fft.jl create mode 100644 benchmark/src_py/benchmarks/poisson_fft.py create mode 100644 docs/src/examples/poisson_fft.md create mode 100644 docs/src/fft.md create mode 100644 examples/poisson_fft.jl create mode 100644 src/ndarray/detail/fft.jl create mode 100644 src/ndarray/fft.jl create mode 100644 test/gpu_only/fft.jl diff --git a/Project.toml b/Project.toml index 0aa016679..b0b674133 100644 --- a/Project.toml +++ b/Project.toml @@ -3,6 +3,7 @@ uuid = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" version = "0.2.0" [deps] +AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" CUDACore = "bd0ed864-bdfe-4181-a5ed-ce625a5fdea2" CUDATools = "9ec180c6-1c07-47c7-9e6e-ebefa4d1f6d0" @@ -26,6 +27,7 @@ cupynumeric_jll = "2862d674-414d-5b0b-a494-b21f8deca547" libcxxwrap_julia_jll = "3eaa8342-bff7-56a5-9981-c04077f7cee7" [compat] +AbstractFFTs = "1.5" CNPreferences = "0.1.3" CUDACore = "6.2" CUDATools = "6.2" diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml index 747989673..0518b3fc8 100644 --- a/benchmark/benchmarks.toml +++ b/benchmark/benchmarks.toml @@ -104,6 +104,24 @@ fusion = [true, false] N = [50000, 100000, 200000, 400000] M = 512 +################################# +# Spectral Poisson (FFT) # +# M independent N×N Poisson # +# solves (∇²u = f, periodic). # +# Transform is the last 2 axes; # +# batch axis M may split GPUs. # +# Work ~ M N² log N. # +# Weak scaling: M / P constant. # +# M = 8 * P, N = 1024. # +################################# + +[[poisson_fft]] +T = "Float32" +gpus = [1, 2, 4, 8] +cpus = 16 +N = 1024 +M = [8, 16, 32, 64] + ################################# # Monte-Carlo Integration # # Work ~ N. Scale N linearly # diff --git a/benchmark/src/benchmarks/poisson_fft.jl b/benchmark/src/benchmarks/poisson_fft.jl new file mode 100644 index 000000000..afe1344f9 --- /dev/null +++ b/benchmark/src/benchmarks/poisson_fft.jl @@ -0,0 +1,84 @@ +# Spectral Poisson on a stack of periodic N×N grids. +# Each of the M right-hand sides is an independent ∇²u = f solve: +# û = fft(f), û *= −1/|k|², u = ifft(û) +# (electrostatics / gravity, not an FFT round-trip). +# +# The transform is over the last two axes, so the leading batch axis may +# partition across GPUs. A single all-axes 2-d FFT cannot. +# Weak scaling: M ∝ P, N fixed. + +function _integer_fftfreq(n::Int) + n2 = n ÷ 2 + return iseven(n) ? vcat(0:(n2 - 1), (-n2):-1) : vcat(0:n2, (-n2):-1) +end + +function _poisson_inv_laplacian(::Type{T}, n::Int) where {T} + freq = _integer_fftfreq(n) + invk = Matrix{T}(undef, n, n) + s = T(4 * π^2) + for j in 1:n, i in 1:n + k2 = s * T(freq[i]^2 + freq[j]^2) + invk[i, j] = k2 == 0 ? zero(T) : -inv(k2) + end + return invk +end + +Base.@kwdef struct PoissonFFT{T} <: AbstractBenchmark{T} + N::Int + M::Int +end + +name(::PoissonFFT) = "poisson_fft" +dims(b::PoissonFFT) = (b.N, b.M) +data(b::PoissonFFT{T}) where {T} = "Poisson FFT with T=$(T), N=$(b.N), M=$(b.M)" +allowed_types(::Type{PoissonFFT}) = cuNumeric.SUPPORTED_FLOAT_TYPES + +# 2-d C2C FFT: 5 n log2(n) real flops per 1-d line, 2n lines → 10 n² log2(n). +# Forward + inverse, plus n² complex muls (6 real flops each), times M batches. +function total_flops(b::PoissonFFT) + n = b.N + m = b.M + n2 = n * n + return m * (20 * n2 * log2(n) + 6 * n2) +end + +function initialize(b::PoissonFFT{T}; mod=cuNumeric) where {T} + f = mod.rand(T, b.M, b.N, b.N) + CT = Complex{T} + fc = f isa NDArray ? cuNumeric.as_type(f, CT) : CT.(f) + work = copy(fc) + kinv_h = _poisson_inv_laplacian(T, b.N) + kinv = if fc isa NDArray + reshape(NDArray(kinv_h), 1, b.N, b.N) + else + reshape(kinv_h, 1, b.N, b.N) + end + GC.gc() + return fc, work, kinv +end + +function run!(::PoissonFFT, fc, work, kinv) + copyto!(work, fc) + cuNumeric.batched_fft!(work) + work .*= kinv + cuNumeric.batched_ifft!(work) + return work +end + +correctness_supported(::PoissonFFT) = true + +function check_benchmark_correctness( + ::PoissonFFT{T}, gs::GlobalSettings; mod=cuNumeric, atol=1e-3, rtol=1e-3 +) where {T} + mod === cuNumeric || return "skipped" + n = 32 + xs = range(T(0), T(1); length=n + 1)[1:(end - 1)] + u_true = T[sin(2T(π) * x) * sin(2T(π) * y) for x in xs, y in xs] + f_h = (-8 * T(π)^2) .* u_true + f = NDArray(reshape(f_h, 1, n, n)) + kinv = reshape(NDArray(_poisson_inv_laplacian(T, n)), 1, n, n) + u = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f) .* kinv)) + return isapprox(Array(u)[1, :, :], u_true; atol=atol, rtol=rtol) ? "pass" : "fail" +end + +register_benchmark("poisson_fft", PoissonFFT) diff --git a/benchmark/src_py/benchmarks/poisson_fft.py b/benchmark/src_py/benchmarks/poisson_fft.py new file mode 100644 index 000000000..ac952a12c --- /dev/null +++ b/benchmark/src_py/benchmarks/poisson_fft.py @@ -0,0 +1,56 @@ +import math + +import cupynumeric as np +import numpy as onp + +from core import register_benchmark + + +def _integer_fftfreq(n): + n2 = n // 2 + if n % 2 == 0: + return list(range(0, n2)) + list(range(-n2, 0)) + return list(range(0, n2 + 1)) + list(range(-n2, 0)) + + +def _poisson_inv_laplacian(T, n): + freq = onp.array(_integer_fftfreq(n), dtype=T) + fx = freq[:, None] + fy = freq[None, :] + k2 = T(4.0 * math.pi**2) * (fx * fx + fy * fy) + invk = onp.zeros((n, n), dtype=T) + onp.divide(-1.0, k2, out=invk, where=k2 != 0) + return invk + + +class PoissonFFT: + name = "poisson_fft" + + def __init__(self, T, N, M): + self.T, self.N, self.M = T, N, M + + def dims(self): + return self.N, self.M + + def total_flops(self): + # Same breakdown as benchmark/src/benchmarks/poisson_fft.jl + n = self.N + m = self.M + n2 = n * n + return m * (20 * n2 * math.log2(n) + 6 * n2) + + def initialize(self): + f = np.random.rand(self.M, self.N, self.N).astype(self.T) + kinv = np.array(_poisson_inv_laplacian(self.T, self.N)).reshape( + 1, self.N, self.N + ) + return (f, kinv) + + def run(self, state): + f, kinv = state + uhat = np.fft.fftn(f, axes=(1, 2)) + uhat *= kinv + return np.fft.ifftn(uhat, axes=(1, 2)) + + +register_benchmark("poisson_fft", PoissonFFT) diff --git a/docs/make.jl b/docs/make.jl index 5ce5328be..e1ab3b66d 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -39,6 +39,7 @@ makedocs(; "Monte-Carlo" => "examples/montecarlo.md", "Gray-Scott" => "examples/grayscott.md", "Dynamic Mode Decomposition" => "examples/dmd.md", + "Periodic Poisson (FFT)" => "examples/poisson_fft.md", ], "Performance Tips" => [ "Kernel Fusion" => "perf/kernel_fusion.md", @@ -65,6 +66,7 @@ makedocs(; "Unary Operations" => "api_unary.md", "Binary Operations" => "api_binary.md", "Linear Algebra" => "linalg.md", + "FFT" => "fft.md", "HDF5" => "api_hdf5.md", "NDArray Reference" => "api.md", "CUDA.jl Tasking" => "api_cuda.md", diff --git a/docs/src/api.md b/docs/src/api.md index d8197626d..864e8d4f5 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -1,6 +1,6 @@ # NDArray Reference -Indexing, reshaping, reductions, comparisons, memory helpers, acceleration macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). +Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). For `fft` / `ifft` / `fft!` / `ifft!` and `batched_fft`, see [FFT](./fft.md). There is no `plan_fft`: cupynumeric does not expose a cuFFT handle. ```@autodocs Modules = [cuNumeric] diff --git a/docs/src/benchmarks/howto.md b/docs/src/benchmarks/howto.md index efce38239..4d7f540bb 100644 --- a/docs/src/benchmarks/howto.md +++ b/docs/src/benchmarks/howto.md @@ -47,13 +47,12 @@ n_correctness_iter = 5 - `cupynumeric` / `cuda`: optional comparison backends - `check_correctness`: one CPU-reference check per config (not per timed iter), recorded in the CSV -Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, -`dmd_baseline`, `dmd_accelerated`, `grayscott_baseline`, -`grayscott_function_accelerated`, …). Names must match what -`src/benchmarks/*.jl` registers. +Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, `dmd_baseline`, `dmd_lifetimes`, `grayscott_baseline`, `grayscott_lifetimes`, `poisson_fft`, …). Names must match what `src/benchmarks/*.jl` registers. DMD's `N` is the number of spatial degrees of freedom (rows of the snapshot matrix), not a grid side length. The SVD is of the tall-skinny `N × (M-1)` matrix `X1`. Thin SVD plus the rank-`r` lift is `Θ(N)` when `M` and `r` are fixed, so weak scaling is `N ∝ P` (same idea as Monte Carlo, not GEMM's `N ∝ P^{1/3}`). The flop count is in `src/benchmarks/dmd.jl`. +`poisson_fft` solves ``M`` independent periodic Poisson problems on an ``N \times N`` grid (FFT, divide by ``-|k|^2``, inverse FFT). The transform is over the last two axes, so the leading batch axis can split across GPUs. Weak scaling is ``M \propto P`` with ``N`` fixed. A single all-axes 2-d `fft` of one grid is single-GPU and would not scale that way. + ```toml [[gemm]] T = ["Float32"] diff --git a/docs/src/examples/poisson_fft.md b/docs/src/examples/poisson_fft.md new file mode 100644 index 000000000..a466103ec --- /dev/null +++ b/docs/src/examples/poisson_fft.md @@ -0,0 +1,63 @@ +# Periodic Poisson (FFT) + +The Poisson equation + +```math +\nabla^2 u = f +``` + +on the periodic unit square is a pointwise divide in Fourier space. With +``k = 2\pi (m_x, m_y)`` the wavevector of each DFT mode, + +```math +\hat{u}[k] = \frac{\hat{f}[k]}{-|k|^2}, \qquad \hat{u}[0] = 0. +``` + +The zero mode is dropped so ``u`` has mean zero (the potential is only defined +up to a constant). One forward [`fft`](@ref), a broadcasted divide, and one +[`ifft`](@ref) is the whole solve. That is the electrostatics / gravitational +potential of a periodic charge density, not an FFT round-trip. There is no +reusable plan to hold across the two transforms; each call launches +`CUPYNUMERIC_FFT` and cuFFT planning stays inside that task. + +A manufactured solution ``u = \sin(2\pi x)\sin(2\pi y)`` has +``\nabla^2 u = -8\pi^2 u``, which is what the example checks. + +```julia +# found in examples/poisson_fft.jl +using cuNumeric + +function integer_fftfreq(n::Int) + n2 = n ÷ 2 + return iseven(n) ? vcat(0:(n2 - 1), (-n2):-1) : vcat(0:n2, (-n2):-1) +end + +function poisson_inv_laplacian(::Type{T}, n::Int) where {T} + freq = integer_fftfreq(n) + invk = Matrix{T}(undef, n, n) + s = T(4 * π^2) + for j in 1:n, i in 1:n + k2 = s * T(freq[i]^2 + freq[j]^2) + invk[i, j] = k2 == 0 ? zero(T) : -inv(k2) + end + return invk +end + +n = 64 +xs = range(0, 1; length=n + 1)[1:(end - 1)] +u_true = Float32[sin(2π * x) * sin(2π * y) for x in xs, y in xs] +f = NDArray((-8 * Float32(π)^2) .* u_true) + +invk = NDArray(poisson_inv_laplacian(Float32, n)) +u = real(ifft(fft(f) .* invk)) +``` + +A stack of right-hand sides (several charge distributions, several snapshots) +uses [`batched_fft`](@ref). The leading axis is the batch and is the one that +can split across GPUs; a single all-axes `fft` of one grid cannot. + +```julia +f_batch = NDArray(repeat(reshape(Array(f), 1, n, n), 4, 1, 1)) +invk3 = reshape(invk, 1, n, n) +u_batch = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f_batch) .* invk3)) +``` diff --git a/docs/src/fft.md b/docs/src/fft.md new file mode 100644 index 000000000..bcd207110 --- /dev/null +++ b/docs/src/fft.md @@ -0,0 +1,58 @@ +# FFT + +cuNumeric.jl exposes GPU-only [`fft`](@ref) / [`ifft`](@ref) (and in-place +[`fft!`](@ref) / [`ifft!`](@ref)) on `NDArray`. The functions follow +[AbstractFFTs.jl](https://github.com/JuliaMath/AbstractFFTs.jl) / FFTW +conventions, not NumPy's defaults: + +- every dimension is transformed unless you pass `dims` +- `fft` is unnormalized; `ifft` divides by the product of the transformed lengths +- real input is promoted to complex (`Float32` → `ComplexF32`, otherwise `ComplexF64`) + +There is no `plan_fft`, and we cannot control the cuFFT plan. cupynumeric never +returns a `cufftHandle` (or any other plan object). Each `fft` / `ifft` call +launches the `CUPYNUMERIC_FFT` task; cuFFT planning and any internal plan cache +live entirely inside that GPU task. A Julia `Plan` that only stored sizes and +re-called `fft` would not be a real plan, so `plan_fft` is not implemented. +`using AbstractFFTs; fft(A)` dispatches, but `plan_fft` will not. + +```julia +using AbstractFFTs +using cuNumeric + +A = cuNumeric.rand(ComplexF32, 64, 64) +Y = fft(A) # all dimensions +Z = fft(A, 1) # first dimension only +ifft(Y) # ≈ A + +fft!(copy(A)) # overwrites a complex array +``` + +`fft!` / `ifft!` require `ComplexF32` or `ComplexF64`. Real arrays must go +through out-of-place `fft`. + +A stack of independent transforms uses [`batched_fft`](@ref): the leading +dimension is the batch, and every trailing dimension is transformed. That is +the same task as `fft(A, 2:ndims(A))`. The batch axis is the one that can +split across GPUs; an all-axes `fft(A)` cannot. + +```julia +signals = cuNumeric.rand(ComplexF32, 32, 1024) # 32 length-1024 traces +batched_fft(signals) # FFT along dim 2 + +fields = cuNumeric.rand(ComplexF32, 8, 64, 64) # 8 images +batched_fft(fields) # 2-d FFT of each +``` + +FFT is GPU-only. A CPU-only runtime raises an error. Multi-GPU use is limited +to batching over dimensions that are not transformed (the same restriction as +cupynumeric). + +Awkward lengths whose prime factors exceed 131 make cuFFT take the Bluestein +path; a warning is emitted. Padding to a nearby highly composite size avoids +that. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/fft.jl"] +``` diff --git a/docs/src/index.md b/docs/src/index.md index bb266637d..b9b693e6e 100644 --- a/docs/src/index.md +++ b/docs/src/index.md @@ -49,7 +49,7 @@ x = unwrap(s) # T, e.g. Float32 **The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. -For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/dev/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/dev/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/dev/perf/patterns_to_avoid). +For API details see [Initialization](./api_initialization.md), [Random](./api_random.md), [FFT](./fft.md), and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). ### Kernel Fusion diff --git a/examples/poisson_fft.jl b/examples/poisson_fft.jl new file mode 100644 index 000000000..dc1a34809 --- /dev/null +++ b/examples/poisson_fft.jl @@ -0,0 +1,84 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +#= Periodic Poisson via FFT. + +Solve ∇²u = f on the unit square with periodic boundaries. In Fourier space +that is a pointwise divide: û[k] = f̂[k] / (−|k|²), with the k = 0 mode set +to zero so the potential has mean zero. + +A manufactured solution u = sin(2πx) sin(2πy) has ∇²u = −8π² u, so we can +check the residual after one forward FFT, the divide, and one inverse FFT. + +The same kernel on a stack of right-hand sides is `batched_fft` — each slice +is an independent electrostatics / gravity solve, and the batch axis is what +can split across GPUs. +=# + +using cuNumeric +using Printf + +function integer_fftfreq(n::Int) + n2 = n ÷ 2 + return iseven(n) ? vcat(0:(n2 - 1), (-n2):-1) : vcat(0:n2, (-n2):-1) +end + +# Multipliers for û = f̂ / (−|k|²). DC is 0 (mean-zero potential). +function poisson_inv_laplacian(::Type{T}, n::Int) where {T} + freq = integer_fftfreq(n) + invk = Matrix{T}(undef, n, n) + s = T(4 * π^2) + for j in 1:n, i in 1:n + k2 = s * T(freq[i]^2 + freq[j]^2) + invk[i, j] = k2 == 0 ? zero(T) : -inv(k2) + end + return invk +end + +function poisson_solve(f::NDArray) + n, m = size(f) + n == m || throw(ArgumentError("poisson_solve expects a square grid")) + invk = NDArray(poisson_inv_laplacian(eltype(f), n)) + uhat = fft(f) + return real(ifft(uhat .* invk)) +end + +function main() + n = 64 + xs = range(0, 1; length=n + 1)[1:(end - 1)] + u_true = Float32[sin(2π * x) * sin(2π * y) for x in xs, y in xs] + f_h = (-8 * Float32(π)^2) .* u_true + + f = NDArray(f_h) + u = poisson_solve(f) + err = maximum(abs.(u - NDArray(u_true))) + @printf("single grid %dx%d max |u − u_true| = %.3e\n", n, n, unwrap(err)) + + # Four independent charge distributions, one FFT task, batch axis first. + b = 4 + f_batch = NDArray(repeat(reshape(f_h, 1, n, n), b, 1, 1)) + invk = reshape(NDArray(poisson_inv_laplacian(Float32, n)), 1, n, n) + u_batch = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f_batch) .* invk)) + err_b = maximum(abs.(u_batch - NDArray(repeat(reshape(u_true, 1, n, n), b, 1, 1)))) + @printf("batched %d×%dx%d max |u − u_true| = %.3e\n", b, n, n, unwrap(err_b)) + + return u +end + +main() diff --git a/lib/cunumeric_jl_wrapper/include/types.h b/lib/cunumeric_jl_wrapper/include/types.h index 4b21fde4b..6271b60e2 100644 --- a/lib/cunumeric_jl_wrapper/include/types.h +++ b/lib/cunumeric_jl_wrapper/include/types.h @@ -69,5 +69,8 @@ void wrap_binary_ops(jlcxx::Module&); // Linear algebra op codes void wrap_linalg_ops(jlcxx::Module& mod); +// FFT task id and type/direction enums (mirror cupynumeric_c.h) +void wrap_fft_ops(jlcxx::Module& mod); + // BitGenerator op codes / enums (mirror cupynumeric_c.h) void wrap_bitgenerator_ops(jlcxx::Module& mod); diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index 8bf7706fb..df218dcc8 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -179,6 +179,18 @@ void wrap_linalg_ops(jlcxx::Module& mod) { legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEEV}); } +void wrap_fft_ops(jlcxx::Module& mod) { + mod.set_const("FFT", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_FFT}); + mod.set_const("FFT_R2C", int32_t(CUPYNUMERIC_FFT_R2C)); + mod.set_const("FFT_C2R", int32_t(CUPYNUMERIC_FFT_C2R)); + mod.set_const("FFT_C2C", int32_t(CUPYNUMERIC_FFT_C2C)); + mod.set_const("FFT_D2Z", int32_t(CUPYNUMERIC_FFT_D2Z)); + mod.set_const("FFT_Z2D", int32_t(CUPYNUMERIC_FFT_Z2D)); + mod.set_const("FFT_Z2Z", int32_t(CUPYNUMERIC_FFT_Z2Z)); + mod.set_const("FFT_FORWARD", int32_t(CUPYNUMERIC_FFT_FORWARD)); + mod.set_const("FFT_INVERSE", int32_t(CUPYNUMERIC_FFT_INVERSE)); +} + void wrap_bitgenerator_ops(jlcxx::Module& mod) { mod.set_const( "BITGENERATOR", diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index fd1785c0c..22e9540ea 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -82,6 +82,7 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { wrap_binary_ops(mod); wrap_unary_reds(mod); wrap_linalg_ops(mod); + wrap_fft_ops(mod); wrap_bitgenerator_ops(mod); using jlcxx::ParameterList; diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index bd080578f..6102c14ba 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -43,6 +43,8 @@ import Base: axes, convert, copy, copyto!, inv, isfinite, sqrt, -, +, *, ==, !=, using LinearAlgebra import LinearAlgebra: mul! +import AbstractFFTs: fft, ifft, fft!, ifft! + using Random import Random: rand!, randn!, randexp! @@ -160,6 +162,7 @@ const TASK_SCOPE_NAMES = CNPreferences.TASK_SCOPE_NAMES # NDArray internal include("ndarray/detail/ndarray.jl") include("ndarray/detail/linalg.jl") +include("ndarray/detail/fft.jl") # Utilities include("cuda/strided_device_array.jl") @@ -187,6 +190,7 @@ include("ndarray/unary.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") include("ndarray/batched_linalg.jl") +include("ndarray/fft.jl") include("scoping/scoping.jl") # From https://github.com/JuliaGraphics/QML.jl/blob/dca239404135d85fe5d4afe34ed3dc5f61736c63/src/QML.jl#L147 diff --git a/src/ndarray/detail/fft.jl b/src/ndarray/detail/fft.jl new file mode 100644 index 000000000..24046866c --- /dev/null +++ b/src/ndarray/detail/fft.jl @@ -0,0 +1,239 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +# cuFFT mixed-radix kernels only cover lengths whose prime factors are all +# <= 131. A larger factor forces Bluestein (slower, more scratch). +const CUFFT_MAX_EFFICIENT_PRIME = 131 +const _CUFFT_SMALL_PRIMES = ( + 2, + 3, + 5, + 7, + 11, + 13, + 17, + 19, + 23, + 29, + 31, + 37, + 41, + 43, + 47, + 53, + 59, + 61, + 67, + 71, + 73, + 79, + 83, + 89, + 97, + 101, + 103, + 107, + 109, + 113, + 127, + 131, +) + +function _has_large_prime_factor(n::Integer) + n < 2 && return false + m = Int(n) + for p in _CUFFT_SMALL_PRIMES + while m % p == 0 + m ÷= p + end + m == 1 && return false + end + return true +end + +function _unique_axes(axes0::NTuple{R,Int64}) where {R} + seen = zero(UInt64) + out = Vector{Int64}(undef, R) + n = 0 + for ax in axes0 + bit = one(UInt64) << ax + if seen & bit == 0 + seen |= bit + n += 1 + out[n] = ax + end + end + return resize!(out, n) +end + +function _axes_sorted(axes0::NTuple{R,Int64}) where {R} + for i in 2:R + axes0[i] < axes0[i - 1] && return false + end + return true +end + +function _operate_over_axes(axes0::NTuple{R,Int64}, ndim::Int) where {R} + unique_axes = _unique_axes(axes0) + return length(unique_axes) != R || R != ndim || !_axes_sorted(axes0) +end + +function _bluestein_mask( + axes0::NTuple{R,Int64}, in_size::NTuple{N,Int}, out_size::NTuple{N,Int} +) where {R,N} + mask = Int32(0) + slow = Tuple{Int,Int}[] + seen = zero(UInt64) + for ax in axes0 + bit = one(UInt64) << ax + seen & bit != 0 && continue + seen |= bit + jdim = Int(ax) + 1 + length_ax = max(in_size[jdim], out_size[jdim]) + if _has_large_prime_factor(length_ax) + mask |= Int32(1) << ax + push!(slow, (ax, length_ax)) + end + end + if !isempty(slow) + details = join(("axis $ax (length $len)" for (ax, len) in slow), ", ") + @warn "cuNumeric is computing an FFT over $details whose length has a prime factor > $CUFFT_MAX_EFFICIENT_PRIME, so cuFFT falls back to the Bluestein algorithm. You may notice significantly decreased performance and much higher GPU memory usage. Zero-padding the transformed axis to a length whose prime factors are all <= $CUFFT_MAX_EFFICIENT_PRIME (e.g. the next power of two) avoids this." + end + return mask +end + +function _assert_fft_gpu() + Legate.num_gpus() > 0 && return nothing + return throw( + ErrorException( + "FFT requires a CUDA GPU; cupynumeric's FFT task has no CPU variant" + ), + ) +end + +_fft_kind(::Type{ComplexF32}) = Int32(cuNumeric.FFT_C2C) +_fft_kind(::Type{ComplexF64}) = Int32(cuNumeric.FFT_Z2Z) + +function _fft_dims(::NDArray{<:Any,N}) where {N} + N >= 1 || throw(ArgumentError("fft does not support 0-dimensional arrays")) + return ntuple(identity, Val(N)) +end + +function _fft_dims(A::NDArray, dim::Integer) + return _fft_dims(A, (Int(dim),)) +end + +function _fft_dims(A::NDArray{T,N}, dims) where {T,N} + N >= 1 || throw(ArgumentError("fft does not support 0-dimensional arrays")) + R = length(dims) + R >= 1 || throw(ArgumentError("fft dims must contain at least one dimension")) + region = ntuple(i -> Int(dims[i]), R) + seen = zero(UInt64) + for d in region + (1 <= d <= N) || throw(ArgumentError("fft dim $d is out of range for a $N-d array")) + bit = one(UInt64) << d + seen & bit != 0 && throw(ArgumentError("fft dims must be unique; got $region")) + seen |= bit + end + return region +end + +# Leading dimension is the batch; every remaining dim is transformed. +# Same CUPYNUMERIC_FFT task as `fft` — the batch axis is the one that may split. +function _fft_batch_dims(::NDArray{<:Any,N}) where {N} + N >= 2 || throw( + ArgumentError( + "batched_fft requires a leading batch dimension; got a $N-d array. Use fft for a single transform." + ), + ) + return ntuple(i -> i + 1, Val(N - 1)) +end + +function _ifft_scale(::Type{T}, sz::NTuple{N,Int}, dims::NTuple{R,Int}) where {T,N,R} + n = one(real(T)) + for d in dims + n *= real(T)(sz[d]) + end + return T(inv(n)) +end + +_fft_scope_name(direction::Int32) = + direction == Int32(cuNumeric.FFT_INVERSE) ? "ifft" : "fft" + +""" + fft_task!(out, inp, dims, direction; scale=false) + +Launch cupynumeric's `CUPYNUMERIC_FFT` auto task. `dims` are 1-based Julia +dimensions. `out` and `inp` must have the same shape and complex eltype; they +may be the same array (in-place C2C). When `scale` is true the inverse is +normalized in this same task scope (`out .*= 1/N`). +""" +function fft_task!( + out::NDArray{T,N}, + inp::NDArray{T,N}, + dims::NTuple{R,Int}, + direction::Int32; + scale::Bool=false, +) where {T<:SUPPORTED_COMPLEX_TYPES,N,R} + _assert_fft_gpu() + size(out) == size(inp) || + throw(DimensionMismatch("FFT output size $(size(out)) != input size $(size(inp))")) + + axes0 = ntuple(i -> Int64(dims[i] - 1), Val(R)) + unique_axes = _unique_axes(axes0) + operate_over = _operate_over_axes(axes0, N) + # Warn on awkward lengths. cuFFT may take Bluestein internally; the + # 26.06 task has no bluestein_mask scalar, so we do not send one. + _bluestein_mask(axes0, size(inp), size(out)) + kind = _fft_kind(T) + + @task_scope _fft_scope_name(direction) begin + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + task = Legate.create_auto_task(rt, lib, cuNumeric.FFT) + cuNumeric.task_throws_exception(task, true) + + l_out = nda_to_logical_array(out) + l_in = inp === out ? l_out : nda_to_logical_array(inp) + + out_var = Legate.add_output(task, l_out) + in_var = Legate.add_input(task, l_in) + + # 26.06 fft_template.inl: kind, direction, operate_over_axes, then axes. + Legate.add_scalar(task, Legate.Scalar(kind)) + Legate.add_scalar(task, Legate.Scalar(direction)) + Legate.add_scalar(task, Legate.Scalar(operate_over)) + for ax in axes0 + Legate.add_scalar(task, Legate.Scalar(ax)) + end + + Legate.add_constraint(task, Legate.align(out_var, in_var)) + if N > length(unique_axes) + Legate.add_broadcast(task, l_in, CxxWrap.StdVector(UInt32.(unique_axes))) + else + Legate.add_broadcast(task, l_in) + end + + Legate.submit_auto_task(rt, task) + if scale + out .*= _ifft_scale(T, size(out), dims) + end + end + return out +end diff --git a/src/ndarray/fft.jl b/src/ndarray/fft.jl new file mode 100644 index 000000000..108006228 --- /dev/null +++ b/src/ndarray/fft.jl @@ -0,0 +1,188 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +export fft, ifft, fft!, ifft!, batched_fft, batched_ifft, batched_fft!, batched_ifft! + +const _FFT_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool,SUPPORTED_FLOAT_TYPES} +const _FFT_ACCEPTED = Union{SUPPORTED_COMPLEX_TYPES,_FFT_PROMOTABLE} + +_fft_eltype(::Type{ComplexF32}) = ComplexF32 +_fft_eltype(::Type{ComplexF64}) = ComplexF64 +_fft_eltype(::Type{Float32}) = ComplexF32 +_fft_eltype(::Type{Float64}) = ComplexF64 +_fft_eltype(::Type{T}) where {T<:_FFT_PROMOTABLE} = ComplexF64 + +function _fft_complex(A::NDArray{T}) where {T<:_FFT_ACCEPTED} + return unchecked_promote_arr(A, _fft_eltype(T)) +end + +function _fft_out_of_place( + A::NDArray{<:_FFT_ACCEPTED}, dims::NTuple{R,Int}, direction::Int32, scale::Bool +) where {R} + inp = _fft_complex(A) + out = cuNumeric.zeros(eltype(inp), size(inp)) + fft_task!(out, inp, dims, direction; scale) + # Promotion allocated a new complex buffer; the task already holds the store. + inp !== A && destroy!(inp) + return out +end + +""" + fft(A::NDArray, [dims]) + +Unnormalized complex discrete Fourier transform of `A`, matching +`AbstractFFTs.fft`. By default every dimension is transformed. `dims` selects a +subset (Julia 1-based dimensions). + +Real and integer inputs are converted to complex (`Float32` → `ComplexF32`, +everything else → `ComplexF64`). The result has the same shape as `A`. + +This is GPU-only: cupynumeric's FFT task has no CPU variant. Multi-GPU +execution is limited to batching over dimensions that are not transformed. + +There is no reusable cuFFT plan; each call launches the `CUPYNUMERIC_FFT` task. +""" +function fft(A::NDArray{<:_FFT_ACCEPTED}) + return _fft_out_of_place(A, _fft_dims(A), Int32(cuNumeric.FFT_FORWARD), false) +end +function fft(A::NDArray{<:_FFT_ACCEPTED}, dims) + return _fft_out_of_place(A, _fft_dims(A, dims), Int32(cuNumeric.FFT_FORWARD), false) +end + +function fft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in fft")) +end +function fft(A::NDArray, ::Any) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in fft")) +end + +""" + ifft(A::NDArray, [dims]) + +Normalized inverse discrete Fourier transform of `A`, matching +`AbstractFFTs.ifft`. Equivalent to the unnormalized inverse scaled by `1/N`, +where `N` is the product of the transformed lengths. + +See [`fft`](@ref). +""" +function ifft(A::NDArray{<:_FFT_ACCEPTED}) + return _fft_out_of_place(A, _fft_dims(A), Int32(cuNumeric.FFT_INVERSE), true) +end +function ifft(A::NDArray{<:_FFT_ACCEPTED}, dims) + return _fft_out_of_place(A, _fft_dims(A, dims), Int32(cuNumeric.FFT_INVERSE), true) +end + +function ifft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in ifft")) +end +function ifft(A::NDArray, ::Any) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in ifft")) +end + +""" + fft!(A::NDArray, [dims]) + +In-place [`fft`](@ref). `A` must already be `ComplexF32` or `ComplexF64`. +""" +function fft!(A::NDArray{T,N}) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(A, A, _fft_dims(A), Int32(cuNumeric.FFT_FORWARD)) +end +function fft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(A, A, _fft_dims(A, dims), Int32(cuNumeric.FFT_FORWARD)) +end + +function fft!(A::NDArray) + return throw(ArgumentError("fft! requires a complex NDArray; got $(eltype(A))")) +end +function fft!(A::NDArray, ::Any) + return throw(ArgumentError("fft! requires a complex NDArray; got $(eltype(A))")) +end + +""" + ifft!(A::NDArray, [dims]) + +In-place [`ifft`](@ref). `A` must already be `ComplexF32` or `ComplexF64`. +""" +function ifft!(A::NDArray{T,N}) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return ifft!(A, _fft_dims(A)) +end +function ifft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} + region = _fft_dims(A, dims) + return fft_task!(A, A, region, Int32(cuNumeric.FFT_INVERSE); scale=true) +end + +function ifft!(A::NDArray) + return throw(ArgumentError("ifft! requires a complex NDArray; got $(eltype(A))")) +end +function ifft!(A::NDArray, ::Any) + return throw(ArgumentError("ifft! requires a complex NDArray; got $(eltype(A))")) +end + +""" + cuNumeric.batched_fft(A) + +FFT every trailing dimension of `A`, treating `size(A, 1)` as a batch. + +`A` must be at least 2-d. A `(b, n)` stack is `b` independent 1-d transforms; +`(b, n, m)` is `b` independent 2-d transforms. This is the same +`CUPYNUMERIC_FFT` task as [`fft`](@ref); the leading axis is the one that may +be partitioned across GPUs. +""" +function batched_fft(A::NDArray{<:_FFT_ACCEPTED}) + return fft(A, _fft_batch_dims(A)) +end +function batched_fft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in batched_fft")) +end + +""" + cuNumeric.batched_ifft(A) + +Normalized inverse of [`batched_fft`](@ref). +""" +function batched_ifft(A::NDArray{<:_FFT_ACCEPTED}) + return ifft(A, _fft_batch_dims(A)) +end +function batched_ifft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in batched_ifft")) +end + +""" + cuNumeric.batched_fft!(A) + +In-place [`batched_fft`](@ref). `A` must already be complex. +""" +function batched_fft!(A::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return fft!(A, _fft_batch_dims(A)) +end +function batched_fft!(A::NDArray) + return throw(ArgumentError("batched_fft! requires a complex NDArray; got $(eltype(A))")) +end + +""" + cuNumeric.batched_ifft!(A) + +In-place [`batched_ifft`](@ref). `A` must already be complex. +""" +function batched_ifft!(A::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return ifft!(A, _fft_batch_dims(A)) +end +function batched_ifft!(A::NDArray) + return throw(ArgumentError("batched_ifft! requires a complex NDArray; got $(eltype(A))")) +end diff --git a/test/Project.toml b/test/Project.toml index 6e9f7cd1a..07239dc50 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -1,7 +1,7 @@ [deps] CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" -CUDACore = "bd0ed864-bdfe-4181-a5ed-ce625a5fdea2" +FFTW = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341" InteractiveUtils = "b77e0a4c-d291-57a0-90e8-8db25a27a240" LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc" diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index ce81c1cb0..887148d1d 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -242,6 +242,17 @@ end end end +@testset verbose = true "fft" begin + if cuNumeric.HAS_CUDA + a = cuNumeric.zeros(ComplexF32, 8) + b = cuNumeric.zeros(ComplexF32, 4, 6) + @test @inferred(fft(a)) !== nothing + @test @inferred(ifft(a)) !== nothing + @test @inferred(fft(b, 1)) !== nothing + @test @inferred(cuNumeric.batched_fft(b)) !== nothing + end +end + @testset verbose = true "batched_solve" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) A = cuNumeric.NDArray(reshape(T[2 1; 5 7], 1, 2, 2)) diff --git a/test/cuda.jl/fusion_compare.jl b/test/cuda.jl/fusion_compare.jl index be0d5090e..8d4c556c3 100644 --- a/test/cuda.jl/fusion_compare.jl +++ b/test/cuda.jl/fusion_compare.jl @@ -1,6 +1,5 @@ -using CUDA: CUDA, @cuda -using CUDACore: blockDim, blockIdx, threadIdx -import CUDACore: i32 +using CUDA: CUDA, @cuda, blockDim, blockIdx, threadIdx +import CUDA: i32 cuNumeric.Experimental(true) diff --git a/test/cuda.jl/fusion_compare_1d.jl b/test/cuda.jl/fusion_compare_1d.jl index 8efab2989..1fa018308 100644 --- a/test/cuda.jl/fusion_compare_1d.jl +++ b/test/cuda.jl/fusion_compare_1d.jl @@ -1,7 +1,6 @@ -using CUDA: CUDA, @cuda -using CUDACore: blockDim, blockIdx, threadIdx -import CUDACore: i32 +using CUDA: CUDA, @cuda, blockDim, blockIdx, threadIdx +import CUDA: i32 cuNumeric.Experimental(true) diff --git a/test/cuda.jl/padding.jl b/test/cuda.jl/padding.jl index f38e875a8..e4ee31d05 100644 --- a/test/cuda.jl/padding.jl +++ b/test/cuda.jl/padding.jl @@ -21,8 +21,8 @@ -- Validate custom-kernel padding, synchronization, and lifetime management =# -using CUDACore: blockDim, blockIdx, threadIdx -import CUDACore: i32 +using CUDA: blockDim, blockIdx, threadIdx +import CUDA: i32 cuNumeric.Experimental(true) diff --git a/test/cuda.jl/vecadd.jl b/test/cuda.jl/vecadd.jl index e2b5de552..723d110c8 100644 --- a/test/cuda.jl/vecadd.jl +++ b/test/cuda.jl/vecadd.jl @@ -21,8 +21,8 @@ -- Register various custom kernels using CUDA.jl =# -using CUDACore: blockDim, blockIdx, threadIdx -import CUDACore: i32 +using CUDA: blockDim, blockIdx, threadIdx +import CUDA: i32 cuNumeric.Experimental(true) diff --git a/test/gpu_only/fft.jl b/test/gpu_only/fft.jl new file mode 100644 index 000000000..4362aa192 --- /dev/null +++ b/test/gpu_only/fft.jl @@ -0,0 +1,166 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +using FFTW + +const _FFT_TEST_TYPES = Base.uniontypes(cuNumeric._FFT_ACCEPTED) +const _FFT_INPLACE_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_COMPLEX_TYPES) + +_fft_out_eltype(::Type{T}) where {T} = cuNumeric._fft_eltype(T) + +function _reference_fft(x::AbstractArray, dims=ntuple(identity, ndims(x))) + return FFTW.fft(_fft_out_eltype(eltype(x)).(x), dims) +end + +function _reference_ifft(x::AbstractArray, dims=ntuple(identity, ndims(x))) + return FFTW.ifft(_fft_out_eltype(eltype(x)).(x), dims) +end + +function _fft_compare(got::NDArray, expected; rtol, atol) + return @allowscalar isapprox(Array(got), expected; rtol=rtol, atol=atol) +end + +@testset verbose = true "fft/ifft" begin + @testset verbose = true for T in _FFT_TEST_TYPES + OT = _fft_out_eltype(T) + rtol_t = rtol(OT) + atol_t = atol(OT) + + @testset "1d" begin + x_cpu = my_rand(T, 16) + x = cuNumeric.NDArray(x_cpu) + y = fft(x) + @test eltype(y) === OT + @test _fft_compare(y, _reference_fft(x_cpu); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y), _reference_ifft(_reference_fft(x_cpu)); rtol=rtol_t, atol=atol_t + ) + end + + @testset "2d all dims" begin + x_cpu = my_rand(T, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = fft(x) + @test eltype(y) === OT + @test _fft_compare(y, _reference_fft(x_cpu); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y), _reference_ifft(_reference_fft(x_cpu)); rtol=rtol_t, atol=atol_t + ) + end + + @testset "2d dims=1" begin + x_cpu = my_rand(T, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = fft(x, 1) + @test _fft_compare(y, _reference_fft(x_cpu, (1,)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y, 1), _reference_ifft(_reference_fft(x_cpu, (1,)), (1,)); + rtol=rtol_t, atol=atol_t, + ) + end + + @testset "2d dims=2" begin + x_cpu = my_rand(T, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = fft(x, 2) + @test _fft_compare(y, _reference_fft(x_cpu, (2,)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y, 2), _reference_ifft(_reference_fft(x_cpu, (2,)), (2,)); + rtol=rtol_t, atol=atol_t, + ) + end + end +end + +@testset verbose = true "fft!/ifft!" begin + @testset verbose = true for T in _FFT_INPLACE_TYPES + rtol_t = rtol(T) + atol_t = atol(T) + x_cpu = my_rand(T, 16) + expected = _reference_fft(x_cpu) + + x = cuNumeric.NDArray(copy(x_cpu)) + y = fft!(x) + @test y === x + @test _fft_compare(x, expected; rtol=rtol_t, atol=atol_t) + + z = ifft!(x) + @test z === x + @test _fft_compare(x, x_cpu; rtol=rtol_t, atol=atol_t) + end + + @testset "fft! rejects non-complex" begin + for T in (Float32, Float64, Int32, Bool) + x = cuNumeric.NDArray(my_rand(T, 8)) + @test_throws ArgumentError fft!(x) + @test_throws ArgumentError ifft!(x) + end + end +end + +@testset verbose = true "batched_fft" begin + @testset verbose = true for T in _FFT_TEST_TYPES + OT = _fft_out_eltype(T) + rtol_t = rtol(OT) + atol_t = atol(OT) + + @testset "1d signals (b, n)" begin + x_cpu = my_rand(T, 4, 16) + x = cuNumeric.NDArray(x_cpu) + y = batched_fft(x) + @test eltype(y) === OT + @test _fft_compare(y, _reference_fft(x_cpu, (2,)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + batched_ifft(y), _reference_ifft(_reference_fft(x_cpu, (2,)), (2,)); + rtol=rtol_t, atol=atol_t, + ) + end + + @testset "2d fields (b, n, m)" begin + x_cpu = my_rand(T, 3, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = batched_fft(x) + @test _fft_compare(y, _reference_fft(x_cpu, (2, 3)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + batched_ifft(y), _reference_ifft(_reference_fft(x_cpu, (2, 3)), (2, 3)); + rtol=rtol_t, atol=atol_t, + ) + end + end + + @testset verbose = true for T in _FFT_INPLACE_TYPES + rtol_t = rtol(T) + atol_t = atol(T) + x_cpu = my_rand(T, 4, 16) + expected = _reference_fft(x_cpu, (2,)) + x = cuNumeric.NDArray(copy(x_cpu)) + y = batched_fft!(x) + @test y === x + @test _fft_compare(x, expected; rtol=rtol_t, atol=atol_t) + z = batched_ifft!(x) + @test z === x + @test _fft_compare(x, x_cpu; rtol=rtol_t, atol=atol_t) + end + + @testset "rejects 1d" begin + x = cuNumeric.NDArray(my_rand(ComplexF32, 8)) + @test_throws ArgumentError batched_fft(x) + @test_throws ArgumentError batched_fft!(x) + end +end diff --git a/test/runtests.jl b/test/runtests.jl index a7ca7bbb7..1ba7b10f4 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -1,10 +1,10 @@ using cuNumeric -using CUDACore: CUDACore +using CUDA: CUDA using ParallelTestRunner using Pkg using InteractiveUtils: versioninfo -run_gpu_tests = CUDACore.functional() +run_gpu_tests = CUDA.functional() @info "Julia information:\n" * sprint(io -> versioninfo(io)) @info "cuNumeric information:\n" * sprint(io -> cuNumeric.versioninfo(io)) @@ -17,6 +17,7 @@ const init_code = quote using LinearAlgebra using Random using StatsBase + using FFTW import Random: rand ENV["LEGATE_SKIP_RUNTIME"] = "false" From 2c5bb7c712cf232de6f9749e9d27db55cccd3c26 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sat, 22 Aug 2026 22:37:57 -0400 Subject: [PATCH 15/49] Support tensor contraction (#186) * Add pairwise tensor contract with mode labels. Wrap cupynumeric's binary contract kernel and the helpers it needs so a later TensorOperations backend can map Index2Tuple onto contract! without einsum strings. --- docs/src/linalg.md | 39 +- .../include/ndarray_c_api.h | 9 + lib/cunumeric_jl_wrapper/src/ndarray.cpp | 43 +++ src/cuNumeric.jl | 1 + src/ndarray/contract.jl | 352 ++++++++++++++++++ src/ndarray/detail/ndarray.jl | 56 +++ src/ndarray/diagonal.jl | 24 ++ src/ndarray/ndarray.jl | 75 +++- test/Project.toml | 1 + test/array/contract.jl | 266 +++++++++++++ test/array/diagonal.jl | 18 + test/array/permutedims.jl | 67 ++++ test/array/unary/tests.jl | 32 ++ 13 files changed, 981 insertions(+), 2 deletions(-) create mode 100644 src/ndarray/contract.jl create mode 100644 test/array/contract.jl create mode 100644 test/array/permutedims.jl diff --git a/docs/src/linalg.md b/docs/src/linalg.md index a47daf389..fac52b770 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -46,6 +46,41 @@ Pages = ["ndarray/binary.jl"] Filter = t -> t isa Function && nameof(t) === :mul! ``` +## Tensor contractions + +`contract` / `contract!` are the pairwise primitive behind a future +TensorOperations.jl backend. They take **mode labels**, not einsum strings: +`"ik"` with `"kj"` is a GEMM; `"ijk"` with `"ikl"` and explicit output `"ijl"` +is a batched product. Labels may be ASCII strings, `Char` tuples, or integers +(`1` maps to `'a'`). + +```julia +A = cuNumeric.rand(Float32, 64, 32) +B = cuNumeric.rand(Float32, 32, 16) +C = contract(A, "ik", B, "kj") # allocates, Einstein output order +contract!(similar(C), "ij", A, "ik", B, "kj"; α=2, β=0) + +# batched: keep the shared 'i' on the output +AA = cuNumeric.rand(Float32, 8, 16, 32) +BB = cuNumeric.rand(Float32, 8, 32, 4) +CC = cuNumeric.zeros(Float32, 8, 16, 4) +contract!(CC, "ijl", AA, "ijk", BB, "ikl") + +tensordot(AA, BB, ([3], [2])) # same contraction, axes form +``` + +`α` and `β` implement `C = β*C + α*(A ⋆ B)` in Julia. The C++ kernel always +writes the unscaled product; `β ≠ 0` uses a temporary. Duplicate labels inside +one array are not allowed — use `cuNumeric.diagonal` first. + +This path is multi-GPU via Legate tiling (per-tile cuTENSOR or TBLIS), not +cuTensorMp. Integer and `Bool` inputs promote to `Float64` like other linalg. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/contract.jl"] +``` + ## Solve `cuNumeric.solve(A, b)` solves a linear system and returns an array with the same @@ -268,7 +303,9 @@ must be `NDArray` unless noted. **Helpers on dense `NDArray`** -- `cuNumeric.diag` / `LinearAlgebra.diag` (2D → 1D), `cuNumeric.trace` / `LinearAlgebra.tr` (2D square → 0D) +- `cuNumeric.diag` / `LinearAlgebra.diag` (2D → 1D), `cuNumeric.diagonal` (two + axes of an `N`-D array → rank `N-1`), `cuNumeric.trace` / `LinearAlgebra.tr` + (2D square → 0D) - `iszero(A)` — all elements `== zero(T)` → 0-dimensional `NDArray{Bool}` - `isone(A)` — square 2D vs `_eye(T, n)` → 0-dimensional `NDArray{Bool}` (non-square → `false`) diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index dd612d188..b8af15bdc 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -72,6 +72,15 @@ CN_NDArray* nda_multiply_scalar(CN_NDArray* rhs1, CN_Type type, CN_NDArray* nda_add_scalar(CN_NDArray* rhs1, CN_Type type, const void* value); CN_NDArray* nda_dot(CN_NDArray* rhs1, CN_NDArray* rhs2); void nda_three_dot_arg(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out); +CN_NDArray* nda_transpose_axes(CN_NDArray* arr, const int32_t* axes, int32_t n); +CN_NDArray* nda_squeeze(CN_NDArray* arr, const int32_t* axes, int32_t n); +CN_NDArray* nda_diagonal(CN_NDArray* arr, int32_t offset, int32_t axis1, + int32_t axis2); +void nda_contract(CN_NDArray* out, const char* lhs_modes, int32_t n_lhs, + CN_NDArray* rhs1, const char* rhs1_modes, int32_t n_rhs1, + CN_NDArray* rhs2, const char* rhs2_modes, int32_t n_rhs2, + const char* extent_keys, const int32_t* extents, + int32_t n_extents); CN_NDArray* nda_copy(CN_NDArray* arr); void nda_assign(CN_NDArray* arr, CN_NDArray* other); diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index 5fb5190bf..e93ff7af7 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -30,7 +30,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -176,6 +178,47 @@ void nda_three_dot_arg(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out) { out->obj.dot(rhs1->obj, rhs2->obj); } +CN_NDArray* nda_transpose_axes(CN_NDArray* arr, const int32_t* axes, + int32_t n) { + std::vector axis_vec(axes, axes + n); + NDArray result = cupynumeric::transpose(arr->obj, std::move(axis_vec)); + return new CN_NDArray{NDArray(std::move(result))}; +} + +CN_NDArray* nda_squeeze(CN_NDArray* arr, const int32_t* axes, int32_t n) { + if (n <= 0) { + NDArray result = cupynumeric::squeeze(arr->obj, std::nullopt); + return new CN_NDArray{NDArray(std::move(result))}; + } + std::vector axis_vec(axes, axes + n); + NDArray result = cupynumeric::squeeze( + arr->obj, + std::optional const>>{ + axis_vec}); + return new CN_NDArray{NDArray(std::move(result))}; +} + +CN_NDArray* nda_diagonal(CN_NDArray* arr, int32_t offset, int32_t axis1, + int32_t axis2) { + NDArray result = cupynumeric::diagonal(arr->obj, offset, axis1, axis2, true); + return new CN_NDArray{NDArray(std::move(result))}; +} + +void nda_contract(CN_NDArray* out, const char* lhs_modes, int32_t n_lhs, + CN_NDArray* rhs1, const char* rhs1_modes, int32_t n_rhs1, + CN_NDArray* rhs2, const char* rhs2_modes, int32_t n_rhs2, + const char* extent_keys, const int32_t* extents, + int32_t n_extents) { + std::vector lhs(lhs_modes, lhs_modes + n_lhs); + std::vector r1(rhs1_modes, rhs1_modes + n_rhs1); + std::vector r2(rhs2_modes, rhs2_modes + n_rhs2); + std::map mode2extent; + for (int32_t i = 0; i < n_extents; ++i) { + mode2extent.emplace(extent_keys[i], extents[i]); + } + out->obj.contract(lhs, rhs1->obj, r1, rhs2->obj, r2, mode2extent); +} + CN_NDArray* nda_copy(CN_NDArray* arr) { NDArray result = arr->obj.copy(); return new CN_NDArray{NDArray(std::move(result))}; diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 6102c14ba..c57fd5b6a 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -190,6 +190,7 @@ include("ndarray/unary.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") include("ndarray/batched_linalg.jl") +include("ndarray/contract.jl") include("ndarray/fft.jl") include("scoping/scoping.jl") diff --git a/src/ndarray/contract.jl b/src/ndarray/contract.jl new file mode 100644 index 000000000..d5179ff5c --- /dev/null +++ b/src/ndarray/contract.jl @@ -0,0 +1,352 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +export contract!, contract, tensordot + +const _CONTRACT_NATIVE = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const _CONTRACT_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool} +const _CONTRACT_ACCEPTED = Union{_CONTRACT_NATIVE,_CONTRACT_PROMOTABLE} + +_contract_eltype(::Type{T}) where {T<:_CONTRACT_NATIVE} = T +_contract_eltype(::Type{<:_CONTRACT_PROMOTABLE}) = Float64 + +function _modes_as_chars(modes::AbstractString) + chars = Vector{UInt8}(undef, length(modes)) + for (i, c) in enumerate(modes) + isascii(c) || throw(ArgumentError("contract mode labels must be ASCII, got $(repr(c))")) + chars[i] = UInt8(c) + end + return chars +end + +function _modes_as_chars(modes::AbstractVector{UInt8}) + return collect(UInt8, modes) +end + +function _modes_as_chars(modes::AbstractVector{Char}) + chars = Vector{UInt8}(undef, length(modes)) + for (i, c) in enumerate(modes) + isascii(c) || throw(ArgumentError("contract mode labels must be ASCII, got $(repr(c))")) + chars[i] = UInt8(c) + end + return chars +end + +function _modes_as_chars(modes::AbstractVector{<:Integer}) + chars = Vector{UInt8}(undef, length(modes)) + for (i, m) in enumerate(modes) + v = Int(m) + (Int('a') - 1) + (0 <= v <= 255) || throw(ArgumentError("integer mode label $m is out of range")) + chars[i] = UInt8(v) + end + return chars +end + +function _modes_as_chars(modes::Tuple) + return _modes_as_chars(collect(modes)) +end + +function _require_modes(arr::NDArray, modes) + chars = _modes_as_chars(modes) + length(chars) == ndims(arr) || throw( + ArgumentError("expected $(ndims(arr)) mode labels, got $(length(chars))") + ) + if length(Base.unique(chars)) != length(chars) + throw(ArgumentError("duplicate mode labels are not allowed: $(modes)")) + end + return chars +end + +function _mode_extents(A::NDArray, Am, B::NDArray, Bm) + extents = Dict{UInt8,Int}() + for (m, s) in zip(Am, size(A)) + prev = get(extents, m, s) + prev == s || throw( + DimensionMismatch("mode $(Char(m)) has incompatible extents $prev and $s") + ) + extents[m] = s + end + for (m, s) in zip(Bm, size(B)) + prev = get(extents, m, s) + prev == s || throw( + DimensionMismatch("mode $(Char(m)) has incompatible extents $prev and $s") + ) + extents[m] = s + end + return extents +end + +function _check_mode_counts(Cm, Am, Bm) + counts = Dict{UInt8,Int}() + for m in Cm + counts[m] = get(counts, m, 0) + 1 + end + for m in Am + counts[m] = get(counts, m, 0) + 1 + end + for m in Bm + counts[m] = get(counts, m, 0) + 1 + end + for (m, c) in counts + (c == 2 || c == 3) || throw( + ArgumentError( + "mode $(Char(m)) appears $c times; each label must appear twice or three times across the output and inputs" + ), + ) + end + return nothing +end + +function _extent_arrays(extents) + n = length(extents) + keys_out = Vector{UInt8}(undef, n) + vals_out = Vector{Int32}(undef, n) + i = 0 + for (k, v) in extents + i += 1 + keys_out[i] = k + vals_out[i] = Int32(v) + end + return keys_out, vals_out +end + +# cupynumeric's MM shortcut asserts on untransposed NDArray shapes. Present a +# physical `ik,kj->ij` via Legate store transpose (no data copy) when needed. +function _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + if length(Cm) == 2 && length(Am) == 2 && length(Bm) == 2 + i, j = Cm[1], Cm[2] + k = (Am[1] == i || Am[1] == j) ? Am[2] : Am[1] + if k != i && k != j + left, Lm, right, Rm = (i == Am[1] || i == Am[2]) ? (A, Am, B, Bm) : (B, Bm, A, Am) + oL = Lm[1] != i + oR = Rm[2] != j + Lp = oL ? permutedims(left) : left + Rp = oR ? permutedims(right) : right + nda_contract(C, Cm, Lp, UInt8[i, k], Rp, UInt8[k, j], extent_keys, extent_vals) + oL && destroy!(Lp) + oR && destroy!(Rp) + return nothing + end + end + nda_contract(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + return nothing +end + +function _free_output_modes(A::NDArray, Am, B::NDArray, Bm) + counts = Dict{UInt8,Int}() + for m in Am + counts[m] = get(counts, m, 0) + 1 + end + for m in Bm + counts[m] = get(counts, m, 0) + 1 + end + Cm = UInt8[] + cshape = Int[] + for (m, s) in zip(Am, size(A)) + if counts[m] == 1 + push!(Cm, m) + push!(cshape, s) + end + end + for (m, s) in zip(Bm, size(B)) + if counts[m] == 1 + push!(Cm, m) + push!(cshape, s) + end + end + return Cm, Tuple(cshape) +end + +""" + contract!(C, Cmodes, A, Amodes, B, Bmodes; α=1, β=0) + +In-place pairwise tensor contraction `C = β * C + α * (A ⋆ B)`. + +`Amodes`, `Bmodes`, and `Cmodes` are mode labels for `A`, `B`, and `C`: an +ASCII `AbstractString`, a tuple/vector of `Char`, or a vector of integers +(`1` maps to `'a'`). Each label appears twice (a contracted or free index) or +three times (a batched / Hadamard index). Duplicate labels inside one array +are not allowed — extract a diagonal first with `cuNumeric.diagonal`. + +Supported element types are `Float32`, `Float64`, `ComplexF32`, and +`ComplexF64`. Integer and `Bool` inputs are converted to `Float64` under the +usual promotion rules. + +This is the pairwise primitive TensorOperations.jl can call later. It does not +parse einsum strings. Multi-GPU execution uses Legate tiling (not cuTensorMp). +""" +function contract!( + C::NDArray{TC}, + Cmodes, + A::NDArray{TA}, + Amodes, + B::NDArray{TB}, + Bmodes; + α=1, + β=0, +) where {TC<:_CONTRACT_NATIVE,TA<:_CONTRACT_ACCEPTED,TB<:_CONTRACT_ACCEPTED} + T = promote_type(_contract_eltype(TA), _contract_eltype(TB)) + T === TC || throw( + ArgumentError("contract! output has type $TC, but inputs promote to $T") + ) + + Ap = checked_promote_arr(contract!, A, T) + Bp = checked_promote_arr(contract!, B, T) + try + return _contract_same_type!(C, Cmodes, Ap, Amodes, Bp, Bmodes, convert(T, α), convert(T, β)) + finally + Ap !== A && destroy!(Ap) + Bp !== B && destroy!(Bp) + end +end + +function contract!(C::NDArray, Cmodes, A::NDArray, Amodes, B::NDArray, Bmodes; α=1, β=0) + bad = if eltype(C) <: _CONTRACT_NATIVE + (eltype(A) <: _CONTRACT_ACCEPTED ? eltype(B) : eltype(A)) + else + eltype(C) + end + return throw(ArgumentError("array type $bad is unsupported in contract!")) +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::T, + β::T, +) where {T} + (C.ptr === A.ptr || C.ptr === B.ptr) && throw( + ArgumentError("contract! output must not alias either input") + ) + + Cm = _require_modes(C, Cmodes) + Am = _require_modes(A, Amodes) + Bm = _require_modes(B, Bmodes) + _check_mode_counts(Cm, Am, Bm) + extents = _mode_extents(A, Am, B, Bm) + for (m, s) in zip(Cm, size(C)) + expected = get(extents, m, -1) + expected == s || throw( + DimensionMismatch("output mode $(Char(m)) has size $s, expected $expected") + ) + end + extent_keys, extent_vals = _extent_arrays(extents) + + if isone(α) && iszero(β) + _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + return C + end + if iszero(β) + _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= α .* C + return C + end + tmp = similar(C) + _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= β .* C .+ α .* tmp + destroy!(tmp) + return C +end + +""" + contract(A, Amodes, B, Bmodes; α=1) + +Allocate and return `α * (A ⋆ B)`. Output modes are the labels that appear +once, in the order they occur on `A` then `B` (classical Einstein). For a +batched or Hadamard product, allocate `C` yourself and call [`contract!`](@ref) +with explicit `Cmodes`. +""" +function contract( + A::NDArray{TA}, Amodes, B::NDArray{TB}, Bmodes; α=1 +) where {TA<:_CONTRACT_ACCEPTED,TB<:_CONTRACT_ACCEPTED} + T = promote_type(_contract_eltype(TA), _contract_eltype(TB)) + T <: _CONTRACT_NATIVE || + throw(ArgumentError("array type $T is unsupported in contract")) + + Am = _require_modes(A, Amodes) + Bm = _require_modes(B, Bmodes) + Cm, cshape = _free_output_modes(A, Am, B, Bm) + C = cuNumeric.zeros(T, cshape) + return contract!(C, Cm, A, Am, B, Bm; α=α, β=zero(T)) +end + +function contract(A::NDArray, Amodes, B::NDArray, Bmodes; α=1) + bad = eltype(A) <: _CONTRACT_ACCEPTED ? eltype(B) : eltype(A) + return throw(ArgumentError("array type $bad is unsupported in contract")) +end + +""" + tensordot(A, B, axes=2; α=1) + tensordot(A, B, (a_axes, b_axes); α=1) + +Contract `A` with `B` along the given 1-based axes. `axes::Integer` contracts +the last `axes` dimensions of `A` with the first `axes` of `B`. A tuple of axis +collections names the axes on each input. Remaining axes of `A` then `B` become +the output. +""" +function tensordot(A::NDArray, B::NDArray, axes::Integer=2; α=1) + n = Int(axes) + n < 0 && throw(ArgumentError("axes must be non-negative, got $n")) + na = ndims(A) + nb = ndims(B) + n > na && throw(ArgumentError("cannot contract $n axes of a $(na)-d array")) + n > nb && throw(ArgumentError("cannot contract $n axes of a $(nb)-d array")) + a_axes = ntuple(i -> na - n + i, n) + b_axes = ntuple(identity, n) + return tensordot(A, B, (a_axes, b_axes); α=α) +end + +function tensordot(A::NDArray, B::NDArray, axes::Tuple; α=1) + a_raw, b_raw = axes + a_axes = collect(Int, a_raw isa Integer ? (a_raw,) : a_raw) + b_axes = collect(Int, b_raw isa Integer ? (b_raw,) : b_raw) + length(a_axes) == length(b_axes) || + throw(ArgumentError("tensordot axis lists must have the same length")) + length(Base.unique(a_axes)) == length(a_axes) || + throw(ArgumentError("duplicate axes on first input: $a_axes")) + length(Base.unique(b_axes)) == length(b_axes) || + throw(ArgumentError("duplicate axes on second input: $b_axes")) + + na = ndims(A) + nb = ndims(B) + Am = Vector{UInt8}(undef, na) + Bm = Vector{UInt8}(undef, nb) + for i in 1:na + Am[i] = UInt8('a' + (i - 1)) + end + for i in 1:nb + Bm[i] = UInt8('A' + (i - 1)) + end + for (ai, bi) in zip(a_axes, b_axes) + (1 <= ai <= na) || throw(ArgumentError("axis $ai is out of range for $(na)-d array")) + (1 <= bi <= nb) || throw(ArgumentError("axis $bi is out of range for $(nb)-d array")) + size(A, ai) == size(B, bi) || throw( + DimensionMismatch( + "tensordot axes $ai and $bi have sizes $(size(A, ai)) and $(size(B, bi))" + ), + ) + Bm[bi] = Am[ai] + end + return contract(A, Am, B, Bm; α=α) +end diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index a8b6f8118..76aaad99e 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -521,6 +521,62 @@ function nda_transpose(arr::NDArray{T,N}) where {T,N} return NDArray(ptr, T, Val(N)) end +# Arbitrary axis permutation; rank is preserved. `axes` are 0-based. +function nda_transpose_axes(arr::NDArray{T,N}, axes::Vector{Int32}) where {T,N} + axes_c = collect(Int32, axes) + ptr = @task_scope "permutedims" begin + ccall((:nda_transpose_axes, libnda), + NDArray_t, (NDArray_t, Ptr{Int32}, Int32), + arr.ptr, axes_c, Int32(length(axes_c))) + end + return NDArray(ptr, T, Val(N)) +end + +# Rank-changing: drop size-1 axes. Empty `axes` drops every size-1 axis. +function nda_squeeze(arr::NDArray, axes::Vector{Int32}) + axes_c = collect(Int32, axes) + ptr = @task_scope "squeeze" begin + ccall((:nda_squeeze, libnda), + NDArray_t, (NDArray_t, Ptr{Int32}, Int32), + arr.ptr, axes_c, Int32(length(axes_c))) + end + return NDArray(ptr) +end + +function nda_diagonal(arr::NDArray, offset::Int32, axis1::Int32, axis2::Int32) + ptr = @task_scope "diagonal" begin + ccall((:nda_diagonal, libnda), + NDArray_t, (NDArray_t, Int32, Int32, Int32), + arr.ptr, offset, axis1, axis2) + end + return NDArray(ptr) +end + +function nda_contract( + out::NDArray, + lhs_modes::Vector{UInt8}, + rhs1::NDArray, + rhs1_modes::Vector{UInt8}, + rhs2::NDArray, + rhs2_modes::Vector{UInt8}, + extent_keys::Vector{UInt8}, + extents::Vector{Int32}, +) + @task_scope "contract" begin + ccall((:nda_contract, libnda), + Cvoid, + ( + NDArray_t, Ptr{UInt8}, Int32, NDArray_t, Ptr{UInt8}, Int32, + NDArray_t, Ptr{UInt8}, Int32, Ptr{UInt8}, Ptr{Int32}, Int32, + ), + out.ptr, lhs_modes, Int32(length(lhs_modes)), + rhs1.ptr, rhs1_modes, Int32(length(rhs1_modes)), + rhs2.ptr, rhs2_modes, Int32(length(rhs2_modes)), + extent_keys, extents, Int32(length(extents))) + end + return out +end + function nda_attach_external(arr::Array{T,N}; shape::Dims{N}=size(arr)) where {T,N} st = Legate.attach_external_row_major(arr; shape) # Use the CxxWrap method for type-safe interaction diff --git a/src/ndarray/diagonal.jl b/src/ndarray/diagonal.jl index ae1d306ad..97d34ea57 100644 --- a/src/ndarray/diagonal.jl +++ b/src/ndarray/diagonal.jl @@ -1,5 +1,7 @@ ###### diag / _eye / trace ###### +export diagonal + @doc""" cuNumeric.diag(arr::NDArray; k=0) @@ -11,6 +13,28 @@ end LinearAlgebra.diag(arr::NDArray{<:Any,2}, k::Integer=0) = nda_diag(arr, Int32(k)) +""" + cuNumeric.diagonal(arr::NDArray; offset=0, dims=(1, 2)) + +Extract the diagonal of `arr` along two 1-based axes `dims`. The result has rank +`ndims(arr) - 1`, with the diagonal stored as the last axis. `offset` selects a +superdiagonal (`> 0`) or subdiagonal (`< 0`), matching `LinearAlgebra.diag`. + +For a 2D matrix this is the same 1D diagonal as [`diag`](@ref). For `N > 2` the +non-diagonal axes are kept in order, followed by the extracted diagonal. +""" +function diagonal(arr::NDArray; offset::Integer=0, dims::NTuple{2,Integer}=(1, 2)) + nd = ndims(arr) + nd < 2 && throw(ArgumentError("diagonal requires ndims >= 2")) + d1 = Int(dims[1]) + d2 = Int(dims[2]) + (1 <= d1 <= nd && 1 <= d2 <= nd) || throw( + ArgumentError("dims $dims are out of range for $(nd)-d array") + ) + d1 == d2 && throw(ArgumentError("diagonal dims must be distinct, got $dims")) + return nda_diagonal(arr, Int32(offset), Int32(d1 - 1), Int32(d2 - 1)) +end + # Internal dense identity used by UniformScaling / Diagonal densify helpers. # Prefer `LinearAlgebra.I` / `NDArray{T}(I, n, n)` / `one(A)` in user code. function _eye(::Type{T}, rows::Int) where {T} diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 437d69d5c..f6fc181d8 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -18,7 +18,7 @@ * Nader Rahhal =# -export unwrap +export unwrap, squeeze # See TODO.md (Base / LinearAlgebra sections) for AbstractArray and LA gaps. @@ -31,6 +31,79 @@ function transpose(arr::NDArray) return nda_transpose(arr) end +""" + Base.permutedims(arr::NDArray, perm) + +Permute the dimensions of `arr` according to `perm`, a 1-based permutation of +`1:ndims(arr)`. Rank is preserved. See also [`transpose`](@ref). +""" +function Base.permutedims(arr::NDArray{T,N}, perm) where {T,N} + length(perm) == N || throw( + ArgumentError("permutation length $(length(perm)) does not match ndims = $N") + ) + p = ntuple(i -> Int(perm[i]), Val(N)) + used = Base.falses(N) + axes = Vector{Int32}(undef, N) + for i in 1:N + d = p[i] + (1 <= d <= N) || throw(ArgumentError("permutation index $d is out of range for ndims = $N")) + used[d] && throw(ArgumentError("permutation $perm is not a permutation of 1:$N")) + used[d] = true + axes[i] = Int32(d - 1) + end + return nda_transpose_axes(arr, axes) +end + +Base.permutedims(arr::NDArray{<:Any,2}) = permutedims(arr, (2, 1)) + +""" + squeeze(arr::NDArray) + squeeze(arr::NDArray, dims) + +Drop size-1 dimensions of `arr`. With no `dims`, every size-1 axis is removed. +With `dims` (a 1-based integer or collection), only those axes are removed and +each must have size 1. + +See also `Base.dropdims`. +""" +function squeeze(arr::NDArray) + any(==(1), size(arr)) || return arr + return nda_squeeze(arr, Int32[]) +end + +function squeeze(arr::NDArray, dims) + axes = _squeeze_axes(arr, dims) + isempty(axes) && return arr + return nda_squeeze(arr, axes) +end + +function Base.dropdims(arr::NDArray; dims) + return squeeze(arr, dims) +end + +function _squeeze_axes(arr::NDArray, dims) + nd = ndims(arr) + axes = Int32[] + seen = Set{Int}() + for d in _as_dims(dims) + ax = Int(d) + (1 <= ax <= nd) || + throw(ArgumentError("dimension $ax is out of range for $(nd)-d array")) + ax in seen && throw(ArgumentError("duplicate dimension $ax")) + push!(seen, ax) + size(arr, ax) == 1 || throw( + DimensionMismatch( + "cannot drop dimension $ax of size $(size(arr, ax)); expected size 1" + ), + ) + push!(axes, Int32(ax - 1)) + end + return axes +end + +_as_dims(d::Integer) = (Int(d),) +_as_dims(dims) = Tuple(Int(d) for d in dims) + @doc""" cuNumeric.ravel(arr::NDArray) diff --git a/test/Project.toml b/test/Project.toml index 07239dc50..6b09222cd 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -8,6 +8,7 @@ ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" diff --git a/test/array/contract.jl b/test/array/contract.jl new file mode 100644 index 000000000..888a1394e --- /dev/null +++ b/test/array/contract.jl @@ -0,0 +1,266 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +using TensorOperations + +# Contractions sum products; TBLIS/cuTENSOR/BLAS may associate differently than +# the host ref. Default (`cancel=true`): Higham floor like unary reductions +# (`n` = contracted length, `scale` ≈ max|A| * max|B|). All-positive inputs +# set `cancel=false` and drop that floor. No-sum cases keep n=1. +function _host_contract_compare( + ref, out, ::Type{T}; n::Integer=1, scale=1, cancel::Bool=true +) where {T} + atolv, rtolv = if n > 1 && cancel + reduction_atol(T, n, scale), reduction_rtol(T, n) + elseif n > 1 + atol(T) * n, reduction_rtol(T, n) + else + atol(T), rtol(T) + end + allowscalar() do + @test safe_compare(ref, out, atolv, rtolv) + end +end + +_contract_scale(A, B) = maximum(abs, A) * maximum(abs, B) + +# TensorOperations forbids an index on both inputs and the output (batched / +# Hadamard). Those refs are explicit loops or broadcasting. +function _batched_ref(A::Array{T,3}, B::Array{T,3}) where {T} + ni, nj, nk = size(A) + nl = size(B, 3) + C = zeros(T, ni, nj, nl) + for i in 1:ni, j in 1:nj, l in 1:nl + s = zero(T) + for k in 1:nk + s += A[i, j, k] * B[i, k, l] + end + C[i, j, l] = s + end + return C +end + +@testset "contract GEMM" begin + @testset for T in (Float32, Float64, ComplexF32, ComplexF64) + A = my_rand(T, 5, 4) + B = my_rand(T, 4, 6) + nda = NDArray(A) + ndb = NDArray(B) + nk = size(A, 2) + scale = _contract_scale(A, B) + @tensor ref_ij[i, j] := A[i, k] * B[k, j] + @tensor ref_ji[j, i] := A[i, k] * B[k, j] + + C = contract(nda, "ik", ndb, "kj") + @test size(C) == (5, 6) + _host_contract_compare(ref_ij, C, T; n=nk, scale) + + out = cuNumeric.zeros(T, 5, 6) + contract!(out, "ij", nda, "ik", ndb, "kj") + _host_contract_compare(ref_ij, out, T; n=nk, scale) + + C_int = contract(nda, (1, 2), ndb, (2, 3)) + _host_contract_compare(ref_ij, C_int, T; n=nk, scale) + + # Every MM layout: A/B axis order, operand swap, output permutation. + ndaT = permutedims(nda) + ndbT = permutedims(ndb) + for (Ause, Am) in ((nda, "ik"), (ndaT, "ki")) + for (Buse, Bm) in ((ndb, "kj"), (ndbT, "jk")) + for swap in (false, true) + X, Xm, Y, Ym = swap ? (Buse, Bm, Ause, Am) : (Ause, Am, Buse, Bm) + Cij = cuNumeric.zeros(T, 5, 6) + contract!(Cij, "ij", X, Xm, Y, Ym) + _host_contract_compare(ref_ij, Cij, T; n=nk, scale) + Cji = cuNumeric.zeros(T, 6, 5) + contract!(Cji, "ji", X, Xm, Y, Ym) + _host_contract_compare(ref_ji, Cji, T; n=nk, scale) + end + end + end + end +end + +@testset "contract MV / VV / outer" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 5, 4) + x = my_rand(T, 4) + y = my_rand(T, 5) + nda = NDArray(A) + ndx = NDArray(x) + ndy = NDArray(y) + + @tensor ref_mv[i] := A[i, j] * x[j] + _host_contract_compare( + ref_mv, contract(nda, "ij", ndx, "j"), T; n=length(x), scale=_contract_scale(A, x) + ) + _host_contract_compare( + ref_mv, contract(ndx, "j", nda, "ij"), T; n=length(x), scale=_contract_scale(A, x) + ) + + @tensor ref_vm[i] := A[j, i] * y[j] + _host_contract_compare( + ref_vm, contract(nda, "ji", ndy, "j"), T; n=length(y), scale=_contract_scale(A, y) + ) + _host_contract_compare( + ref_vm, contract(ndy, "j", nda, "ji"), T; n=length(y), scale=_contract_scale(A, y) + ) + + u = my_rand(T, 6) + v = my_rand(T, 6) + @tensor ref_dot[] := u[i] * v[i] + dot = contract(NDArray(u), "i", NDArray(v), "i") + @test ndims(dot) == 0 + _host_contract_compare(ref_dot, dot, T; n=length(u), scale=_contract_scale(u, v)) + + @tensor ref_outer[i, j] := u[i] * v[j] + outer = contract(NDArray(u), "i", NDArray(v), "j") + @test size(outer) == (6, 6) + _host_contract_compare(ref_outer, outer, T) + end +end + +@testset "contract batched" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 3, 4, 5) + B = my_rand(T, 3, 5, 6) + nda = NDArray(A) + ndb = NDArray(B) + ref = _batched_ref(A, B) + nk = size(A, 3) + scale = _contract_scale(A, B) + # All 6 output orders of (i, j, l) + for (cm, p) in ( + ("ijl", (1, 2, 3)), + ("ilj", (1, 3, 2)), + ("jil", (2, 1, 3)), + ("jli", (2, 3, 1)), + ("lij", (3, 1, 2)), + ("lji", (3, 2, 1)), + ) + C = cuNumeric.zeros(T, map(d -> size(ref, d), p)...) + contract!(C, cm, nda, "ijk", ndb, "ikl") + _host_contract_compare(permutedims(ref, p), C, T; n=nk, scale) + end + # Input axis orders (general path, not MM) + C = cuNumeric.zeros(T, 3, 4, 6) + contract!(C, "ijl", permutedims(nda, (2, 1, 3)), "jik", ndb, "ikl") + _host_contract_compare(ref, C, T; n=nk, scale) + contract!(C, "ijl", nda, "ijk", permutedims(ndb, (2, 1, 3)), "kil") + _host_contract_compare(ref, C, T; n=nk, scale) + end +end + +@testset "contract Hadamard" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 4, 5) + B = my_rand(T, 4, 5) + ref = A .* B + C = cuNumeric.zeros(T, 4, 5) + contract!(C, "ij", NDArray(A), "ij", NDArray(B), "ij") + _host_contract_compare(ref, C, T) + Cji = cuNumeric.zeros(T, 5, 4) + contract!(Cji, "ji", NDArray(A), "ij", NDArray(B), "ij") + _host_contract_compare(permutedims(ref), Cji, T) + end +end + +@testset "contract alpha/beta" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 4, 3) + B = my_rand(T, 3, 5) + nda = NDArray(A) + ndb = NDArray(B) + α2, α3, β3 = T(2), T(3), T(3) + nk = size(A, 2) + scale = _contract_scale(A, B) + @tensor prod[i, j] := A[i, k] * B[k, j] + + C = contract(nda, "ik", ndb, "kj"; α=α2) + _host_contract_compare(α2 * prod, C, T; n=nk, scale=α2 * scale) + + out = cuNumeric.zeros(T, 4, 5) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α3, β=0) + _host_contract_compare(α3 * prod, out, T; n=nk, scale=α3 * scale) + + seed = my_rand(T, 4, 5) + out = NDArray(copy(seed)) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α2, β=β3) + _host_contract_compare(β3 * seed + α2 * prod, out, T; n=nk, scale=α2 * scale) + + # α/β on a non-canonical MM layout + out = NDArray(copy(seed)) + contract!(out, "ij", permutedims(nda), "ki", ndb, "kj"; α=α2, β=β3) + _host_contract_compare(β3 * seed + α2 * prod, out, T; n=nk, scale=α2 * scale) + + seed_ji = permutedims(seed) + out_ji = NDArray(copy(seed_ji)) + contract!(out_ji, "ji", nda, "ik", ndb, "kj"; α=α2, β=β3) + _host_contract_compare( + β3 * seed_ji + α2 * permutedims(prod), out_ji, T; n=nk, scale=α2 * scale + ) + end +end + +@testset "tensordot" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 3, 4, 5) + B = my_rand(T, 5, 6) + nda = NDArray(A) + ndb = NDArray(B) + @tensor ref[i, j, l] := A[i, j, k] * B[k, l] + + C = tensordot(nda, ndb, 1) + @test size(C) == (3, 4, 6) + _host_contract_compare(ref, C, T; n=size(A, 3), scale=_contract_scale(A, B)) + + D = tensordot(nda, ndb, ([3], [1])) + _host_contract_compare(ref, D, T; n=size(A, 3), scale=_contract_scale(A, B)) + end +end + +@testset "contract no cancellation" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 5, 4; L=one(T), R=T(1000)) + B = my_rand(T, 4, 6; L=one(T), R=T(1000)) + nk = size(A, 2) + @tensor ref[i, j] := A[i, k] * B[k, j] + _host_contract_compare( + ref, contract(NDArray(A), "ik", NDArray(B), "kj"), T; n=nk, cancel=false + ) + + A3 = my_rand(T, 3, 4, 5; L=one(T), R=T(1000)) + B3 = my_rand(T, 3, 5, 6; L=one(T), R=T(1000)) + C = cuNumeric.zeros(T, 3, 4, 6) + contract!(C, "ijl", NDArray(A3), "ijk", NDArray(B3), "ikl") + _host_contract_compare(_batched_ref(A3, B3), C, T; n=size(A3, 3), cancel=false) + end +end + +@testset "contract errors" begin + A = cuNumeric.ones(Float32, 3, 4) + B = cuNumeric.ones(Float32, 4, 5) + @test_throws ArgumentError contract(A, "ii", B, "jk") + @test_throws ArgumentError contract(A, "ik", B, "kjx") + @test_throws DimensionMismatch contract(A, "ik", cuNumeric.ones(Float32, 3, 5), "kj") + C = cuNumeric.zeros(Float64, 3, 5) + @test_throws ArgumentError contract!(C, "ij", A, "ik", B, "kj") + @test_throws ArgumentError contract!(A, "ij", A, "ik", B, "kj") +end diff --git a/test/array/diagonal.jl b/test/array/diagonal.jl index a04e16291..52f2ee21d 100644 --- a/test/array/diagonal.jl +++ b/test/array/diagonal.jl @@ -69,6 +69,24 @@ end end end +@testset "N-D diagonal extract" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 4, 4, 3) + nda = NDArray(A) + ref = [A[i, i, k] for k in 1:3, i in 1:4] + _host_diag_compare(ref, cuNumeric.diagonal(nda; dims=(1, 2)), T) + A2 = copy(A[:, :, 1]) + _host_diag_compare(diag(A2), cuNumeric.diagonal(NDArray(A2)), T) + + B = my_rand(T, 3, 5, 5) + ndb = NDArray(B) + ref_b = [B[i, j, j] for i in 1:3, j in 1:5] + _host_diag_compare(ref_b, cuNumeric.diagonal(ndb; dims=(2, 3)), T) + end + @test_throws ArgumentError cuNumeric.diagonal(cuNumeric.ones(4); dims=(1, 1)) + @test_throws ArgumentError cuNumeric.diagonal(cuNumeric.ones(2, 3); dims=(1, 1)) +end + @testset "identity via I / _eye" begin @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES n = 4 diff --git a/test/array/permutedims.jl b/test/array/permutedims.jl new file mode 100644 index 000000000..526d0fa43 --- /dev/null +++ b/test/array/permutedims.jl @@ -0,0 +1,67 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +@testset "permutedims" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 3, 4) + nda = NDArray(A) + allowscalar() do + @test safe_compare(permutedims(A), cuNumeric.transpose(nda), atol(T), rtol(T)) + @test safe_compare(permutedims(A, (2, 1)), permutedims(nda, (2, 1)), atol(T), rtol(T)) + @test safe_compare(permutedims(A), permutedims(nda), atol(T), rtol(T)) + end + + B = my_rand(T, 2, 3, 4) + ndb = NDArray(B) + allowscalar() do + @test safe_compare( + permutedims(B, (3, 1, 2)), permutedims(ndb, (3, 1, 2)), atol(T), rtol(T) + ) + @test safe_compare( + permutedims(B, (1, 2, 3)), permutedims(ndb, (1, 2, 3)), atol(T), rtol(T) + ) + end + end + @test_throws ArgumentError permutedims(cuNumeric.ones(2, 3), (1,)) + @test_throws ArgumentError permutedims(cuNumeric.ones(2, 3), (1, 1)) + @test_throws ArgumentError permutedims(cuNumeric.ones(2, 3), (1, 3)) +end + +@testset "squeeze / dropdims" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 2, 3) + nda = NDArray(reshape(A, 2, 1, 3)) + out = squeeze(nda) + @test size(out) == (2, 3) + allowscalar() do + @test safe_compare(A, out, atol(T), rtol(T)) + end + @test size(squeeze(NDArray(A))) == (2, 3) + + dropped = dropdims(nda; dims=2) + @test size(dropped) == (2, 3) + allowscalar() do + @test safe_compare(A, dropped, atol(T), rtol(T)) + end + @test size(squeeze(nda, 2)) == (2, 3) + @test_throws DimensionMismatch squeeze(nda, 1) + @test_throws ArgumentError dropdims(nda; dims=4) + end +end diff --git a/test/array/unary/tests.jl b/test/array/unary/tests.jl index 509ee9eb3..13f35af02 100644 --- a/test/array/unary/tests.jl +++ b/test/array/unary/tests.jl @@ -333,6 +333,38 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) end end + # All-positive: |sum| ≈ Σ|xᵢ|, so the Higham input-magnitude floor is + # unnecessary. Keep O(n) rtol for association; this still flags a + # wrong kernel that cancellation tols would swallow. + @testset "sum no cancellation" begin + @testset for T in types + T <: AbstractFloat || continue + x = my_rand(T, N; L=one(T), R=T(1000)) + ndx = @allowscalar NDArray(x) + allowscalar() do + @test isapprox( + sum(x), + unwrap(sum(ndx)); + atol=atol(T) * N, + rtol=reduction_rtol(T, N), + ) + end + x2 = my_rand(T, isqrt(N), isqrt(N); L=one(T), R=T(1000)) + nd2 = @allowscalar NDArray(x2) + n1 = size(x2, 1) + allowpromotion(true) do + allowscalar() do + @test safe_compare( + sum(x2; dims=1), + sum(nd2; dims=1), + atol(T) * n1, + reduction_rtol(T, n1), + ) + end + end + end + end + if include_bool_reductions # Test things that only work on Booleans julia_bools = rand(Bool, N) From 2d48197804425f80bfe68304f46f4f9a2876bf58 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sun, 23 Aug 2026 22:59:16 -0400 Subject: [PATCH 16/49] Add sort, sort!, sortperm, searchsortedfirst, searchsortedlast (#181) * sort, sort!, searchsorted --------- Co-authored-by: krasow Co-authored-by: David Krasowska --- .buildkite/run_developer_ci.sh | 2 +- README.md | 2 +- benchmark/__plot_results.jl | 301 ++++++++++++++++++ docs/src/api.md | 2 +- .../include/ndarray_c_api.h | 9 + lib/cunumeric_jl_wrapper/src/ndarray.cpp | 41 +++ src/cuNumeric.jl | 1 + src/ndarray/detail/ndarray.jl | 36 +++ src/ndarray/ndarray.jl | 9 - src/ndarray/sort.jl | 136 ++++++++ src/ndarray/unary.jl | 7 +- src/scoping/inter_broadcast_fusion.jl | 2 +- src/scoping/scoping.jl | 2 +- test/analysis/type_stability.jl | 16 + test/array/linalg.jl | 12 - test/array/sort.jl | 147 +++++++++ 16 files changed, 698 insertions(+), 27 deletions(-) create mode 100644 benchmark/__plot_results.jl create mode 100644 src/ndarray/sort.jl create mode 100644 test/array/sort.jl diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 4d130393c..bba74a57c 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -20,7 +20,7 @@ sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" export PATH="$CMAKE_ROOT/bin:$PATH" cmake --version -# Clean slate so cached state doesn't leak across Julia versions. +# Clean slate so cached state doesn't leak across Julia versions. rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml diff --git a/README.md b/README.md index cad89042a..8c8132e51 100644 --- a/README.md +++ b/README.md @@ -91,4 +91,4 @@ More worked examples (initialization, Gray-Scott, …) are in the documentation ### Known Limitations -- There is no support for `Float16` or `ComplexF16` +- There is no support for `Float16` or `ComplexF16` or `Complex{<:Integer}` diff --git a/benchmark/__plot_results.jl b/benchmark/__plot_results.jl new file mode 100644 index 000000000..fd1ed646a --- /dev/null +++ b/benchmark/__plot_results.jl @@ -0,0 +1,301 @@ +#!/usr/bin/env julia +# Weak-scaling plots (1/2/4/8 GPUs) for the benchmark result CSVs. +# One figure per benchmark, three panels: throughput, time/step, parallel efficiency. +# +# CSV schema (see src/core.jl save_result): +# implementation,gpus,N,M,trial,time_ms,throughput,correctness +# `throughput` is the benchmark's `total_flops` divided by elapsed time. For +# Gray-Scott that unit is Gpoint-updates/s; other benchmarks report GFLOP/s. +# +# save_result appends and does NOT encode the code-path variant, so a cuNumeric CSV +# holds alternating runs: baseline, @accelerate, baseline, @accelerate, ... +# (a run boundary = the GPU count resetting downward). +# cuPyNumeric / CUDA.jl have no accelerated path -> a single block. +# +# Encoding: color = implementation; line style = code path +# solid = baseline, dashed = @accelerate. + +using Plots +using Statistics + +gr() + +function parse_args(args) + results_dir = "results" + out_dir = nothing + single_cunumeric_run = :baseline + hide_baseline = false + output_suffix = "" + + for arg in args + if startswith(arg, "--single-cu=") + value = Symbol(lowercase(last(split(arg, "="; limit=2)))) + value in (:baseline, :accelerated) || + error("--single-cu must be baseline or accelerated") + single_cunumeric_run = value + elseif arg == "--hide-baseline" + hide_baseline = true + elseif startswith(arg, "--out=") + out_dir = last(split(arg, "="; limit=2)) + elseif startswith(arg, "--suffix=") + output_suffix = last(split(arg, "="; limit=2)) + else + results_dir = arg + end + end + isempty(output_suffix) && hide_baseline && (output_suffix = "_no_baseline") + + results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) + if out_dir === nothing + out_dir = if basename(normpath(results_dir)) == "results" + joinpath(@__DIR__, "plots") + else + joinpath(@__DIR__, "plots", basename(normpath(results_dir))) + end + else + out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) + end + return (; results_dir, out_dir, single_cunumeric_run, hide_baseline, output_suffix) +end + +const CONFIG = parse_args(ARGS) +const RESULTS_DIR = CONFIG.results_dir +const OUT_DIR = CONFIG.out_dir +const SINGLE_CUNUMERIC_RUN = CONFIG.single_cunumeric_run +const HIDE_BASELINE = CONFIG.hide_baseline +const OUTPUT_SUFFIX = CONFIG.output_suffix + +# filekey, family label, color, marker, can_contain_accelerated_blocks +const FAMILIES = [ + ("cunumeric", "cuNumeric.jl (fused)", "#2a78d6", :circle, true), + ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond, true), + ("cupynumeric", "cuPyNumeric", "#eb6834", :rect, false), + ("CUDA.jl", "CUDA.jl", "#008300", :utriangle, false), +] + +const INK = "#0b0b0b" +const MUTED = "#898781" +const GRIDCOL = "#e1e0d9" +const IDEALCOL = "#c3c2b7" + +struct Row + gpus::Int + time_ms::Float64 + thr::Float64 +end + +# Parse a CSV into runs, split wherever the GPU count resets to a smaller value. +function load_runs(path) + rows = Row[] + for line in eachline(path) + isempty(strip(line)) && continue + f = split(line, ',') + push!(rows, Row(parse(Int, f[2]), parse(Float64, f[6]), parse(Float64, f[7]))) + end + isempty(rows) && return Vector{Row}[] + runs = [Row[]] + for (i, r) in enumerate(rows) + i > 1 && r.gpus < rows[i - 1].gpus && push!(runs, Row[]) + push!(runs[end], r) + end + return runs +end + +# Aggregate trials per GPU count -> sorted vector of (gpus, t, tsd, h, hsd). +function aggregate(rows) + by = Dict{Int,Vector{Row}}() + for r in rows + push!(get!(by, r.gpus, Row[]), r) + end + sd(x) = length(x) > 1 ? std(x) : 0.0 + return [ + (gpus=g, t=mean(getfield.(by[g], :time_ms)), tsd=sd(getfield.(by[g], :time_ms)), + h=mean(getfield.(by[g], :thr)), hsd=sd(getfield.(by[g], :thr))) + for g in sort(collect(keys(by))) + ] +end + +# Build the series (color+marker+linestyle+agg) present for one benchmark. +function series_for(bench) + series = [] # NamedTuple(label,color,marker,ls,agg) + for (key, fam, color, marker, splits) in FAMILIES + path = joinpath(RESULTS_DIR, "$(bench)_$(key).csv") + isfile(path) || continue + runs = load_runs(path) + isempty(runs) && continue + if splits && bench == "grayscott" + if length(runs) == 1 + label = + SINGLE_CUNUMERIC_RUN === :accelerated ? "$fam · accelerated" : + "$fam · baseline" + ls = SINGLE_CUNUMERIC_RUN === :accelerated ? :dash : :solid + push!( + series, (label=label, color=color, marker=marker, ls=ls, + agg=aggregate(runs[1])) + ) + else + # Repeated harness runs append alternating baseline/accelerated blocks. + baseline_rows = reduce(vcat, runs[1:2:end]) + accelerated_rows = reduce(vcat, runs[2:2:end]) + push!( + series, + (label="$fam · baseline", color=color, marker=marker, + ls=:solid, agg=aggregate(baseline_rows)), + ) + push!(series, + (label="$fam · accelerated", color=color, marker=marker, + ls=:dash, agg=aggregate(accelerated_rows))) + end + else + push!( + series, + (label=fam, color=color, marker=marker, ls=:solid, + agg=aggregate(reduce(vcat, runs))), + ) + end + end + return series +end + +function throughput_label(bench) + return bench == "grayscott" ? + "Throughput (Gpoint-updates/s)" : "Throughput (GFLOP/s)" +end + +function addline!(p, s, y; kw...) + return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, + lw=2.2, ls=s.ls, marker=s.marker, ms=6, msc=s.color, markerstrokewidth=0.8, + label=s.label, kw...) +end + +# One legend key: a short line sample (+ optional marker) with a text label. +function swatch!(p, x, y, color, ls, marker, label) + plot!(p, [x, x + 0.032], [y, y]; color=color, lw=2.6, ls=ls, label="") + marker !== nothing && scatter!(p, [x + 0.016], [y]; color=color, marker=marker, + ms=6, msc=color, markerstrokewidth=0.8, label="") + return annotate!(p, x + 0.045, y, text(label, 9, INK, :left)) +end + +# Grouped legend: color/marker = implementation, line style = code path. +function build_legend(series) + pl = plot(; framestyle=:none, legend=false, xlims=(0, 1), ylims=(0, 1), + grid=false, ticks=false) + # implementations present, in FAMILIES order, matched by color + present = [ + (fam, color, marker) for (key, fam, color, marker, _) in FAMILIES + if any(s.color == color for s in series) + ] + annotate!(pl, 0.015, 0.74, text("Implementation", 10, INK, :left)) + xs = range(0.18, 0.80; length=max(length(present), 1)) + for ((fam, color, marker), x) in zip(present, xs) + swatch!(pl, x, 0.74, color, :solid, marker, fam) + end + annotate!(pl, 0.015, 0.26, text("Line style", 10, INK, :left)) + has_baseline = any(endswith(s.label, "· baseline") for s in series) + has_accelerated = any(endswith(s.label, "· accelerated") for s in series) + if has_baseline && has_accelerated + swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") + swatch!(pl, 0.40, 0.26, MUTED, :dash, nothing, "@accelerate") + swatch!(pl, 0.70, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + elseif has_accelerated + swatch!(pl, 0.18, 0.26, MUTED, :dash, nothing, "@accelerate") + swatch!(pl, 0.52, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + elseif has_baseline + swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") + swatch!(pl, 0.48, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + else + swatch!(pl, 0.18, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + end + return pl +end + +function positive_ylim(vals; pad=0.12) + isempty(vals) && return (0, 1) + hi = maximum(vals) + hi > 0 || return (0, 1) + return (0, hi * (1 + pad)) +end + +function main() + mkpath(OUT_DIR) + files = filter(f -> endswith(f, ".csv"), readdir(RESULTS_DIR)) + benches = unique( + String[ + m.captures[1] for f in files for (key, _, _, _, _) in FAMILIES + for m in (match(Regex("^(.*)_" * replace(key, "." => "\\.") * "\\.csv\$"), f),) + if m !== nothing + ], + ) + + for bench in benches + series = series_for(bench) + if HIDE_BASELINE + series = filter(s -> !endswith(s.label, "· baseline"), series) + end + isempty(series) && continue + + common = (xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", + framestyle=:box, grid=true, gridcolor=GRIDCOL, gridalpha=1.0, + foreground_color_text=INK, tickfontcolor=MUTED, legend=false, + xlims=(0.85, 9.4)) + + # Panel 1: throughput (higher better) + throughput = [x.h for s in series for x in s.agg] + p1 = plot(; ylabel=throughput_label(bench), title="Throughput", + ylims=positive_ylim(throughput), common...) + for s in series + addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) + end + + # Panel 2: time per step (lower better; ideal = flat) + p2 = plot(; ylabel="Time / step (ms)", title="Time per step", common...) + for s in series + addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) + end + + # Panel 3: parallel efficiency = thr(p)/(p*thr(1)); ideal = 1.0 + efficiencies = Float64[] + for s in series + i1 = findfirst(x -> x.gpus == 1, s.agg) + i1 === nothing && continue + base = s.agg[i1].h + append!(efficiencies, [x.h/(x.gpus*base) for x in s.agg]) + end + p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", + ylims=positive_ylim(vcat(efficiencies, [1.0])), common...) + hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.4, label="") + for s in series + i1 = findfirst(x -> x.gpus == 1, s.agg) + i1 === nothing && continue + base = s.agg[i1].h + addline!(p3, s, [x.h/(x.gpus*base) for x in s.agg]) + end + + # grouped legend panel: color/marker = implementation, style = code path + pl = build_legend(series) + has_baseline = any(endswith(s.label, "· baseline") for s in series) + has_accelerated = any(endswith(s.label, "· accelerated") for s in series) + style_title = if has_baseline && has_accelerated + "solid = baseline · dashed = @accelerate" + elseif has_accelerated + "dashed = @accelerate" + elseif has_baseline + "solid = baseline" + else + "implementation comparison" + end + + fig = plot(p1, p2, p3, pl; layout=@layout([grid(1, 3); leg{0.16h}]), + size=(1400, 600), dpi=200, + plot_title=titlecase(bench) * " — weak scaling ($style_title)", + plot_titlefontsize=12, left_margin=6Plots.mm, + bottom_margin=6Plots.mm, top_margin=4Plots.mm, + background_color="#fcfcfb") + + out = joinpath(OUT_DIR, "$(bench)_weak_scaling$(OUTPUT_SUFFIX).png") + savefig(fig, out) + println("wrote $out") + end +end + +main() diff --git a/docs/src/api.md b/docs/src/api.md index 864e8d4f5..4e58b7d47 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -4,6 +4,6 @@ Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, a ```@autodocs Modules = [cuNumeric] -Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl", "scoping/accelerate.jl"] +Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "ndarray/sort.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl", "scoping/accelerate.jl"] Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!, :randn, :randn!, :randexp, :randexp!, :default_rng, :random, :random!)) ``` diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index b8af15bdc..ab78b2f2f 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -108,6 +108,15 @@ CN_NDArray* nda_get_slice(CN_NDArray* arr, const CN_Slice* slices, CN_NDArray* nda_attach_external(const void* ptr, size_t size, int dim, const uint64_t* shape, CN_Type type); +// axis is 0-based. stable=true maps to cupynumeric kind="stable". +CN_NDArray* nda_sort(CN_NDArray* arr, int32_t axis, bool stable); +void nda_sort_inplace(CN_NDArray* arr, int32_t axis, bool stable); +CN_NDArray* nda_argsort(CN_NDArray* arr, int32_t axis, bool stable); +// a is 1-D sorted; v is the needle array (any rank, same dtype). +// left=true is NumPy side='left'; left=false is side='right'. +// Returns int64 indices with v's shape (0-based). +CN_NDArray* nda_searchsorted(CN_NDArray* a, CN_NDArray* v, bool left); + #ifdef __cplusplus } #endif diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index e93ff7af7..91ba89617 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -30,6 +30,7 @@ #include #include #include +#include #include #include #include @@ -130,6 +131,24 @@ CN_NDArray* nda_unique(CN_NDArray* arr) { return new CN_NDArray{NDArray(std::move(result))}; } +CN_NDArray* nda_sort(CN_NDArray* arr, int32_t axis, bool stable) { + const char* kind = stable ? "stable" : "quicksort"; + NDArray result = + cupynumeric::sort(arr->obj, std::optional{axis}, kind); + return new CN_NDArray{NDArray(std::move(result))}; +} + +void nda_sort_inplace(CN_NDArray* arr, int32_t axis, bool stable) { + arr->obj.sort(arr->obj, false, std::optional{axis}, stable); +} + +CN_NDArray* nda_argsort(CN_NDArray* arr, int32_t axis, bool stable) { + const char* kind = stable ? "stable" : "quicksort"; + NDArray result = + cupynumeric::argsort(arr->obj, std::optional{axis}, kind); + return new CN_NDArray{NDArray(std::move(result))}; +} + CN_NDArray* nda_ravel(CN_NDArray* arr) { NDArray result = cupynumeric::ravel(arr->obj, "C"); return new CN_NDArray{NDArray(std::move(result))}; @@ -476,4 +495,26 @@ CN_NDArray* nda_get_slice(CN_NDArray* arr, const CN_Slice* slices, CN_NDArray* nda_store_to_ndarray(CN_Store* st) { return new CN_NDArray{cupynumeric::as_array(st->obj)}; } + +// Mirrors cupynumeric deferred.searchsorted: fill + MIN/MAX reduction. +CN_NDArray* nda_searchsorted(CN_NDArray* a, CN_NDArray* v, bool left) { + auto* runtime = cupynumeric::CuPyNumericRuntime::get_runtime(); + NDArray out = runtime->create_array(v->obj.shape(), legate::int64()); + const int64_t n = static_cast(a->obj.size()); + out.fill(Scalar(left ? n : int64_t{0})); + + auto task = runtime->create_task(CuPyNumericOpCode::CUPYNUMERIC_SEARCHSORTED); + auto p_out = + task.add_reduction(out.get_store(), left ? legate::ReductionOpKind::MIN + : legate::ReductionOpKind::MAX); + task.add_input(a->obj.get_store()); + auto p_v = task.add_input(v->obj.get_store()); + task.add_constraint(legate::broadcast(p_v)); + task.add_constraint(legate::broadcast(p_out)); + task.add_constraint(legate::align(p_out, p_v)); + task.add_scalar_arg(Scalar(left)); + task.add_scalar_arg(Scalar(n)); + runtime->submit(std::move(task)); + return new CN_NDArray{NDArray(std::move(out))}; +} } // extern "C" diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index c57fd5b6a..7a01faa52 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -189,6 +189,7 @@ include("ndarray/random/random.jl") include("ndarray/unary.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") +include("ndarray/sort.jl") include("ndarray/batched_linalg.jl") include("ndarray/contract.jl") include("ndarray/fft.jl") diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 76aaad99e..ac029934f 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -430,6 +430,42 @@ function nda_unique(arr::NDArray{T}) where {T} return NDArray(ptr, T, Val(1)) end +function nda_sort(arr::NDArray{T,N}, axis::Int32, stable::Bool) where {T,N} + ptr = @task_scope "sort" begin + ccall((:nda_sort, libnda), + NDArray_t, (NDArray_t, Int32, Bool), + arr.ptr, axis, stable) + end + return NDArray(ptr, T, Val(N)) +end + +function nda_sort_inplace(arr::NDArray, axis::Int32, stable::Bool) + @task_scope "sort!" begin + ccall((:nda_sort_inplace, libnda), + Cvoid, (NDArray_t, Int32, Bool), + arr.ptr, axis, stable) + end + return arr +end + +function nda_argsort(arr::NDArray{<:Any,N}, axis::Int32, stable::Bool) where {N} + ptr = @task_scope "argsort" begin + ccall((:nda_argsort, libnda), + NDArray_t, (NDArray_t, Int32, Bool), + arr.ptr, axis, stable) + end + return NDArray(ptr, Int64, Val(N)) +end + +function nda_searchsorted(a::NDArray, v::NDArray{<:Any,N}, left::Bool) where {N} + ptr = @task_scope "searchsorted" begin + ccall((:nda_searchsorted, libnda), + NDArray_t, (NDArray_t, NDArray_t, Bool), + a.ptr, v.ptr, left) + end + return NDArray(ptr, Int64, Val(N)) +end + function nda_ravel(arr::NDArray) ptr = @task_scope "ravel" begin ccall((:nda_ravel, libnda), diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index f6fc181d8..9e517e552 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -113,15 +113,6 @@ function ravel(arr::NDArray) return nda_ravel(arr) end -@doc""" - cuNumeric.unique(arr::NDArray) - -Return a new `NDArray` containing the unique elements of the input `arr`. -""" -function unique(arr::NDArray) - return nda_unique(arr) -end - @doc""" Base.copy(arr::NDArray) diff --git a/src/ndarray/sort.jl b/src/ndarray/sort.jl new file mode 100644 index 000000000..72164e9d0 --- /dev/null +++ b/src/ndarray/sort.jl @@ -0,0 +1,136 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +# Julia 1-based dims -> 0-based cupynumeric axis. +function _checked_sort_axis(N::Int, dims::Integer) + 1 <= dims <= N || throw(ArgumentError("dims=$dims is invalid for a $N-d array")) + return Int32(dims - 1) +end + +@doc""" + cuNumeric.sort(v::NDArray{T,1}; stable::Bool=false) + cuNumeric.sort(A::NDArray; dims::Integer, stable::Bool=false) + +Return a copy of `A` sorted in ascending order. + +This is **not** `Base.sort`. Call it as `cuNumeric.sort`. `alg`, `lt`, `by`, +`rev`, and `order` are not accepted. + +For rank greater than 1, `dims` is required (Julia 1-based). `stable=true` +sets the cupynumeric stable-sort flag (`kind="stable"`); the default is +unstable (`kind="quicksort"`). These names do not run Julia's QuickSort or +MergeSort. Complex values are ordered lexicographically by `(real, imag)`. +""" +function sort(arr::NDArray{T,1}; stable::Bool=false) where {T} + return nda_sort(arr, Int32(-1), stable) +end + +function sort(arr::NDArray{T,N}; dims::Integer, stable::Bool=false) where {T,N} + return nda_sort(arr, _checked_sort_axis(N, dims), stable) +end + +@doc""" + cuNumeric.sort!(v::NDArray{T,1}; stable::Bool=false) + cuNumeric.sort!(A::NDArray; dims::Integer, stable::Bool=false) + +Sort `A` in place. Same kwargs as [`cuNumeric.sort`](@ref). Not `Base.sort!`. +""" +function sort!(arr::NDArray{T,1}; stable::Bool=false) where {T} + return nda_sort_inplace(arr, Int32(-1), stable) +end + +function sort!(arr::NDArray{T,N}; dims::Integer, stable::Bool=false) where {T,N} + return nda_sort_inplace(arr, _checked_sort_axis(N, dims), stable) +end + +function _searchsorted_impl(a::NDArray{TA,1}, v::NDArray{TV,N}, left::Bool) where {TA,TV,N} + (TA <: Complex || TV <: Complex) && + throw(ArgumentError("searchsorted is not supported for complex arrays")) + U = promote_type(TA, TV) + a2 = unchecked_promote_arr(a, U) + v2 = unchecked_promote_arr(v, U) + raw = nda_searchsorted(a2, v2, left) + a2 !== a && destroy!(a2) + v2 !== v && destroy!(v2) + return left ? _indices_to_one_based(raw) : raw +end + +@doc""" + cuNumeric.searchsortedfirst(a::NDArray{T,1}, x) + cuNumeric.searchsortedlast(a::NDArray{T,1}, x) + +Insertion indices into a 1-d sorted `a`. `x` may be a `Number` or an +`NDArray` of needles. + +Scalar queries return a 0-d `NDArray{Int64}` (not a Julia `Int`); use +`unwrap` for an `Int`. Array queries return an `NDArray{Int64}` with the +shape of the needles. Indices are 1-based. + +Not `Base.searchsortedfirst` / `searchsortedlast`. `a` must already be sorted +ascending. `lt` / `by` / `rev` are not accepted. Complex arrays are not +supported. +""" +function searchsortedfirst(a::NDArray{T,1}, v::NDArray) where {T} + return _searchsorted_impl(a, v, true) +end + +function searchsortedfirst(a::NDArray{T,1}, x::Number) where {T} + needle = NDArray(convert(T, x)) + result = searchsortedfirst(a, needle) + destroy!(needle) + return result +end + +function searchsortedlast(a::NDArray{T,1}, v::NDArray) where {T} + return _searchsorted_impl(a, v, false) +end + +function searchsortedlast(a::NDArray{T,1}, x::Number) where {T} + needle = NDArray(convert(T, x)) + result = searchsortedlast(a, needle) + destroy!(needle) + return result +end + +@doc""" + cuNumeric.searchsorted(a::NDArray{T,1}, x::Number) + +`searchsortedfirst(a, x):searchsortedlast(a, x)` as a `UnitRange`, matching +Base's scalar search. Materializes two 0-d index arrays via `unwrap`. +Not `Base.searchsorted`. +""" +function searchsorted(a::NDArray{T,1}, x::Number) where {T} + lo = searchsortedfirst(a, x) + hi = searchsortedlast(a, x) + return unwrap(lo):unwrap(hi) +end + +@doc""" + cuNumeric.unique(A::NDArray) -> NDArray{T,1} + +Sorted unique elements of `A`, flattened to 1-d. + +This is **not** `Base.unique`, which keeps first-occurrence order and does +not sort. `dims`, `return_index`, `return_inverse`, and `return_counts` +are not accepted; cupynumeric does not implement them. +""" +function unique(arr::NDArray) + return nda_unique(arr) +end diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index d27a78951..2d980cb42 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -524,7 +524,6 @@ end """ argmax(A::NDArray{<:Any,1}) - argmin(A::NDArray{<:Any,1}) 1-based index of the first extremum, as a 0-d `NDArray{Int64}` (not a Julia `Int`). 1-d only. Complex arrays are not supported. @@ -535,6 +534,12 @@ function argmax(arr::NDArray{T,1}) where {T} return _indices_to_one_based(raw) end +""" + argmin(A::NDArray{<:Any,1}) + +1-based index of the first extremum, as a 0-d `NDArray{Int64}` (not a Julia +`Int`). 1-d only. Complex arrays are not supported. +""" function argmin(arr::NDArray{T,1}) where {T} T <: Complex && throw(ArgumentError("argmax/argmin are not supported for complex arrays")) raw = nda_unary_reduction_axes(cuNumeric.ARGMIN, arr, Int32[], false) diff --git a/src/scoping/inter_broadcast_fusion.jl b/src/scoping/inter_broadcast_fusion.jl index d69bff4bd..86730055b 100644 --- a/src/scoping/inter_broadcast_fusion.jl +++ b/src/scoping/inter_broadcast_fusion.jl @@ -70,7 +70,7 @@ function _source_indices(expr, replacement_sources) append!(indices, get(replacement_sources, symbol, Int[])) end unique!(indices) - sort!(indices) + Base.sort!(indices) return indices end diff --git a/src/scoping/scoping.jl b/src/scoping/scoping.jl index dec17913f..eac7b7744 100644 --- a/src/scoping/scoping.jl +++ b/src/scoping/scoping.jl @@ -89,7 +89,7 @@ function _assigned_symbols(expr) end function _lexical_scope(body, bindings::Set{Symbol}) - ordered = sort!(collect(bindings); by=string) + ordered = Base.sort!(collect(bindings); by=string) return Expr(:let, Expr(:block, ordered...), body) end diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index 887148d1d..5160ade55 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -270,6 +270,22 @@ end @test @inferred(cuNumeric.transpose(M)) !== nothing @test @inferred(cuNumeric.trace(sq)) !== nothing @test @inferred(cuNumeric.diag(sq)) !== nothing + end +end + +@testset verbose = true "sort" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) + v = cuNumeric.zeros(T, 8) + @test @inferred(cuNumeric.sort(v)) !== nothing + if !(T <: Complex) + @test @inferred(cuNumeric.searchsortedfirst(v, zero(T))) !== nothing + end + end +end + +@testset verbose = true "unique" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) + v = cuNumeric.zeros(T, 8) @test @inferred(cuNumeric.unique(v)) !== nothing end end diff --git a/test/array/linalg.jl b/test/array/linalg.jl index 3a204c1bf..55640b67f 100644 --- a/test/array/linalg.jl +++ b/test/array/linalg.jl @@ -174,18 +174,6 @@ end # end # end -@testset "unique" begin - @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) - A = T[1, 2, 2, 3, 4, 4, 4, 5] - nda = cuNumeric.NDArray(A) - - ref = unique(A) - out = cuNumeric.unique(nda) - - @test Set(Array(out)) == Set(ref) - end -end - @testset "solve diagonal" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) n = 4 diff --git a/test/array/sort.jl b/test/array/sort.jl new file mode 100644 index 000000000..ee96378ff --- /dev/null +++ b/test/array/sort.jl @@ -0,0 +1,147 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +const SORT_TYPES = (Bool, Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES)...) + +# Julia has no isless(::Complex, ::Complex). Use this only as Base.sort's `by` +# so the reference matches cupynumeric's (real, imag) order. +_lex_complex(x) = (real(x), imag(x)) + +function _sort_fixture(::Type{T}) where {T} + T <: Bool && return Bool[1, 0, 1, 0, 0, 1, 1, 0, 1, 0, 0, 1] + T <: Complex && + return T[ + 10 + 3im, + 2 + 4im, + 1 + 5im, + 8 + 9im, + 3 + 1im, + 2 + 4im, + 10 + 3im, + 1, + 5 + 2im, + 0, + 4, + 7 + im, + ] + return T[10, 3, 12, 5, 2, 4, 8, 9, 7, 6, 11, 1] +end + +function _search_haystack(::Type{T}) where {T} + T <: Bool && return Bool[0, 0, 1, 1] + return T[1, 2, 2, 4, 5, 7] +end + +function _search_needles(::Type{T}) where {T} + T <: Bool && return Bool[0, 1] + return T[0, 1, 2, 3, 7, 8] +end + +function _unique_fixture(::Type{T}) where {T} + T <: Bool && return Bool[1, 0, 1, 0, 1] + T <: Complex && return T[1, 2, 2, 3 + im, 3 + im, 1] + return T[1, 2, 2, 3, 4, 4, 4, 5] +end + +@testset "sort 1-d" begin + @testset verbose = true for T in SORT_TYPES + A = _sort_fixture(T) + nda = cuNumeric.NDArray(A) + @test Array(cuNumeric.sort(nda)) == + (T <: Complex ? Base.sort(A; by=_lex_complex) : Base.sort(A)) + @test Array(nda) == A # cuNumeric.sort is not in-place + end +end + +@testset "sort! 1-d" begin + @testset verbose = true for T in SORT_TYPES + A = _sort_fixture(T) + nda = cuNumeric.NDArray(copy(A)) + cuNumeric.sort!(nda) + @test Array(nda) == (T <: Complex ? Base.sort(A; by=_lex_complex) : Base.sort(A)) + end +end + +@testset "sort dims" begin + @testset verbose = true for T in SORT_TYPES + A = reshape(_sort_fixture(T), 3, 4) + nda = cuNumeric.NDArray(A) + @test Array(cuNumeric.sort(nda; dims=1)) == + (T <: Complex ? Base.sort(A; dims=1, by=_lex_complex) : Base.sort(A; dims=1)) + @test Array(cuNumeric.sort(nda; dims=2)) == + (T <: Complex ? Base.sort(A; dims=2, by=_lex_complex) : Base.sort(A; dims=2)) + @test_throws "invalid for a" cuNumeric.sort(nda; dims=3) + @test_throws "invalid for a" cuNumeric.sort(nda; dims=0) + @test_throws UndefKeywordError cuNumeric.sort(nda) # dims required for N>1 + end +end + +@testset "unsupported kwargs and not Base.sort" begin + v = cuNumeric.NDArray(Int32[3, 1, 2]) + @test_throws MethodError cuNumeric.sort(v; alg=Base.QuickSort) + @test_throws MethodError cuNumeric.sort(v; rev=true) + @test_throws MethodError cuNumeric.sort(v; lt=(!)) + # cuNumeric.sort is not Base.sort: Julia arrays are a MethodError + @test_throws MethodError cuNumeric.sort([3, 1, 2]) + @test_throws MethodError cuNumeric.sort!([3, 1, 2]) + @test_throws MethodError cuNumeric.searchsortedfirst([1, 2, 3], 2) + @test_throws MethodError cuNumeric.unique([1, 1, 2]) +end + +@testset "searchsorted" begin + @testset verbose = true for T in SORT_TYPES + A = _search_haystack(T) + nda = cuNumeric.sort(cuNumeric.NDArray(A)) + if T <: Complex + @test_throws "not supported for complex" cuNumeric.searchsortedfirst(nda, zero(T)) + @test_throws "not supported for complex" cuNumeric.searchsortedlast(nda, zero(T)) + @test_throws "not supported for complex" cuNumeric.searchsorted(nda, zero(T)) + continue + end + for x in _search_needles(T) + @test cuNumeric.unwrap(cuNumeric.searchsortedfirst(nda, x)) == + Base.searchsortedfirst(A, x) + @test cuNumeric.unwrap(cuNumeric.searchsortedlast(nda, x)) == + Base.searchsortedlast(A, x) + @test cuNumeric.searchsorted(nda, x) == Base.searchsorted(A, x) + end + needles = _search_needles(T) + firsts = Array(cuNumeric.searchsortedfirst(nda, cuNumeric.NDArray(needles))) + lasts = Array(cuNumeric.searchsortedlast(nda, cuNumeric.NDArray(needles))) + @test firsts == Base.searchsortedfirst.(Ref(A), needles) + @test lasts == Base.searchsortedlast.(Ref(A), needles) + M = cuNumeric.NDArray(reshape(A, 2, :)) + @test_throws MethodError cuNumeric.searchsortedfirst(M, zero(T)) + end +end + +@testset "unique" begin + @testset verbose = true for T in SORT_TYPES + A = _unique_fixture(T) + out = Array(cuNumeric.unique(cuNumeric.NDArray(A))) + @test Set(out) == Set(Base.unique(A)) + if !(T <: Complex) + @test Base.issorted(out) + end + B = reshape(_sort_fixture(T), 3, 4) + @test Set(Array(cuNumeric.unique(cuNumeric.NDArray(B)))) == Set(Base.unique(vec(B))) + end + @test_throws MethodError cuNumeric.unique(cuNumeric.NDArray(Int32[1, 1]); dims=1) +end From b7aa2fc78293ff8debee8f0346c26aee17ec8cb5 Mon Sep 17 00:00:00 2001 From: ejmeitz Date: Mon, 24 Aug 2026 12:38:28 -0500 Subject: [PATCH 17/49] remvoe some unecessary global keywords --- src/cuNumeric.jl | 2 -- src/ndarray/binary.jl | 4 ++-- src/ndarray/unary.jl | 6 +++--- 3 files changed, 5 insertions(+), 7 deletions(-) diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 7a01faa52..3e636a7f7 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -213,8 +213,6 @@ function my_on_exit() return drain_pending_frees!() # flush before Legate tears down end -global cuNumeric_config_str::String = "" - ### These functions guard against a user trying ### to start multiple runtimes and also to allow ## package extensions which always try to re-load diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 5428f701a..b16dd4fd1 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -8,7 +8,7 @@ # # truncated `div`. Do not ship as `div`: `-7 ÷ 2` is -3 in Julia, -4 for fld. # Binary ops which are equivalent to Julia's broadcast syntax -global const binary_op_map = Dict{Function,BinaryOpCode}( +const binary_op_map = Dict{Function,BinaryOpCode}( Base.:+ => cuNumeric.ADD, Base.:* => cuNumeric.MULTIPLY, Base.:(-) => cuNumeric.SUBTRACT, @@ -38,7 +38,7 @@ global const binary_op_map = Dict{Function,BinaryOpCode}( # Base.:(||) => (cuNumeric.LOGICAL_OR, Bool, :same_as_input), # cannot overload ) -global const floaty_binary_op_map = Dict{Function,BinaryOpCode}( +const floaty_binary_op_map = Dict{Function,BinaryOpCode}( Base.:/ => cuNumeric.DIVIDE, Base.hypot => cuNumeric.HYPOT, Base.atan => cuNumeric.ARCTAN2, diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 2d980cb42..79f7a2edd 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -1,4 +1,4 @@ -global const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( +const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( Base.acos => cuNumeric.ARCCOS, Base.acosh => cuNumeric.ARCCOSH, Base.asin => cuNumeric.ARCSIN, @@ -24,7 +24,7 @@ global const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( Base.tanh => cuNumeric.TANH, ) -global const unary_op_map_no_args = Dict{Function,UnaryOpCode}( +const unary_op_map_no_args = Dict{Function,UnaryOpCode}( Base.abs => cuNumeric.ABSOLUTE, # Base.conj => cuNumeric.CONJ, # handled as a special case below Base.:(-) => cuNumeric.NEGATIVE, @@ -289,7 +289,7 @@ sum(B, dims=2) # 3×1 result sum(B, dims=(1,2)) # 1×1 result ``` """ -global const unary_reduction_map = Dict{Function,UnaryRedCode}( +const unary_reduction_map = Dict{Function,UnaryRedCode}( # ARGMAX/ARGMIN: 1-d Base.argmax/argmin below, not this map. #missing => cuNumeric.CONTAINS, # strings or also integral types Base.maximum => cuNumeric.MAX, From 52c97834ae7b65c69444e11a21620c17641461a4 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Mon, 24 Aug 2026 15:03:07 -0400 Subject: [PATCH 18/49] Implement == and != as 0-d Bool reductions. (#187) Match Julia array equality (including mixed dtypes and shape mismatch) without falling back to cupynumeric's length-1 1-d array_equal result. --- .../include/ndarray_c_api.h | 3 + src/ndarray/binary.jl | 12 ---- src/ndarray/detail/ndarray.jl | 11 ++++ src/ndarray/ndarray.jl | 57 ++++++++++++++----- test/array/binary/float.jl | 1 + test/array/binary/tests.jl | 39 ++++++++++++- test/util.jl | 6 +- 7 files changed, 99 insertions(+), 30 deletions(-) diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index ab78b2f2f..3fe6b0896 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -96,6 +96,9 @@ uint64_t nda_nbytes(CN_NDArray* arr); void nda_binary_op(CN_NDArray* out, CuPyNumericBinaryOpCode op_code, const CN_NDArray* rhs1, const CN_NDArray* rhs2); +void nda_binary_reduction(CN_NDArray* out, CuPyNumericBinaryOpCode op_code, + const CN_NDArray* rhs1, const CN_NDArray* rhs2); +CN_NDArray* nda_array_equal(const CN_NDArray* rhs1, const CN_NDArray* rhs2); void nda_unary_op(CN_NDArray* out, CuPyNumericUnaryOpCode op_code, CN_NDArray* input); void nda_unary_reduction(CN_NDArray* out, CuPyNumericUnaryRedCode op_code, diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index b16dd4fd1..7f4231e02 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -299,18 +299,6 @@ end return result end -# function Base.:(==)(lhs::NDArray{A}, rhs::NDArray{B}) where {A,B} -# error("Not implemented yet") -# #! REPLACE WITH ARRAY_EQUAL ONCE THAT IS WRAPPED -# #! or explicit call to nda_binary_reduction -# end - -# function Base.:(!=)(lhs::NDArray{A}, rhs::NDArray{B}) where {A,B} -# error("Not implemented yet") -# #! REPLACE WITH ARRAY_EQUAL ONCE THAT IS WRAPPED -# #! or explicit call to nda_binary_reduction -# end - # Specializations for 2 and -1 in unary.jl @inline function __broadcast( f::typeof(Base.literal_pow), out::NDArray, _, input::NDArray{T}, power::NDArray{T} diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index ac029934f..66f9e4cd0 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -370,6 +370,17 @@ function nda_binary_op!(out::NDArray, op_code::BinaryOpCode, rhs1::NDArray, rhs2 return out end +function nda_binary_reduction!( + out::NDArray, op_code::BinaryOpCode, rhs1::NDArray, rhs2::NDArray +) + @task_scope _scope_op("binary_red", op_code) begin + ccall((:nda_binary_reduction, libnda), + Cvoid, (NDArray_t, BinaryOpCode, NDArray_t, NDArray_t), + out.ptr, op_code, rhs1.ptr, rhs2.ptr) + end + return out +end + function nda_unary_op!(out::NDArray, op_code::UnaryOpCode, input::NDArray) @task_scope _scope_op("unary", op_code) begin ccall((:nda_unary_op, libnda), diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 9e517e552..9329817b0 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -839,16 +839,12 @@ unwrap(x::NDArray) = only(x) @doc""" ==(arr1::NDArray, arr2::NDArray) + !=(arr1::NDArray, arr2::NDArray) -Check if two NDArrays are equal element-wise. - -Returns `true` if both arrays have the same shape and all corresponding elements are equal. -Currently supports arrays up to 3 dimensions. For higher dimensions, returns `false` with a warning. - -!!! warning - - This function uses scalar indexing and should not be used in production code. This is meant for testing. - +Element-wise equality reduced to a 0-d `NDArray{Bool}` (not a Julia `Bool`). +Same shape and values yields true; mismatched shape or rank yields false. +Mixed dtypes follow Julia `==` (promote, then compare). Use `unwrap` or `A[]` +(with `allowscalar`) for a host value. Broadcast `.==` / `.!=` stay elementwise. # Examples ```@repl @@ -859,12 +855,47 @@ c = cuNumeric.zeros(2, 2) a == c ``` """ -function Base.:(==)(arr1::NDArray{T,N}, arr2::NDArray{T,N}) where {T,N} - return nda_array_equal(arr1, arr2) #DOESNT RETURN SCALAR +function Base.:(==)(a::NDArray, b::NDArray) + size(a) == size(b) || return NDArray(false) + return _array_equal_impl(a, b) +end + +function Base.:(!=)(a::NDArray, b::NDArray) + return !(a == b) +end + +function _array_equal_impl(a::NDArray{T}, b::NDArray{T}) where {T} + out = cuNumeric.zeros(Bool) + return nda_binary_reduction!(out, cuNumeric.EQUAL, a, b) +end + +function _array_equal_impl(a::NDArray{A}, b::NDArray{B}) where {A,B} + T = promote_type(A, B) + return _array_equal_promoted(a, b, T) end -function Base.:(!=)(arr1::NDArray{T,N}, arr2::NDArray{T,N}) where {T,N} - return !(arr1 == arr2) +function _array_equal_promoted(a::NDArray{T}, b::NDArray{T}, ::Type{T}) where {T} + return _array_equal_impl(a, b) +end +function _array_equal_promoted(a::NDArray, b::NDArray{T}, ::Type{T}) where {T} + p1 = unchecked_promote_arr(a, T) + result = _array_equal_impl(p1, b) + destroy!(p1) + return result +end +function _array_equal_promoted(a::NDArray{T}, b::NDArray, ::Type{T}) where {T} + p2 = unchecked_promote_arr(b, T) + result = _array_equal_impl(a, p2) + destroy!(p2) + return result +end +function _array_equal_promoted(a::NDArray, b::NDArray, ::Type{T}) where {T} + p1 = unchecked_promote_arr(a, T) + p2 = unchecked_promote_arr(b, T) + result = _array_equal_impl(p1, p2) + destroy!(p1) + destroy!(p2) + return result end @doc""" diff --git a/test/array/binary/float.jl b/test/array/binary/float.jl index 490d92773..b3250e5bf 100644 --- a/test/array/binary/float.jl +++ b/test/array/binary/float.jl @@ -25,3 +25,4 @@ run_binary_ops_tests( Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES), ), ) +run_array_equal_tests() diff --git a/test/array/binary/tests.jl b/test/array/binary/tests.jl index de1d2f13a..87887b86a 100644 --- a/test/array/binary/tests.jl +++ b/test/array/binary/tests.jl @@ -178,9 +178,13 @@ function run_binary_ops_tests(types) end allowscalar() do - @test unwrap(arr_cn == arr_cn) + eq = arr_cn == arr_cn + neq = arr_cn != arr_cn2 + @test ndims(eq) == 0 + @test ndims(neq) == 0 + @test unwrap(eq) @test !unwrap(arr_cn == arr_cn2) - @test unwrap(arr_cn != arr_cn2) + @test unwrap(neq) @test !unwrap(arr_cn != arr_cn) @test unwrap(all(arr_cn .== arr_cn)) end @@ -196,3 +200,34 @@ function run_binary_copyto_tests() @test is_same(a, b) end end + +function run_array_equal_tests() + @testset "0-d == / !=" begin + a32 = Float32[1, 2, 3] + a64 = Float64[1, 2, 3] + n32 = @allowscalar NDArray(a32) + n64 = @allowscalar NDArray(a64) + + allowscalar() do + @test ndims(n32 == n64) == 0 + @test unwrap(n32 == n64) == (a32 == a64) + @test unwrap(n32 != n64) == (a32 != a64) + + b64 = Float64[1, 2, 4] + m64 = NDArray(b64) + @test unwrap(n32 == m64) == (a32 == b64) + @test unwrap(n32 != m64) == (a32 != b64) + end + + same = cuNumeric.ones(2, 2) + other_shape = cuNumeric.ones(3, 3) + other_rank = cuNumeric.ones(4) + allowscalar() do + @test ndims(same == other_shape) == 0 + @test !unwrap(same == other_shape) + @test unwrap(same != other_shape) + @test !unwrap(same == other_rank) + @test unwrap(same != other_rank) + end + end +end diff --git a/test/util.jl b/test/util.jl index 6654765ad..582fcd2eb 100644 --- a/test/util.jl +++ b/test/util.jl @@ -56,9 +56,9 @@ function reduction_atol(::Type{T}, n, scale=1) where {T} return max(atol(T) * n, n * eps(FT) * abs(scale)) end -is_same(arr1::NDArray, arr2::NDArray) = @allowscalar (arr1 == arr2)[1] -is_same(arr1::NDArray, arr2::Array) = @allowscalar (arr1 == arr2)[1] -is_same(arr1::Array, arr2::NDArray) = @allowscalar (arr1 == arr2)[1] +is_same(arr1::NDArray, arr2::NDArray) = unwrap(arr1 == arr2) +is_same(arr1::NDArray, arr2::Array) = @allowscalar (arr1 == arr2) +is_same(arr1::Array, arr2::NDArray) = @allowscalar (arr1 == arr2) is_same(arr1::Array, arr2::Array) = (arr1 == arr2) function my_rand(::Type{F}, dims...; L=F(-1000), R=F(1000)) where {F<:AbstractFloat} From 1e0bfb89830dbf9c11dca2cf0bd8eccf1655017c Mon Sep 17 00:00:00 2001 From: ejmeitz Date: Mon, 24 Aug 2026 15:48:17 -0500 Subject: [PATCH 19/49] support new JLL wrapper --- Project.toml | 4 ++-- lib/cunumeric_jl_wrapper/VERSION | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Project.toml b/Project.toml index b0b674133..3d8a481f3 100644 --- a/Project.toml +++ b/Project.toml @@ -1,6 +1,6 @@ name = "cuNumeric" uuid = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" -version = "0.2.0" +version = "0.3.0" [deps] AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" @@ -44,7 +44,7 @@ Preferences = "1" Random = "1" StaticArrays = "1" StatsBase = "0.34" -cunumeric_jl_wrapper_jll = "26.6.0" +cunumeric_jl_wrapper_jll = "26.6.1" cupynumeric_jll = "26.6.0" julia = "1.10" diff --git a/lib/cunumeric_jl_wrapper/VERSION b/lib/cunumeric_jl_wrapper/VERSION index 2e6fb0f27..e8c5e343a 100644 --- a/lib/cunumeric_jl_wrapper/VERSION +++ b/lib/cunumeric_jl_wrapper/VERSION @@ -1 +1 @@ -26.6.0 +26.6.1 From fda74c7d9f733a043911e88c62673400806cb1c9 Mon Sep 17 00:00:00 2001 From: ejmeitz Date: Mon, 24 Aug 2026 15:55:32 -0500 Subject: [PATCH 20/49] fix docs project compat --- docs/Project.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/Project.toml b/docs/Project.toml index 078ced337..d302b4cd0 100644 --- a/docs/Project.toml +++ b/docs/Project.toml @@ -1,7 +1,7 @@ [compat] CNPreferences = "0.1.3" Documenter = "1.5" -cuNumeric = "0.2.0" +cuNumeric = "0.3.0" [deps] CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" From 28f65e902da5965febaf0130b4e7786b3fc3ed87 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Mon, 31 Aug 2026 11:34:43 -0400 Subject: [PATCH 21/49] TensorOperations.jl Package Extension (#188) * tensor oeprations extension * force tensors which reduce to scalars to hit scalar indexing --- .github/workflows/ci.yml | 2 + Project.toml | 7 + benchmark/Project.toml | 2 + benchmark/README.md | 27 ++ benchmark/benchmarks.toml | 33 +++ benchmark/plot_results.jl | 3 +- benchmark/run.jl | 8 +- .../src/benchmarks/tensor_contractions.jl | 136 ++++++++++ benchmark/src/core.jl | 19 +- benchmark/src/parse_benchmarks.jl | 12 + benchmark/src/single.jl | 26 +- .../src_py/benchmarks/tensor_contractions.py | 68 +++++ docs/make.jl | 1 + docs/src/examples/tensor_network.md | 78 ++++++ docs/src/linalg.md | 27 +- examples/Project.toml | 1 + examples/tensor_network.jl | 48 ++++ ext/cuNumericTensorOperationsExt.jl | 238 ++++++++++++++++++ src/ndarray/ndarray.jl | 1 + test/array/tensoroperations.jl | 237 +++++++++++++++++ 20 files changed, 954 insertions(+), 20 deletions(-) create mode 100644 benchmark/src/benchmarks/tensor_contractions.jl create mode 100644 benchmark/src_py/benchmarks/tensor_contractions.py create mode 100644 docs/src/examples/tensor_network.md create mode 100644 examples/tensor_network.jl create mode 100644 ext/cuNumericTensorOperationsExt.jl create mode 100644 test/array/tensoroperations.jl diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a4eb983fc..d7b63a6e7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -15,6 +15,7 @@ on: push: paths: - 'src/**' + - 'ext/**' - 'scripts/**' - 'deps/build.jl' - 'Project.toml' @@ -29,6 +30,7 @@ on: pull_request: paths: - 'src/**' + - 'ext/**' - 'scripts/**' - 'deps/build.jl' - 'Project.toml' diff --git a/Project.toml b/Project.toml index 3d8a481f3..0358c7d0b 100644 --- a/Project.toml +++ b/Project.toml @@ -26,6 +26,12 @@ cunumeric_jl_wrapper_jll = "49048992-29d2-5fd1-994f-9cecf112d624" cupynumeric_jll = "2862d674-414d-5b0b-a494-b21f8deca547" libcxxwrap_julia_jll = "3eaa8342-bff7-56a5-9981-c04077f7cee7" +[weakdeps] +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" + +[extensions] +cuNumericTensorOperationsExt = "TensorOperations" + [compat] AbstractFFTs = "1.5" CNPreferences = "0.1.3" @@ -44,6 +50,7 @@ Preferences = "1" Random = "1" StaticArrays = "1" StatsBase = "0.34" +TensorOperations = "5.8" cunumeric_jl_wrapper_jll = "26.6.1" cupynumeric_jll = "26.6.0" julia = "1.10" diff --git a/benchmark/Project.toml b/benchmark/Project.toml index 805db8aad..793787ae1 100644 --- a/benchmark/Project.toml +++ b/benchmark/Project.toml @@ -6,7 +6,9 @@ Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" +cuTENSOR = "011b41b2-24ef-40a8-b3eb-fa098493e9e1" [extras] LegatePreferences = "8028f36a-2b64-49e9-aa04-2d0933fd2ed9" diff --git a/benchmark/README.md b/benchmark/README.md index eb3b346eb..0d6bc8a74 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -16,6 +16,11 @@ cuNumeric always runs; extra comparison backends are toggled in `[Global]`: single-device). - `cupynumeric = true` → also run under cupynumeric (see below). +Individual `[[benchmark]]` blocks may override `cuda`, `n_warmup`, `n_iter`, +and `n_trial`. Unspecified values inherit from `[Global]`. This is useful for +enabling a single-GPU CUDA comparison only for compatible benchmarks or reducing +the iteration count for expensive kernels. + ### Comparing against cupynumeric cupynumeric runs in a conda env whose major.minor matches this project's @@ -87,6 +92,28 @@ When `T = ["Float32", "Float64"]` and a length-2 `N`/`M` sweep you get all **4** combinations, not a paired `Float32 -> N[1], Float64 -> N[2]`. To pin a type to a specific size, use separate `[[name]]` blocks. +## Tensor contractions + +Two direct TensorOperations benchmarks compare the same mathematical kernel +across cuNumeric.jl, cuPyNumeric, and—when `cuda = true` on a one-GPU +entry—TensorOperations.jl's cuTENSOR backend: + +- `tensor_projection3` computes + `D[n,m,l] = A[i,j,k] * B[n,i] * B[m,j] * B[l,k]`. TensorOperations performs + three pairwise contractions with rank-3 intermediates. For equal index extent + `N`, the counted work is `3N^3(2N-1)`, asymptotically `6N^4`. +- `tensor_contract4` computes + `C[a,b,c,d] = X[a,i,c,j] * Y[i,b,j,d]`. This single contraction isolates the + primitive high-rank backend path and counts `N^4(2N^2-1)` operations. + +The Julia implementations use `@tensor` on both `NDArray` and `CuArray`; the +latter activates TensorOperations' cuTENSOR extension and is recorded as +`TensorOperations.jl / cuTENSOR`. The cuPyNumeric implementations use equivalent +`einsum` expressions. Final outputs are preallocated, and contraction-order +selection is performed before timing (at macro expansion in Julia and during +initialization in Python). Required intermediate allocation and release remain +part of each timed projection iteration. + ## Plotting ```bash diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml index 0518b3fc8..13eb5418c 100644 --- a/benchmark/benchmarks.toml +++ b/benchmark/benchmarks.toml @@ -132,3 +132,36 @@ T = "Float32" gpus = [1, 2, 4, 8] cpus = 16 N = [1_000_000, 2_000_000, 4_000_000, 8_000_000] + +######################################### +# Three-mode tensor projection # +# D[n,m,l] = A[i,j,k] B[n,i] B[m,j] # +# B[l,k] # +# Optimal pairwise work ~ 6*N^4. # +######################################### + +[[tensor_projection3]] +T = "Float32" +gpus = 1 +cpus = 16 +N = [32, 64, 128] +cuda = true +n_warmup = 3 +n_iter = 20 +n_trial = 5 + +######################################### +# Rank-4 two-index tensor contraction # +# C[a,b,c,d] = X[a,i,c,j] Y[i,b,j,d] # +# Work ~ 2*N^6. # +######################################### + +[[tensor_contract4]] +T = "Float32" +gpus = 1 +cpus = 16 +N = [12, 16, 24] +cuda = true +n_warmup = 3 +n_iter = 20 +n_trial = 5 diff --git a/benchmark/plot_results.jl b/benchmark/plot_results.jl index b53ac3618..b24098003 100644 --- a/benchmark/plot_results.jl +++ b/benchmark/plot_results.jl @@ -45,6 +45,7 @@ const FAMILIES = [ ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond), ("cupynumeric", "cuPyNumeric", "#eb6834", :rect), ("CUDA.jl", "CUDA.jl", "#008300", :utriangle), + ("tensoroperations_cuda", "TensorOperations.jl / cuTENSOR", "#159a9c", :star5), ] const INK = "#0b0b0b" @@ -173,7 +174,7 @@ function build_legend(series) grid=false, ticks=false) present = [(s.label, s.color, s.marker) for s in series] annotate!(pl, 0.015, 0.74, text("Series", 10, INK, :left)) - xs = range(0.18, 0.80; length=max(length(present), 1)) + xs = length(present) <= 1 ? (0.18,) : range(0.18, 0.80; length=length(present)) for ((fam, color, marker), x) in zip(present, xs) swatch!(pl, x, 0.74, color, :solid, marker, fam) end diff --git a/benchmark/run.jl b/benchmark/run.jl index 36c4c2661..4debdbdcf 100644 --- a/benchmark/run.jl +++ b/benchmark/run.jl @@ -106,11 +106,11 @@ function run_all_benchmarks(config="benchmarks.toml") T=spec.T, N=N, M=M, fusion=spec.fusion, - n_iter=gs.n_iter, - n_warmup=gs.n_warmup, - n_trial=gs.n_trial, + n_iter=spec.n_iter, + n_warmup=spec.n_warmup, + n_trial=spec.n_trial, cupynumeric=gs.cupynumeric, - cudajl=gs.cuda, + cudajl=spec.cuda, check_correctness=gs.check_correctness, n_correctness_iter=gs.n_correctness_iter, ) diff --git a/benchmark/src/benchmarks/tensor_contractions.jl b/benchmark/src/benchmarks/tensor_contractions.jl new file mode 100644 index 000000000..825b8b565 --- /dev/null +++ b/benchmark/src/benchmarks/tensor_contractions.jl @@ -0,0 +1,136 @@ +using Random +using TensorOperations + +abstract type AbstractTensorContraction{T} <: AbstractBenchmark{T} end + +Base.@kwdef struct TensorProjection3{T} <: AbstractTensorContraction{T} + N::Int +end + +Base.@kwdef struct TensorContract4{T} <: AbstractTensorContraction{T} + N::Int +end + +name(::TensorProjection3) = "tensor_projection3" +name(::TensorContract4) = "tensor_contract4" +dims(b::AbstractTensorContraction) = (b.N, 1) +function data(b::AbstractTensorContraction{T}) where {T} + return "$(name(b)) with T=$(T), N=$(b.N)" +end + +allowed_types(::Type{<:AbstractTensorContraction}) = cuNumeric.SUPPORTED_FLOAT_TYPES + +# Three rank-3 outputs, each containing an N-term dot product. +total_flops(b::TensorProjection3) = 3 * b.N^3 * (2 * b.N - 1) + +# N^4 output elements, each containing an N^2-term dot product. +total_flops(b::TensorContract4) = b.N^4 * (2 * b.N^2 - 1) + +function build_benchmark( + ::Type{TensorProjection3}, ::Type{T}, N, M +) where {T} + return TensorProjection3{T}(; N=N) +end + +function build_benchmark( + ::Type{TensorContract4}, ::Type{T}, N, M +) where {T} + return TensorContract4{T}(; N=N) +end + +function _tensor_rand(mod, ::Type{T}, dims...) where {T} + mod === CUDACore && return CUDACore.CuArray(rand(T, dims...)) + return mod.rand(T, dims...) +end + +function initialize(b::TensorProjection3{T}; mod=cuNumeric) where {T} + A = _tensor_rand(mod, T, b.N, b.N, b.N) + B = _tensor_rand(mod, T, b.N, b.N) + D = mod.zeros(T, b.N, b.N, b.N) + GC.gc() + return D, A, B +end + +function initialize(b::TensorContract4{T}; mod=cuNumeric) where {T} + X = _tensor_rand(mod, T, b.N, b.N, b.N, b.N) + Y = _tensor_rand(mod, T, b.N, b.N, b.N, b.N) + C = mod.zeros(T, b.N, b.N, b.N, b.N) + GC.gc() + return C, X, Y +end + +function run!(::TensorProjection3, D, A, B) + @tensor opt=true D[n, m, l] = + A[i, j, k] * B[n, i] * B[m, j] * B[l, k] + return D +end + +function run!(::TensorContract4, C, X, Y) + @tensor C[a, b, c, d] = X[a, i, c, j] * Y[i, b, j, d] + return C +end + +correctness_supported(::AbstractTensorContraction) = true + +function _backend_array(mod, A) + mod === cuNumeric && return NDArray(A) + mod === CUDACore && return CUDACore.CuArray(A) + return throw(ArgumentError("unsupported tensor contraction backend $mod")) +end + +function _tensor_isapprox(actual, expected, ::Type{T}) where {T} + tol = T === Float32 ? 2e-4 : 1e-11 + return isapprox(Array(actual), expected; atol=tol, rtol=tol) +end + +function check_benchmark_correctness( + b::TensorProjection3{T}, gs::GlobalSettings; mod=cuNumeric +) where {T} + mod in (cuNumeric, CUDACore) || return "skipped" + n = min(b.N, 4) + rng = MersenneTwister(0x3b8a7c21) + Ah = rand(rng, T, n, n, n) + Bh = rand(rng, T, n, n) + ref = zeros(T, n, n, n) + @tensor opt=true ref[n, m, l] = + Ah[i, j, k] * Bh[n, i] * Bh[m, j] * Bh[l, k] + + D = _backend_array(mod, zeros(T, n, n, n)) + A = _backend_array(mod, Ah) + B = _backend_array(mod, Bh) + run!(b, D, A, B) + return _tensor_isapprox(D, ref, T) ? "pass" : "fail" +end + +function check_benchmark_correctness( + b::TensorContract4{T}, gs::GlobalSettings; mod=cuNumeric +) where {T} + mod in (cuNumeric, CUDACore) || return "skipped" + n = min(b.N, 4) + rng = MersenneTwister(0xa3c97d42) + Xh = rand(rng, T, n, n, n, n) + Yh = rand(rng, T, n, n, n, n) + ref = zeros(T, n, n, n, n) + @tensor ref[a, b, c, d] = Xh[a, i, c, j] * Yh[i, b, j, d] + + C = _backend_array(mod, zeros(T, n, n, n, n)) + X = _backend_array(mod, Xh) + Y = _backend_array(mod, Yh) + run!(b, C, X, Y) + return _tensor_isapprox(C, ref, T) ? "pass" : "fail" +end + +function benchmark_backend_label( + ::AbstractTensorContraction, backend::String, default::String +) + return backend == "cudajl" ? "TensorOperations.jl / cuTENSOR" : default +end + +function benchmark_backend_save_as( + ::AbstractTensorContraction, backend::String, default::String +) + return backend == "cudajl" ? "tensoroperations_cuda" : default +end + +register_benchmark("tensor_projection3", TensorProjection3) +register_benchmark("tensor_contract4", TensorContract4) diff --git a/benchmark/src/core.jl b/benchmark/src/core.jl index 7a0aacbc4..9511ba65e 100644 --- a/benchmark/src/core.jl +++ b/benchmark/src/core.jl @@ -56,6 +56,9 @@ function register_benchmark(key::AbstractString, ::Type{B}) where {B<:AbstractBe return BENCHMARKS[key] = B end +benchmark_backend_label(::AbstractBenchmark, backend::String, default::String) = default +benchmark_backend_save_as(::AbstractBenchmark, backend::String, default::String) = default + function build_benchmark(::Type{B}, ::Type{T}, N, M) where {B<:AbstractBenchmark,T} return B{T}(; N=N, M=M) end @@ -80,18 +83,21 @@ function check_benchmark_correctness(b::AbstractBenchmark, gs::GlobalSettings; m end # One timed trial: warmup, then time `n_iter` iterations of `run!`. -function _trial(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) +function _trial( + b::AbstractBenchmark, gs::GlobalSettings; + mod=cuNumeric, clock=get_time_microseconds, +) GC.gc(true) state = initialize(b; mod=mod) start_time = nothing for idx in 1:(gs.n_warmup + gs.n_iter) if idx == gs.n_warmup + 1 - start_time = get_time_microseconds() + start_time = clock() end run!(b, state...) end - total_time_μs = get_time_microseconds() - start_time + total_time_μs = clock() - start_time mean_time_ms = total_time_μs / (gs.n_iter * 1e3) gflops = total_flops(b) / (mean_time_ms * 1e6) @@ -100,7 +106,10 @@ end # Run `n_trial` independent trials and collect their per-trial measurements. # Correctness (if enabled) runs once before timing, not per trial/iteration. -function run_benchmark(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) +function run_benchmark( + b::AbstractBenchmark, gs::GlobalSettings; + mod=cuNumeric, clock=get_time_microseconds, +) correctness = "skipped" if gs.check_correctness if correctness_supported(b) @@ -113,7 +122,7 @@ function run_benchmark(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) times_ms = Float64[] gflops = Float64[] for _ in 1:gs.n_trial - t, g = _trial(b, gs; mod=mod) + t, g = _trial(b, gs; mod=mod, clock=clock) push!(times_ms, t) push!(gflops, g) end diff --git a/benchmark/src/parse_benchmarks.jl b/benchmark/src/parse_benchmarks.jl index 28cad96d4..3f30e44ee 100644 --- a/benchmark/src/parse_benchmarks.jl +++ b/benchmark/src/parse_benchmarks.jl @@ -11,6 +11,10 @@ struct BenchmarkSpec gpus::Int cpus::Int fusion::Bool + cuda::Bool + n_warmup::Int + n_iter::Int + n_trial::Int args::Vector{Int} end @@ -76,6 +80,10 @@ function parse_config(path) fusion = aslist(get(e, "fusion", true)) N = aslist(e["N"]) M = aslist(get(e, "M", 1)) + cuda = get(e, "cuda", global_settings.cuda) + n_warmup = get(e, "n_warmup", global_settings.n_warmup) + n_iter = get(e, "n_iter", global_settings.n_iter) + n_trial = get(e, "n_trial", global_settings.n_trial) n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus, "N" => N, "M" => M]) @@ -88,6 +96,10 @@ function parse_config(path) sweep_value(gpus, i), sweep_value(cpus, i), parse_fusion(fuse), + cuda, + n_warmup, + n_iter, + n_trial, [sweep_value(N, i), sweep_value(M, i)], ), ) diff --git a/benchmark/src/single.jl b/benchmark/src/single.jl index 56ecd171b..ec672014c 100644 --- a/benchmark/src/single.jl +++ b/benchmark/src/single.jl @@ -8,6 +8,8 @@ using cuNumeric using CUDACore using LinearAlgebra +using TensorOperations +using cuTENSOR include("core.jl") const BENCHMARK_DIR = joinpath(@__DIR__, "benchmarks") @@ -16,10 +18,16 @@ include.(filter(contains(r".jl$"), readdir(BENCHMARK_DIR; join=true))) # Resolve a TOML type string like "Float32" to the actual Julia type. parse_type(s) = getfield(Base, Symbol(s))::DataType +cuda_clock() = (CUDACore.synchronize(; blocking=true); time_ns() / 1e3) + # mod runs the kernels; label tags stdout; save_as names the results CSV. const BACKENDS = Dict( - "cunumeric" => (mod=cuNumeric, label="cuNumeric", save_as="cunumeric"), - "cudajl" => (mod=CUDACore, label="CUDA.jl", save_as="CUDA.jl"), + "cunumeric" => ( + mod=cuNumeric, label="cuNumeric", save_as="cunumeric", clock=get_time_microseconds + ), + "cudajl" => ( + mod=CUDACore, label="CUDA.jl", save_as="CUDA.jl", clock=cuda_clock + ), ) function run_single( @@ -34,13 +42,15 @@ function run_single( ) bk = BACKENDS[backend] - # unfused cuNumeric runs land in their own CSV so they stay a distinct series - fused = cuNumeric.FUSE_BROADCAST_EXPRS - save_as = fused ? bk.save_as : "$(bk.save_as)_nofusion" - label = fused ? bk.label : "$(bk.label) (no fusion)" - T = parse_type(T_str) b = build_benchmark(BENCHMARKS[name], T, N, M) + + # unfused cuNumeric runs land in their own CSV so they stay a distinct series + fused = cuNumeric.FUSE_BROADCAST_EXPRS + default_save_as = fused ? bk.save_as : "$(bk.save_as)_nofusion" + default_label = fused ? bk.label : "$(bk.label) (no fusion)" + save_as = benchmark_backend_save_as(b, backend, default_save_as) + label = benchmark_backend_label(b, backend, default_label) gs = GlobalSettings(; n_warmup=n_warmup, n_iter=n_iter, @@ -53,7 +63,7 @@ function run_single( "[$(label)] $(name) benchmark ($(T)) on $(N)x$(M) for $(n_iter) " * "iterations ($(n_warmup) warmup) x $(n_trial) trials", ) - br = run_benchmark(b, gs; mod=bk.mod) + br = run_benchmark(b, gs; mod=bk.mod, clock=bk.clock) @printf("[%s] Mean Run Time: %.5f ± %.5f ms\n", label, mean(br.times_ms), _std(br.times_ms)) @printf("[%s] FLOPS: %.5f ± %.5f GFLOPS\n", label, mean(br.gflops), _std(br.gflops)) println("[$(label)] Correctness: $(br.correctness)") diff --git a/benchmark/src_py/benchmarks/tensor_contractions.py b/benchmark/src_py/benchmarks/tensor_contractions.py new file mode 100644 index 000000000..6f3cd1e0d --- /dev/null +++ b/benchmark/src_py/benchmarks/tensor_contractions.py @@ -0,0 +1,68 @@ +import cupynumeric as np + +from core import register_benchmark + + +def _optimal_path(expression, *operands): + path, _ = np.einsum_path( + expression, *operands, optimize="optimal" + ) + # cuPyNumeric returns NumPy's sentinel-prefixed path, but its einsum forwards + # paths directly to opt_einsum, which expects only contraction tuples. + return path[1:] if path and path[0] == "einsum_path" else path + + +class TensorProjection3: + name = "tensor_projection3" + expression = "ijk,ni,mj,lk->nml" + + def __init__(self, T, N, M): + self.T, self.N = T, N + + def dims(self): + return self.N, 1 + + def total_flops(self): + return 3 * self.N**3 * (2 * self.N - 1) + + def initialize(self): + A = np.random.rand(self.N, self.N, self.N).astype(self.T) + B = np.random.rand(self.N, self.N).astype(self.T) + D = np.zeros((self.N, self.N, self.N), dtype=self.T) + path = _optimal_path(self.expression, A, B, B, B) + return D, A, B, path + + def run(self, state): + D, A, B, path = state + np.einsum( + self.expression, A, B, B, B, out=D, optimize=path + ) + + +class TensorContract4: + name = "tensor_contract4" + expression = "aicj,ibjd->abcd" + + def __init__(self, T, N, M): + self.T, self.N = T, N + + def dims(self): + return self.N, 1 + + def total_flops(self): + return self.N**4 * (2 * self.N**2 - 1) + + def initialize(self): + X = np.random.rand(self.N, self.N, self.N, self.N).astype(self.T) + Y = np.random.rand(self.N, self.N, self.N, self.N).astype(self.T) + C = np.zeros((self.N, self.N, self.N, self.N), dtype=self.T) + path = _optimal_path(self.expression, X, Y) + return C, X, Y, path + + def run(self, state): + C, X, Y, path = state + np.einsum(self.expression, X, Y, out=C, optimize=path) + + +register_benchmark("tensor_projection3", TensorProjection3) +register_benchmark("tensor_contract4", TensorContract4) diff --git a/docs/make.jl b/docs/make.jl index e1ab3b66d..99dc2c5a8 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -40,6 +40,7 @@ makedocs(; "Gray-Scott" => "examples/grayscott.md", "Dynamic Mode Decomposition" => "examples/dmd.md", "Periodic Poisson (FFT)" => "examples/poisson_fft.md", + "Tensor Network Contraction" => "examples/tensor_network.md", ], "Performance Tips" => [ "Kernel Fusion" => "perf/kernel_fusion.md", diff --git a/docs/src/examples/tensor_network.md b/docs/src/examples/tensor_network.md new file mode 100644 index 000000000..4389b2430 --- /dev/null +++ b/docs/src/examples/tensor_network.md @@ -0,0 +1,78 @@ +# Tensor Network Contraction + +[TensorOperations.jl](https://quantumkithub.github.io/TensorOperations.jl/stable/) +provides Einstein-index notation through `@tensor`. Loading it together with +cuNumeric activates the `NDArray` extension, so contractions and their +intermediate tensors remain in Legate-managed memory. + +The two-site spin-1/2 Heisenberg interaction is + +```math +H = S^x_1 S^x_2 + S^y_1 S^y_2 + S^z_1 S^z_2. +``` + +The examples below apply it to an MPS-like state with tensors ``A^{(1)}``, +``A^{(2)}`` and boundary environments ``L``, ``R``. The first contractions keep +open bond indices and stay `NDArray`s. The second contracts every index to a +host scalar. + +```julia +# found in examples/tensor_network.jl +using cuNumeric +using LinearAlgebra +using Random +using TensorOperations + +Random.seed!(1234) + +physical_dim = 2 +bond_dim = 16 + +σx = ComplexF64[0 1; 1 0] +σy = ComplexF64[0 -im; im 0] +σz = ComplexF64[1 0; 0 -1] +Hmatrix = (kron(σx, σx) + kron(σy, σy) + kron(σz, σz)) / 4 +H = NDArray(reshape(Hmatrix, 2, 2, 2, 2)) + +A1 = NDArray(randn(ComplexF64, bond_dim, physical_dim, bond_dim)) +A2 = NDArray(randn(ComplexF64, bond_dim, physical_dim, bond_dim)) +L = NDArray(randn(ComplexF64, bond_dim, bond_dim)) +R = NDArray(randn(ComplexF64, bond_dim, bond_dim)) +``` + +## Apply a two-site operator + +These contractions have free indices ``a, s_1, s_2, b``, so the results are +`NDArray`s. TensorOperations lowers the network to pairwise contractions and +releases the intermediate `NDArray`s. Nothing here indexes a single element. + +```julia +@tensor ψ[a, s1, s2, b] := + L[a, ap] * A1[ap, s1, c] * A2[c, s2, bp] * R[bp, b] +@tensor Hψ[a, s1, s2, b] := H[s1, s2, t1, t2] * ψ[a, t1, t2, b] + +println("ψ is a ", typeof(ψ), " of size ", size(ψ)) +println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) +``` + +## Expectation value + +A fully contracted `@tensor` assignment is a Julia scalar. TensorOperations +gets that value with `tensorscalar`, which indexes the rank-zero `NDArray`. +That is scalar indexing, so it needs `@allowscalar`. + +```julia +energy, norm² = @allowscalar begin + @tensor begin + e = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] + n = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] + end + e, n +end + +println("two-site energy = ", real(energy / norm²)) +``` + +The extension also supports output-index permutations, traces, conjugation, and +accumulation into an existing tensor. See [Tensor contractions](../linalg.md#Tensor-contractions) +for the lower-level `contract` and `contract!` interfaces. diff --git a/docs/src/linalg.md b/docs/src/linalg.md index fac52b770..522249ab1 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -48,8 +48,31 @@ Filter = t -> t isa Function && nameof(t) === :mul! ## Tensor contractions -`contract` / `contract!` are the pairwise primitive behind a future -TensorOperations.jl backend. They take **mode labels**, not einsum strings: +Loading [TensorOperations.jl](https://quantumkithub.github.io/TensorOperations.jl/stable/) +alongside cuNumeric activates the package extension for `NDArray`. Its `@tensor` +API supports additions and permutations, traces, pairwise contractions, complex +conjugation, scalar results, and multi-step expressions while keeping intermediate +tensors in cuNumeric: + +```julia +using cuNumeric +using TensorOperations + +A = cuNumeric.rand(Float64, 64, 32) +B = cuNumeric.rand(Float64, 32, 16) + +@tensor opt=true C[i, j] := A[i, k] * B[k, j] +@allowscalar @tensor squared_norm = conj(C[i, j]) * C[i, j] +``` + +A fully contracted `@tensor` result is a Julia scalar, so it goes through +`tensorscalar` and needs `@allowscalar`. Tensor results stay `NDArray`s. + +The extension is optional: TensorOperations is not loaded by cuNumeric itself. +TensorOperations-created intermediate `NDArray`s are released eagerly after use. + +The lower-level `contract` / `contract!` functions take **mode labels**, not +einsum strings: `"ik"` with `"kj"` is a GEMM; `"ijk"` with `"ikl"` and explicit output `"ijl"` is a batched product. Labels may be ASCII strings, `Char` tuples, or integers (`1` maps to `'a'`). diff --git a/examples/Project.toml b/examples/Project.toml index 44251c5ef..fa6e87e75 100644 --- a/examples/Project.toml +++ b/examples/Project.toml @@ -4,6 +4,7 @@ Legate = "1238f2cf-6593-4d60-9aca-2f5364e49909" LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" [extras] diff --git a/examples/tensor_network.jl b/examples/tensor_network.jl new file mode 100644 index 000000000..5defd1a81 --- /dev/null +++ b/examples/tensor_network.jl @@ -0,0 +1,48 @@ +#= Contract a two-site tensor network, then optionally pull a host scalar. + +The first contractions build an MPS-like two-site state and apply a Heisenberg +operator. Those results stay on device as NDArrays. Fully contracting to +⟨ψ|H|ψ⟩ / ⟨ψ|ψ⟩ goes through tensorscalar and therefore needs @allowscalar. +=# + +using cuNumeric +using LinearAlgebra +using Random +using TensorOperations + +Random.seed!(1234) + +const PHYSICAL_DIM = 2 +const BOND_DIM = 16 + +σx = ComplexF64[0 1; 1 0] +σy = ComplexF64[0 -im; im 0] +σz = ComplexF64[1 0; 0 -1] +Hmatrix = (kron(σx, σx) + kron(σy, σy) + kron(σz, σz)) / 4 +H = NDArray(reshape(Hmatrix, PHYSICAL_DIM, PHYSICAL_DIM, PHYSICAL_DIM, PHYSICAL_DIM)) + +# MPS tensors A1[ap, s1, c], A2[c, s2, bp] and boundary environments. +A1 = NDArray(randn(ComplexF64, BOND_DIM, PHYSICAL_DIM, BOND_DIM)) +A2 = NDArray(randn(ComplexF64, BOND_DIM, PHYSICAL_DIM, BOND_DIM)) +L = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) +R = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) + +# Open bonds remain, so both results are NDArrays. +@tensor ψ[a, s1, s2, b] := + L[a, ap] * A1[ap, s1, c] * A2[c, s2, bp] * R[bp, b] +@tensor Hψ[a, s1, s2, b] := H[s1, s2, t1, t2] * ψ[a, t1, t2, b] + +println("ψ is a ", typeof(ψ), " of size ", size(ψ)) +println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) + +# Fully contracted @tensor results go through tensorscalar, which indexes the +# rank-zero NDArray. That is scalar indexing and needs @allowscalar. +energy, norm² = @allowscalar begin + @tensor begin + e = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] + n = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] + end + e, n +end + +println("two-site energy = ", real(energy / norm²)) diff --git a/ext/cuNumericTensorOperationsExt.jl b/ext/cuNumericTensorOperationsExt.jl new file mode 100644 index 000000000..e83e6d339 --- /dev/null +++ b/ext/cuNumericTensorOperationsExt.jl @@ -0,0 +1,238 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): Ethan Meitz +=# + +module cuNumericTensorOperationsExt + +using cuNumeric +using TensorOperations + +const CN = cuNumeric +const TO = TensorOperations + +struct CuNumericBackend <: TO.AbstractBackend end + +TO.select_backend(::typeof(TO.tensoradd!), C::CN.NDArray, A::CN.NDArray) = + CuNumericBackend() +TO.select_backend(::typeof(TO.tensortrace!), C::CN.NDArray, A::CN.NDArray) = + CuNumericBackend() +function TO.select_backend( + ::typeof(TO.tensorcontract!), C::CN.NDArray, A::CN.NDArray, B::CN.NDArray +) + return CuNumericBackend() +end + +function TO.tensoradd_type( + TC, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool +) + return CN.NDArray{TC,TO.numind(pA)} +end + +function TO.tensorcontract_type( + TC, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, +) + Tout = TC <: Union{Integer,Bool} ? Float64 : TC + return CN.NDArray{Tout,TO.numind(pAB)} +end + +function TO.tensoralloc( + ::Type{<:CN.NDArray{T,N}}, + structure, + ::Val{istemp}=Val(false), + allocator=TO.DefaultAllocator(), +) where {T,N,istemp} + return CN.zeros(T, Tuple(structure)) +end + +function TO.tensorfree!(C::CN.NDArray, allocator=TO.DefaultAllocator()) + CN.destroy!(C) + return nothing +end + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α, β) where {T} + α′ = convert(T, α) + if iszero(β) + C .= α′ .* A + else + β′ = convert(T, β) + C .= β′ .* C .+ α′ .* A + end + return C +end + +function _convert_eltype(A::CN.NDArray, ::Type{T}) where {T} + return eltype(A) === T ? A : CN.as_type(A, T) +end + +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + TO.argcheck_tensoradd(C, A, pA) + TO.dimcheck_tensoradd(C, A, pA) + if C.ptr === A.ptr && !TO.istrivialpermutation(pA) + throw(ArgumentError("output tensor must not alias a permuted input tensor")) + end + + opA = conjA ? conj(A) : A + converted = _convert_eltype(opA, eltype(C)) + permutation = TO.linearize(pA) + permuted = + TO.istrivialpermutation(permutation) ? converted : + permutedims(converted, permutation) + + _accumulate!(C, permuted, α, β) + + permuted !== converted && CN.destroy!(permuted) + converted !== opA && CN.destroy!(converted) + opA !== A && CN.destroy!(opA) + return C +end + +function _trace_pairs(A::CN.NDArray, q::TO.Index2Tuple) + current = A + current_owned = false + labels = collect(1:ndims(A)) + + for (left, right) in zip(q[1], q[2]) + left_position = findfirst(==(left), labels)::Int + right_position = findfirst(==(right), labels)::Int + diagonal = CN.diagonal(current; dims=(left_position, right_position)) + current_owned && CN.destroy!(current) + + reduced = sum(diagonal; dims=ndims(diagonal)) + CN.destroy!(diagonal) + current = dropdims(reduced; dims=ndims(reduced)) + CN.destroy!(reduced) + current_owned = true + + deleteat!(labels, max(left_position, right_position)) + deleteat!(labels, min(left_position, right_position)) + end + return current, current_owned, labels +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + TO.argcheck_tensortrace(C, A, p, q) + TO.dimcheck_tensortrace(C, A, p, q) + C.ptr === A.ptr && throw(ArgumentError("output tensor must not alias input tensor")) + + opA = conjA ? conj(A) : A + traced, traced_owned, labels = _trace_pairs(opA, q) + output_labels = TO.linearize(p) + permutation = Tuple(findfirst(==(label), labels) for label in output_labels) + ordered = TO.istrivialpermutation(permutation) ? traced : + permutedims(traced, permutation) + converted = _convert_eltype(ordered, eltype(C)) + + _accumulate!(C, converted, α, β) + + converted !== ordered && CN.destroy!(converted) + ordered !== traced && CN.destroy!(ordered) + traced_owned && CN.destroy!(traced) + opA !== A && CN.destroy!(opA) + return C +end + +function _contract_modes( + A::CN.NDArray, + pA::TO.Index2Tuple, + B::CN.NDArray, + pB::TO.Index2Tuple, + pAB::TO.Index2Tuple, +) + nlabels = ndims(A) + length(pB[2]) + nlabels <= typemax(UInt8) || + throw(ArgumentError("tensor contraction requires $nlabels distinct mode labels")) + + Amodes = collect(UInt8, 1:ndims(A)) + Bmodes = Vector{UInt8}(undef, ndims(B)) + for (a, b) in zip(pA[2], pB[1]) + Bmodes[b] = Amodes[a] + end + nextlabel = ndims(A) + for b in pB[2] + nextlabel += 1 + Bmodes[b] = UInt8(nextlabel) + end + + free_modes = (map(i -> Amodes[i], pA[1])..., map(i -> Bmodes[i], pB[2])...) + Cmodes = UInt8[free_modes[i] for i in TO.linearize(pAB)] + return Cmodes, Amodes, Bmodes +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + TO.argcheck_tensorcontract(C, A, pA, B, pB, pAB) + TO.dimcheck_tensorcontract(C, A, pA, B, pB, pAB) + + opA = conjA ? conj(A) : A + opB = conjB ? conj(B) : B + convertedA = _convert_eltype(opA, eltype(C)) + convertedB = _convert_eltype(opB, eltype(C)) + Cmodes, Amodes, Bmodes = _contract_modes(convertedA, pA, convertedB, pB, pAB) + + CN.contract!(C, Cmodes, convertedA, Amodes, convertedB, Bmodes; α=α, β=β) + + convertedB !== opB && CN.destroy!(convertedB) + convertedA !== opA && CN.destroy!(convertedA) + opB !== B && CN.destroy!(opB) + opA !== A && CN.destroy!(opA) + return C +end + +function TO.tensorscalar(C::CN.NDArray) + ndims(C) == 0 || throw(DimensionMismatch("tensorscalar requires a rank-zero tensor")) + return C[] +end + +end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 9329817b0..683b3f8bd 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -209,6 +209,7 @@ function _copy_to_julia_array(arr::NDArray{T,N}) where {T,N} attached = NDArray(ptr, T, Val(N), out) copyto!(attached, arr) get_ptr(attached) # Block until the copy into `out` completes. + destroy!(attached) return out end diff --git a/test/array/tensoroperations.jl b/test/array/tensoroperations.jl new file mode 100644 index 000000000..63ffd8bf2 --- /dev/null +++ b/test/array/tensoroperations.jl @@ -0,0 +1,237 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): Ethan Meitz +=# + +using TensorOperations +using TensorOperations: TensorOperations as TO + +@testset "TensorOperations extension" begin + @test Base.get_extension(cuNumeric, :cuNumericTensorOperationsExt) !== nothing + + @testset "@tensor permutations and accumulation" begin + hostA = reshape(collect(Float64, 1:24), 2, 3, 4) + hostB = reshape(collect(Float64, 1:24), 3, 4, 2) + A = NDArray(hostA) + B = NDArray(hostB) + + @tensor permuted[k, i, j] := A[i, j, k] + @test permuted isa NDArray + @test Array(permuted) == permutedims(hostA, (3, 1, 2)) + + @tensor combined[k, i, j] := 2 * A[i, j, k] - 3 * B[j, k, i] + ref = 2 .* permutedims(hostA, (3, 1, 2)) .- + 3 .* permutedims(hostB, (2, 3, 1)) + @test Array(combined) == ref + + accumulated = cuNumeric.zeros(Float64, 4, 2, 3) + @tensor accumulated[k, i, j] = A[i, j, k] + @tensor accumulated[k, i, j] += 2 * B[j, k, i] + @tensor accumulated[k, i, j] -= 3 * A[i, j, k] + @test Array(accumulated) == + -2 .* permutedims(hostA, (3, 1, 2)) .+ + 2 .* permutedims(hostB, (2, 3, 1)) + end + + @testset "@tensor traces and output permutations" begin + host = reshape(collect(Float64, 1:(2 * 3 * 4 * 3 * 5 * 4)), 2, 3, 4, 3, 5, 4) + A = NDArray(host) + + @tensor traced[d, a] := A[a, b, c, b, d, c] + @tensor ref[d, a] := host[a, b, c, b, d, c] + @test traced isa NDArray + @test Array(traced) == ref + + destination = cuNumeric.ones(Float64, 5, 2) + @tensor destination[d, a] += 0.5 * A[a, b, c, b, d, c] + @test Array(destination) == ones(5, 2) .+ 0.5 .* ref + end + + @testset "@tensor contractions and outer products" begin + hostA = reshape(collect(Float64, 1:(2 * 3 * 4 * 5)), 2, 3, 4, 5) + hostB = reshape(collect(Float64, 1:(4 * 6 * 3 * 7)), 4, 6, 3, 7) + A = NDArray(hostA) + B = NDArray(hostB) + + @tensor contracted[a, g, d, f] := A[a, b, c, d] * B[c, f, b, g] + @tensor ref[a, g, d, f] := hostA[a, b, c, d] * hostB[c, f, b, g] + @test contracted isa NDArray + @test Array(contracted) == ref + + hostX = reshape( + ComplexF64.(1:6) .+ im .* ComplexF64.(6:-1:1), 2, 3 + ) + hostY = reshape( + ComplexF64.(7:26) .- im .* ComplexF64.(1:20), 4, 5 + ) + X = NDArray(hostX) + Y = NDArray(hostY) + @tensor outer[a, c, b, d] := conj(X[a, b]) * conj(Y[c, d]) + @test Array(outer) == reshape(conj(hostX), 2, 1, 3, 1) .* + reshape(conj(hostY), 1, 4, 1, 5) + end + + @testset "@tensor block and tensor network" begin + hostA = reshape( + collect(Float64, 1:(2 * 3 * 4 * 5 * 4 * 6)), 2, 3, 4, 5, 4, 6 + ) + hostB = reshape(collect(Float64, 1:(6 * 7 * 3)), 6, 7, 3) + hostC = reshape(collect(Float64, 1:(5 * 2 * 7)), 5, 2, 7) + A = NDArray(hostA) + B = NDArray(hostB) + C = NDArray(hostC) + D = cuNumeric.zeros(Float64, 2, 7, 5) + α = 0.25 + + @tensor begin + D[a, b, c] = A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + E[a, b, c] := A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + end + @tensor ref[a, b, c] := hostA[a, e, f, c, f, g] * hostB[g, b, e] + + α * hostC[c, a, b] + @test D isa NDArray + @test E isa NDArray + @test Array(D) == ref + @test Array(E) == ref + + hostA1 = reshape(collect(Float64, 1:24), 2, 3, 4) + hostA2 = reshape(collect(Float64, 1:120), 4, 5, 6) + hostL = reshape(collect(Float64, 1:4), 2, 2) + hostR = reshape(collect(Float64, 1:36), 6, 6) + hostH = reshape(collect(Float64, 1:225), 3, 5, 3, 5) + A1, A2 = NDArray(hostA1), NDArray(hostA2) + L, R, H = NDArray(hostL), NDArray(hostR), NDArray(hostH) + + @tensor network[a, s1, s2, c] := + L[a, ap] * A1[ap, t1, b] * + A2[b, t2, cp] * R[cp, c] * + H[s1, s2, t1, t2] + @tensor network_ref[a, s1, s2, c] := + hostL[a, ap] * hostA1[ap, t1, b] * + hostA2[b, t2, cp] * hostR[cp, c] * + hostH[s1, s2, t1, t2] + @test network isa NDArray + @test Array(network) == network_ref + end + + @testset "tensoradd!" begin + host = ComplexF64.(reshape(1:6, 2, 3)) .+ im .* reshape(7:12, 2, 3) + A = NDArray(host) + C = NDArray(fill(ComplexF64(2 - im), 3, 2)) + + TO.tensoradd!(C, A, ((2, 1), ()), true, 2, 3) + @test Array(C) ≈ 3 .* fill(ComplexF64(2 - im), 3, 2) .+ + 2 .* permutedims(conj(host)) + + @tensor allocated[j, i] := conj(A[i, j]) + @test allocated isa NDArray + @test Array(allocated) == permutedims(conj(host)) + + overwrite = NDArray(fill(ComplexF64(NaN), 3, 2)) + TO.tensoradd!(overwrite, A, ((2, 1), ()), false, 1, 0) + @test Array(overwrite) == permutedims(host) + end + + @testset "tensortrace!" begin + host = reshape(collect(Float64, 1:(2 * 3 * 3)), 2, 3, 3) + A = NDArray(host) + @tensor traced[i] := A[i, j, j] + ref = [sum(host[i, j, j] for j in axes(host, 2)) for i in axes(host, 1)] + @test traced isa NDArray + @test Array(traced) == ref + + host5 = reshape(collect(Float64, 1:(2 * 3 * 3 * 4 * 4)), 2, 3, 3, 4, 4) + A5 = NDArray(host5) + @tensor traced2[i] := A5[i, j, j, k, k] + ref2 = [ + sum(host5[i, j, j, k, k] for j in axes(host5, 2), k in axes(host5, 4)) + for i in axes(host5, 1) + ] + @test Array(traced2) == ref2 + + matrix = reshape(ComplexF64.(1:9) .+ im .* ComplexF64.(9:-1:1), 3, 3) + M = NDArray(matrix) + @test_throws ErrorException (@tensor t = M[i, i]) + scalar_trace = @allowscalar @tensor(t = M[i, i]) + @test scalar_trace isa ComplexF64 + @test scalar_trace == sum(matrix[i, i] for i in axes(matrix, 1)) + + conjugated_trace = TO.tensortrace(M, ((), ()), ((1,), (2,)), true) + @test_throws ErrorException TO.tensorscalar(conjugated_trace) + @test (@allowscalar TO.tensorscalar(conjugated_trace)) == + sum(conj(matrix[i, i]) for i in axes(matrix, 1)) + TO.tensorfree!(conjugated_trace) + end + + @testset "tensorcontract!" begin + hostA = ComplexF64.(reshape(1:12, 3, 4)) .+ im .* reshape(13:24, 3, 4) + hostB = ComplexF64.(reshape(1:20, 4, 5)) .- im .* reshape(21:40, 4, 5) + A = NDArray(hostA) + B = NDArray(hostB) + + @tensor C[i, j] := conj(A[i, k]) * B[k, j] + @test C isa NDArray + @test Array(C) ≈ conj(hostA) * hostB + + seed = fill(ComplexF64(1 + 2im), 3, 5) + accumulated = NDArray(copy(seed)) + TO.tensorcontract!( + accumulated, + A, + ((1,), (2,)), + true, + B, + ((1,), (2,)), + false, + ((1, 2), ()), + 2, + 3, + ) + @test Array(accumulated) ≈ 3 .* seed .+ 2 .* (conj(hostA) * hostB) + + uhost = ComplexF64.(1:4) .+ im .* ComplexF64.(4:-1:1) + vhost = ComplexF64.(5:8) .- im .* ComplexF64.(1:4) + u = NDArray(uhost) + v = NDArray(vhost) + scalar_product = @allowscalar @tensor(s = u[k] * v[k]) + @test scalar_product isa ComplexF64 + @test scalar_product ≈ sum(uhost .* vhost) + end + + @testset "temporary destruction" begin + hostA = reshape(collect(Float64, 1:12), 3, 4) + hostB = reshape(collect(Float64, 1:20), 4, 5) + hostD = reshape(collect(Float64, 1:10), 5, 2) + A = NDArray(hostA) + B = NDArray(hostB) + D = NDArray(hostD) + GC.gc() + cuNumeric.drain_pending_frees!() + current_bytes = + cuNumeric.HAS_CUDA ? + cuNumeric.current_device_bytes : + cuNumeric.current_host_bytes + baseline = current_bytes[] + + @tensor C[i, l] := A[i, j] * B[j, k] * D[k, l] + @test current_bytes[] == baseline + C.nbytes + @test Array(C) ≈ hostA * hostB * hostD + + TO.tensorfree!(C) + @test C.ptr == C_NULL + @test current_bytes[] == baseline + end +end From cf5dc4da30eb62c56ecba187f016d022104dc14d Mon Sep 17 00:00:00 2001 From: ejmeitz Date: Tue, 1 Sep 2026 14:00:34 -0500 Subject: [PATCH 22/49] update some docs --- README.md | 34 ++++++++-- docs/make.jl | 3 +- docs/src/api_preferences.md | 2 +- docs/src/api_random.md | 7 +- docs/src/api_tensor.md | 101 ++++++++++++++++++++++++++++ docs/src/examples/tensor_network.md | 9 ++- docs/src/install.md | 2 +- docs/src/linalg.md | 61 +---------------- docs/src/perf/kernel_fusion.md | 53 ++++++++------- docs/src/perf/patterns_to_avoid.md | 2 + docs/src/perf/reduce_allocations.md | 31 +++++---- examples/daxpy.jl | 11 --- examples/dmd.jl | 4 +- examples/tensor_network.jl | 9 +-- 14 files changed, 198 insertions(+), 131 deletions(-) create mode 100644 docs/src/api_tensor.md delete mode 100644 examples/daxpy.jl diff --git a/README.md b/README.md index 8c8132e51..cdc31fbf1 100644 --- a/README.md +++ b/README.md @@ -34,18 +34,18 @@ For more details, see [Hardware](https://julialegate.github.io/cuNumeric.jl/dev/ The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it is a drop-in replacement. You can use the same constructors (i.e., `zeros`, `ones`, `rand`), broadcasting, slicing, and linear algebra. Under the hood a few details differ from Base, and knowing them can help you write fast code. -**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be partitioned over GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. +**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be stored across multiple GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. -**Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. +**Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutating an `NDArray` slice mutates the parent and all other aliases of the underlying data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communicate with the Julia runtime. When you need a plain Julia number, call `unwrap`: +**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the task graph asynchronous instead of forcing synchronization to communicate with the Julia runtime. When you need a plain Julia number, call `unwrap` or `only`: ```julia s = sum(A) # NDArray{T,0} x = unwrap(s) # T, e.g. Float32 ``` -**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. +**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into a task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them [for example `println`, `unwrap`, or communicating with the Julia runtime (i.e., `Array(A)`)]. Hiding latency enables performant code. For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/dev/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/dev/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/dev/perf/patterns_to_avoid). @@ -67,25 +67,45 @@ See [Kernel Fusion](https://julialegate.github.io/cuNumeric.jl/dev/perf/kernel_f Results and reproduction instructions live under [Benchmark Results](https://julialegate.github.io/cuNumeric.jl/dev/benchmarks/results) and [How to Benchmark](https://julialegate.github.io/cuNumeric.jl/dev/benchmarks/howto). +### TensorOperations.jl Integration + +We implement a package extension for [TensorOperations.jl](https://github.com/QuantumKitHub/TensorOperations.jl) to enable clean syntax and optimal contraction order for tensor contractions. Simply load TensorOperations alongside cuNumeric to take advantage! Some useful macros if you are unfamiliar are [@tensor](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#The-@tensor-macro), [@tensoropt](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@tensoropt) and [@notensor](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@notensor). + +```julia +using TensorOperations +using cuNumeric + +α = randn() # Must be Julia scalar, not NDArray currently +A = cuNumeric.randn(5, 5, 5, 5, 5, 5) +B = cuNumeric.randn(5, 5, 5) +C = cuNumeric.randn(5, 5, 5) +D = cuNumeric.zeros(5, 5, 5) + +@tensor begin + D[a, b, c] = A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + E[a, b, c] := A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] +end +``` + ### Try an example ```julia using cuNumeric -integrand(x) = @. exp(-x^2) +integrand(x) = exp(-x^2) @accelerate function monte_carlo(N, x_max) Ω = 2 * x_max raw_samples = cuNumeric.rand(N) samples = @. Ω * raw_samples - x_max - return (Ω / N) * sum(integrand(samples)) + return (Ω / N) * sum(integrand.(samples)) end N = 1_000_000 x_max = 10.0f0 estimate = monte_carlo(N, x_max) -println("Monte-Carlo Estimate: $(estimate)") +println("Monte-Carlo Estimate: $(estimate)") # Should be ~ sqrt(pi) ``` More worked examples (initialization, Gray-Scott, …) are in the documentation sidebar under **Examples**. diff --git a/docs/make.jl b/docs/make.jl index 99dc2c5a8..7ec8a7f44 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -44,7 +44,7 @@ makedocs(; ], "Performance Tips" => [ "Kernel Fusion" => "perf/kernel_fusion.md", - "The @accelerate Macro" => "perf/reduce_allocations.md", + "@accelerate" => "perf/reduce_allocations.md", "Patterns to Avoid" => "perf/patterns_to_avoid.md", ], "Configuration" => [ @@ -67,6 +67,7 @@ makedocs(; "Unary Operations" => "api_unary.md", "Binary Operations" => "api_binary.md", "Linear Algebra" => "linalg.md", + "Tensor Contractions" => "api_tensor.md", "FFT" => "fft.md", "HDF5" => "api_hdf5.md", "NDArray Reference" => "api.md", diff --git a/docs/src/api_preferences.md b/docs/src/api_preferences.md index 12892687a..2973f7377 100644 --- a/docs/src/api_preferences.md +++ b/docs/src/api_preferences.md @@ -11,7 +11,7 @@ Out of the box (no `LocalPreferences.toml` changes): | `FUSE_BROADCAST_MIN_OPS` | **2** (single-op broadcasts stay unfused) | | Task scope names | **off** | -Build-mode setup (JLL / conda / developer) is documented under [Build Modes](./install.md). Fusion usage tips live under [Kernel Fusion](./perf/kernel_fusion.md). +Build-mode setup (JLL / conda / developer) is documented under [Build Modes](./install.md). ## Build mode diff --git a/docs/src/api_random.md b/docs/src/api_random.md index 5a9923c86..9d6775710 100644 --- a/docs/src/api_random.md +++ b/docs/src/api_random.md @@ -1,16 +1,15 @@ # Random Module-level [`rand`](@ref cuNumeric.rand), [`randn`](@ref cuNumeric.randn), -and [`randexp`](@ref cuNumeric.randexp) are listed under +and [`randexp`](@ref cuNumeric.randexp) are documented under [Initialization](./api_initialization.md). This page covers the cuPyNumeric RNG stack: BitGenerators, `Generator`, and `default_rng`. -Draws go through cuRAND with the [`XORWOW`](@ref cuNumeric.XORWOW) random +Random draws use cuRAND with the [`XORWOW`](@ref cuNumeric.XORWOW) random number generator by default. We support `Float32` and `Float64` uniforms, normals, and exponentials (`randexp`), `ComplexF32` / `ComplexF64` uniforms and normals (independent real/imag parts; no `randexp`), `Bool` coin flips, -and signed `Int16` / `Int32` / `Int64`. Ranged integers use Julia -`rand(1:10, dims...)` (inclusive). There is no native Bool or complex +and signed `Int16` / `Int32` / `Int64`. There is no native Bool or complex distribution, so `rand(Bool, …)` draws `Int16` values in `{0,1}` and compares them to zero, and complex draws two real arrays packed as `re + i*imag`. diff --git a/docs/src/api_tensor.md b/docs/src/api_tensor.md new file mode 100644 index 000000000..d647f9c08 --- /dev/null +++ b/docs/src/api_tensor.md @@ -0,0 +1,101 @@ +# Tensor Contractions + +We extend [TensorOperations.jl](https://quantumkithub.github.io/TensorOperations.jl/stable/) to work with `NDArray`s. Loading TensorOperations alongside cuNumeric activates the package extension. + +Useful macros if you are new to TensorOperations are +[`@tensor`](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#The-@tensor-macro), +[`@tensoropt`](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@tensoropt), +and +[`@notensor`](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@notensor). + +The extension supports additions and permutations, traces, pairwise +contractions, complex conjugation, scale factors, accumulation into an existing +tensor, and multi-step `@tensor` blocks. Prefer `@tensor` (with `opt=true` for +larger networks) over the low-level `contract` / `contract!` API at the bottom +of this page. + +```julia +using cuNumeric +using TensorOperations + +α = randn() # Julia scalar, not a 0D NDArray +A = cuNumeric.randn(5, 5, 5, 5, 5, 5) +B = cuNumeric.randn(5, 5, 5) +C = cuNumeric.randn(5, 5, 5) +D = cuNumeric.zeros(5, 5, 5) + +@tensor begin + D[a, b, c] = A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + E[a, b, c] := A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] +end +``` + +## Scalars + +`@tensor` does **not** use the same 0D-`NDArray` convention as the rest of cuNumeric. In the future we will break the TensorOperatoins API so that scalars are 0D Arrays instead. + +```@raw html +
    +
  1. Scale factors must be Julia scalars. Coefficients such as α in α * C[i, j] must be a Julia Number. A 0D NDArray (for example the result of sum) is not accepted.
  2. +
  3. A fully contracted result is a Julia scalar. When every index is contracted, @tensor returns a host Number, not a 0D NDArray. TensorOperations gets that value with tensorscalar, which indexes the rank-zero array and therefore needs @allowscalar.
  4. +
+``` + +```julia +A = cuNumeric.rand(Float64, 64, 32) +B = cuNumeric.rand(Float64, 32, 16) + +@tensor opt=true C[i, j] := A[i, k] * B[k, j] # NDArray + +α = 2.0 # OK +# α = sum(C) # 0D NDArray, not a scale factor +@tensor D[i, j] := α * C[i, j] + +@allowscalar @tensor s = conj(C[i, j]) * C[i, j] # Julia Number, blocks +``` + +Full contractions also force a host synchronization, which stalls the Legate +runtime. If you can keep a 0D `NDArray` instead, use an ordinary reduction: + +```julia +s = sum(conj(C) .* C) # NDArray{T,0}, stays asynchronous +x = unwrap(s) # host Number, only when you need it +``` + +## Low-level `contract` / `contract!` + +> [!WARNING] +> Prefer `@tensor`. `contract` and `contract!` are the pairwise primitive the +> extension calls. They do not parse einsum strings, pick a contraction order, +> or free intermediates for you. + +Mode labels are not einsum strings: `"ik"` with `"kj"` is a GEMM; `"ijk"` with +`"ikl"` and explicit output `"ijl"` is a batched product. Labels may be ASCII +strings, `Char` tuples, or integers (`1` maps to `'a'`). + +```julia +A = cuNumeric.rand(Float32, 64, 32) +B = cuNumeric.rand(Float32, 32, 16) +C = contract(A, "ik", B, "kj") # allocates, Einstein output order +contract!(similar(C), "ij", A, "ik", B, "kj"; α=2, β=0) + +# batched: keep the shared 'i' on the output +AA = cuNumeric.rand(Float32, 8, 16, 32) +BB = cuNumeric.rand(Float32, 8, 32, 4) +CC = cuNumeric.zeros(Float32, 8, 16, 4) +contract!(CC, "ijl", AA, "ijk", BB, "ikl") + +tensordot(AA, BB, ([3], [2])) # same contraction, axes form +``` + +`α` and `β` implement `C = β*C + α*(A ⋆ B)` in Julia. The C++ kernel always +writes the unscaled product; `β ≠ 0` uses a temporary. Duplicate labels inside +one array are not allowed — use `cuNumeric.diagonal` first. + +This path is multi-GPU via Legate tiling (per-tile cuTENSOR or TBLIS), not +cuTensorMp. Integer and `Bool` inputs promote to `Float64` like other linalg. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/contract.jl"] +``` diff --git a/docs/src/examples/tensor_network.md b/docs/src/examples/tensor_network.md index 4389b2430..21bd490a9 100644 --- a/docs/src/examples/tensor_network.md +++ b/docs/src/examples/tensor_network.md @@ -57,9 +57,9 @@ println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) ## Expectation value -A fully contracted `@tensor` assignment is a Julia scalar. TensorOperations -gets that value with `tensorscalar`, which indexes the rank-zero `NDArray`. -That is scalar indexing, so it needs `@allowscalar`. +A fully contracted `@tensor` assignment is a Julia scalar, not a 0D `NDArray`. +That path needs `@allowscalar`. See [Scalars](../api_tensor.md#Scalars) for +the full rules, including that scale factors must be Julia `Number`s. ```julia energy, norm² = @allowscalar begin @@ -74,5 +74,4 @@ println("two-site energy = ", real(energy / norm²)) ``` The extension also supports output-index permutations, traces, conjugation, and -accumulation into an existing tensor. See [Tensor contractions](../linalg.md#Tensor-contractions) -for the lower-level `contract` and `contract!` interfaces. +accumulation into an existing tensor. See [Tensor Contractions](../api_tensor.md). diff --git a/docs/src/install.md b/docs/src/install.md index a4dc9ef37..429d92cc5 100644 --- a/docs/src/install.md +++ b/docs/src/install.md @@ -1,6 +1,6 @@ # Build Modes -cuNumeric.jl gets its cupynumeric / Legate binaries from one of three providers, chosen through `CNPreferences` (writes `LocalPreferences.toml`; **restart Julia** after changing mode): +cuNumeric.jl gets its cupynumeric / Legate binaries from one of three providers, chosen through `CNPreferences` (writes `LocalPreferences.toml`, **restart Julia** after changing mode): | Mode | When to use | |---|---| diff --git a/docs/src/linalg.md b/docs/src/linalg.md index 522249ab1..b3a0f337a 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -1,7 +1,8 @@ # Linear Algebra cuNumeric.jl provides matrix multiplication, solves, Cholesky, eigen, SVD, QR, -and related helpers for `NDArray`. +and related helpers for `NDArray`. For Einstein-index contractions, see +[Tensor Contractions](./api_tensor.md). All of the decompositions accept `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. Integer and `Bool` inputs are converted to `Float64`. As @@ -46,64 +47,6 @@ Pages = ["ndarray/binary.jl"] Filter = t -> t isa Function && nameof(t) === :mul! ``` -## Tensor contractions - -Loading [TensorOperations.jl](https://quantumkithub.github.io/TensorOperations.jl/stable/) -alongside cuNumeric activates the package extension for `NDArray`. Its `@tensor` -API supports additions and permutations, traces, pairwise contractions, complex -conjugation, scalar results, and multi-step expressions while keeping intermediate -tensors in cuNumeric: - -```julia -using cuNumeric -using TensorOperations - -A = cuNumeric.rand(Float64, 64, 32) -B = cuNumeric.rand(Float64, 32, 16) - -@tensor opt=true C[i, j] := A[i, k] * B[k, j] -@allowscalar @tensor squared_norm = conj(C[i, j]) * C[i, j] -``` - -A fully contracted `@tensor` result is a Julia scalar, so it goes through -`tensorscalar` and needs `@allowscalar`. Tensor results stay `NDArray`s. - -The extension is optional: TensorOperations is not loaded by cuNumeric itself. -TensorOperations-created intermediate `NDArray`s are released eagerly after use. - -The lower-level `contract` / `contract!` functions take **mode labels**, not -einsum strings: -`"ik"` with `"kj"` is a GEMM; `"ijk"` with `"ikl"` and explicit output `"ijl"` -is a batched product. Labels may be ASCII strings, `Char` tuples, or integers -(`1` maps to `'a'`). - -```julia -A = cuNumeric.rand(Float32, 64, 32) -B = cuNumeric.rand(Float32, 32, 16) -C = contract(A, "ik", B, "kj") # allocates, Einstein output order -contract!(similar(C), "ij", A, "ik", B, "kj"; α=2, β=0) - -# batched: keep the shared 'i' on the output -AA = cuNumeric.rand(Float32, 8, 16, 32) -BB = cuNumeric.rand(Float32, 8, 32, 4) -CC = cuNumeric.zeros(Float32, 8, 16, 4) -contract!(CC, "ijl", AA, "ijk", BB, "ikl") - -tensordot(AA, BB, ([3], [2])) # same contraction, axes form -``` - -`α` and `β` implement `C = β*C + α*(A ⋆ B)` in Julia. The C++ kernel always -writes the unscaled product; `β ≠ 0` uses a temporary. Duplicate labels inside -one array are not allowed — use `cuNumeric.diagonal` first. - -This path is multi-GPU via Legate tiling (per-tile cuTENSOR or TBLIS), not -cuTensorMp. Integer and `Bool` inputs promote to `Float64` like other linalg. - -```@autodocs -Modules = [cuNumeric] -Pages = ["ndarray/contract.jl"] -``` - ## Solve `cuNumeric.solve(A, b)` solves a linear system and returns an array with the same diff --git a/docs/src/perf/kernel_fusion.md b/docs/src/perf/kernel_fusion.md index cb715fadf..8e2fb2d59 100644 --- a/docs/src/perf/kernel_fusion.md +++ b/docs/src/perf/kernel_fusion.md @@ -1,8 +1,12 @@ # Kernel Fusion -When CUDA is available, cuNumeric can compile an eligible nested broadcast tree into one PTX kernel instead of launching one operation at a time. CPU execution follows the normal unfused path. +When CUDA is available cuNumeric.jl provides several methods to fuse operations into a single CUDA kernel. This can greatly improve performance and should be used whenever possible. -Prefer Julia's `@.` macro for multi-operation elementwise expressions so every operator is dotted: +## Automatic Broadcast Fusion + +Eligible broadcast expressions (i.e., `z .= cos.(x) .+ y`) are automatically fused into a single CUDA kernel. + +We reccomend using Julia's `@.` macro to ensure the entire expression gets fused. A missing dot can result in poor performance! ```julia # A missing dot on unary negation changes the expression and can prevent fusion. @@ -12,11 +16,27 @@ y .= .-a .+ b .* c y .= @. -a + b * c ``` -Fusion requires array leaves with the same shape and at least `FUSE_BROADCAST_MIN_OPS` broadcast operations (default: 2). Shape-mismatched broadcasts such as `matrix .+ vector` use the unfused path. +Automatic kernel fusion requires that: + +```@raw html +
    +
  1. Arrays have the same shape. Shape-mismatched broadcasts such as matrix .+ vector use the unfused path.
  2. +
  3. Expressions have at least two broadcast operations. Expressions like y .= cos.(x) with a single operation are unfused to reduce compilation overhead. This setting can be modified by calling CNPreferences.set_broadcast_fusion_min_ops!(x::Int) before launching cuNumeric. The default is x = 2.
  4. +
  5. A CUDA device is available.
  6. +
+``` + +Broadcast fusion can be disabled or re-enabled with CNPreferences as well: +```julia +CNPreferences.enable_broadcast_fusion!() # default +CNPreferences.disable_broadcast_fusion!() +``` + +For more information on setting preferences see the CNPreferences [CNPreferences documentation](../api_preferences.md). -## Fuse across statements with `@accelerate` +## Fuse Multiple Expressions with `@accelerate` -An ordinary assignment lets `@accelerate` substitute a single-use producer into its consumer: +One function of the `@accelerate` macro is to analyze code and find temporary variables which can be elided via kernel fusion. In the example below `tmp` is unused outside of the `update!` function, so `@accelerate` will merge the two lines together. ```julia @accelerate function update!(result, A, B, C) @@ -26,30 +46,17 @@ An ordinary assignment lets `@accelerate` substitute a single-use producer into end ``` -This can become the equivalent of `result .= @. (A + B) * C + 1.0f0`. The rewrite is conservative: the producer must have one use, no intervening statement may invalidate its inputs, and the normal fusion requirements still apply. +The equivalent code is `result .= @. (A + B) * C + 1.0f0`. The rewrite is conservative: the producer must have one use, no intervening statement may invalidate its inputs, and the normal fusion requirements still apply. -Do not preallocate a single-use intermediate with `.=` merely to avoid allocation: +This pattern requires the intermediate object to be temporary, and not pre-allocated. For example, if `tmp` was externally managed storage whose values were modified with `.=`, `@accelerate` would not be able to combine the expressions. For example, the following code would remain as two kernels. ```julia tmp .= @. A + B result .= @. tmp * C + 1.0f0 ``` -The mutation of `tmp` is observable, so it remains a kernel boundary. Keep preallocation when the intermediate is reused or its mutation must be visible. - -The `begin` form has different ownership semantics: named bindings remain live. On CUDA, an eligible same-shape chain can still be emitted as one multi-output kernel that materializes each binding. See [Accelerate Array Code](./reduce_allocations.md) for all macro forms. - -## Configure and inspect fusion - -Set preferences in one Julia process, then restart Julia: +More details on `@accelerate` can be found in the [memory management](./reduce_allocations.md) docs. -```julia -using CNPreferences - -CNPreferences.enable_broadcast_fusion!() # default -CNPreferences.disable_broadcast_fusion!() -CNPreferences.set_broadcast_fusion_min_ops!(2) # default -CNPreferences.set_broadcast_fusion_min_ops!(1) # include single-op broadcasts -``` +## Other -The default threshold avoids PTX compilation overhead when there is little work to combine. See [CNPreferences](../api_preferences.md) for preference details and [`BCAST_FUSION_DEBUG`](../debugging.md#inspect-fused-broadcasts-with-bcast_fusion_debug) to confirm which path ran. +To inspect the kernels emitted by broadcast fusion see these docs: [`BCAST_FUSION_DEBUG`](../debugging.md#inspect-fused-broadcasts-with-bcast_fusion_debug). diff --git a/docs/src/perf/patterns_to_avoid.md b/docs/src/perf/patterns_to_avoid.md index 3fe864acc..b393f1542 100644 --- a/docs/src/perf/patterns_to_avoid.md +++ b/docs/src/perf/patterns_to_avoid.md @@ -4,6 +4,8 @@ Accessing elements of an NDArray one at a time (e.g., `arr[5]`) is slow and should be avoided. Indexing like this requires data to be transferred between device and host and maybe even communicated across nodes. Scalar indexing will emit an error which can be opted out of with `@allowscalar` or `allowscalar() do ... end`. Several functions in the existing API invoke scalar indexing and are intended for testing (e.g., the `==` operator). +Scalar indexing can also appear through the [TensorOperations.jl](https://github.com/QuantumKitHub/TensorOperations.jl) extension when contractions reduce to a scalar. See [Scalars](../api_tensor.md#Scalars). In the future, we may break their API and return 0D NDArrays instead, to avoid blocking the runtime. Today, this can usually be avoided by using other reductions like `sum`, which do return 0D NDArrays, inplace of the einsum notation. + ## Implicit promotion Mixing integral types of different size (e.g., `Float64` and `Float32`) will result in implicit promotion of the smaller type to the larger types. This creates a copy of the data and hurts performance. Implicit promotion from a smaller integral type to a larger integral type will emit an error which can be opted out of with `@allowpromotion` or `allowpromotion() do ... end`. This error is common when mixing literals with `NDArrays`. By default a floating point literal (i.e., 1.0) is `Float64` but the default type of an `NDArray` is `Float32`. diff --git a/docs/src/perf/reduce_allocations.md b/docs/src/perf/reduce_allocations.md index f36ab803b..a36eabb86 100644 --- a/docs/src/perf/reduce_allocations.md +++ b/docs/src/perf/reduce_allocations.md @@ -1,24 +1,21 @@ -# The `@accelerate` Macro +# `@accelerate` -`@accelerate` optimizes *straight-line* array code: a fixed sequence of statements -with no branches, loops, jumps, `try`, or nested functions. Ordinary calls are -opaque boundaries; general control flow is not rewritten. +`@accelerate` optimizes array operations with no branches, loops, jumps, `try`, or nested functions (i.e., straight-line code). Ordinary function calls are opaque boundaries, general control flow is not rewritten. -Within that restricted body, it performs: +Given those contraints, `@accelerate` will: ```@raw html
    -
  1. Fusion within a broadcast expression. On CUDA, an eligible dotted expression such as @. A + B * C can run as one kernel. CPU execution uses the normal unfused path.
  2. -
  3. Fusion across broadcast statements. A single-use broadcast result can be substituted into its consumer, producing fewer GPU kernel launches.
  4. +
  5. Fuse broadcast expressions. On CUDA, an eligible dotted expression such as @. A + B * C can run as one kernel. CPU execution uses the normal unfused path.
  6. +
  7. Fuse across broadcast statements. A single-use broadcast result can be substituted into its consumer, producing fewer GPU kernel launches.
  8. Temporary lifetime analysis. After rewriting the code, the macro releases materialized, non-returned NDArrays after their final use on CPU or GPU.
``` - These jobs must happen together: an intermediate that fuses into its consumer is never allocated, while an intermediate that cannot fuse is materialized and then released after its last use. -## Use the function form by default +## `@accelerate` on Functions -Annotate a reusable straight-line function: +We reccomend using `@accelerate` on function definitions as they naturally separate inputs/ouputs (i.e., what should be kept) from everything else (i.e., something we can eagerly delete). We provide other forms (see below), but find this the most natural. ```julia @accelerate function update!(C, A, B) @@ -28,11 +25,11 @@ Annotate a reusable straight-line function: end ``` -On an eligible GPU path, `combined` can be folded into the second broadcast so the chain runs as one kernel. On CPU, or when fusion is ineligible, `combined` is materialized and released after the update. Function arguments belong to the caller and are never released by `@accelerate`; returned values also remain valid. +On an eligible GPU path, `combined` will be folded into the second broadcast so the chain runs as one kernel. On CPU, or when fusion is ineligible, `combined` is materialized and *immediately* released after the update (we do not wait for Julia GC). Function arguments belong to the caller and are never released by `@accelerate`. Returned values also remain valid. -## Choose a form +## Other Forms -The forms differ in which values must remain available, which determines how aggressively the macro may fuse or release intermediates. +The other forms of `@accelerate` provide other mechanisms to indicate which values must remain available, which in turn determines how aggressively the macro may fuse or release intermediates. | Form | Use it when | Fusion and lifetime behavior | | :--- | :--- | :--- | @@ -79,7 +76,15 @@ result = @accelerate (@. A + B * C) - Keep control flow outside the accelerated body. Loops, conditionals, `try`, short-circuit operators, and nested functions are rejected. - Ordinary function calls run in program order and form rewrite boundaries. Annotate the called function separately if its body should also be accelerated. +The `@accelerate` macro is great for fusing hot-loops where GC or kernel launch overhead should be minimized. + ```julia +@accelerate function update!(C, A, B) + combined = @. A + B + C .= @. 2.0f0 * combined + return C +end + for _ in 1:nsteps update!(C, A, B) end diff --git a/examples/daxpy.jl b/examples/daxpy.jl deleted file mode 100644 index 7ecb82886..000000000 --- a/examples/daxpy.jl +++ /dev/null @@ -1,11 +0,0 @@ -# found in examples/daxpy.jl -using cuNumeric - -arr = cuNumeric.rand(20) - -α = 1.32f0 -b = 2.0f0 - -arr2 = @accelerate @. α * arr + b - -println(arr2) diff --git a/examples/dmd.jl b/examples/dmd.jl index c8f2680a5..03b20f855 100644 --- a/examples/dmd.jl +++ b/examples/dmd.jl @@ -39,9 +39,9 @@ function dmd(X::NDArray{Float32,2}, r::Int) # X2 V Σ⁻¹ appears in both the projected operator and the exact modes. B = X2 * cuNumeric.transpose(Vt) * Sinv - à = cuNumeric.transpose(U) * B + A_tilde = cuNumeric.transpose(U) * B - E = eigen(Ã) # always complex, even for a real à + E = eigen(A_tilde) # always complex, even for a real Ã Φ = cuNumeric.as_type(B, ComplexF32) * E.vectors return E.values, Φ diff --git a/examples/tensor_network.jl b/examples/tensor_network.jl index 5defd1a81..71ed6519e 100644 --- a/examples/tensor_network.jl +++ b/examples/tensor_network.jl @@ -21,15 +21,14 @@ const BOND_DIM = 16 Hmatrix = (kron(σx, σx) + kron(σy, σy) + kron(σz, σz)) / 4 H = NDArray(reshape(Hmatrix, PHYSICAL_DIM, PHYSICAL_DIM, PHYSICAL_DIM, PHYSICAL_DIM)) -# MPS tensors A1[ap, s1, c], A2[c, s2, bp] and boundary environments. +# Initialize Random Tensors A1 = NDArray(randn(ComplexF64, BOND_DIM, PHYSICAL_DIM, BOND_DIM)) A2 = NDArray(randn(ComplexF64, BOND_DIM, PHYSICAL_DIM, BOND_DIM)) L = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) R = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) -# Open bonds remain, so both results are NDArrays. -@tensor ψ[a, s1, s2, b] := - L[a, ap] * A1[ap, s1, c] * A2[c, s2, bp] * R[bp, b] +# Perform Initial Tensor Contraction +@tensor ψ[a, s1, s2, b] := L[a, ap] * A1[ap, s1, c] * A2[c, s2, bp] * R[bp, b] @tensor Hψ[a, s1, s2, b] := H[s1, s2, t1, t2] * ψ[a, t1, t2, b] println("ψ is a ", typeof(ψ), " of size ", size(ψ)) @@ -37,6 +36,8 @@ println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) # Fully contracted @tensor results go through tensorscalar, which indexes the # rank-zero NDArray. That is scalar indexing and needs @allowscalar. +# This code could be replicated by calling `sum` to get a 0D-NDArray results +# that does NOT block the runtime. energy, norm² = @allowscalar begin @tensor begin e = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] From f03ede3207bf7ac31a4f9428cc781e56b4799d87 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:46:54 -0400 Subject: [PATCH 23/49] return 0D NDArray from TensorOperations API (#190) --- README.md | 2 +- docs/src/api_tensor.md | 37 +- docs/src/examples/tensor_network.md | 32 +- docs/src/perf/patterns_to_avoid.md | 5 +- examples/tensor_network.jl | 25 +- ext/cuNumericTensorOperationsExt.jl | 514 +++++++++++++++++++++++++--- src/ndarray/contract.jl | 122 ++++++- test/array/contract.jl | 10 + test/array/tensoroperations.jl | 43 ++- 9 files changed, 666 insertions(+), 124 deletions(-) diff --git a/README.md b/README.md index cdc31fbf1..7942c54f6 100644 --- a/README.md +++ b/README.md @@ -75,7 +75,7 @@ We implement a package extension for [TensorOperations.jl](https://github.com/Qu using TensorOperations using cuNumeric -α = randn() # Must be Julia scalar, not NDArray currently +α = randn() # prefer a Julia Number when you already have one A = cuNumeric.randn(5, 5, 5, 5, 5, 5) B = cuNumeric.randn(5, 5, 5) C = cuNumeric.randn(5, 5, 5) diff --git a/docs/src/api_tensor.md b/docs/src/api_tensor.md index d647f9c08..8fffa2cc6 100644 --- a/docs/src/api_tensor.md +++ b/docs/src/api_tensor.md @@ -18,7 +18,7 @@ of this page. using cuNumeric using TensorOperations -α = randn() # Julia scalar, not a 0D NDArray +α = randn() # prefer a Julia Number when you already have one A = cuNumeric.randn(5, 5, 5, 5, 5, 5) B = cuNumeric.randn(5, 5, 5) C = cuNumeric.randn(5, 5, 5) @@ -32,14 +32,11 @@ end ## Scalars -`@tensor` does **not** use the same 0D-`NDArray` convention as the rest of cuNumeric. In the future we will break the TensorOperatoins API so that scalars are 0D Arrays instead. +Prefer a Julia `Number` for scale factors when possible (i.e., `2.0f0`, `randn()`, …). This allows TensorOperations.jl to perform optimizatoins for special values like zero and one. -```@raw html -
    -
  1. Scale factors must be Julia scalars. Coefficients such as α in α * C[i, j] must be a Julia Number. A 0D NDArray (for example the result of sum) is not accepted.
  2. -
  3. A fully contracted result is a Julia scalar. When every index is contracted, @tensor returns a host Number, not a 0D NDArray. TensorOperations gets that value with tensorscalar, which indexes the rank-zero array and therefore needs @allowscalar.
  4. -
-``` +A 0D `NDArray` is also accepted as a scale — for example `sum(C)`, or a +fully contracted `@tensor` result. If you intend to use these results in down-stream tensor contractions keep those on device. `unwrap` copies the value to the host and **blocks the Legate runtime**, so only unwrap +when you truly need a Julia `Number` (i.e. printing of host-side if-else). ```julia A = cuNumeric.rand(Float64, 64, 32) @@ -47,19 +44,12 @@ B = cuNumeric.rand(Float64, 32, 16) @tensor opt=true C[i, j] := A[i, k] * B[k, j] # NDArray -α = 2.0 # OK -# α = sum(C) # 0D NDArray, not a scale factor -@tensor D[i, j] := α * C[i, j] - -@allowscalar @tensor s = conj(C[i, j]) * C[i, j] # Julia Number, blocks -``` +α = 2.0 +@tensor D[i, j] := α * C[i, j] # Julia Number, no sync -Full contractions also force a host synchronization, which stalls the Legate -runtime. If you can keep a 0D `NDArray` instead, use an ordinary reduction: - -```julia -s = sum(conj(C) .* C) # NDArray{T,0}, stays asynchronous -x = unwrap(s) # host Number, only when you need it +@tensor s = conj(C[i, j]) * C[i, j] # 0D NDArray, stays asynchronous +@tensor E[i, j] := s * C[i, j] # reuse 0D as a scale, still no unwrap +x = unwrap(s) # host Number; blocks ``` ## Low-level `contract` / `contract!` @@ -88,9 +78,10 @@ contract!(CC, "ijl", AA, "ijk", BB, "ikl") tensordot(AA, BB, ([3], [2])) # same contraction, axes form ``` -`α` and `β` implement `C = β*C + α*(A ⋆ B)` in Julia. The C++ kernel always -writes the unscaled product; `β ≠ 0` uses a temporary. Duplicate labels inside -one array are not allowed — use `cuNumeric.diagonal` first. +`α` and `β` implement `C = β*C + α*(A ⋆ B)` in Julia. Prefer a Julia `Number`; +a 0-d `NDArray` is also accepted. The C++ kernel always writes the unscaled +product; `β ≠ 0` uses a temporary. Duplicate labels inside one array are not +allowed — use `cuNumeric.diagonal` first. This path is multi-GPU via Legate tiling (per-tile cuTENSOR or TBLIS), not cuTensorMp. Integer and `Bool` inputs promote to `Float64` like other linalg. diff --git a/docs/src/examples/tensor_network.md b/docs/src/examples/tensor_network.md index 21bd490a9..6caff7289 100644 --- a/docs/src/examples/tensor_network.md +++ b/docs/src/examples/tensor_network.md @@ -13,8 +13,8 @@ H = S^x_1 S^x_2 + S^y_1 S^y_2 + S^z_1 S^z_2. The examples below apply it to an MPS-like state with tensors ``A^{(1)}``, ``A^{(2)}`` and boundary environments ``L``, ``R``. The first contractions keep -open bond indices and stay `NDArray`s. The second contracts every index to a -host scalar. +open bond indices and stay `NDArray`s. The second contracts every index; the +ratio stays on device until `unwrap` for printing. ```julia # found in examples/tensor_network.jl @@ -23,7 +23,7 @@ using LinearAlgebra using Random using TensorOperations -Random.seed!(1234) +Random.seed!(1234) # for Random.randn, cuNumeric expects seed via default_rng physical_dim = 2 bond_dim = 16 @@ -44,7 +44,7 @@ R = NDArray(randn(ComplexF64, bond_dim, bond_dim)) These contractions have free indices ``a, s_1, s_2, b``, so the results are `NDArray`s. TensorOperations lowers the network to pairwise contractions and -releases the intermediate `NDArray`s. Nothing here indexes a single element. +releases the intermediate `NDArray`s. ```julia @tensor ψ[a, s1, s2, b] := @@ -57,21 +57,15 @@ println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) ## Expectation value -A fully contracted `@tensor` assignment is a Julia scalar, not a 0D `NDArray`. -That path needs `@allowscalar`. See [Scalars](../api_tensor.md#Scalars) for -the full rules, including that scale factors must be Julia `Number`s. +A fully contracted `@tensor` assignment is a 0D `NDArray`, like `sum`. +Prefer to keep it that way and do device arithmetic (`./`) until you +need a host `Number`. `unwrap` (here, only to print) **blocks**. See +[Scalars](../api_tensor.md#Scalars). ```julia -energy, norm² = @allowscalar begin - @tensor begin - e = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] - n = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] - end - e, n -end - -println("two-site energy = ", real(energy / norm²)) -``` +@tensor energy = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] +@tensor norm² = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] -The extension also supports output-index permutations, traces, conjugation, and -accumulation into an existing tensor. See [Tensor Contractions](../api_tensor.md). +println("two-site energy = ", real(unwrap(energy ./ norm²))) +``` +See [Tensor Contractions](../api_tensor.md) documentation for more details. diff --git a/docs/src/perf/patterns_to_avoid.md b/docs/src/perf/patterns_to_avoid.md index b393f1542..ef99566b2 100644 --- a/docs/src/perf/patterns_to_avoid.md +++ b/docs/src/perf/patterns_to_avoid.md @@ -4,7 +4,10 @@ Accessing elements of an NDArray one at a time (e.g., `arr[5]`) is slow and should be avoided. Indexing like this requires data to be transferred between device and host and maybe even communicated across nodes. Scalar indexing will emit an error which can be opted out of with `@allowscalar` or `allowscalar() do ... end`. Several functions in the existing API invoke scalar indexing and are intended for testing (e.g., the `==` operator). -Scalar indexing can also appear through the [TensorOperations.jl](https://github.com/QuantumKitHub/TensorOperations.jl) extension when contractions reduce to a scalar. See [Scalars](../api_tensor.md#Scalars). In the future, we may break their API and return 0D NDArrays instead, to avoid blocking the runtime. Today, this can usually be avoided by using other reductions like `sum`, which do return 0D NDArrays, inplace of the einsum notation. +`unwrap` (and `A[]`) pulls a host `Number` out of a 0D `NDArray` and +**blocks the runtime**. Prefer a Julia `Number` when you already have one, +and leave reductions / fully contracted `@tensor` results as 0D arrays +until you actually need the host value. See [Scalars](../api_tensor.md#Scalars). ## Implicit promotion diff --git a/examples/tensor_network.jl b/examples/tensor_network.jl index 71ed6519e..2937e8025 100644 --- a/examples/tensor_network.jl +++ b/examples/tensor_network.jl @@ -1,8 +1,8 @@ -#= Contract a two-site tensor network, then optionally pull a host scalar. +#= Contract a two-site tensor network, then print a host scalar. The first contractions build an MPS-like two-site state and apply a Heisenberg -operator. Those results stay on device as NDArrays. Fully contracting to -⟨ψ|H|ψ⟩ / ⟨ψ|ψ⟩ goes through tensorscalar and therefore needs @allowscalar. +operator. Those results stay on device as NDArrays. The fully contracted +⟨ψ|H|ψ⟩ / ⟨ψ|ψ⟩ ratio is computed on device; unwrap only to print (it blocks). =# using cuNumeric @@ -34,16 +34,9 @@ R = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) println("ψ is a ", typeof(ψ), " of size ", size(ψ)) println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) -# Fully contracted @tensor results go through tensorscalar, which indexes the -# rank-zero NDArray. That is scalar indexing and needs @allowscalar. -# This code could be replicated by calling `sum` to get a 0D-NDArray results -# that does NOT block the runtime. -energy, norm² = @allowscalar begin - @tensor begin - e = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] - n = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] - end - e, n -end - -println("two-site energy = ", real(energy / norm²)) +# Fully contracted @tensor results are 0D NDArrays. Keep them on device +# until a host Number is required; unwrap blocks. +@tensor energy = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] +@tensor norm² = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] + +println("two-site energy = ", real(unwrap(energy ./ norm²))) diff --git a/ext/cuNumericTensorOperationsExt.jl b/ext/cuNumericTensorOperationsExt.jl index e83e6d339..8fb4c0a6e 100644 --- a/ext/cuNumericTensorOperationsExt.jl +++ b/ext/cuNumericTensorOperationsExt.jl @@ -23,6 +23,8 @@ using TensorOperations const CN = cuNumeric const TO = TensorOperations +const One = TO.One +const Zero = TO.Zero struct CuNumericBackend <: TO.AbstractBackend end @@ -70,31 +72,102 @@ function TO.tensorfree!(C::CN.NDArray, allocator=TO.DefaultAllocator()) return nothing end -function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α, β) where {T} - α′ = convert(T, α) - if iszero(β) - C .= α′ .* A - else - β′ = convert(T, β) - C .= β′ .* C .+ α′ .* A +# ------------------------------------------------------------------------------------------ +# Scale factors: One/Zero dispatch, Number wraps to 0D and re-dispatches +# ------------------------------------------------------------------------------------------ + +function _require_0d_scale(x::CN.NDArray) + ndims(x) == 0 || throw( + DimensionMismatch("tensor scale factor must be a Julia Number or a 0-d NDArray") + ) + return x +end + +_as_scale(x::One, ::Type) = x +_as_scale(x::Zero, ::Type) = x +function _as_scale(x::CN.NDArray, ::Type{T}) where {T} + _require_0d_scale(x) + return _convert_eltype(x, T) +end +function _as_scale(x::Number, ::Type{T}) where {T} + isone(x) && return One() + iszero(x) && return Zero() + return CN.NDArray(convert(T, x)) +end + +function _free_scale!(orig, scaled) + if scaled isa CN.NDArray && !(orig isa CN.NDArray && orig === scaled) + CN.destroy!(scaled) end + return nothing +end + +_accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, ::Zero) = (C.=A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, ::Zero) = (C.=α .* A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, ::One) = (C.=C .+ A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, ::One) = (C.=C .+ α .* A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, β::CN.NDArray) = (C.=β .* C .+ A; C) +function _accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, β::CN.NDArray) + C .= β .* C .+ α .* A return C end +# α = 0: product does not contribute. Strong-zero for β = 0 does not read C. +_accumulate!(C::CN.NDArray, ::CN.NDArray, ::Zero, ::Zero) = (C.=zero(eltype(C)); C) +_accumulate!(C::CN.NDArray, ::CN.NDArray, ::Zero, ::One) = C +_accumulate!(C::CN.NDArray, ::CN.NDArray, ::Zero, β::CN.NDArray) = (C.=β .* C; C) + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α::Number, β::Number) where {T} + α′ = _as_scale(α, T) + β′ = _as_scale(β, T) + try + return _accumulate!(C, A, α′, β′) + finally + _free_scale!(α, α′) + _free_scale!(β, β′) + end +end + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α::CN.NDArray, β::Number) where {T} + α′ = _as_scale(α, T) + β′ = _as_scale(β, T) + try + return _accumulate!(C, A, α′, β′) + finally + _free_scale!(α, α′) + _free_scale!(β, β′) + end +end + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α::Number, β::CN.NDArray) where {T} + α′ = _as_scale(α, T) + β′ = _as_scale(β, T) + try + return _accumulate!(C, A, α′, β′) + finally + _free_scale!(α, α′) + _free_scale!(β, β′) + end +end function _convert_eltype(A::CN.NDArray, ::Type{T}) where {T} return eltype(A) === T ? A : CN.as_type(A, T) end -function TO.tensoradd!( - C::CN.NDArray, - A::CN.NDArray, - pA::TO.Index2Tuple, - conjA::Bool, - α::Number, - β::Number, - ::CuNumericBackend, - allocator=TO.DefaultAllocator(), -) +function _ensure_backend(op, backend, C, select_args...) + if backend isa TO.DefaultBackend + return TO.select_backend(op, C, select_args...) + elseif backend isa CuNumericBackend + return backend + else + throw(ArgumentError("Unknown backend $backend for $op and NDArray")) + end +end + +# ------------------------------------------------------------------------------------------ +# tensoradd! +# ------------------------------------------------------------------------------------------ + +function _tensoradd_impl!(C::CN.NDArray, A::CN.NDArray, pA, conjA, α, β) TO.argcheck_tensoradd(C, A, pA) TO.dimcheck_tensoradd(C, A, pA) if C.ptr === A.ptr && !TO.istrivialpermutation(pA) @@ -116,6 +189,105 @@ function TO.tensoradd!( return C end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + return _tensoradd_impl!(C, A, pA, conjA, α, β) +end + +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β::Union{Number,CN.NDArray}, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(α) + β isa CN.NDArray && _require_0d_scale(β) + return _tensoradd_impl!(C, A, pA, conjA, α, β) +end + +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(β) + return _tensoradd_impl!(C, A, pA, conjA, α, β) +end + +function TO.tensoradd!( + C::CN.NDArray, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool, α::CN.NDArray, β +) + return TO.tensoradd!(C, A, pA, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensoradd!( + C::CN.NDArray, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool, α::Number, β::CN.NDArray +) + return TO.tensoradd!(C, A, pA, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensoradd!( + C::CN.NDArray, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool, α::CN.NDArray, β, backend +) + return TO.tensoradd!(C, A, pA, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, +) + return TO.tensoradd!(C, A, pA, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, + backend, + allocator, +) + b = _ensure_backend(TO.tensoradd!, backend, C, A) + return TO.tensoradd!(C, A, pA, conjA, α, β, b, allocator) +end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, + allocator, +) + b = _ensure_backend(TO.tensoradd!, backend, C, A) + return TO.tensoradd!(C, A, pA, conjA, α, β, b, allocator) +end + +# ------------------------------------------------------------------------------------------ +# tensortrace! +# ------------------------------------------------------------------------------------------ + function _trace_pairs(A::CN.NDArray, q::TO.Index2Tuple) current = A current_owned = false @@ -139,17 +311,7 @@ function _trace_pairs(A::CN.NDArray, q::TO.Index2Tuple) return current, current_owned, labels end -function TO.tensortrace!( - C::CN.NDArray, - A::CN.NDArray, - p::TO.Index2Tuple, - q::TO.Index2Tuple, - conjA::Bool, - α::Number, - β::Number, - ::CuNumericBackend, - allocator=TO.DefaultAllocator(), -) +function _tensortrace_impl!(C::CN.NDArray, A::CN.NDArray, p, q, conjA, α, β) TO.argcheck_tensortrace(C, A, p, q) TO.dimcheck_tensortrace(C, A, p, q) C.ptr === A.ptr && throw(ArgumentError("output tensor must not alias input tensor")) @@ -171,6 +333,130 @@ function TO.tensortrace!( return C end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + return _tensortrace_impl!(C, A, p, q, conjA, α, β) +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β::Union{Number,CN.NDArray}, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(α) + β isa CN.NDArray && _require_0d_scale(β) + return _tensortrace_impl!(C, A, p, q, conjA, α, β) +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(β) + return _tensortrace_impl!(C, A, p, q, conjA, α, β) +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, + backend, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, + backend, + allocator, +) + b = _ensure_backend(TO.tensortrace!, backend, C, A) + return TO.tensortrace!(C, A, p, q, conjA, α, β, b, allocator) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, + allocator, +) + b = _ensure_backend(TO.tensortrace!, backend, C, A) + return TO.tensortrace!(C, A, p, q, conjA, α, β, b, allocator) +end + +# ------------------------------------------------------------------------------------------ +# tensorcontract! +# ------------------------------------------------------------------------------------------ + function _contract_modes( A::CN.NDArray, pA::TO.Index2Tuple, @@ -198,19 +484,8 @@ function _contract_modes( return Cmodes, Amodes, Bmodes end -function TO.tensorcontract!( - C::CN.NDArray, - A::CN.NDArray, - pA::TO.Index2Tuple, - conjA::Bool, - B::CN.NDArray, - pB::TO.Index2Tuple, - conjB::Bool, - pAB::TO.Index2Tuple, - α::Number, - β::Number, - ::CuNumericBackend, - allocator=TO.DefaultAllocator(), +function _tensorcontract_impl!( + C::CN.NDArray, A::CN.NDArray, pA, conjA, B::CN.NDArray, pB, conjB, pAB, α, β ) TO.argcheck_tensorcontract(C, A, pA, B, pB, pAB) TO.dimcheck_tensorcontract(C, A, pA, B, pB, pAB) @@ -230,9 +505,164 @@ function TO.tensorcontract!( return C end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + return _tensorcontract_impl!(C, A, pA, conjA, B, pB, conjB, pAB, α, β) +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β::Union{Number,CN.NDArray}, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(α) + β isa CN.NDArray && _require_0d_scale(β) + return _tensorcontract_impl!(C, A, pA, conjA, B, pB, conjB, pAB, α, β) +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(β) + return _tensorcontract_impl!(C, A, pA, conjA, B, pB, conjB, pAB, α, β) +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, TO.DefaultBackend() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, TO.DefaultBackend() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β, + backend, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, backend, TO.DefaultAllocator() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, + backend, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, backend, TO.DefaultAllocator() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β, + backend, + allocator, +) + b = _ensure_backend(TO.tensorcontract!, backend, C, A, B) + return TO.tensorcontract!(C, A, pA, conjA, B, pB, conjB, pAB, α, β, b, allocator) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, + backend, + allocator, +) + b = _ensure_backend(TO.tensorcontract!, backend, C, A, B) + return TO.tensorcontract!(C, A, pA, conjA, B, pB, conjB, pAB, α, β, b, allocator) +end + function TO.tensorscalar(C::CN.NDArray) ndims(C) == 0 || throw(DimensionMismatch("tensorscalar requires a rank-zero tensor")) - return C[] + return copy(C) end end diff --git a/src/ndarray/contract.jl b/src/ndarray/contract.jl index d5179ff5c..2cf7a9ed5 100644 --- a/src/ndarray/contract.jl +++ b/src/ndarray/contract.jl @@ -179,6 +179,8 @@ end In-place pairwise tensor contraction `C = β * C + α * (A ⋆ B)`. +`α` and `β` may be a Julia `Number` or a 0-d `NDArray`. + `Amodes`, `Bmodes`, and `Cmodes` are mode labels for `A`, `B`, and `C`: an ASCII `AbstractString`, a tuple/vector of `Char`, or a vector of integers (`1` maps to `'a'`). Each label appears twice (a contracted or free index) or @@ -210,7 +212,7 @@ function contract!( Ap = checked_promote_arr(contract!, A, T) Bp = checked_promote_arr(contract!, B, T) try - return _contract_same_type!(C, Cmodes, Ap, Amodes, Bp, Bmodes, convert(T, α), convert(T, β)) + return _contract_same_type!(C, Cmodes, Ap, Amodes, Bp, Bmodes, α, β) finally Ap !== A && destroy!(Ap) Bp !== B && destroy!(Bp) @@ -226,15 +228,15 @@ function contract!(C::NDArray, Cmodes, A::NDArray, Amodes, B::NDArray, Bmodes; return throw(ArgumentError("array type $bad is unsupported in contract!")) end -function _contract_same_type!( - C::NDArray{T}, - Cmodes, - A::NDArray{T}, - Amodes, - B::NDArray{T}, - Bmodes, - α::T, - β::T, +function _require_0d_scale(x::NDArray) + ndims(x) == 0 || throw( + ArgumentError("contract! scale factor must be a Number or a 0-d NDArray") + ) + return x +end + +function _contract_prepare( + C::NDArray{T}, Cmodes, A::NDArray{T}, Amodes, B::NDArray{T}, Bmodes ) where {T} (C.ptr === A.ptr || C.ptr === B.ptr) && throw( ArgumentError("contract! output must not alias either input") @@ -252,23 +254,119 @@ function _contract_same_type!( ) end extent_keys, extent_vals = _extent_arrays(extents) + return Cm, Am, Bm, extent_keys, extent_vals +end +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::Number, + β::Number, +) where {T} + Cm, Am, Bm, extent_keys, extent_vals = _contract_prepare(C, Cmodes, A, Amodes, B, Bmodes) if isone(α) && iszero(β) _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) return C end + αT = convert(T, α) if iszero(β) _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) - C .= α .* C + C .= αT .* C return C end + βT = convert(T, β) tmp = similar(C) _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) - C .= β .* C .+ α .* tmp + C .= βT .* C .+ αT .* tmp destroy!(tmp) return C end +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::NDArray, + β::Number, +) where {T} + _require_0d_scale(α) + Cm, Am, Bm, extent_keys, extent_vals = _contract_prepare(C, Cmodes, A, Amodes, B, Bmodes) + α′ = eltype(α) === T ? α : as_type(α, T) + try + if iszero(β) + _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= α′ .* C + return C + end + tmp = similar(C) + _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) + if isone(β) + C .= C .+ α′ .* tmp + else + βa = NDArray(convert(T, β)) + C .= βa .* C .+ α′ .* tmp + destroy!(βa) + end + destroy!(tmp) + return C + finally + α′ !== α && destroy!(α′) + end +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::Number, + β::NDArray, +) where {T} + _require_0d_scale(β) + αa = NDArray(convert(T, α)) + try + return _contract_same_type!(C, Cmodes, A, Amodes, B, Bmodes, αa, β) + finally + destroy!(αa) + end +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::NDArray, + β::NDArray, +) where {T} + _require_0d_scale(α) + _require_0d_scale(β) + Cm, Am, Bm, extent_keys, extent_vals = _contract_prepare(C, Cmodes, A, Amodes, B, Bmodes) + α′ = eltype(α) === T ? α : as_type(α, T) + β′ = eltype(β) === T ? β : as_type(β, T) + try + tmp = similar(C) + _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= β′ .* C .+ α′ .* tmp + destroy!(tmp) + return C + finally + α′ !== α && destroy!(α′) + β′ !== β && destroy!(β′) + end +end + """ contract(A, Amodes, B, Bmodes; α=1) diff --git a/test/array/contract.jl b/test/array/contract.jl index 888a1394e..4d829a585 100644 --- a/test/array/contract.jl +++ b/test/array/contract.jl @@ -216,6 +216,16 @@ end _host_contract_compare( β3 * seed_ji + α2 * permutedims(prod), out_ji, T; n=nk, scale=α2 * scale ) + + α0 = NDArray(α2) + out = cuNumeric.zeros(T, 4, 5) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α0, β=0) + _host_contract_compare(α2 * prod, out, T; n=nk, scale=α2 * scale) + + β0 = NDArray(β3) + out = NDArray(copy(seed)) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α0, β=β0) + _host_contract_compare(β3 * seed + α2 * prod, out, T; n=nk, scale=α2 * scale) end end diff --git a/test/array/tensoroperations.jl b/test/array/tensoroperations.jl index 63ffd8bf2..4532ca39e 100644 --- a/test/array/tensoroperations.jl +++ b/test/array/tensoroperations.jl @@ -164,15 +164,16 @@ using TensorOperations: TensorOperations as TO matrix = reshape(ComplexF64.(1:9) .+ im .* ComplexF64.(9:-1:1), 3, 3) M = NDArray(matrix) - @test_throws ErrorException (@tensor t = M[i, i]) - scalar_trace = @allowscalar @tensor(t = M[i, i]) - @test scalar_trace isa ComplexF64 - @test scalar_trace == sum(matrix[i, i] for i in axes(matrix, 1)) + scalar_trace = @tensor(t = M[i, i]) + @test scalar_trace isa NDArray + @test ndims(scalar_trace) == 0 + @test only(Array(scalar_trace)) == sum(matrix[i, i] for i in axes(matrix, 1)) conjugated_trace = TO.tensortrace(M, ((), ()), ((1,), (2,)), true) - @test_throws ErrorException TO.tensorscalar(conjugated_trace) - @test (@allowscalar TO.tensorscalar(conjugated_trace)) == - sum(conj(matrix[i, i]) for i in axes(matrix, 1)) + ts = TO.tensorscalar(conjugated_trace) + @test ts isa NDArray + @test ndims(ts) == 0 + @test only(Array(ts)) == sum(conj(matrix[i, i]) for i in axes(matrix, 1)) TO.tensorfree!(conjugated_trace) end @@ -206,9 +207,31 @@ using TensorOperations: TensorOperations as TO vhost = ComplexF64.(5:8) .- im .* ComplexF64.(1:4) u = NDArray(uhost) v = NDArray(vhost) - scalar_product = @allowscalar @tensor(s = u[k] * v[k]) - @test scalar_product isa ComplexF64 - @test scalar_product ≈ sum(uhost .* vhost) + scalar_product = @tensor(s = u[k] * v[k]) + @test scalar_product isa NDArray + @test ndims(scalar_product) == 0 + @test only(Array(scalar_product)) ≈ sum(uhost .* vhost) + end + + @testset "0D NDArray scale factors" begin + hostC = reshape(collect(Float64, 1:6), 2, 3) + C = NDArray(hostC) + α = sum(C) + @test α isa NDArray + @test ndims(α) == 0 + + @tensor scaled[i, j] := α * C[i, j] + @test Array(scaled) ≈ only(Array(α)) .* hostC + + @tensor s = C[i, j] * C[i, j] + @test s isa NDArray + @test ndims(s) == 0 + @tensor scaled2[i, j] := s * C[i, j] + @test Array(scaled2) ≈ only(Array(s)) .* hostC + + dest = NDArray(fill(2.0, 2, 3)) + @tensor dest[i, j] += α * C[i, j] + @test Array(dest) ≈ fill(2.0, 2, 3) .+ only(Array(α)) .* hostC end @testset "temporary destruction" begin From 3ecab6f1514e82cc061f56aa41c9e23c887ae069 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Thu, 10 Sep 2026 18:07:06 -0400 Subject: [PATCH 24/49] revamp benchmark suite (#191) * revamp benchmark suite * automate selection of N * fix bug in 0D benchmark correctness, add more logging * int overflow bug * add execution fences on certain benchmarks --- .gitignore | 1 - benchmark/MEMORY.md | 80 ++++ benchmark/Project.toml | 5 +- benchmark/README.md | 92 ++++- benchmark/__plot_results.jl | 301 --------------- benchmark/benchmarks.toml | 137 +++---- benchmark/cpp_matmul/CMakeLists.txt | 17 - benchmark/cpp_matmul/build.sh | 7 - benchmark/cpp_matmul/main.cpp | 50 --- benchmark/diagnose_montecarlo.jl | 34 ++ benchmark/install_cupynumeric.sh | 11 +- benchmark/mwe_fusion_indexing.jl | 78 ++++ benchmark/plot_results.jl | 365 +++++++++--------- benchmark/run.jl | 131 +------ benchmark/run_benchmark.sh | 12 +- benchmark/src/autosize.jl | 108 ++++++ benchmark/src/benchmarks/dmd.jl | 71 ++-- benchmark/src/benchmarks/gemm.jl | 32 +- benchmark/src/benchmarks/grayscott.jl | 100 ++--- .../benchmarks/grayscott_accelerate_forms.jl | 70 ++-- benchmark/src/benchmarks/montecarlo.jl | 31 +- benchmark/src/benchmarks/poisson_fft.jl | 74 ++-- .../src/benchmarks/tensor_contractions.jl | 109 +++--- benchmark/src/core.jl | 218 +++++++++-- benchmark/src/memory.jl | 147 +++++++ benchmark/src/parse_benchmarks.jl | 122 +++++- benchmark/src/planning.jl | 140 +++++++ benchmark/src/result_rows.jl | 47 +++ benchmark/src/runner.jl | 166 ++++++++ benchmark/src/single.jl | 69 ++-- benchmark/src_py/benchmarks/dmd.py | 19 +- benchmark/src_py/benchmarks/gemm.py | 11 +- benchmark/src_py/benchmarks/grayscott.py | 21 +- benchmark/src_py/benchmarks/montecarlo.py | 7 +- benchmark/src_py/benchmarks/poisson_fft.py | 11 +- .../src_py/benchmarks/tensor_contractions.py | 20 +- benchmark/src_py/core.py | 33 +- benchmark/src_py/single.py | 17 +- benchmark/test/autosize.jl | 26 ++ benchmark/test/runtests.jl | 198 ++++++++++ benchmark/test/test_timing.py | 63 +++ benchmark/test/timing.jl | 34 ++ docs/src/benchmarks/howto.md | 74 +++- lib/cunumeric_jl_wrapper/include/accessors.h | 8 +- .../include/cuda_macros.h | 3 +- lib/cunumeric_jl_wrapper/src/cuda.cpp | 96 +---- lib/cunumeric_jl_wrapper/src/memory.cpp | 6 +- src/ndarray/broadcast_fusion.jl | 72 +++- src/ndarray/detail/ndarray.jl | 2 +- test/regressions.jl | 16 + 50 files changed, 2344 insertions(+), 1218 deletions(-) create mode 100644 benchmark/MEMORY.md delete mode 100644 benchmark/__plot_results.jl delete mode 100644 benchmark/cpp_matmul/CMakeLists.txt delete mode 100755 benchmark/cpp_matmul/build.sh delete mode 100644 benchmark/cpp_matmul/main.cpp create mode 100644 benchmark/diagnose_montecarlo.jl create mode 100644 benchmark/mwe_fusion_indexing.jl create mode 100644 benchmark/src/autosize.jl create mode 100644 benchmark/src/memory.jl create mode 100644 benchmark/src/planning.jl create mode 100644 benchmark/src/result_rows.jl create mode 100644 benchmark/src/runner.jl create mode 100644 benchmark/test/autosize.jl create mode 100644 benchmark/test/runtests.jl create mode 100644 benchmark/test/test_timing.py create mode 100644 benchmark/test/timing.jl create mode 100644 test/regressions.jl diff --git a/.gitignore b/.gitignore index cf02abbe8..045e0e303 100644 --- a/.gitignore +++ b/.gitignore @@ -33,7 +33,6 @@ benchmark/results/* benchmark/plots** compile_wrapper.sh -__plot_results.jl *.tar.gz # generated by CMake diff --git a/benchmark/MEMORY.md b/benchmark/MEMORY.md new file mode 100644 index 000000000..0f7d2d5f1 --- /dev/null +++ b/benchmark/MEMORY.md @@ -0,0 +1,80 @@ +# Memory accounting and supported runs + +`src/memory.jl` is the authoritative preflight model. The older `total_space` +helpers describe storage and are not used by the sweep planner. Byte arithmetic +uses `BigInt`; GPU count, backend, dtype, fusion, benchmark type, dimensions, +and warmup/iteration count all enter the planning calculation. + +## Lifetime policy + +Estimates are conservative upper bounds, not measured allocator peaks. They +include initialization and the entire trial. Julia tracing GC is not assumed +to run between iterations. In the baseline Monte Carlo kernel an unreferenced +broadcast output can therefore remain for every iteration. Gray-Scott baseline, +begin and expression forms can similarly retain named buffers until GC. The +function/let acceleration forms explicitly destroy last-used local arrays and +are bounded independently of iteration count. Fusion-disabled execution still +uses the accelerated lifetime rewrite. + +Fused Gray-Scott is bounded by four persistent grids plus four named interior +results and an assignment output. Unfused execution reserves three additional +interior buffers for nested operands and the outer broadcast destination. +The function/let bound conservatively retains named intermediates within an +iteration even when inter-statement fusion may eliminate them. It does not +claim an exact optimized peak. Full parent grids are counted rather than +assuming slice/halo partition placement. + +Python reference counting avoids the Julia retention allowance, but an input +and output of an unfused operation must coexist. The random helper generates +Float64 values before casting, and this conversion is counted. These models +assume the current kernels and standard supported runtime execution; physical +Legate instance lifetimes still require validation on the target machine. + +## Native workspace + +GEMM, DMD, FFT and tensor contraction scratch/packing bounds depend on native +libraries and their algorithms. The previous arbitrary extra-array allowances +are not treated as verified bounds. Supply a verified **per-GPU byte bound**: + +```toml +[workspace.gemm] +cunumeric = 268435456 +cudajl = 268435456 +cupynumeric = 268435456 +``` + +The numbers above demonstrate syntax only, not recommended bounds. Use a bound +verified for the maximum dimensions in the sweep and installed library versions. +Other keys are the registered names, e.g. `workspace.dmd_baseline` and +`workspace.tensor_contract4`. Each enabled native backend needs an entry; +missing entries fail preflight, including for pinned dimensions. Zero is valid +only when no extra workspace/packing is required. No calibration, OOM retry, +or automatic problem reduction is performed during execution. + +DMD counts the entire SVD on one GPU regardless of P. Its shared baseline is +limited by the largest requested GPU count's scaled problem. This is a memory +guard, not a distributed SVD implementation. Full factors/parent stores and +complex outputs are included in addition to the supplied workspace bound. + +GEMM and contractions conservatively count full operands on each GPU until +mapper replication bounds are verified. Poisson partitions batches and retains +the inverse Laplacian per GPU; Python FFT outputs are conservatively treated +as complex128. Native workspace remains separate. + +## Budgets and validation + +`mem_frac` applies to the smallest visible GPU capacity, capped by +`CUNUMERIC_BENCH_FBMEM_MB` when provided. That cap is also passed to Legate. +`CUDA_VISIBLE_DEVICES` numeric indices and GPU UUIDs are resolved explicitly; +unresolvable/MIG identities are rejected rather than guessed. Free memory must +cover the budget before starting. External GPU users can invalidate that check. + +Use `--dry-run` to see initialization, iteration, workspace and peak bytes for +each configuration. CPU tests verify dispatch and planner invariants; they do +not verify native allocator bounds. Before calling a configuration GPU-validated, +check its runtime allocation trace and execute the requested GPU sweep with the +installed native libraries. Record the verified workspace bounds in its config. + +```bash +julia --project=. test/runtests.jl +``` diff --git a/benchmark/Project.toml b/benchmark/Project.toml index 793787ae1..5b279307b 100644 --- a/benchmark/Project.toml +++ b/benchmark/Project.toml @@ -1,9 +1,10 @@ [deps] -BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" +AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" -CUDACore = "bd0ed864-bdfe-4181-a5ed-ce625a5fdea2" +CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" +ProgressMeter = "92933f4c-e287-5a05-a399-4b506db050ca" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76" TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" diff --git a/benchmark/README.md b/benchmark/README.md index 0d6bc8a74..95982b9b6 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -2,20 +2,61 @@ Benchmarks are declared in `benchmarks.toml`. `run.jl` parses it. +Each independent iteration completes before the next is submitted, including +warmups, for cuNumeric, CUDA.jl, and cuPyNumeric. Blocking synchronization is +included in timed iterations. Gray–Scott is the exception: its timesteps form +one trajectory, so all variants retain batch synchronization at timing boundaries. +Initialization remains outside timing. Earlier non-Gray–Scott results used batch +synchronization and should be rerun for comparison. Fences do not force GC. + ## Running +Run a complete selected sweep and plot it without editing other TOML blocks: + +```bash +julia --project=. run.jl --only=montecarlo +julia --project=. run.jl --only=grayscott +julia --project=. run.jl --only=grayscott --fusion=both +julia --project=. run.jl --only=grayscott --dry-run +``` + +`--only` accepts benchmark or plot-group names (comma-separated). `--fusion=on`, +`off`, or `both` overrides the selected blocks. `--config=path` selects another +configuration. The shipped configuration uses fusion enabled. Positional +single-run arguments remain supported and do not plot automatically. + +Automatic sizing shares one baseline per comparison group and dtype, accounting +for every selected backend, fusion setting and GPU count. Incompatible pinned +constraints fail preflight. Comparison backends run once even if only unfused +Julia configurations are selected. Read [memory accounting](MEMORY.md) before +running native-library benchmarks: verified workspace bounds are required where +the harness cannot infer them. Unexpected failures are reported, not retried. + +Each invocation writes `results///` plus a `manifest.toml` with +resolved dimensions, memory estimates, package versions and worker statuses. +Plots go to `plots///`; failed sweeps are marked incomplete. +Different sizes at the same GPU count cannot be silently merged into a plot. + ```bash julia --project=. run.jl # runs whatever benchmarks.toml configures ``` `run.jl` runs each (benchmark, backend) pair in its own process via `run_benchmark.sh`, so backends never share a GPU/runtime within a measurement. +Julia workers log correctness checking and show a trial progress meter with the +latest trial's mean time and GFLOP/s. The meter advances only after a trial +finishes, outside the timed loop; initialization and warmup can also take time +before the next update. cuNumeric always runs; extra comparison backends are toggled in `[Global]`: - `cuda = true` → also run under CUDA.jl (single-GPU configs only; CUDA.jl is - single-device). + single-device). Every kernel has a `CuArray` implementation. - `cupynumeric = true` → also run under cupynumeric (see below). +On a single GPU, the cuNumeric worker also compares a tiny problem against +CUDA.jl (`pass` / `fail` in the CSV). The timed CUDA.jl run still happens; +its correctness column is `skipped`. Multi-GPU and cupynumeric skip the check. + Individual `[[benchmark]]` blocks may override `cuda`, `n_warmup`, `n_iter`, and `n_trial`. Unspecified values inherit from `[Global]`. This is useful for enabling a single-GPU CUDA comparison only for compatible benchmarks or reducing @@ -26,6 +67,11 @@ the iteration count for expensive kernels. cupynumeric runs in a conda env whose major.minor matches this project's resolved `cupynumeric_jll`. Build it once: +The runner checks that conda and the requested environment are available before +starting any timed workers. If conda is not on the worker's PATH, set +`CUNUMERIC_BENCH_CONDA` to its executable path (or use `CONDA_EXE`). The installer +honors the same setting. + ```bash ./install_cupynumeric.sh # creates env cupynumeric-bench- ``` @@ -39,6 +85,8 @@ resolved `cupynumeric_jll`. Build it once: n_warmup = 5 n_iter = 1000 n_trial = 5 +auto_size = true +mem_frac = 0.5 # fraction of the smallest visible GPU's total RAM [[gemm]] # name registered under src/benchmarks/ T = "Float32" # element type @@ -60,6 +108,17 @@ two axes: - **`gpus`, `cpus`, `N`, `M` zip** into a single lockstep sweep — element `i` of each is paired together. +When `[Global] auto_size = true` and a block **omits** `N` (or sets `N = "auto"`), +the harness RAM-fits the 1-GPU problem from `total_space` on that benchmark type, +then maps to `P` GPUs. Pin an explicit `N` list to keep paper sizes. `mem_frac` +is the fraction of the *smallest* visible GPU's total RAM (`CUNUMERIC_BENCH_MEM_FRAC` +overrides it). DMD keeps `M` as the intensity knob; Poisson holds grid `N` across +the GPU sweep and scales batch `M`. +Peak estimates are now backend- and variant-aware. Monte Carlo's fused +iteration needs the samples and broadcast output, but the planner also accounts +for outputs that may await Julia GC across a trial. See [memory accounting](MEMORY.md) +for the supported bounds and native-library workspace configuration. + `fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, default `true`); it only affects cuNumeric, so comparison backends run once, not per variant. @@ -106,21 +165,32 @@ entry—TensorOperations.jl's cuTENSOR backend: `C[a,b,c,d] = X[a,i,c,j] * Y[i,b,j,d]`. This single contraction isolates the primitive high-rank backend path and counts `N^4(2N^2-1)` operations. -The Julia implementations use `@tensor` on both `NDArray` and `CuArray`; the +The Julia implementations use `@tensor opt=true` on both `NDArray` and `CuArray`; the latter activates TensorOperations' cuTENSOR extension and is recorded as `TensorOperations.jl / cuTENSOR`. The cuPyNumeric implementations use equivalent -`einsum` expressions. Final outputs are preallocated, and contraction-order -selection is performed before timing (at macro expansion in Julia and during -initialization in Python). Required intermediate allocation and release remain -part of each timed projection iteration. +`einsum` expressions with `einsum_path(optimize="optimal")`. Final outputs are +preallocated. Required intermediate allocation (projection3: two rank-3 temps; +contract4: an `N²×N²` GEMM workspace) is part of each timed iteration and of +Julia `total_space`. The orchestrator computes flop counts in Julia and passes +them to the Python worker so the formulas live in one place. ## Plotting +A full `run.jl` pass (no extra args) plots at the end. One-off CLI runs do +not. To plot existing CSVs: + ```bash -julia --project=benchmark benchmark/plot_results.jl +julia --project=. plot_results.jl ``` -The plotter reads the result files in the selected results directory and writes -one weak-scaling figure per benchmark, plus aggregate fusion and no-fusion -figures when those result groups are present. Outputs are grouped under a -subdirectory named for the shared benchmark prefix. +`[plot.groups]` in `benchmarks.toml` puts related kernels on one figure. +Gray-Scott's baseline and `@accelerate` forms share `grayscott`; DMD baseline +and accelerated share `dmd`. Every other `[[benchmark]]` table is its own +figure. Each figure overlays CUDA.jl (1 GPU) and cupynumeric from that +group's baseline CSV (`*_baseline`, else the only / first name). Accelerated +kernels have no CUDA.jl or Python CSVs; the overlay still comes from the +baseline. + +cuNumeric fused vs unfused uses the same color with solid vs dashed lines. +Outputs are `plots/_weak_scaling.png`. Optional flags: `--out=`, +`--suffix=`, `--config=`, or a results-directory path. diff --git a/benchmark/__plot_results.jl b/benchmark/__plot_results.jl deleted file mode 100644 index fd1ed646a..000000000 --- a/benchmark/__plot_results.jl +++ /dev/null @@ -1,301 +0,0 @@ -#!/usr/bin/env julia -# Weak-scaling plots (1/2/4/8 GPUs) for the benchmark result CSVs. -# One figure per benchmark, three panels: throughput, time/step, parallel efficiency. -# -# CSV schema (see src/core.jl save_result): -# implementation,gpus,N,M,trial,time_ms,throughput,correctness -# `throughput` is the benchmark's `total_flops` divided by elapsed time. For -# Gray-Scott that unit is Gpoint-updates/s; other benchmarks report GFLOP/s. -# -# save_result appends and does NOT encode the code-path variant, so a cuNumeric CSV -# holds alternating runs: baseline, @accelerate, baseline, @accelerate, ... -# (a run boundary = the GPU count resetting downward). -# cuPyNumeric / CUDA.jl have no accelerated path -> a single block. -# -# Encoding: color = implementation; line style = code path -# solid = baseline, dashed = @accelerate. - -using Plots -using Statistics - -gr() - -function parse_args(args) - results_dir = "results" - out_dir = nothing - single_cunumeric_run = :baseline - hide_baseline = false - output_suffix = "" - - for arg in args - if startswith(arg, "--single-cu=") - value = Symbol(lowercase(last(split(arg, "="; limit=2)))) - value in (:baseline, :accelerated) || - error("--single-cu must be baseline or accelerated") - single_cunumeric_run = value - elseif arg == "--hide-baseline" - hide_baseline = true - elseif startswith(arg, "--out=") - out_dir = last(split(arg, "="; limit=2)) - elseif startswith(arg, "--suffix=") - output_suffix = last(split(arg, "="; limit=2)) - else - results_dir = arg - end - end - isempty(output_suffix) && hide_baseline && (output_suffix = "_no_baseline") - - results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) - if out_dir === nothing - out_dir = if basename(normpath(results_dir)) == "results" - joinpath(@__DIR__, "plots") - else - joinpath(@__DIR__, "plots", basename(normpath(results_dir))) - end - else - out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) - end - return (; results_dir, out_dir, single_cunumeric_run, hide_baseline, output_suffix) -end - -const CONFIG = parse_args(ARGS) -const RESULTS_DIR = CONFIG.results_dir -const OUT_DIR = CONFIG.out_dir -const SINGLE_CUNUMERIC_RUN = CONFIG.single_cunumeric_run -const HIDE_BASELINE = CONFIG.hide_baseline -const OUTPUT_SUFFIX = CONFIG.output_suffix - -# filekey, family label, color, marker, can_contain_accelerated_blocks -const FAMILIES = [ - ("cunumeric", "cuNumeric.jl (fused)", "#2a78d6", :circle, true), - ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond, true), - ("cupynumeric", "cuPyNumeric", "#eb6834", :rect, false), - ("CUDA.jl", "CUDA.jl", "#008300", :utriangle, false), -] - -const INK = "#0b0b0b" -const MUTED = "#898781" -const GRIDCOL = "#e1e0d9" -const IDEALCOL = "#c3c2b7" - -struct Row - gpus::Int - time_ms::Float64 - thr::Float64 -end - -# Parse a CSV into runs, split wherever the GPU count resets to a smaller value. -function load_runs(path) - rows = Row[] - for line in eachline(path) - isempty(strip(line)) && continue - f = split(line, ',') - push!(rows, Row(parse(Int, f[2]), parse(Float64, f[6]), parse(Float64, f[7]))) - end - isempty(rows) && return Vector{Row}[] - runs = [Row[]] - for (i, r) in enumerate(rows) - i > 1 && r.gpus < rows[i - 1].gpus && push!(runs, Row[]) - push!(runs[end], r) - end - return runs -end - -# Aggregate trials per GPU count -> sorted vector of (gpus, t, tsd, h, hsd). -function aggregate(rows) - by = Dict{Int,Vector{Row}}() - for r in rows - push!(get!(by, r.gpus, Row[]), r) - end - sd(x) = length(x) > 1 ? std(x) : 0.0 - return [ - (gpus=g, t=mean(getfield.(by[g], :time_ms)), tsd=sd(getfield.(by[g], :time_ms)), - h=mean(getfield.(by[g], :thr)), hsd=sd(getfield.(by[g], :thr))) - for g in sort(collect(keys(by))) - ] -end - -# Build the series (color+marker+linestyle+agg) present for one benchmark. -function series_for(bench) - series = [] # NamedTuple(label,color,marker,ls,agg) - for (key, fam, color, marker, splits) in FAMILIES - path = joinpath(RESULTS_DIR, "$(bench)_$(key).csv") - isfile(path) || continue - runs = load_runs(path) - isempty(runs) && continue - if splits && bench == "grayscott" - if length(runs) == 1 - label = - SINGLE_CUNUMERIC_RUN === :accelerated ? "$fam · accelerated" : - "$fam · baseline" - ls = SINGLE_CUNUMERIC_RUN === :accelerated ? :dash : :solid - push!( - series, (label=label, color=color, marker=marker, ls=ls, - agg=aggregate(runs[1])) - ) - else - # Repeated harness runs append alternating baseline/accelerated blocks. - baseline_rows = reduce(vcat, runs[1:2:end]) - accelerated_rows = reduce(vcat, runs[2:2:end]) - push!( - series, - (label="$fam · baseline", color=color, marker=marker, - ls=:solid, agg=aggregate(baseline_rows)), - ) - push!(series, - (label="$fam · accelerated", color=color, marker=marker, - ls=:dash, agg=aggregate(accelerated_rows))) - end - else - push!( - series, - (label=fam, color=color, marker=marker, ls=:solid, - agg=aggregate(reduce(vcat, runs))), - ) - end - end - return series -end - -function throughput_label(bench) - return bench == "grayscott" ? - "Throughput (Gpoint-updates/s)" : "Throughput (GFLOP/s)" -end - -function addline!(p, s, y; kw...) - return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, - lw=2.2, ls=s.ls, marker=s.marker, ms=6, msc=s.color, markerstrokewidth=0.8, - label=s.label, kw...) -end - -# One legend key: a short line sample (+ optional marker) with a text label. -function swatch!(p, x, y, color, ls, marker, label) - plot!(p, [x, x + 0.032], [y, y]; color=color, lw=2.6, ls=ls, label="") - marker !== nothing && scatter!(p, [x + 0.016], [y]; color=color, marker=marker, - ms=6, msc=color, markerstrokewidth=0.8, label="") - return annotate!(p, x + 0.045, y, text(label, 9, INK, :left)) -end - -# Grouped legend: color/marker = implementation, line style = code path. -function build_legend(series) - pl = plot(; framestyle=:none, legend=false, xlims=(0, 1), ylims=(0, 1), - grid=false, ticks=false) - # implementations present, in FAMILIES order, matched by color - present = [ - (fam, color, marker) for (key, fam, color, marker, _) in FAMILIES - if any(s.color == color for s in series) - ] - annotate!(pl, 0.015, 0.74, text("Implementation", 10, INK, :left)) - xs = range(0.18, 0.80; length=max(length(present), 1)) - for ((fam, color, marker), x) in zip(present, xs) - swatch!(pl, x, 0.74, color, :solid, marker, fam) - end - annotate!(pl, 0.015, 0.26, text("Line style", 10, INK, :left)) - has_baseline = any(endswith(s.label, "· baseline") for s in series) - has_accelerated = any(endswith(s.label, "· accelerated") for s in series) - if has_baseline && has_accelerated - swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") - swatch!(pl, 0.40, 0.26, MUTED, :dash, nothing, "@accelerate") - swatch!(pl, 0.70, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - elseif has_accelerated - swatch!(pl, 0.18, 0.26, MUTED, :dash, nothing, "@accelerate") - swatch!(pl, 0.52, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - elseif has_baseline - swatch!(pl, 0.18, 0.26, MUTED, :solid, nothing, "baseline") - swatch!(pl, 0.48, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - else - swatch!(pl, 0.18, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") - end - return pl -end - -function positive_ylim(vals; pad=0.12) - isempty(vals) && return (0, 1) - hi = maximum(vals) - hi > 0 || return (0, 1) - return (0, hi * (1 + pad)) -end - -function main() - mkpath(OUT_DIR) - files = filter(f -> endswith(f, ".csv"), readdir(RESULTS_DIR)) - benches = unique( - String[ - m.captures[1] for f in files for (key, _, _, _, _) in FAMILIES - for m in (match(Regex("^(.*)_" * replace(key, "." => "\\.") * "\\.csv\$"), f),) - if m !== nothing - ], - ) - - for bench in benches - series = series_for(bench) - if HIDE_BASELINE - series = filter(s -> !endswith(s.label, "· baseline"), series) - end - isempty(series) && continue - - common = (xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", - framestyle=:box, grid=true, gridcolor=GRIDCOL, gridalpha=1.0, - foreground_color_text=INK, tickfontcolor=MUTED, legend=false, - xlims=(0.85, 9.4)) - - # Panel 1: throughput (higher better) - throughput = [x.h for s in series for x in s.agg] - p1 = plot(; ylabel=throughput_label(bench), title="Throughput", - ylims=positive_ylim(throughput), common...) - for s in series - addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) - end - - # Panel 2: time per step (lower better; ideal = flat) - p2 = plot(; ylabel="Time / step (ms)", title="Time per step", common...) - for s in series - addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) - end - - # Panel 3: parallel efficiency = thr(p)/(p*thr(1)); ideal = 1.0 - efficiencies = Float64[] - for s in series - i1 = findfirst(x -> x.gpus == 1, s.agg) - i1 === nothing && continue - base = s.agg[i1].h - append!(efficiencies, [x.h/(x.gpus*base) for x in s.agg]) - end - p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", - ylims=positive_ylim(vcat(efficiencies, [1.0])), common...) - hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.4, label="") - for s in series - i1 = findfirst(x -> x.gpus == 1, s.agg) - i1 === nothing && continue - base = s.agg[i1].h - addline!(p3, s, [x.h/(x.gpus*base) for x in s.agg]) - end - - # grouped legend panel: color/marker = implementation, style = code path - pl = build_legend(series) - has_baseline = any(endswith(s.label, "· baseline") for s in series) - has_accelerated = any(endswith(s.label, "· accelerated") for s in series) - style_title = if has_baseline && has_accelerated - "solid = baseline · dashed = @accelerate" - elseif has_accelerated - "dashed = @accelerate" - elseif has_baseline - "solid = baseline" - else - "implementation comparison" - end - - fig = plot(p1, p2, p3, pl; layout=@layout([grid(1, 3); leg{0.16h}]), - size=(1400, 600), dpi=200, - plot_title=titlecase(bench) * " — weak scaling ($style_title)", - plot_titlefontsize=12, left_margin=6Plots.mm, - bottom_margin=6Plots.mm, top_margin=4Plots.mm, - background_color="#fcfcfb") - - out = joinpath(OUT_DIR, "$(bench)_weak_scaling$(OUTPUT_SUFFIX).png") - savefig(fig, out) - println("wrote $out") - end -end - -main() diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml index 13eb5418c..24f8f3590 100644 --- a/benchmark/benchmarks.toml +++ b/benchmark/benchmarks.toml @@ -1,12 +1,30 @@ [Global] -n_warmup = 5 -n_iter = 1000 +n_warmup = 2 +n_iter = 250 n_trial = 5 cupynumeric = true # (needs install_cupynumeric.sh) -cuda = false # compare against CUDA.jl (single-GPU configs only) -# One CPU-reference check per config (not per timed iter). Written to CSV. +cuda = true # also run CUDA.jl (single-GPU configs only) +# When gpus == 1, cuNumeric compares a tiny problem against CUDA.jl. +# CUDA.jl is still timed; its CSV correctness column is skipped. check_correctness = true n_correctness_iter = 5 +# RAM-fit the 1-GPU problem, then scale with P. Pin N (and M) on a block +# to keep paper sizes. CUNUMERIC_BENCH_MEM_FRAC overrides mem_frac. +auto_size = true +mem_frac = 0.75 + +# Names in a list share one weak-scaling figure. Unlisted [[benchmark]] +# tables each get their own. CUDA.jl and cupynumeric overlay from the +# group's baseline member (`*_baseline`, else the only / first name). +[plot.groups] +grayscott = [ + "grayscott_baseline", + "grayscott_function_accelerated", + "grayscott_begin_accelerated", + "grayscott_let_accelerated", + "grayscott_expression_accelerated", +] +dmd = ["dmd_baseline", "dmd_accelerated"] #################################### # GEMM # @@ -14,14 +32,15 @@ n_correctness_iter = 5 # Work ~ 2*N^2*M = 2*N^3. # # N^3 / P --> constant. # # N = baseline * P^(1/3). # +# Paper sizes (A100 80GB, frac~0.5): +# N = [20000, 25200, 31752, 40000] #################################### [[gemm]] T = ["Float32"] gpus = [1, 2, 4, 8] -cpus = 16 -N = [20000, 25200, 31752, 40000] -M = [20000, 25200, 31752, 40000] +cpus = 8 +n_iter = 50 ################################# # Gray-Scott # @@ -29,80 +48,63 @@ M = [20000, 25200, 31752, 40000] # Work ~ N*M = N^2. # # N^2 / P --> constant. # # N = baseline * P^(1/2). # +# Paper: N = [24000, 33944, 48000, 67888] ################################# [[grayscott_baseline]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [24000, 33944, 48000, 67888] -M = [24000, 33944, 48000, 67888] +cpus = 8 +fusion = true [[grayscott_function_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [24000, 33944, 48000, 67888] -M = [24000, 33944, 48000, 67888] +cpus = 8 +fusion = true [[grayscott_begin_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [24000, 33944, 48000, 67888] -M = [24000, 33944, 48000, 67888] +cpus = 8 +fusion = true [[grayscott_let_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [24000, 33944, 48000, 67888] -M = [24000, 33944, 48000, 67888] +cpus = 8 +fusion = true [[grayscott_expression_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [24000, 33944, 48000, 67888] -M = [24000, 33944, 48000, 67888] +cpus = 8 +fusion = true ################################# # DMD # # Snapshot matrix N × M # # (space × time), r=min(20,M-1).# -# FLOPs (m=N, n=M-1): # -# 2mn² + 11n³ thin SVD # -# 2mnr + mr B = X2 V Σ⁻¹ # -# 2mr² Ã = U' B # -# 25r³ eigen(Ã) # -# 2mr² Φ = B W # -# M fixed ⇒ n,r fixed ⇒ W ~ N. # -# Weak scaling: N / P constant. # -# N = 50000 * P. # -# M is the intensity knob: AI # -# ~ n²/M ~ M, independent of P. # +# M is the intensity knob (fixed). +# Weak scaling: N ∝ P if N ≫ M. # +# Paper: N = [50000, 100000, 200000, 400000] ################################# [[dmd_baseline]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [50000, 100000, 200000, 400000] +cpus = 8 +fusion = true M = 512 +n_iter = 50 [[dmd_accelerated]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [50000, 100000, 200000, 400000] +cpus = 8 +fusion = true M = 512 +n_iter = 50 ################################# # Spectral Poisson (FFT) # @@ -110,58 +112,63 @@ M = 512 # solves (∇²u = f, periodic). # # Transform is the last 2 axes; # # batch axis M may split GPUs. # -# Work ~ M N² log N. # -# Weak scaling: M / P constant. # -# M = 8 * P, N = 1024. # +# N is held across the P sweep # +# (changing N changes FFT size).# +# Omit N to RAM-fit one grid, # +# then M = P. Pin e.g. N = 1024 # +# and M = "auto" to fit batch. # +# Paper: N = 1024, M = 8*P # ################################# [[poisson_fft]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -N = 1024 -M = [8, 16, 32, 64] +cpus = 8 +n_iter = 50 ################################# # Monte-Carlo Integration # # Work ~ N. Scale N linearly # +# Paper: N = 1e6 * P # ################################# [[montecarlo]] T = "Float32" gpus = [1, 2, 4, 8] -cpus = 16 -N = [1_000_000, 2_000_000, 4_000_000, 8_000_000] +cpus = 8 +n_iter = 50 ######################################### # Three-mode tensor projection # # D[n,m,l] = A[i,j,k] B[n,i] B[m,j] # # B[l,k] # # Optimal pairwise work ~ 6*N^4. # +# Weak scaling: N ∝ P^{1/4}. # +# Peak RAM includes two N³ temps. # +# CUDA.jl / cuTENSOR overlays at 1 GPU. # ######################################### [[tensor_projection3]] T = "Float32" -gpus = 1 -cpus = 16 -N = [32, 64, 128] -cuda = true -n_warmup = 3 +gpus = [1, 2, 4, 8] +cpus = 8 +n_warmup = 2 n_iter = 20 -n_trial = 5 ######################################### # Rank-4 two-index tensor contraction # # C[a,b,c,d] = X[a,i,c,j] Y[i,b,j,d] # # Work ~ 2*N^6. # +# Weak scaling: N ∝ P^{1/6}. # +# Peak RAM includes GEMM workspace. # +# Keep n_iter small: RAM-fit N makes # +# each step much heavier than GEMM. # ######################################### [[tensor_contract4]] T = "Float32" -gpus = 1 -cpus = 16 -N = [12, 16, 24] -cuda = true -n_warmup = 3 -n_iter = 20 -n_trial = 5 +gpus = [1, 2, 4, 8] +cpus = 8 +n_warmup = 2 +n_iter = 5 + diff --git a/benchmark/cpp_matmul/CMakeLists.txt b/benchmark/cpp_matmul/CMakeLists.txt deleted file mode 100644 index d547f62dd..000000000 --- a/benchmark/cpp_matmul/CMakeLists.txt +++ /dev/null @@ -1,17 +0,0 @@ - -cmake_minimum_required(VERSION 3.22.1 FATAL_ERROR) - -project(cuNumericWrapper VERSION 0.01 LANGUAGES C CXX) - -# Specify C++ standard -set(CMAKE_CXX_STANDARD_REQUIRED True) - -if (NOT CMAKE_CXX_STANDARD) - set(CMAKE_CXX_STANDARD 20) -endif() - -find_package(cupynumeric REQUIRED) - -add_executable(matmulfp32_test main.cpp) -target_link_libraries(matmulfp32_test PRIVATE cupynumeric::cupynumeric) -install(TARGETS matmulfp32_test DESTINATION "${CMAKE_CURRENT_BINARY_DIR}/cmake-install") diff --git a/benchmark/cpp_matmul/build.sh b/benchmark/cpp_matmul/build.sh deleted file mode 100755 index d7463c345..000000000 --- a/benchmark/cpp_matmul/build.sh +++ /dev/null @@ -1,7 +0,0 @@ -legate_root=`python -c 'import legate.install_info as i; from pathlib import Path; print(Path(i.libpath).parent.resolve())'` -echo "Using Legate at $legate_root" -cupynumeric_root=`python -c 'import cupynumeric.install_info as i; from pathlib import Path; print(Path(i.libpath).parent.resolve())'` -echo "Using cuPyNumeric at $cupynumeric_root" - -cmake -S . -B build -D legate_ROOT="$legate_root" -D cupynumeric_ROOT="$cupynumeric_root" -D CMAKE_BUILD_TYPE=Debug -cmake --build build --parallel 8 --verbose diff --git a/benchmark/cpp_matmul/main.cpp b/benchmark/cpp_matmul/main.cpp deleted file mode 100644 index a549e02ec..000000000 --- a/benchmark/cpp_matmul/main.cpp +++ /dev/null @@ -1,50 +0,0 @@ -/* Copyright 2025 Northwestern University, - * Carnegie Mellon University University - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - * - * Author(s): David Krasowska - * Ethan Meitz - */ - -#include - -#include "cupynumeric.h" -#include "legate.h" -// #include - -// using cupynumeric::slice; - -void matmul_fp32(size_t N) { - std::vector dims = {N, N}; - auto A = cupynumeric::random(dims).as_type(legate::float32()); - auto B = cupynumeric::random(dims).as_type(legate::float32()); - std::optional T = legate::float32(); - auto C = cupynumeric::zeros(dims, T); - - C.dot(A, B); - return; - // std::cout << C[{slice(0,0), slice(0,0)}] << std::endl; -} - -int main(int argc, char** argv) { - auto result = legate::start(argc, argv); - // assert(result == 0); - - cupynumeric::initialize(argc, argv); - - const size_t N = 10000; - matmul_fp32(N); - - return legate::finish(); -} diff --git a/benchmark/diagnose_montecarlo.jl b/benchmark/diagnose_montecarlo.jl new file mode 100644 index 000000000..8a3868665 --- /dev/null +++ b/benchmark/diagnose_montecarlo.jl @@ -0,0 +1,34 @@ +# From benchmark/: bash run_benchmark.sh diagnose_montecarlo.jl --gpus 8 --cpus 8 4266645824 +# Diagnostic only: fences deliberately change execution timing. +using cuNumeric + +length(ARGS) == 2 || error("Use run_benchmark.sh with --gpus

--cpus ") +const N = parse(Int, ARGS[2]) +N > 0 || error("N must be positive") +println("Monte Carlo diagnostic: GPUs=$(ARGS[1]), N=$N, fusion=$(cuNumeric.FUSE_BROADCAST_EXPRS)") + +function stage(f, label) + println("START: $label") + flush(stdout) + result = f() + cuNumeric.issue_execution_fence(; block=true) + println("PASS: $label") + flush(stdout) + return result +end + +x = stage("Float32 random generation") do + cuNumeric.rand(Float32, N) +end +x = stage("scale samples") do + 10.0f0 .* x +end +y = stage("fused square / negate / exponential") do + exp.(.-(x .^ 2)) +end +s = stage("sum reduction") do + sum(y) +end +stage("scale reduction result") do + (10.0f0 / N) * s +end diff --git a/benchmark/install_cupynumeric.sh b/benchmark/install_cupynumeric.sh index 6140cf64a..eb7d0bc9a 100755 --- a/benchmark/install_cupynumeric.sh +++ b/benchmark/install_cupynumeric.sh @@ -8,6 +8,11 @@ # ./install_cupynumeric.sh --name myenv # override the env name # ./install_cupynumeric.sh --into existing # install into an existing env instead of creating one set -euo pipefail +CONDA="${CUNUMERIC_BENCH_CONDA:-${CONDA_EXE:-conda}}" +command -v "$CONDA" >/dev/null 2>&1 || { + echo "Error: conda not found. Add it to PATH or set CUNUMERIC_BENCH_CONDA to its executable path." + exit 1 +} SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -57,20 +62,20 @@ NUMPY_SPEC="numpy<2.3" if [[ -n "$INTO_ENV" ]]; then echo "Installing $SPEC into existing env '$INTO_ENV'..." - conda install -y -n "$INTO_ENV" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" + "$CONDA" install -y -n "$INTO_ENV" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" echo "Done. Activate with: conda activate $INTO_ENV" exit 0 fi [[ -z "$ENV_NAME" ]] && ENV_NAME="cupynumeric-bench-$VER" -if conda env list | awk '{print $1}' | grep -qx "$ENV_NAME"; then +if "$CONDA" env list | awk '{print $1}' | grep -qx "$ENV_NAME"; then echo "Env '$ENV_NAME' already exists with $SPEC; nothing to do." echo "Activate with: conda activate $ENV_NAME" exit 0 fi echo "Creating env '$ENV_NAME' with $SPEC..." -conda create -y -n "$ENV_NAME" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" +"$CONDA" create -y -n "$ENV_NAME" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" echo "Done. Activate with: conda activate $ENV_NAME" diff --git a/benchmark/mwe_fusion_indexing.jl b/benchmark/mwe_fusion_indexing.jl new file mode 100644 index 000000000..db8ef7542 --- /dev/null +++ b/benchmark/mwe_fusion_indexing.jl @@ -0,0 +1,78 @@ +# From benchmark/, run each case in a fresh process: +# bash run_benchmark.sh mwe_fusion_indexing.jl --gpus 8 --cpus 8 2454267008 fused +# bash run_benchmark.sh mwe_fusion_indexing.jl --gpus 8 --cpus 8 2454267072 fused +# bash run_benchmark.sh mwe_fusion_indexing.jl --gpus 8 --cpus 8 2454267072 native +# These straddle 2^31 at the last partition's START (7*N/8), not global N. +# With N near 2^31 and eight equal partitions, all starts still fit Int32. +# Also test the original failing N=4266645824, and repeat with --gpus 1. +# No RNG, sum, benchmark harness, or preference changes. Two persistent arrays. +using cuNumeric + +function stage(f, label) + println("START: $label") + flush(stdout) + result = f() + cuNumeric.issue_execution_fence(; block=true) + println("PASS: $label") + flush(stdout) + return result +end + +function landmarks(n, p) + points = Int[1,n] + for boundary in (Int(2)^31, (cld(n,p)*k for k in 1:(p-1))...) + append!(points,filter(i->1<=i<=n,(boundary-1,boundary,boundary+1))) + end + return sort!(unique!(points)) +end + +function main(args=ARGS) + length(args)==3 || error("Use run_benchmark.sh with --gpus P --cpus C N fused|native") + p,n = parse.(Int,args[1:2]) + mode = args[3] + p>0 && n>0 || error("P and N must be positive") + mode in ("fused","native") || error("Mode must be fused or native") + println("P=$p N=$n mode=$mode; persistent data=$(2big(n)*sizeof(Float32)) bytes globally") + println("Expected last partition start (zero-based, equal tiles): $((p-1)*cld(n,p)); Int32 limit=$(typemax(Int32))") + x,y = stage("allocate and fill input/output") do + cuNumeric.ones(Float32,n),cuNumeric.zeros(Float32,n) + end + points = landmarks(n,p) + values = Dict(i=>Float32(0.125+0.025*j) for (j,i) in enumerate(points)) + stage("write boundary markers") do + for i in points + x[i:i] .= values[i] + end + end + stage("verify input markers (before fusion)") do + for i in points + @assert only(Array(x[i:i])) == values[i] "input marker mismatch at $i" + end + end + stage("$mode square / negate / exp") do + # Exact broadcast tree for exp.(.-(x .^ 2)), including literal_pow. + square = Base.Broadcast.broadcasted(Base.literal_pow,Ref(^),x,Ref(Val(2))) + tree = Base.Broadcast.instantiate(Base.Broadcast.broadcasted(exp,Base.Broadcast.broadcasted(-,square))) + if mode == "fused" + @assert cuNumeric.can_fuse_linear_broadcast(y,tree) + cuNumeric.fuse_broadcast_tree!(y,tree) + else + # Force native tasks regardless of the user's fusion preference. + tmp = cuNumeric.unravel_broadcast_tree(tree) + copyto!(y,tmp) + cuNumeric.destroy!(tmp) + end + end + stage("verify output markers (no reduction)") do + for i in points + actual = only(Array(y[i:i])) + expected = exp(-values[i]^2) + @assert isapprox(actual,expected;rtol=2f-5,atol=2f-6) "output mismatch at $i: got $actual expected $expected" + end + end + println("PASS: all $(length(points)) markers; P=$p N=$n mode=$mode") +end + +if abspath(PROGRAM_FILE)==abspath(@__FILE__) + main() +end diff --git a/benchmark/plot_results.jl b/benchmark/plot_results.jl index b24098003..d27ae866e 100644 --- a/benchmark/plot_results.jl +++ b/benchmark/plot_results.jl @@ -1,27 +1,47 @@ #!/usr/bin/env julia # Generate weak-scaling plots from benchmark CSVs. -# Each figure shows throughput, time per step, and parallel efficiency. +# Each figure is one `[plot.groups]` entry (or a singleton [[benchmark]]). using Plots using Statistics +include(joinpath(@__DIR__, "src", "parse_benchmarks.jl")) + +# GPU nodes often have no display; 100 = PNG. Override with GKSwstype if needed. +get!(ENV, "GKSwstype", "100") gr() +default(; + fontfamily="sans-serif", + fg=:black, + fg_text=:black, + fg_axis=:black, + fg_border=:black, + grid=false, + gridalpha=0, + gridlinewidth=0, + minorgrid=false, + legend=false, +) function parse_args(args) results_dir = "results" out_dir = nothing output_suffix = "" + config = joinpath(@__DIR__, "benchmarks.toml") for arg in args if startswith(arg, "--out=") out_dir = last(split(arg, "="; limit=2)) elseif startswith(arg, "--suffix=") output_suffix = last(split(arg, "="; limit=2)) + elseif startswith(arg, "--config=") + config = last(split(arg, "="; limit=2)) else results_dir = arg end end results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) + config = isabspath(config) ? config : joinpath(@__DIR__, config) if out_dir === nothing out_dir = if basename(normpath(results_dir)) == "results" joinpath(@__DIR__, "plots") @@ -31,66 +51,44 @@ function parse_args(args) else out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) end - return (; results_dir, out_dir, output_suffix) + return (; results_dir, out_dir, output_suffix, config) end -const CONFIG = parse_args(ARGS) -const RESULTS_DIR = CONFIG.results_dir -const OUT_DIR = CONFIG.out_dir -const OUTPUT_SUFFIX = CONFIG.output_suffix - -# file key, display label, color, marker -const FAMILIES = [ - ("cunumeric", "cuNumeric.jl (fused)", "#2a78d6", :circle), - ("cunumeric_nofusion", "cuNumeric.jl (unfused)", "#4a3aa7", :diamond), - ("cupynumeric", "cuPyNumeric", "#eb6834", :rect), - ("CUDA.jl", "CUDA.jl", "#008300", :utriangle), - ("tensoroperations_cuda", "TensorOperations.jl / cuTENSOR", "#159a9c", :star5), +# Fixed across every figure so GEMM / Gray-Scott / DMD read as one set. +const COLOR_CUNUMERIC = "#2a78d6" +const COLOR_CUPYNUMERIC = "#eb6834" +const COLOR_CUDA = "#1a7f37" +const COLOR_CUTENSOR = "#0d7377" +const MARKER_CUNUMERIC = :circle +const MARKER_CUPYNUMERIC = :rect +const MARKER_CUDA = :utriangle +const MARKER_CUTENSOR = :star5 + +# Extra cuNumeric variants (Gray-Scott forms, DMD accelerated). Avoid the +# reference orange/green so CUDA.jl and cuPyNumeric stay unique. +const VARIANT_COLORS = [COLOR_CUNUMERIC, "#7b2d8e", "#b8860b", "#3d5a80", "#a23b72", "#2f6f4e"] +const VARIANT_MARKERS = [:circle, :diamond, :hexagon, :dtriangle, :star4, :pentagon] + +const REF_FAMILIES = [ + ("cupynumeric", "cuPyNumeric", COLOR_CUPYNUMERIC, MARKER_CUPYNUMERIC), + ("CUDA.jl", "CUDA.jl", COLOR_CUDA, MARKER_CUDA), + ("tensoroperations_cuda", "TensorOperations.jl / cuTENSOR", COLOR_CUTENSOR, MARKER_CUTENSOR), ] -const INK = "#0b0b0b" -const MUTED = "#898781" -const GRIDCOL = "#e1e0d9" -const IDEALCOL = "#c3c2b7" -const GROUP_COLORS = ["#2a78d6", "#eb6834", "#008300", "#8b3fb0", "#c47f00", "#159a9c"] -const GROUP_MARKERS = [:circle, :diamond, :utriangle, :rect, :star5, :hexagon] - -struct Row - gpus::Int - time_ms::Float64 - thr::Float64 -end +const INK = "#111111" +const IDEALCOL = "#6e6e6e" -# Parse a CSV into runs, split wherever the GPU count resets to a smaller value. -function load_runs(path) - rows = Row[] - for line in eachline(path) - isempty(strip(line)) && continue - f = split(line, ',') - push!(rows, Row(parse(Int, f[2]), parse(Float64, f[6]), parse(Float64, f[7]))) - end - isempty(rows) && return Vector{Row}[] - runs = [Row[]] - for (i, r) in enumerate(rows) - i > 1 && r.gpus < rows[i - 1].gpus && push!(runs, Row[]) - push!(runs[end], r) - end - return runs -end +const GROUP_TITLES = Dict( + "grayscott" => "Gray-Scott", + "dmd" => "DMD", + "gemm" => "GEMM", + "poisson_fft" => "Poisson FFT", + "montecarlo" => "Monte Carlo", + "tensor_projection3" => "Tensor projection (3-mode)", + "tensor_contract4" => "Tensor contraction (rank-4)", +) -# Aggregate trials per GPU count -> sorted vector of (gpus, t, tsd, h, hsd). -function aggregate(rows) - by = Dict{Int,Vector{Row}}() - for r in rows - push!(get!(by, r.gpus, Row[]), r) - end - sd(x) = length(x) > 1 ? std(x) : 0.0 - return [ - (gpus=g, t=mean(getfield.(by[g], :time_ms)), tsd=sd(getfield.(by[g], :time_ms)), - h=mean(getfield.(by[g], :thr)), hsd=sd(getfield.(by[g], :thr))) - for g in sort(collect(keys(by))) - ] -end +include(joinpath(@__DIR__, "src", "result_rows.jl")) _all_rows(runs) = reduce(vcat, runs; init=Row[]) @@ -100,186 +98,205 @@ function make_series(label, color, marker, ls, runs) agg=aggregate(_all_rows(runs))) end -# Build the available implementation series for one benchmark. -function load_series(bench, specs; filename=(b, k) -> "$(b)_$(k).csv") - series = [] - for (key, label, color, marker) in specs - path = joinpath(RESULTS_DIR, filename(bench, key)) - isfile(path) || continue - runs = load_runs(path) - isempty(runs) && continue - - push!(series, make_series(label, color, marker, :solid, runs)) - end - return filter(!isnothing, series) +function load_csv_series(results_dir, bench, key, label, color, marker, ls) + path = joinpath(results_dir, "$(bench)_$(key).csv") + isfile(path) || return nothing + runs = load_runs(path) + isempty(runs) && return nothing + return make_series(label, color, marker, ls, runs) end -series_for(bench) = load_series(bench, FAMILIES) +function group_title(group) + return get(GROUP_TITLES, group, titlecase(replace(group, '_' => ' '))) +end -function common_prefix(names) - words = split.(names, '_') - n = minimum(length, words) - i = 0 - while i < n && all(w -> w[i + 1] == words[1][i + 1], words) - i += 1 - end - return i == 0 ? String[] : words[1][1:i] +function variant_label(group, member) + member == group && return "cuNumeric.jl" + prefix = group * "_" + stem = startswith(member, prefix) ? member[(length(prefix) + 1):end] : member + return replace(stem, '_' => ' ') end -function family_stem(name, prefix) - words = split(name, '_') - length(words) > length(prefix) || return name - return join(words[(length(prefix) + 1):end], "_") +function cunumeric_series_label(group, member, fused) + base = variant_label(group, member) + return fused ? base : "$(base) (unfused)" end -function grouped_series(benches, key) +function overlay_refs(results_dir, members) series = [] - prefix = common_prefix(benches) - for (i, bench) in enumerate(benches) - path = joinpath(RESULTS_DIR, "$(bench)_$(key).csv") - isfile(path) || continue - runs = load_runs(path) - isempty(runs) && continue - color = GROUP_COLORS[mod1(i, length(GROUP_COLORS))] - marker = GROUP_MARKERS[mod1(i, length(GROUP_MARKERS))] - words = split(bench, '_') - length(words) > length(prefix) && (words = words[(length(prefix) + 1):end]) - label = titlecase(join(words, ' ')) - push!(series, make_series(label, color, marker, :solid, runs)) + seen = Set{String}() + order = unique!(vcat([plot_baseline(members)], members)) + for (key, label, color, marker) in REF_FAMILIES + key in seen && continue + for member in order + s = load_csv_series(results_dir, member, key, label, color, marker, :solid) + s === nothing && continue + push!(series, s) + push!(seen, key) + break + end end - return filter(!isnothing, series) + return series end -function throughput_label() - return "Throughput" +function group_series(results_dir, group, members) + series = [] + n_members = length(members) + for (i, member) in enumerate(members) + color, marker = if n_members == 1 + COLOR_CUNUMERIC, MARKER_CUNUMERIC + else + VARIANT_COLORS[mod1(i, length(VARIANT_COLORS))], + VARIANT_MARKERS[mod1(i, length(VARIANT_MARKERS))] + end + for (key, fused, ls) in ( + ("cunumeric", true, :solid), + ("cunumeric_nofusion", false, :dash), + ) + label = cunumeric_series_label(group, member, fused) + s = load_csv_series(results_dir, member, key, label, color, marker, ls) + s === nothing || push!(series, s) + end + end + append!(series, overlay_refs(results_dir, members)) + return filter(!isnothing, series) end function addline!(p, s, y; kw...) return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, - lw=2.2, ls=s.ls, marker=s.marker, ms=6, msc=s.color, markerstrokewidth=0.8, + lw=2.6, ls=s.ls, marker=s.marker, ms=7, msc=s.color, markerstrokewidth=0.7, label=s.label, kw...) end -# One legend key: a short line sample (+ optional marker) with a text label. -function swatch!(p, x, y, color, ls, marker, label) - plot!(p, [x, x + 0.032], [y, y]; color=color, lw=2.6, ls=ls, label="") - marker !== nothing && scatter!(p, [x + 0.016], [y]; color=color, marker=marker, - ms=6, msc=color, markerstrokewidth=0.8, label="") - return annotate!(p, x + 0.045, y, text(label, 9, INK, :left)) -end - -# Build a legend from the series currently being plotted. function build_legend(series) - pl = plot(; framestyle=:none, legend=false, xlims=(0, 1), ylims=(0, 1), - grid=false, ticks=false) - present = [(s.label, s.color, s.marker) for s in series] - annotate!(pl, 0.015, 0.74, text("Series", 10, INK, :left)) - xs = length(present) <= 1 ? (0.18,) : range(0.18, 0.80; length=length(present)) - for ((fam, color, marker), x) in zip(present, xs) - swatch!(pl, x, 0.74, color, :solid, marker, fam) + n = length(series) + cols = min(max(n, 1), 4) + rows = cld(n, cols) + slot = min(0.32, 0.92 / cols) + x0 = (1 - cols * slot) / 2 + y0 = 0.50 + 0.18 * (rows - 1) / 2 + + pl = plot(; + framestyle=:none, grid=false, ticks=false, legend=false, + xlims=(0, 1), ylims=(0, 1), widen=false, + left_margin=0Plots.mm, right_margin=0Plots.mm, + top_margin=0Plots.mm, bottom_margin=0Plots.mm, + background_color=:transparent, + ) + # Pin the coordinate system so later scatter/annotate cannot rescale it. + scatter!(pl, [0.0, 1.0], [0.0, 1.0]; ms=0, msw=0, mc=:white, label="") + + for (i, s) in enumerate(series) + r, c = divrem(i - 1, cols) + x = x0 + c * slot + y = y0 - r * 0.36 + plot!(pl, [x, x + 0.028], [y, y]; color=s.color, lw=2.8, ls=s.ls, label="") + scatter!(pl, [x + 0.014], [y]; color=s.color, marker=s.marker, + ms=7, msc=s.color, markerstrokewidth=0.6, label="") + annotate!(pl, x + 0.036, y, text(s.label, 12, :black, :left)) end - swatch!(pl, 0.18, 0.26, IDEALCOL, :dashdot, nothing, "efficiency = 1") + plot!(pl; xlims=(0, 1), ylims=(0, 1), widen=false) return pl end -function positive_ylim(vals; pad=0.12) - isempty(vals) && return (0, 1) - hi = maximum(vals) +function series_ymax(series, yfield, efield) + m = 0.0 + for s in series, x in s.agg + m = max(m, getfield(x, yfield) + getfield(x, efield)) + end + return m +end + +function positive_ylim(hi; pad=0.18) hi > 0 || return (0, 1) return (0, hi * (1 + pad)) end -function weak_scaling_figure(bench, series, legend_panel; plot_title) - common = (xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", - framestyle=:box, grid=true, gridcolor=GRIDCOL, gridalpha=1.0, - foreground_color_text=INK, tickfontcolor=MUTED, legend=false, - xlims=(0.85, 9.4)) +function weak_scaling_figure(series; plot_title) + common = ( + xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", + framestyle=:box, grid=false, gridalpha=0, gridlinewidth=0, minorgrid=false, + foreground_color_grid=:white, + foreground_color_axis=:black, foreground_color_border=:black, + foreground_color_text=:black, foreground_color_guide=:black, + tickfontcolor=:black, guidefontcolor=:black, titlefontcolor=:black, + tickfontsize=14, guidefontsize=16, titlefontsize=16, + xtickfontsize=14, ytickfontsize=14, + xguidefontsize=16, yguidefontsize=16, + legend=false, xlims=(0.85, 9.4), widen=false, + left_margin=10Plots.mm, right_margin=6Plots.mm, + top_margin=5Plots.mm, bottom_margin=10Plots.mm, + ) - # Panel 1: throughput (higher better) - throughput = [x.h for s in series for x in s.agg] - p1 = plot(; ylabel=throughput_label(), title="Throughput", - ylims=positive_ylim(throughput), common...) + p1 = plot(; ylabel="Throughput", title="Throughput", + ylims=positive_ylim(series_ymax(series, :h, :hsd); pad=0.28), common...) for s in series addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) end - # Panel 2: time per step (lower better; ideal = flat) - p2 = plot(; ylabel="Time / step (ms)", title="Time per step", common...) + p2 = plot(; ylabel="Time / step (ms)", title="Time per step", + ylims=positive_ylim(series_ymax(series, :t, :tsd); pad=0.28), + common..., left_margin=28Plots.mm, yguidefontsize=15) for s in series addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) end - # Panel 3: parallel efficiency = thr(p)/(p*thr(1)); ideal = 1.0 efficiencies = Float64[] for s in series i1 = findfirst(x -> x.gpus == 1, s.agg) i1 === nothing && continue base = s.agg[i1].h - append!(efficiencies, [x.h/(x.gpus*base) for x in s.agg]) + append!(efficiencies, [x.h / (x.gpus * base) for x in s.agg]) end p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", - ylims=positive_ylim(vcat(efficiencies, [1.0])), common...) - hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.4, label="") + ylims=positive_ylim(max(1.0, isempty(efficiencies) ? 0.0 : maximum(efficiencies)); pad=0.12), + common..., left_margin=16Plots.mm) + hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.6, label="") for s in series i1 = findfirst(x -> x.gpus == 1, s.agg) i1 === nothing && continue base = s.agg[i1].h - addline!(p3, s, [x.h/(x.gpus*base) for x in s.agg]) + addline!(p3, s, [x.h / (x.gpus * base) for x in s.agg]) end + nrows = cld(length(series), 4) + layout = if nrows > 1 + @layout([grid(1, 3); leg{0.16h}]) + else + @layout([grid(1, 3); leg{0.14h}]) + end return plot( - p1, p2, p3, legend_panel; - layout=@layout([grid(1, 3); leg{0.16h}]), - size=(1400, 600), dpi=200, plot_title, - plot_titlefontsize=12, left_margin=6Plots.mm, - bottom_margin=6Plots.mm, top_margin=4Plots.mm, - background_color="#fcfcfb", + p1, p2, p3, build_legend(series); + layout, + size=(1760, nrows > 1 ? 680 : 620), dpi=220, plot_title, + plot_titlefontsize=20, plot_titlefontcolor=:black, + background_color=:white, ) end -function main() - mkpath(OUT_DIR) - files = filter(f -> endswith(f, ".csv"), readdir(RESULTS_DIR)) - benches = unique( - String[ - m.captures[1] for f in files for (key, _, _, _) in FAMILIES - for m in (match(Regex("^(.*)_" * replace(key, "." => "\\.") * "\\.csv\$"), f),) - if m !== nothing - ], - ) - family = common_prefix(benches) - benchmark_out = joinpath(OUT_DIR, isempty(family) ? "benchmarks" : join(family, "_")) - mkpath(benchmark_out) - - for bench in benches - series = series_for(bench) - isempty(series) && continue - - fig = weak_scaling_figure( - bench, series, build_legend(series); - plot_title=titlecase(bench) * " — weak scaling", - ) - - stem = family_stem(bench, family) - out = joinpath(benchmark_out, "$(stem)_weak_scaling$(OUTPUT_SUFFIX).png") - savefig(fig, out) - println("wrote $out") +function main(args=ARGS) + cfg = parse_args(args) + if !isdir(cfg.results_dir) + println("no results directory at $(cfg.results_dir)") + return nothing end + isfile(cfg.config) || error("plot config not found: $(cfg.config)") - # Add aggregate views that compare all benchmark variants for each mode. - for (key, title, stem) in (("cunumeric", "Fusion enabled", "fusion"), - ("cunumeric_nofusion", "Fusion disabled", "no_fusion")) - series = grouped_series(benches, key) + mkpath(cfg.out_dir) + for (group, members) in parse_plot_groups(cfg.config) + series = group_series(cfg.results_dir, group, members) isempty(series) && continue + validate_series_sizes(series) fig = weak_scaling_figure( - stem, series, build_legend(series); - plot_title=title * " — weak scaling", + series; plot_title=group_title(group) * " — weak scaling" ) - out = joinpath(benchmark_out, "$(stem)_weak_scaling$(OUTPUT_SUFFIX).png") + out = joinpath(cfg.out_dir, "$(group)_weak_scaling$(cfg.output_suffix).png") savefig(fig, out) println("wrote $out") end return nothing end -main() +if abspath(PROGRAM_FILE) == abspath(@__FILE__) + main() +end diff --git a/benchmark/run.jl b/benchmark/run.jl index 4debdbdcf..dc6beaff1 100644 --- a/benchmark/run.jl +++ b/benchmark/run.jl @@ -1,44 +1,14 @@ -# run.jl: orchestrator. Builds one run_benchmark.sh command per benchmark and -# dispatches it; the script sets LEGATE_CONFIG (from --gpus/--cpus) before -# launching the worker (single.jl) that actually runs the benchmark. -# no args -> one command per benchmarks.toml entry -# with args -> one command from [fusion] - -# Orchestrator stays off the GPU: it only needs GlobalSettings + parse_config, -# both cuNumeric-free. The worker (single.jl) loads cuNumeric and the kernels. - +# Orchestrator remains off the GPU; each timed backend uses its own process. using Pkg -include("src/core.jl") -include("src/parse_benchmarks.jl") - -const RUNNER = joinpath(@__DIR__, "run_benchmark.sh") -const WORKER = joinpath(@__DIR__, "src/single.jl") -const PY_WORKER = joinpath(@__DIR__, "src_py/single.py") - -# quiet by default (banner + results only); -v/--verbose shows the plumbing -const VERBOSE_FLAGS = ("-v", "--verbose") -const VERBOSE = any(in(ARGS), VERBOSE_FLAGS) -const POSARGS = filter(a -> a ∉ VERBOSE_FLAGS, ARGS) - -banner(msg) = println("\n", "="^128, "\n", msg, "\n", "="^128) - -# `_accelerated` is a cuNumeric-only code-path variant (`@accelerate`). -cunumeric_only(name) = endswith(name, "_accelerated") - -const LAST_FUSION_TOGGLE = Ref{Union{Nothing,Bool}}(nothing) - -# dev CNPreferences && cuNumeric function ensure_project_ready() Pkg.develop([ Pkg.PackageSpec(; path=joinpath(@__DIR__, "..", "lib", "CNPreferences")), Pkg.PackageSpec(; path=joinpath(@__DIR__, "..")), ]) - return Pkg.instantiate() + Pkg.instantiate() end -# default env name mirrors install_cupynumeric.sh: cupynumeric-bench-. -# CUPYNUMERIC_ENV overrides it. function cupynumeric_env_name() haskey(ENV, "CUPYNUMERIC_ENV") && return ENV["CUPYNUMERIC_ENV"] for (_, info) in Pkg.dependencies() @@ -49,91 +19,14 @@ function cupynumeric_env_name() return error("could not resolve cupynumeric_jll version; set CUPYNUMERIC_ENV explicitly") end -function dispatch(; gpus, cpus, name, T, N, M, n_iter, n_warmup, n_trial, - fusion=true, cupynumeric=false, cudajl=false, - check_correctness=false, n_correctness_iter=5) - fstr = fusion ? "enabled" : "disabled" - banner( - "$(name): T=$(T) gpus=$(gpus) cpus=$(cpus) N=$(N) M=$(M) fusion=$(fstr) " * - "n_iter=$(n_iter) n_warmup=$(n_warmup) n_trial=$(n_trial)", - ) - - # precompile in the orchestrator so the worker loads a warm cache quietly - CNPreferences.set_broadcast_fusion!(fusion) - if LAST_FUSION_TOGGLE[] != fusion - VERBOSE && println("Precompiling cuNumeric (fusion=$(fstr))") - Pkg.precompile("cuNumeric"; io=devnull) - LAST_FUSION_TOGGLE[] = fusion - end - - # each backend runs in its own worker process - vflag = VERBOSE ? `--verbose` : `` - args = `--gpus $gpus --cpus $cpus $name $T $N $M $n_iter $n_warmup $n_trial` - # trailing: backend check_correctness n_correctness_iter - corr_args = `$check_correctness $n_correctness_iter` - cmds = [`bash $RUNNER $WORKER $vflag $args cunumeric $corr_args`] - - # comparison backends have no fusion knob, so run them once instead of per - # fusion variant; the fused pass (the default) is that single run - run_comparison_backends = fusion - if run_comparison_backends - # CUDA.jl is single-GPU only - if cudajl && gpus == 1 && !cunumeric_only(name) - push!(cmds, `bash $RUNNER $WORKER $vflag $args cudajl $corr_args`) - end - if cupynumeric && !cunumeric_only(name) - push!(cmds, `bash $RUNNER $PY_WORKER $vflag --pyenv $(cupynumeric_env_name()) $args`) - end - end - - for cmd in cmds - try - run(cmd) - catch e - @error "Benchmark '$(name)' failed; continuing." exception = e - end - end -end - -function run_all_benchmarks(config="benchmarks.toml") - gs, specs = parse_config(joinpath(@__DIR__, config)) - for spec in specs - N, M = spec.args - dispatch(; - gpus=spec.gpus, - cpus=spec.cpus, - name=spec.name, - T=spec.T, - N=N, M=M, - fusion=spec.fusion, - n_iter=spec.n_iter, - n_warmup=spec.n_warmup, - n_trial=spec.n_trial, - cupynumeric=gs.cupynumeric, - cudajl=spec.cuda, - check_correctness=gs.check_correctness, - n_correctness_iter=gs.n_correctness_iter, - ) - end -end - -ensure_project_ready() -using CNPreferences: CNPreferences -if isempty(POSARGS) - run_all_benchmarks() -else # dispatch on args - dispatch(; - gpus=parse(Int, POSARGS[1]), - cpus=parse(Int, POSARGS[2]), - name=POSARGS[3], - T=POSARGS[4], - N=parse(Int, POSARGS[5]), - M=parse(Int, POSARGS[6]), - n_iter=parse(Int, POSARGS[7]), - n_warmup=parse(Int, POSARGS[8]), - n_trial=parse(Int, POSARGS[9]), - fusion=length(POSARGS) >= 10 ? parse_fusion(POSARGS[10]) : true, - check_correctness=length(POSARGS) >= 11 ? parse(Bool, POSARGS[11]) : false, - n_correctness_iter=length(POSARGS) >= 12 ? parse(Int, POSARGS[12]) : 5, - ) +if abspath(PROGRAM_FILE) == abspath(@__FILE__) + ensure_project_ready() + include("src/core.jl") + include_benchmarks() + include("src/parse_benchmarks.jl") + include("src/memory.jl") + include("src/planning.jl") + include("src/runner.jl") + using CNPreferences: CNPreferences + exit(main()) end diff --git a/benchmark/run_benchmark.sh b/benchmark/run_benchmark.sh index 8fc47aa8d..e9ef39307 100755 --- a/benchmark/run_benchmark.sh +++ b/benchmark/run_benchmark.sh @@ -59,6 +59,9 @@ fi export LEGATE_AUTO_CONFIG=1 export LEGATE_CONFIG="--cpus=$CPUS --gpus=$GPUS" +if [[ -n ${CUNUMERIC_BENCH_FBMEM_MB:-} ]]; then + export LEGATE_CONFIG="$LEGATE_CONFIG --fbmem=$CUNUMERIC_BENCH_FBMEM_MB" +fi export LEGATE_SHOW_CONFIG=$VERBOSE export LD_LIBRARY_PATH="" @@ -72,10 +75,11 @@ if [[ $FILENAME == *.py ]]; then echo "Error: running a .py worker requires --pyenv (run install_cupynumeric.sh first)." exit 1 fi - CMD="conda run --no-capture-output -n $PYENV python $FILENAME $GPUS ${EXTRA_ARGS[@]}" + CMD=("${CUNUMERIC_BENCH_CONDA:-${CONDA_EXE:-conda}}" run --no-capture-output -n "$PYENV" python "$FILENAME" "$GPUS" "${EXTRA_ARGS[@]}") else - CMD="julia --project $FILENAME $GPUS ${EXTRA_ARGS[@]}" + CMD=("${CUNUMERIC_BENCH_JULIA:-julia}" --project "$FILENAME" "$GPUS" "${EXTRA_ARGS[@]}") fi -[[ $VERBOSE == 1 ]] && printf "Running: %s\n" "$CMD" -eval "$CMD" +[[ $VERBOSE == 1 ]] && printf "Running: %q " "${CMD[@]}" +[[ $VERBOSE == 1 ]] && printf '\n' +"${CMD[@]}" diff --git a/benchmark/src/autosize.jl b/benchmark/src/autosize.jl new file mode 100644 index 000000000..43223325e --- /dev/null +++ b/benchmark/src/autosize.jl @@ -0,0 +1,108 @@ +# RAM query + binary search. Peak bytes and P-scaling live on each benchmark +# type (`total_space`, `estimate_scaling`, `fit_one_gpu`) so the formulas sit +# next to the kernel. Python never sees them: run.jl resolves N/M/flops and +# passes integers on the worker command line. + +align8(n::Integer) = max(8, 8 * fld(Int(n), 8)) +align2(n::Integer) = max(2, 2 * fld(Int(n), 2)) + +function scale_axis(n1::Integer, P::Integer, α::Real) + return align8(floor(Int, n1 * Float64(P)^Float64(α))) +end + +is_auto_size(x) = x isa AbstractString && lowercase(strip(string(x))) == "auto" + +function largest_feasible(lo::Int, hi::Int, pred) + hi < lo && return nothing + pred(lo) || return nothing + best = lo + while lo <= hi + mid = lo + (hi - lo) >> 1 + if pred(mid) + best = mid + lo = mid + 1 + else + hi = mid - 1 + end + end + return best +end + +function min_gpu_memory_bytes() + out = read(`nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits`, String) + mibs = Int[] + for line in split(out, '\n'; keepempty=false) + tok = strip(split(line, ',')[1]) + push!(mibs, parse(Int, tok)) + end + isempty(mibs) && error("nvidia-smi reported no GPUs") + return minimum(mibs) * 1024^2 +end + +function autosize_budget(mem_frac::Real) + frac = parse(Float64, get(ENV, "CUNUMERIC_BENCH_MEM_FRAC", string(mem_frac))) + (0 < frac <= 1) || error("mem_frac must be in (0, 1], got $frac") + bytes = Int(floor(frac * min_gpu_memory_bytes())) + bytes > 0 || error("GPU memory budget is 0") + return bytes, frac +end + +# Query only the GPUs the worker is allowed to see. Tests inject rows/env so +# device selection and budget calculation do not require a GPU. +function selected_gpu_budget(mem_frac, count; + inventory=read(`nvidia-smi --query-gpu=index,uuid,memory.total,memory.free --format=csv,noheader,nounits`,String), + visibility=get(ENV,"CUDA_VISIBLE_DEVICES",nothing), + fraction=get(ENV,"CUNUMERIC_BENCH_MEM_FRAC",string(mem_frac)), + fbmem=get(ENV,"CUNUMERIC_BENCH_FBMEM_MB",nothing)) + frac = parse(Float64,fraction) + 0 < frac <= 1 || error("mem_frac must be in (0,1]") + rows = [strip.(split(line,',')) for line in split(inventory,'\n') if !isempty(strip(line))] + devices = if visibility === nothing + rows + else + tokens = split(visibility,',') + any(t->isempty(strip(t)),tokens) && error("No usable CUDA_VISIBLE_DEVICES entries") + length(unique(tokens)) == length(tokens) || error("Duplicate CUDA_VISIBLE_DEVICES entries") + map(tokens) do token + matches = filter(r->r[1]==strip(token) || startswith(r[2],strip(token)),rows) + length(matches)==1 || error("Cannot resolve visible GPU '$token'; MIG requires a device-specific inventory") + only(matches) + end + end + 0 < count <= length(devices) || error("Requested $count GPUs, but only $(length(devices)) are visible") + # All visible devices are eligible for the runtime; use the smallest pool. + pools = [parse(Int,r[3])*1024^2 for r in devices] + free = minimum(parse(Int,r[4])*1024^2 for r in devices) + if fbmem !== nothing + cap = parse(Int,fbmem)*1024^2 + cap > 0 || error("CUNUMERIC_BENCH_FBMEM_MB must be positive") + pools = min.(pools,cap) + end + budget = floor(Int,frac*minimum(pools)) + # Do not silently shrink for transient other workloads. + free >= budget || error("Selected GPUs have only $free free bytes; planned budget is $budget. Free the devices or explicitly reduce mem_frac.") + return budget,frac +end + +function parse_bench_type(T_str::AbstractString) + return getfield(Base, Symbol(T_str))::DataType +end + +function resolve_autosize( + name::AbstractString, T_str::AbstractString, P::Integer; + mem_frac::Real, N_hint, M_hint, +) + haskey(BENCHMARKS, name) || error( + "No benchmark registered for '$(name)'. Known: $(join(sort(collect(keys(BENCHMARKS))), ", "))", + ) + B = BENCHMARKS[name] + T = parse_bench_type(T_str) + budget, frac = autosize_budget(mem_frac) + N1, M1 = fit_one_gpu(B, T; budget, N_hint, M_hint) + scaled = estimate_scaling(build_benchmark(B, T, N1, M1), P) + if scaled === nothing + return nothing + end + N, M = scaled + return (; N, M, N1, M1, budget, frac) +end diff --git a/benchmark/src/benchmarks/dmd.jl b/benchmark/src/benchmarks/dmd.jl index 7f5b41fea..717406269 100644 --- a/benchmark/src/benchmarks/dmd.jl +++ b/benchmark/src/benchmarks/dmd.jl @@ -36,6 +36,9 @@ _dmd_rank(b::AbstractDMD) = min(20, b.M - 1) # 2mr² Ã = U_r' B # 25r³ eigen(Ã) (LAPACK xGEEV) # 2mr² Φ = B W +const DMD_TALL_RATIO = 10 +const DEFAULT_DMD_M = 512 + function total_flops(b::AbstractDMD) m = b.N n = b.M - 1 @@ -49,12 +52,37 @@ function total_flops(b::AbstractDMD) ) end +# X (N×M), thin-SVD U ~ N×(M-1), Vt ~ n×n, B, plus a cuSOLVER-sized fudge. +function total_space(b::AbstractDMD{T}) where {T} + n = max(b.M - 1, 1) + s = sizeof(T) + return (b.N * b.M + b.N * n + n * n + b.N * n) * s + 2 * b.N * b.M * s +end + +function estimate_scaling(b::AbstractDMD, P::Integer) + P == 1 && return (b.N, b.M) + b.N >= DMD_TALL_RATIO * b.M || return nothing + return (b.N * P, b.M) +end + +function fit_one_gpu( + ::Type{B}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {B<:AbstractDMD,T} + M = something(M_hint, DEFAULT_DMD_M) + lo = max(M, 8) + hi = max(lo, Int(fld(budget, max(6 * M * sizeof(T), 1)))) + N = largest_feasible(lo, hi, n -> total_space(B{T}(; N=n, M=M)) <= budget) + N === nothing && error("dmd M=$M does not fit in $(budget) bytes") + return (max(align8(N), M), M) +end + function initialize(b::AbstractDMD{T}; mod=cuNumeric) where {T} # X1 is N×(M-1); the SVD backend requires m >= n. b.N >= b.M - 1 || throw( ArgumentError("DMD snapshot matrix is N×M with N ≥ M-1 (got N=$(b.N), M=$(b.M))") ) - X = mod.rand(T, b.N, b.M) + X = rand_array(mod, T, b.N, b.M) GC.gc() return (X,) end @@ -84,10 +112,12 @@ let body = quote (B, Ã) end @eval _dmd_project(::DMDBaseline, X, X2, U, Vt, S) = $body - @eval @accelerate function _dmd_project( - ::DMDAccelerated, X, X2, U, Vt, S - ) - $body + if CUNUMERIC_BENCH_RUNTIME + @eval @accelerate function _dmd_project( + ::DMDAccelerated, X, X2, U, Vt, S + ) + $body + end end end @@ -96,30 +126,23 @@ function _dmd_compute!(b::AbstractDMD, X, r) B, Ã = _dmd_project(b, X, X2, U, Vt, S) E = eigen(Ã) CT = Complex{eltype(X)} - Bc = X isa NDArray ? cuNumeric.as_type(B, CT) : CT.(B) + Bc = astype_array(B, CT) return E.values, Bc * E.vectors end run!(b::AbstractDMD, X) = _dmd_compute!(b, X, _dmd_rank(b)) -correctness_supported(::AbstractDMD) = true - -function check_benchmark_correctness( - b::AbstractDMD{T}, gs::GlobalSettings; mod=cuNumeric, atol=1e-3, rtol=1e-3 -) where {T} - mod === cuNumeric || return "skipped" - - Xh = rand(T, b.N, b.M) - X = NDArray(Xh) - r = _dmd_rank(b) - # Values, not lifetimes: compare the selected device path against the host baseline. - ref = DMDBaseline{T}(; N=b.N, M=b.M) - λ, _ = _dmd_compute!(b, X, r) - λh, _ = _dmd_compute!(ref, Xh, r) - - mag = sort(abs.(Array(λ)); rev=true) - magh = sort(abs.(λh); rev=true) - return isapprox(mag, magh; atol=atol, rtol=rtol) ? "pass" : "fail" +function correctness_problem(b::AD) where {AD <: AbstractDMD} + m = min(b.M, 16) + n = max(min(b.N, 64), m - 1) + return AD(; N=n, M=m) +end +# Eigenvectors have a phase; compare sorted |λ| only. +function correctness_result(::AbstractDMD, _, out) + return sort(abs.(vec(to_host(out[1]))); rev=true) +end +function cuda_runnable(b::DMDAccelerated{T}) where {T} + return DMDBaseline{T}(; N=b.N, M=b.M) end register_benchmark("dmd_baseline", DMDBaseline) diff --git a/benchmark/src/benchmarks/gemm.jl b/benchmark/src/benchmarks/gemm.jl index 4f85d3e24..886033b5e 100644 --- a/benchmark/src/benchmarks/gemm.jl +++ b/benchmark/src/benchmarks/gemm.jl @@ -1,3 +1,7 @@ +# Interface: `name`, `dims`, `total_flops`, `total_space`, `estimate_scaling`, +# `fit_one_gpu`, `initialize`, `run!`. Peak-byte and P-scaling formulas live +# here so the orchestrator can dispatch without a name switch. + Base.@kwdef struct GEMM{T} <: AbstractBenchmark{T} N::Int M::Int @@ -12,16 +16,36 @@ function allowed_types(::Type{GEMM}) end total_flops(s::GEMM) = s.N * s.N * ((2*s.M) - 1) -total_space(s::GEMM{T}) where {T} = 2 * ((s.N*s.M) * sizeof(T)) + ((s.N*s.N) * sizeof(T)) + +# Live arrays for `mul!(C, A, B)`: A (N×M), B (M×N), C (N×N). +total_space(s::GEMM{T}) where {T} = (2 * s.N * s.M + s.N * s.N) * sizeof(T) + +function estimate_scaling(s::GEMM, P::Integer) + P == 1 && return (s.N, s.M) + return (scale_axis(s.N, P, 1//3), scale_axis(s.M, P, 1//3)) +end + +function fit_one_gpu( + ::Type{GEMM}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {T} + hi = max(8, Int(floor(sqrt(Float64(budget) / sizeof(T))))) + n = largest_feasible(8, hi, k -> total_space(GEMM{T}(; N=k, M=k)) <= budget) + n === nothing && error("gemm does not fit in $(budget) bytes") + n = align8(n) + return (n, n) +end function initialize(s::GEMM{T}; mod=cuNumeric) where {T} - A = mod.rand(T, s.N, s.M) - B = mod.rand(T, s.M, s.N) - C = mod.zeros(T, s.N, s.N) + A = rand_array(mod, T, s.N, s.M) + B = rand_array(mod, T, s.M, s.N) + C = zeros_array(mod, T, s.N, s.N) GC.gc() return C, A, B end run!(::GEMM, C, A, B) = mul!(C, A, B) +correctness_problem(b::GEMM{T}) where {T} = GEMM{T}(; N=min(b.N, 8), M=min(b.M, 8)) + register_benchmark("gemm", GEMM) diff --git a/benchmark/src/benchmarks/grayscott.jl b/benchmark/src/benchmarks/grayscott.jl index 2fee6146e..c05066f52 100644 --- a/benchmark/src/benchmarks/grayscott.jl +++ b/benchmark/src/benchmarks/grayscott.jl @@ -13,6 +13,9 @@ end abstract type AbstractGrayScott{T} <: AbstractBenchmark{T} end +# Timesteps form one trajectory; allow Legate to schedule ahead within it. +fence_each_iteration(::AbstractGrayScott) = false + Base.@kwdef struct GrayScottBaseline{T} <: AbstractGrayScott{T} N::Int M::Int @@ -23,12 +26,32 @@ Base.@kwdef struct GrayScottAccelerated{T} <: AbstractGrayScott{T} M::Int end -name(::AbstractGrayScott) = "grayscott" +name(::GrayScottBaseline) = "grayscott_baseline" +name(::GrayScottAccelerated) = "grayscott_accelerated" dims(b::AbstractGrayScott) = (b.N, b.M) data(b::AbstractGrayScott{T}) where {T} = "GrayScott with T=$(T), N=$(b.N), M=$(b.M)" allowed_types(::Type{AbstractGrayScott}) = cuNumeric.SUPPORTED_FLOAT_TYPES total_flops(b::AbstractGrayScott) = b.N * b.M # grid points updated per step +# Four live grids: u, v, u_new, v_new. Legion halos are not counted. +total_space(b::AbstractGrayScott{T}) where {T} = 4 * b.N * b.M * sizeof(T) + +function estimate_scaling(b::AbstractGrayScott, P::Integer) + P == 1 && return (b.N, b.M) + return (scale_axis(b.N, P, 1//2), scale_axis(b.M, P, 1//2)) +end + +function fit_one_gpu( + ::Type{B}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {B<:AbstractGrayScott,T} + hi = max(8, Int(floor(sqrt(Float64(budget) / sizeof(T))))) + n = largest_feasible(8, hi, k -> total_space(B{T}(; N=k, M=k)) <= budget) + n === nothing && error("$B does not fit in $(budget) bytes") + n = align8(n) + return (n, n) +end + function build_benchmark(::Type{A}, ::Type{T}, N, M) where {A<:AbstractGrayScott,T} return A{T}(; N=N, M=M) end @@ -42,54 +65,43 @@ mutable struct GrayScottState{A,P} end function initialize(b::AbstractGrayScott{T}; mod=cuNumeric, deterministic::Bool=false) where {T} - d = (b.N, b.M) - u = mod.ones(T, d) - v = mod.zeros(T, d) - u_new = mod.zeros(T, d) - v_new = mod.zeros(T, d) + u = ones_array(mod, T, b.N, b.M) + v = zeros_array(mod, T, b.N, b.M) + u_new = zeros_array(mod, T, b.N, b.M) + v_new = zeros_array(mod, T, b.N, b.M) seed = min(150, b.N, b.M) if deterministic - # Fixed host pattern so CPU and GPU (any GPU count) share the same IC. - # Avoids Random streams differing across array backends. - host_u = T[ - T(0.5) + T(0.5) * sin(T(i)) * cos(T(j)) for i in 1:seed, j in 1:seed - ] - host_v = T[ - T(0.25) + T(0.25) * cos(T(i)) * sin(T(j)) for i in 1:seed, j in 1:seed - ] - u[1:seed, 1:seed] = mod === cuNumeric ? NDArray(host_u) : host_u - v[1:seed, 1:seed] = mod === cuNumeric ? NDArray(host_v) : host_v + # Shared host pattern so cuNumeric and CUDA.jl start from the same IC. + host_u = T[T(0.5) + T(0.5) * sin(T(i)) * cos(T(j)) for i in 1:seed, j in 1:seed] + host_v = T[T(0.25) + T(0.25) * cos(T(i)) * sin(T(j)) for i in 1:seed, j in 1:seed] + u[1:seed, 1:seed] = to_backend(mod, host_u) + v[1:seed, 1:seed] = to_backend(mod, host_v) else - u[1:seed, 1:seed] = mod.rand(T, (seed, seed)) - v[1:seed, 1:seed] = mod.rand(T, (seed, seed)) + u[1:seed, 1:seed] = rand_array(mod, T, seed, seed) + v[1:seed, 1:seed] = rand_array(mod, T, seed, seed) end return (GrayScottState(u, v, u_new, v_new, GSParams{T}()),) end -correctness_supported(::AbstractGrayScott) = true - -function check_benchmark_correctness( - b::AbstractGrayScott{T}, gs::GlobalSettings; mod=cuNumeric, atol=1e-4, rtol=1e-4 -) where {T} - # CPU reference compares via cuNumeric.compare (scalar gather). Other backends skip. - mod === cuNumeric || return "skipped" - - n = gs.n_correctness_iter - st_gpu = only(initialize(b; mod=mod, deterministic=true)) - st_cpu = only(initialize(b; mod=Base, deterministic=true)) - - for _ in 1:n - run!(b, st_gpu) - run!(b, st_cpu) - end +function to_backend_state(mod, st::GrayScottState) + return GrayScottState( + to_backend_state(mod, st.u), + to_backend_state(mod, st.v), + to_backend_state(mod, st.u_new), + to_backend_state(mod, st.v_new), + st.params, + ) +end - # Element-wise NDArray indexing gathers across tiles — do not use Array(NDArray) - # for multi-GPU (get_ptr is local-tile only). - u_ok = @allowscalar cuNumeric.compare(st_cpu.u, st_gpu.u, atol, rtol) - v_ok = @allowscalar cuNumeric.compare(st_cpu.v, st_gpu.v, atol, rtol) - return (u_ok && v_ok) ? "pass" : "fail" +function correctness_problem(b::AbstractGrayScott{T}) where {T} + return typeof(b)(; N=min(32, b.N, b.M), M=min(32, b.N, b.M)) +end +correctness_iters(::AbstractGrayScott, gs::GlobalSettings) = gs.n_correctness_iter +correctness_result(::AbstractGrayScott, state, _) = (only(state).u, only(state).v) +function cuda_runnable(b::GrayScottAccelerated{T}) where {T} + return GrayScottBaseline{T}(; N=b.N, M=b.M) end # Shared syntax tree keeps every Gray-Scott variant on the exact same workload. @@ -151,10 +163,12 @@ end # Original baseline and recommended function-form benchmark. let body = deepcopy(GRAYSCOTT_STEP_BODY) @eval _gs_step!(b::GrayScottBaseline, u, v, u_new, v_new, args::GSParams) = $body - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottAccelerated, u, v, u_new, v_new, args::GSParams)), body - ) - @eval $definition + if CUNUMERIC_BENCH_RUNTIME + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottAccelerated, u, v, u_new, v_new, args::GSParams)), body + ) + @eval $definition + end end function run!(b::AbstractGrayScott, st::GrayScottState) diff --git a/benchmark/src/benchmarks/grayscott_accelerate_forms.jl b/benchmark/src/benchmarks/grayscott_accelerate_forms.jl index 187dc299f..2b83342d6 100644 --- a/benchmark/src/benchmarks/grayscott_accelerate_forms.jl +++ b/benchmark/src/benchmarks/grayscott_accelerate_forms.jl @@ -30,37 +30,43 @@ name(::GrayScottBeginAccelerated) = "grayscott_begin_accelerated" name(::GrayScottLetAccelerated) = "grayscott_let_accelerated" name(::GrayScottExpressionAccelerated) = "grayscott_expression_accelerated" +function cuda_runnable(b::AbstractGrayScottAccelerateForm{T}) where {T} + return GrayScottBaseline{T}(; N=b.N, M=b.M) +end + # Function form is the reusable default: arguments and the return value survive, # while non-returned locals may fuse across statements or die after their last use. -let body = deepcopy(GRAYSCOTT_STEP_BODY) - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottFunctionAccelerated, u, v, u_new, v_new, args::GSParams)), - body, - :function, - ) - @eval $definition -end +if CUNUMERIC_BENCH_RUNTIME + let body = deepcopy(GRAYSCOTT_STEP_BODY) + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottFunctionAccelerated, u, v, u_new, v_new, args::GSParams)), + body, + :function, + ) + @eval $definition + end -# `begin` adds no scope. Every named local remains visible, so it measures the -# multi-output/materialized path rather than eliminating named intermediates. -let body = deepcopy(GRAYSCOTT_STEP_BODY) - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottBeginAccelerated, u, v, u_new, v_new, args::GSParams)), - body, - :begin, - ) - @eval $definition -end + # `begin` adds no scope. Every named local remains visible, so it measures the + # multi-output/materialized path rather than eliminating named intermediates. + let body = deepcopy(GRAYSCOTT_STEP_BODY) + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottBeginAccelerated, u, v, u_new, v_new, args::GSParams)), + body, + :begin, + ) + @eval $definition + end -# `let` is a hard one-off scope. Only its result escapes, allowing aggressive -# inter-statement fusion and last-use cleanup for all other local temporaries. -let body = deepcopy(GRAYSCOTT_STEP_BODY) - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottLetAccelerated, u, v, u_new, v_new, args::GSParams)), - body, - :let, - ) - @eval $definition + # `let` is a hard one-off scope. Only its result escapes, allowing aggressive + # inter-statement fusion and last-use cleanup for all other local temporaries. + let body = deepcopy(GRAYSCOTT_STEP_BODY) + definition = _define_accelerated_definition( + :(_gs_step!(b::GrayScottLetAccelerated, u, v, u_new, v_new, args::GSParams)), + body, + :let, + ) + @eval $definition + end end # Expression form has no multi-statement scope. Accelerating each RHS preserves @@ -82,10 +88,12 @@ function accelerate_grayscott_rhs(body::Expr) end let body = accelerate_grayscott_rhs(deepcopy(GRAYSCOTT_STEP_BODY)) - @eval function _gs_step!( - b::GrayScottExpressionAccelerated, u, v, u_new, v_new, args::GSParams - ) - $body + if CUNUMERIC_BENCH_RUNTIME + @eval function _gs_step!( + b::GrayScottExpressionAccelerated, u, v, u_new, v_new, args::GSParams + ) + $body + end end end diff --git a/benchmark/src/benchmarks/montecarlo.jl b/benchmark/src/benchmarks/montecarlo.jl index 978e5906f..c42db0066 100644 --- a/benchmark/src/benchmarks/montecarlo.jl +++ b/benchmark/src/benchmarks/montecarlo.jl @@ -10,22 +10,47 @@ end allowed_types(::Type{MonteCarloIntegration}) = cuNumeric.SUPPORTED_FLOAT_TYPES -total_space(s::MonteCarloIntegration{T}) where {T} = s.n_samples * sizeof(T) total_flops(s::MonteCarloIntegration) = s.n_samples +# Fused broadcast: samples x plus exp.(-x .^ 2), materialized before sum. +# Initialization also needs two arrays: random samples and their scaled output. +# Assumes fusion is enabled; workspace/runtime overhead uses the headroom left +# by mem_frac. This is not a memory estimate for unfused comparison backends. +total_space(s::MonteCarloIntegration{T}) where {T} = 2 * s.n_samples * sizeof(T) + +function estimate_scaling(s::MonteCarloIntegration, P::Integer) + P == 1 && return dims(s) + return (s.n_samples * P, 1) +end + +function fit_one_gpu( + ::Type{MonteCarloIntegration}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {T} + hi = max(8, Int(fld(budget, sizeof(T)))) + n = largest_feasible(8, hi, k -> total_space(MonteCarloIntegration{T}(; n_samples=k)) <= budget) + n === nothing && error("montecarlo does not fit in $(budget) bytes") + return (align8(n), 1) +end function initialize(mci::MonteCarloIntegration{T}; mod=cuNumeric) where {T} # Uniform samples over the integration domain [0, 10]. - x = T(10) .* mod.rand(T, mci.n_samples) + x = T(10) .* rand_array(mod, T, mci.n_samples) GC.gc() return (x,) end _domain_volume(mci::MonteCarloIntegration{T}) where {T} = T(10) / mci.n_samples -run!(mci::MonteCarloIntegration, x) = _domain_volume(mci) * sum(exp.(-x .^ 2)) +# Dot the negation too: plain `-` materializes the squared array and prevents +# the surrounding exponential from sharing one broadcast with the square. +run!(mci::MonteCarloIntegration, x) = _domain_volume(mci) * sum(exp.(.-(x .^ 2))) # n_samples comes in as N; M is unused. function build_benchmark(::Type{MonteCarloIntegration}, ::Type{T}, N, M) where {T} return MonteCarloIntegration{T}(; n_samples=N) end +function correctness_problem(b::MonteCarloIntegration{T}) where {T} + return MonteCarloIntegration{T}(; n_samples=min(b.n_samples, 1024)) +end + register_benchmark("montecarlo", MonteCarloIntegration) diff --git a/benchmark/src/benchmarks/poisson_fft.jl b/benchmark/src/benchmarks/poisson_fft.jl index afe1344f9..34eacb7f1 100644 --- a/benchmark/src/benchmarks/poisson_fft.jl +++ b/benchmark/src/benchmarks/poisson_fft.jl @@ -5,7 +5,8 @@ # # The transform is over the last two axes, so the leading batch axis may # partition across GPUs. A single all-axes 2-d FFT cannot. -# Weak scaling: M ∝ P, N fixed. +# Weak scaling: M ∝ P, N held constant across the GPU sweep (changing N +# would change the FFT size). N itself may be RAM-fitted or pinned. function _integer_fftfreq(n::Int) n2 = n ÷ 2 @@ -42,43 +43,62 @@ function total_flops(b::PoissonFFT) return m * (20 * n2 * log2(n) + 6 * n2) end -function initialize(b::PoissonFFT{T}; mod=cuNumeric) where {T} - f = mod.rand(T, b.M, b.N, b.N) - CT = Complex{T} - fc = f isa NDArray ? cuNumeric.as_type(f, CT) : CT.(f) - work = copy(fc) - kinv_h = _poisson_inv_laplacian(T, b.N) - kinv = if fc isa NDArray - reshape(NDArray(kinv_h), 1, b.N, b.N) +# fc and work as Complex, kinv as T, plus an FFT workspace ~ one extra complex grid. +function total_space(b::PoissonFFT{T}) where {T} + n2 = b.N * b.N + return (3 * b.M * n2) * sizeof(Complex{T}) + n2 * sizeof(T) +end + +function estimate_scaling(b::PoissonFFT, P::Integer) + P == 1 && return (b.N, b.M) + # Grid N is held across P (FFT is the last two axes). Batch M partitions. + return (b.N, b.M * P) +end + +function fit_one_gpu( + ::Type{PoissonFFT}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {T} + if N_hint !== nothing + N = N_hint + hi = max(1, Int(fld(budget, max(3 * N * N * sizeof(Complex{T}), 1)))) + M = largest_feasible(1, hi, m -> total_space(PoissonFFT{T}(; N=N, M=m)) <= budget) + M === nothing && error("poisson_fft N=$N does not fit in $(budget) bytes") + return (N, M) else - reshape(kinv_h, 1, b.N, b.N) + hi = max(8, Int(floor(sqrt(Float64(budget) / (3 * sizeof(Complex{T})))))) + N = largest_feasible(8, hi, n -> total_space(PoissonFFT{T}(; N=n, M=1)) <= budget) + N === nothing && error("poisson_fft does not fit in $(budget) bytes") + return (align2(N), 1) end +end + +function initialize(b::PoissonFFT{T}; mod=cuNumeric) where {T} + f = rand_array(mod, T, b.M, b.N, b.N) + fc = astype_array(f, Complex{T}) + work = copy(fc) + kinv = to_backend(mod, reshape(_poisson_inv_laplacian(T, b.N), 1, b.N, b.N)) GC.gc() return fc, work, kinv end +_trailing_fft_dims(A) = ntuple(i -> i + 1, ndims(A) - 1) + +if CUNUMERIC_BENCH_RUNTIME + _batched_fft!(A::NDArray) = (cuNumeric.batched_fft!(A); A) + _batched_ifft!(A::NDArray) = (cuNumeric.batched_ifft!(A); A) +end +_batched_fft!(A) = (fft!(A, _trailing_fft_dims(A)); A) +_batched_ifft!(A) = (ifft!(A, _trailing_fft_dims(A)); A) + function run!(::PoissonFFT, fc, work, kinv) copyto!(work, fc) - cuNumeric.batched_fft!(work) + _batched_fft!(work) work .*= kinv - cuNumeric.batched_ifft!(work) + _batched_ifft!(work) return work end -correctness_supported(::PoissonFFT) = true - -function check_benchmark_correctness( - ::PoissonFFT{T}, gs::GlobalSettings; mod=cuNumeric, atol=1e-3, rtol=1e-3 -) where {T} - mod === cuNumeric || return "skipped" - n = 32 - xs = range(T(0), T(1); length=n + 1)[1:(end - 1)] - u_true = T[sin(2T(π) * x) * sin(2T(π) * y) for x in xs, y in xs] - f_h = (-8 * T(π)^2) .* u_true - f = NDArray(reshape(f_h, 1, n, n)) - kinv = reshape(NDArray(_poisson_inv_laplacian(T, n)), 1, n, n) - u = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f) .* kinv)) - return isapprox(Array(u)[1, :, :], u_true; atol=atol, rtol=rtol) ? "pass" : "fail" -end +correctness_problem(b::PoissonFFT{T}) where {T} = PoissonFFT{T}(; N=min(b.N, 32), M=1) register_benchmark("poisson_fft", PoissonFFT) diff --git a/benchmark/src/benchmarks/tensor_contractions.jl b/benchmark/src/benchmarks/tensor_contractions.jl index 825b8b565..08ff9a370 100644 --- a/benchmark/src/benchmarks/tensor_contractions.jl +++ b/benchmark/src/benchmarks/tensor_contractions.jl @@ -1,4 +1,3 @@ -using Random using TensorOperations abstract type AbstractTensorContraction{T} <: AbstractBenchmark{T} end @@ -26,6 +25,42 @@ total_flops(b::TensorProjection3) = 3 * b.N^3 * (2 * b.N - 1) # N^4 output elements, each containing an N^2-term dot product. total_flops(b::TensorContract4) = b.N^4 * (2 * b.N^2 - 1) +# opt=true pairwise: T1[n,j,k] and T2[n,m,k] are both N³, plus A, D, B. +total_space(b::TensorProjection3{T}) where {T} = (4 * b.N^3 + b.N^2) * sizeof(T) + +# opt=true is a (N²×N²)×(N²×N²) GEMM: X, Y, C plus one workspace. +total_space(b::TensorContract4{T}) where {T} = 4 * b.N^4 * sizeof(T) + +function estimate_scaling(b::TensorProjection3, P::Integer) + P == 1 && return dims(b) + return (align2(floor(Int, b.N * Float64(P)^(1 / 4))), 1) +end + +function estimate_scaling(b::TensorContract4, P::Integer) + P == 1 && return dims(b) + return (align2(floor(Int, b.N * Float64(P)^(1 / 6))), 1) +end + +function fit_one_gpu( + ::Type{TensorProjection3}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {T} + hi = max(4, Int(floor((Float64(budget) / (4 * sizeof(T)))^(1 / 3)))) + n = largest_feasible(4, hi, k -> total_space(TensorProjection3{T}(; N=k)) <= budget) + n === nothing && error("tensor_projection3 does not fit in $(budget) bytes") + return (n, 1) +end + +function fit_one_gpu( + ::Type{TensorContract4}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {T} + hi = max(4, Int(floor((Float64(budget) / (4 * sizeof(T)))^(1 / 4)))) + n = largest_feasible(4, hi, k -> total_space(TensorContract4{T}(; N=k)) <= budget) + n === nothing && error("tensor_contract4 does not fit in $(budget) bytes") + return (n, 1) +end + function build_benchmark( ::Type{TensorProjection3}, ::Type{T}, N, M ) where {T} @@ -38,23 +73,18 @@ function build_benchmark( return TensorContract4{T}(; N=N) end -function _tensor_rand(mod, ::Type{T}, dims...) where {T} - mod === CUDACore && return CUDACore.CuArray(rand(T, dims...)) - return mod.rand(T, dims...) -end - function initialize(b::TensorProjection3{T}; mod=cuNumeric) where {T} - A = _tensor_rand(mod, T, b.N, b.N, b.N) - B = _tensor_rand(mod, T, b.N, b.N) - D = mod.zeros(T, b.N, b.N, b.N) + A = rand_array(mod, T, b.N, b.N, b.N) + B = rand_array(mod, T, b.N, b.N) + D = zeros_array(mod, T, b.N, b.N, b.N) GC.gc() return D, A, B end function initialize(b::TensorContract4{T}; mod=cuNumeric) where {T} - X = _tensor_rand(mod, T, b.N, b.N, b.N, b.N) - Y = _tensor_rand(mod, T, b.N, b.N, b.N, b.N) - C = mod.zeros(T, b.N, b.N, b.N, b.N) + X = rand_array(mod, T, b.N, b.N, b.N, b.N) + Y = rand_array(mod, T, b.N, b.N, b.N, b.N) + C = zeros_array(mod, T, b.N, b.N, b.N, b.N) GC.gc() return C, X, Y end @@ -66,58 +96,15 @@ function run!(::TensorProjection3, D, A, B) end function run!(::TensorContract4, C, X, Y) - @tensor C[a, b, c, d] = X[a, i, c, j] * Y[i, b, j, d] + @tensor opt=true C[a, b, c, d] = X[a, i, c, j] * Y[i, b, j, d] return C end -correctness_supported(::AbstractTensorContraction) = true - -function _backend_array(mod, A) - mod === cuNumeric && return NDArray(A) - mod === CUDACore && return CUDACore.CuArray(A) - return throw(ArgumentError("unsupported tensor contraction backend $mod")) -end - -function _tensor_isapprox(actual, expected, ::Type{T}) where {T} - tol = T === Float32 ? 2e-4 : 1e-11 - return isapprox(Array(actual), expected; atol=tol, rtol=tol) -end - -function check_benchmark_correctness( - b::TensorProjection3{T}, gs::GlobalSettings; mod=cuNumeric -) where {T} - mod in (cuNumeric, CUDACore) || return "skipped" - n = min(b.N, 4) - rng = MersenneTwister(0x3b8a7c21) - Ah = rand(rng, T, n, n, n) - Bh = rand(rng, T, n, n) - ref = zeros(T, n, n, n) - @tensor opt=true ref[n, m, l] = - Ah[i, j, k] * Bh[n, i] * Bh[m, j] * Bh[l, k] - - D = _backend_array(mod, zeros(T, n, n, n)) - A = _backend_array(mod, Ah) - B = _backend_array(mod, Bh) - run!(b, D, A, B) - return _tensor_isapprox(D, ref, T) ? "pass" : "fail" -end - -function check_benchmark_correctness( - b::TensorContract4{T}, gs::GlobalSettings; mod=cuNumeric -) where {T} - mod in (cuNumeric, CUDACore) || return "skipped" - n = min(b.N, 4) - rng = MersenneTwister(0xa3c97d42) - Xh = rand(rng, T, n, n, n, n) - Yh = rand(rng, T, n, n, n, n) - ref = zeros(T, n, n, n, n) - @tensor ref[a, b, c, d] = Xh[a, i, c, j] * Yh[i, b, j, d] - - C = _backend_array(mod, zeros(T, n, n, n, n)) - X = _backend_array(mod, Xh) - Y = _backend_array(mod, Yh) - run!(b, C, X, Y) - return _tensor_isapprox(C, ref, T) ? "pass" : "fail" +correctness_problem(b::TensorProjection3{T}) where {T} = TensorProjection3{T}(; N=min(b.N, 4)) +correctness_problem(b::TensorContract4{T}) where {T} = TensorContract4{T}(; N=min(b.N, 4)) +function correctness_atol_rtol(::AbstractTensorContraction, ::Type{T}) where {T} + tol = T === Float32 ? 2.0f-4 : 1e-11 + return tol, tol end function benchmark_backend_label( diff --git a/benchmark/src/core.jl b/benchmark/src/core.jl index 9511ba65e..aee95efbd 100644 --- a/benchmark/src/core.jl +++ b/benchmark/src/core.jl @@ -1,18 +1,19 @@ using Printf +using ProgressMeter: ProgressMeter using Statistics """ - `n_warmup::Int` : Number of warmup steps. These are not timed. Intended to avoid pre-compilation cost being timed. -- `n_iter::Int` : Number of iterations to run per trial. Should be large enough - to build up queue depth of tasks such that latency is hidden. +- `n_iter::Int` : Number of completed iterations per trial. Gray–Scott instead + queues timesteps and synchronizes at the trial boundaries. - `n_trial::Int` : Number of independent trials to run. Timing is restarted and legate in between each trial. Sets number of datapoints used to estimated standard deviations/errors. - `n_gpu::Int` : The number of GPUs used by legate. Set through the LEGATE_CONFIG, this value is just bookkeeping. -- `check_correctness::Bool` : If true, run one CPU-reference check per config - (not per timed iteration) before timing; result is recorded in the CSV. +- `check_correctness::Bool` : If true and `n_gpu == 1`, compare a tiny cuNumeric + result against CUDA.jl before timing. CUDA.jl / multi-GPU / Python skip. - `n_correctness_iter::Int` : Steps to run for that single correctness check. """ Base.@kwdef struct GlobalSettings @@ -24,12 +25,24 @@ Base.@kwdef struct GlobalSettings cuda::Bool = false # also run under CUDA.jl for comparison (single-GPU only) check_correctness::Bool = false n_correctness_iter::Int = 5 + auto_size::Bool = false + mem_frac::Float64 = 0.5 end ######################################### abstract type AbstractBenchmark{T} end +# Independent problems must finish before the next repetition is submitted. +fence_each_iteration(::AbstractBenchmark) = true +benchmark_synchronize() = cuNumeric.issue_execution_fence(; block=true) + +# True when this file is included after `using cuNumeric` (the worker). The +# orchestrator includes the same kernel files for types / `total_space` / +# `estimate_scaling` without loading Legion; those files skip `@accelerate` +# and `::NDArray` methods in that case. +const CUNUMERIC_BENCH_RUNTIME = isdefined(@__MODULE__, :cuNumeric) + # Interface each benchmark implements (see benchmarks/gemm.jl for a template). function name end function dims end @@ -39,6 +52,34 @@ function total_flops end function initialize end function run! end +include("autosize.jl") + +function include_benchmarks() + dir = joinpath(@__DIR__, "benchmarks") + for file in sort(filter(f -> endswith(f, ".jl"), readdir(dir; join=true))) + Base.include(@__MODULE__, file) + end + return nothing +end + +total_space(b::AbstractBenchmark) = + error("total_space not defined for $(typeof(b)); add a method in its benchmark file") + +function estimate_scaling(b::AbstractBenchmark, P::Integer) + P < 1 && throw(ArgumentError("P must be ≥ 1, got $P")) + P == 1 && return map(Int, dims(b)) + return error( + "estimate_scaling not defined for $(typeof(b)); add a method in its benchmark file", + ) +end + +function fit_one_gpu( + ::Type{B}, ::Type{T}; + budget::Int, N_hint=nothing, M_hint=nothing, +) where {B,T} + return error("fit_one_gpu not defined for $B; add a method in its benchmark file") +end + # Internal adapter for benchmark generators that share a quoted step body. function _define_accelerated_definition(signature, body, form=:function) if form === :function @@ -63,6 +104,13 @@ function build_benchmark(::Type{B}, ::Type{T}, N, M) where {B<:AbstractBenchmark return B{T}(; N=N, M=M) end +# Optional hooks for the generic CUDA.jl check (initialize + run!). +correctness_problem(b::AbstractBenchmark) = b +correctness_iters(::AbstractBenchmark, gs::GlobalSettings) = 1 +cuda_runnable(b::AbstractBenchmark) = b +correctness_result(::AbstractBenchmark, state, out) = out === nothing ? state : out +correctness_atol_rtol(::AbstractBenchmark, ::Type{T}) where {T} = ref_atol_rtol(T) + ######################################### # Per-trial timings for one benchmark. `times_ms[i]`/`gflops[i]` are the mean @@ -75,20 +123,115 @@ struct BenchmarkResult{B<:AbstractBenchmark} correctness::String end -# Optional per-benchmark correctness vs a CPU/`Array` reference. -# Return "pass", "fail", or "skipped". Default: no check implemented. -correctness_supported(::AbstractBenchmark) = false -function check_benchmark_correctness(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) - return "skipped" +# CUDA.jl 6: the worker may pass `CUDA` or `CUDACore` as `mod`. +is_cuda_backend(mod) = nameof(mod) === :CUDA || nameof(mod) === :CUDACore + +# Timed CUDA.jl is never the thing we check. Oracle compare is cuNumeric vs CUDA +# on a single GPU (the cuNumeric worker loads CUDA for the tiny problem). +function correctness_applies(gs::GlobalSettings, mod) + is_cuda_backend(mod) && return false + return gs.n_gpu == 1 +end + +function cuda_backend() + for (id, mod) in Base.loaded_modules + id.name == "CUDA" && return mod + end + return error("CUDA.jl must be loaded to check cuNumeric against CUDA.jl") +end + +# 1–4D (or more) constructors. `mod` is cuNumeric or CUDA; both expose +# rand/zeros/ones(::Type, dims...). Host `Array` is only a seed for to_backend. +rand_array(mod, ::Type{T}, dims::Integer...) where {T} = mod.rand(T, dims...) +rand_array(mod, ::Type{T}, dims::Tuple) where {T} = rand_array(mod, T, dims...) +zeros_array(mod, ::Type{T}, dims::Integer...) where {T} = mod.zeros(T, dims...) +zeros_array(mod, ::Type{T}, dims::Tuple) where {T} = zeros_array(mod, T, dims...) +ones_array(mod, ::Type{T}, dims::Integer...) where {T} = mod.ones(T, dims...) +ones_array(mod, ::Type{T}, dims::Tuple) where {T} = ones_array(mod, T, dims...) + +# Avoid `::NDArray` in the signature so the orchestrator can include this file +# without loading cuNumeric. +function astype_array(A, ::Type{T}) where {T} + nameof(typeof(A)) === :NDArray && return cuNumeric.as_type(A, T) + return T.(A) +end + +to_host(A::Array) = A +to_host(x::Number) = x +to_host(A) = Array(A) + +function to_backend(mod, A::AbstractArray) + h = A isa Array ? A : Array(A) + mod === cuNumeric && return NDArray(h) + is_cuda_backend(mod) && return mod.CuArray(h) + return h +end + +to_backend_state(mod, x::AbstractArray) = to_backend(mod, x) +to_backend_state(mod, x::Tuple) = map(s -> to_backend_state(mod, s), x) +to_backend_state(mod, x) = x + +function ref_atol_rtol(::Type{T}; atol=nothing, rtol=nothing) where {T} + default = T <: Float32 ? 1.0f-3 : 1e-10 + return something(atol, default), something(rtol, default) +end + +function isapprox_ref(actual, expected, ::Type{T}; atol=nothing, rtol=nothing) where {T} + at, rt = ref_atol_rtol(T; atol, rtol) + return isapprox(to_host(actual), to_host(expected); atol=at, rtol=rt) +end + +# Full cuNumeric reductions return 0-D arrays, while CUDA returns scalars. +# Define this only in the worker; the orchestrator does not load cuNumeric. +if CUNUMERIC_BENCH_RUNTIME + @eval function isapprox_ref( + actual::AbstractArray{<:Any,0}, expected::Number, ::Type{T}; kwargs... + ) where {T} + value = cuNumeric.@allowscalar actual[] + return isapprox_ref(value, expected, T; kwargs...) + end +end + +_all_approx(a, b, ::Type{T}; kwargs...) where {T} = isapprox_ref(a, b, T; kwargs...) +function _all_approx(a::Tuple, b::Tuple, ::Type{T}; kwargs...) where {T} + length(a) == length(b) || return false + return all(_all_approx(x, y, T; kwargs...) for (x, y) in zip(a, b)) +end + +# Host-seed once, upload to both backends, initialize/run! the same way as timing. +function check_benchmark_correctness( + b::AbstractBenchmark{T}, gs::GlobalSettings; mod=cuNumeric +) where {T} + tiny = correctness_problem(b) + seed = initialize(tiny; mod=Base) + atol, rtol = correctness_atol_rtol(b, T) + nstep = correctness_iters(tiny, gs) + return check_vs_cuda(T; atol, rtol) do backend + kernel = backend === cuNumeric ? tiny : cuda_runnable(tiny) + state = to_backend_state(backend, seed) + out = nothing + for _ in 1:nstep + out = run!(kernel, state...) + end + return correctness_result(kernel, state, out) + end +end + +# `f(mod)` runs the tiny problem on one backend and returns the value(s) to compare. +function check_vs_cuda(f, ::Type{T}; atol=nothing, rtol=nothing) where {T} + got = f(cuNumeric) + ref = f(cuda_backend()) + return _all_approx(got, ref, T; atol, rtol) ? "pass" : "fail" end # One timed trial: warmup, then time `n_iter` iterations of `run!`. function _trial( b::AbstractBenchmark, gs::GlobalSettings; - mod=cuNumeric, clock=get_time_microseconds, + mod=cuNumeric, clock=get_time_microseconds, synchronize=benchmark_synchronize, ) GC.gc(true) state = initialize(b; mod=mod) + fence_each = fence_each_iteration(b) start_time = nothing for idx in 1:(gs.n_warmup + gs.n_iter) @@ -96,6 +239,7 @@ function _trial( start_time = clock() end run!(b, state...) + fence_each && synchronize() end total_time_μs = clock() - start_time @@ -108,23 +252,40 @@ end # Correctness (if enabled) runs once before timing, not per trial/iteration. function run_benchmark( b::AbstractBenchmark, gs::GlobalSettings; - mod=cuNumeric, clock=get_time_microseconds, + mod=cuNumeric, clock=get_time_microseconds, synchronize=benchmark_synchronize, ) correctness = "skipped" if gs.check_correctness - if correctness_supported(b) + if correctness_applies(gs, mod) + println("Checking correctness against CUDA.jl on a small problem...") + flush(stdout) correctness = check_benchmark_correctness(b, gs; mod=mod) else correctness = "skipped" end end + println("Correctness: $(correctness)") + println( + "Starting $(gs.n_trial) trials; each includes initialization, " * + "$(gs.n_warmup) warmups, and $(gs.n_iter) timed iterations; " * + (fence_each_iteration(b) ? "per-iteration synchronization." : "batch synchronization."), + ) + flush(stdout) times_ms = Float64[] gflops = Float64[] - for _ in 1:gs.n_trial - t, g = _trial(b, gs; mod=mod, clock=clock) + progress = ProgressMeter.Progress(gs.n_trial; dt=0.0, desc="$(name(b)) trials: ") + ProgressMeter.update!(progress, 0) + for trial in 1:gs.n_trial + t, g = _trial(b, gs; mod=mod, clock=clock, synchronize=synchronize) push!(times_ms, t) push!(gflops, g) + # Update only after _trial has stopped its clock; never inside the kernel loop. + ProgressMeter.next!(progress; showvalues=[ + ("Completed trials", "$(trial)/$(gs.n_trial)"), + ("Last trial mean (ms/iteration)", t), + ("Last trial GFLOP/s", g), + ]) end return BenchmarkResult(times_ms, gflops, b, correctness) end @@ -133,7 +294,8 @@ _std(x) = length(x) > 1 ? std(x) : 0.0 function save_result(br::BenchmarkResult, gpus; mod::String="cunumeric") N, M = dims(br.benchmark) - path = joinpath(@__DIR__, "..", "results", "$(name(br.benchmark))_$(mod).csv") + results = get(ENV, "CUNUMERIC_BENCH_RESULTS_DIR", joinpath(@__DIR__, "..", "results")) + path = joinpath(results, "$(name(br.benchmark))_$(mod).csv") mkpath(dirname(path)) open(path, "a") do io for trial in eachindex(br.times_ms) @@ -146,29 +308,3 @@ function save_result(br::BenchmarkResult, gpus; mod::String="cunumeric") end end end - -######################################### - -# `setup` runs in the worker before the benchmark is built (e.g. flip a runtime -# preference); code-path variants leave it a no-op. -# struct Variant -# name::String -# setup::Function -# end - -# const VARIANTS = Dict{String,Variant}() - -# function register_variant(name, setup=() -> nothing) -# VARIANTS[name] = Variant(name, setup) -# end - -# function variant_setup(name) -# if haskey(VARIANTS, name) -# return VARIANTS[name].setup -# end -# return () -> nothing -# end - -# register_variant("baseline") -# register_variant("fusion_off", cuNumeric.disable_broadcast_fusion!) -# register_variant("fusion_on", cuNumeric.enable_broadcast_fusion!) diff --git a/benchmark/src/memory.jl b/benchmark/src/memory.jl new file mode 100644 index 000000000..28eac5f56 --- /dev/null +++ b/benchmark/src/memory.jl @@ -0,0 +1,147 @@ +# Analytical estimates. This file is included after the benchmark definitions. +# All byte counts use BigInt so preflight cannot wrap on oversized dimensions. +Base.@kwdef struct MemoryContext + backend::Symbol = :cunumeric + fusion::Bool = true + gpus::Int = 1 + steps::Int = 1 + # Explicit upper bound per GPU for opaque native-library scratch/packing. + workspace_bytes::Union{Nothing,Int} = nothing +end + +struct MemoryEstimate + initialization::BigInt + iteration::BigInt + workspace::BigInt + explanation::String +end +peak_bytes(m::MemoryEstimate) = max(m.initialization, m.iteration) + m.workspace + +function library_workspace(b, c) + c.workspace_bytes === nothing && error( + "$(name(b)) / $(c.backend): native workspace bound is unknown. " * + "Set workspace_bytes to a verified per-GPU upper bound for this backend/library " * + "configuration; autosizing will not guess or probe after an OOM.", + ) + c.workspace_bytes >= 0 || error("workspace_bytes must be nonnegative") + return big(c.workspace_bytes) +end + +function validate_memory_context(b::AbstractBenchmark{T}, c) where {T} + c.gpus > 0 || error("GPU count must be positive") + c.steps > 0 || error("Trial steps must be positive") + c.backend in (:cunumeric, :cudajl, :cupynumeric) || error("Unknown backend $(c.backend)") + c.backend == :cudajl && c.gpus != 1 && error("CUDA.jl supports one GPU only") + endswith(name(b),"_accelerated") && c.backend != :cunumeric && error("$(name(b)) is cuNumeric-only") + T in (Float32, Float64) || error("Memory accounting currently supports Float32 and Float64; got $T") + all(>(0), dims(b)) || error("Problem dimensions must be positive") +end + +# A conservative slab bound: rounding a partition up cannot undercount uneven +# dimensions. Stencil halos are counted separately below. DMD is never divided. +slab(n, tail, p) = cld(big(n), p) * big(tail) +random_peak(elements, ::Type{T}, backend) where {T} = + elements * (backend == :cupynumeric ? sizeof(Float64) + sizeof(T) : sizeof(T)) + +function memory_estimate(b::MonteCarloIntegration{T}, c::MemoryContext) where {T} + validate_memory_context(b, c) + e = cld(big(b.n_samples), c.gpus) + bytes = e * sizeof(T) + init = max(2bytes, random_peak(e, T, c.backend)) + # The unfused NDArray copy path allocates an outer destination before + # recursively materializing operations; count it as well as two temporaries. + arrays = c.backend == :cupynumeric ? 3 : c.backend == :cudajl || c.fusion ? 2 : 4 + # Julia has tracing GC, not Python's reference counting. The returned + # broadcast output is not explicitly destroyed by this baseline kernel. + # This assumes the fully dotted expression in montecarlo.jl. On the unfused + # path nested broadcast temporaries are explicitly destroyed by the runtime; + # an undotted operation would instead escape that cleanup and need its own + # per-iteration retention allowance. + # Bound its retention over the complete trial instead of assuming a GC. + retained = c.backend == :cunumeric || c.backend == :cudajl ? c.steps-1 : 0 + return MemoryEstimate(init, (arrays + retained)*bytes, 0, + "samples + broadcast output; unfused temporaries; up to $retained prior Julia outputs awaiting GC; random dtype conversion") +end + +function memory_estimate(b::GEMM{T}, c::MemoryContext) where {T} + validate_memory_context(b, c) + # Without a mapper-specific replication guarantee count the complete inputs + # on each GPU. This also covers broadcast operands in distributed matmul. + a = big(b.N) * b.M * sizeof(T) + out = big(b.N)^2 * sizeof(T) + init = max(2a + out, a + random_peak(big(b.N)*b.M, T, c.backend)) + return MemoryEstimate(init, 2a + out, library_workspace(b, c), + "A, B, C; full operands per GPU (replication-safe); native packing/workspace") +end + +function memory_estimate(b::AbstractGrayScott{T}, c::MemoryContext) where {T} + validate_memory_context(b, c) + b.N >= 3 && b.M >= 3 || error("Gray-Scott requires N and M >= 3") + # Count full grids until the mapper's halo/replication contract is bounded. + # Local expressions can retain parents; never treat a slice as a free copy. + grid = big(b.N) * b.M * sizeof(T) + interior = big(max(b.N-2, 0)) * max(b.M-2, 0) * sizeof(T) + init = 4grid + random_peak(big(min(150,b.N,b.M))^2, T, c.backend) + # Four named RHS results plus an assignment output. Hard-scope acceleration + # may eliminate these, but this remains a valid upper bound for every form. + # Do not assume a lower peak solely from the @accelerate spelling. + fused = c.backend == :cudajl || (c.backend == :cunumeric && c.fusion) + # Unfused Laplacian: first branch survives evaluation of second branch; + # include outer destination and intermediate binary operands. + temps = fused ? 5 : 8 + variant = name(b) + hard_scope = b isa Union{GrayScottAccelerated,GrayScottFunctionAccelerated,GrayScottLetAccelerated} + # Hard scopes insert explicit last-use destruction whether fusion is on or + # off. Baseline/begin/expression leave the named results for tracing GC. + retained = c.backend == :cunumeric && hard_scope || c.backend == :cupynumeric ? 0 : 6*(c.steps-1) + return MemoryEstimate(init, 4grid + (temps+retained)*interior, 0, + "$variant: four persistent grids + $temps active interior buffers + $retained prior buffers awaiting GC; full-parent bound; fusion=$(c.fusion)") +end + +function memory_estimate(b::AbstractDMD{T}, c::MemoryContext) where {T} + validate_memory_context(b, c) + b.M >= 2 && b.N >= b.M-1 || error("DMD requires M >= 2 and N >= M-1") + n, m, r = big(b.N), big(b.M-1), big(_dmd_rank(b)) + # Retained X, copies/views X1/X2, full thin U/Vt/S, projected products, + # transpose copies, eigen inputs/outputs and complex lift. Count retained + # parents even when only r columns are returned from _dmd_factors. + persistent = n*big(b.M) + factors = 3n*m + m*m + m + project = 2m*r + 4n*r + 2r*r + r + complex_lift = 2*(2n*r + 2r*r + r) + # The factorization/output wrappers escape the lifetime rewriter; native + # arrays can remain until GC between trials on Julia backends. + retained_steps = c.backend == :cupynumeric ? 1 : c.steps + iteration = (persistent + retained_steps*(factors + project + complex_lift))*sizeof(T) + init = random_peak(n*big(b.M), T, c.backend) + return MemoryEstimate(init, iteration, library_workspace(b,c), + "full single-task SVD on one GPU (P does not divide memory); factors, retained parents, projections, complex lift") +end + +function memory_estimate(b::PoissonFFT{T}, c::MemoryContext) where {T} + validate_memory_context(b, c) + e = cld(big(b.M), c.gpus)*big(b.N)^2 + realbytes, complexbytes = e*sizeof(T), e*sizeof(Complex{T}) + kinv = big(b.N)^2*sizeof(T) + # Python FFT precision is conservatively bounded by complex128. + pycomplex = e*sizeof(ComplexF64) + init = max(random_peak(e,T,c.backend), realbytes+2complexbytes+kinv) + iteration = c.backend == :cupynumeric ? realbytes+2pycomplex+kinv : 2complexbytes+kinv + return MemoryEstimate(init, iteration, library_workspace(b,c), + "batched grids + replicated inverse Laplacian; Python out-of-place FFT precision; native FFT workspace") +end + +function memory_estimate(b::AbstractTensorContraction{T}, c::MemoryContext) where {T} + validate_memory_context(b,c) + # Full operands and intermediates, without assuming distributed packing. + n = big(b.N) + if b isa TensorProjection3 + live = (4n^3+n^2)*sizeof(T) + init = max((2n^3+n^2)*sizeof(T), random_peak(n^3,T,c.backend)) + else + live = 3n^4*sizeof(T) + init = max(live, n^4*sizeof(T)+random_peak(n^4,T,c.backend)) + end + return MemoryEstimate(init, live, library_workspace(b,c), + "full contraction inputs/outputs and pairwise intermediates; native packing/workspace counted separately") +end diff --git a/benchmark/src/parse_benchmarks.jl b/benchmark/src/parse_benchmarks.jl index 3f30e44ee..83f3d0ea4 100644 --- a/benchmark/src/parse_benchmarks.jl +++ b/benchmark/src/parse_benchmarks.jl @@ -3,7 +3,8 @@ using TOML """ One benchmark invocation parsed from `benchmarks.toml`. `name` selects the benchmark type from `BENCHMARKS`; `T` is the element type (e.g. "Float32"); -`args` are the sizes (currently `N M`). +`args` are the sizes (`N M`) when pinned. When `autosize` is true, `args` is +unused and `N_hint` / `M_hint` feed `fit_one_gpu`. """ struct BenchmarkSpec name::String @@ -16,6 +17,9 @@ struct BenchmarkSpec n_iter::Int n_trial::Int args::Vector{Int} + autosize::Bool + N_hint::Union{Int,Nothing} + M_hint::Union{Int,Nothing} end # A field may be a scalar or a list. @@ -58,7 +62,18 @@ function declared_order(path) return order end -function parse_config(path) +function size_field(raw) + raw === nothing && return (:omitted, Int[]) + vals = collect(aslist(raw)) + autos = [is_auto_size(v) for v in vals] + if any(autos) + all(autos) || error("cannot mix auto and numeric sizes in one field; got $(repr(raw))") + return (:auto, Int[]) + end + return (:pinned, Int[Int(v) for v in vals]) +end + +function parse_config(path; only=nothing, fusion_override=nothing) raw = TOML.parsefile(path) g = raw["Global"] @@ -68,39 +83,78 @@ function parse_config(path) cuda=get(g, "cuda", false), check_correctness=get(g, "check_correctness", false), n_correctness_iter=get(g, "n_correctness_iter", 5), + auto_size=get(g, "auto_size", false), + mem_frac=Float64(get(g, "mem_frac", 0.5)), ) specs = BenchmarkSpec[] + selected = only === nothing ? nothing : Set(split(only, ',')) + if selected !== nothing + groups = Dict(parse_plot_groups(path)) + selected = Set(vcat([get(groups, s, [s]) for s in selected]...)) + unknown = setdiff(selected, Set(declared_order(path))) + isempty(unknown) || error("Unknown benchmark selection: $(join(unknown, ", "))") + end for name in declared_order(path) + selected !== nothing && name ∉ selected && continue entries = raw[name] + entries isa AbstractVector || continue for e in entries types = aslist(get(e, "T", "Float32")) gpus = aslist(e["gpus"]) cpus = aslist(e["cpus"]) - fusion = aslist(get(e, "fusion", true)) - N = aslist(e["N"]) - M = aslist(get(e, "M", 1)) + fusion = aslist(fusion_override === nothing ? get(e, "fusion", true) : fusion_override) + nmode, nvals = size_field(get(e, "N", nothing)) + mmode, mvals = size_field(get(e, "M", nothing)) cuda = get(e, "cuda", global_settings.cuda) n_warmup = get(e, "n_warmup", global_settings.n_warmup) n_iter = get(e, "n_iter", global_settings.n_iter) n_trial = get(e, "n_trial", global_settings.n_trial) + block_auto = get(e, "auto_size", global_settings.auto_size) - n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus, "N" => N, "M" => M]) + n_auto = nmode == :omitted || nmode == :auto + m_auto = mmode == :auto + use_auto = block_auto && (n_auto || m_auto) + if use_auto + nmode == :pinned && length(nvals) != 1 && error( + "benchmark '$(name)': autosize with pinned N requires a scalar N", + ) + mmode == :pinned && length(mvals) != 1 && error( + "benchmark '$(name)': autosize with pinned M requires a scalar M", + ) + N_hint = nmode == :pinned ? nvals[1] : nothing + M_hint = mmode == :pinned ? mvals[1] : nothing + n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus]) + else + nmode == :auto && error("benchmark '$(name)': N = \"auto\" requires auto_size = true") + mmode == :auto && error("benchmark '$(name)': M = \"auto\" requires auto_size = true") + nmode == :omitted && error( + "benchmark '$(name)' is missing N (set N or enable auto_size)", + ) + mmode == :omitted && (mvals = [1]) + N_hint = nothing + M_hint = nothing + n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus, "N" => nvals, "M" => mvals]) + end for T in types, fuse in fusion, i in 1:n + args = use_auto ? Int[0, 0] : Int[sweep_value(nvals, i), sweep_value(mvals, i)] push!( specs, BenchmarkSpec( name, - T, - sweep_value(gpus, i), - sweep_value(cpus, i), + string(T), + Int(sweep_value(gpus, i)), + Int(sweep_value(cpus, i)), parse_fusion(fuse), cuda, n_warmup, n_iter, n_trial, - [sweep_value(N, i), sweep_value(M, i)], + args, + use_auto, + N_hint, + M_hint, ), ) end @@ -109,3 +163,51 @@ function parse_config(path) return global_settings, specs end + +# Group members that share a figure. Explicit `[plot.groups]` lists first; +# every other `[[benchmark]]` table is a singleton group named after itself. +# Group order follows the first listed member in the file. +function parse_plot_groups(path) + raw = TOML.parsefile(path) + declared = declared_order(path) + plot = get(raw, "plot", Dict{String,Any}()) + groups_tbl = get(plot, "groups", Dict{String,Any}()) + + explicit = Dict{String,Vector{String}}() + for (gname, members) in groups_tbl + explicit[string(gname)] = String[string(m) for m in aslist(members)] + end + + assigned = Set{String}() + groups = Pair{String,Vector{String}}[] + remaining = copy(explicit) + for name in declared + name in assigned && continue + gname = nothing + for (g, members) in remaining + if name in members + gname = g + break + end + end + if gname !== nothing + members = remaining[gname] + push!(groups, gname => members) + union!(assigned, members) + delete!(remaining, gname) + else + push!(groups, name => [name]) + push!(assigned, name) + end + end + for (gname, members) in remaining + push!(groups, gname => members) + end + return groups +end + +function plot_baseline(members::Vector{String}) + i = findfirst(m -> endswith(m, "_baseline"), members) + i !== nothing && return members[i] + return first(members) +end diff --git a/benchmark/src/planning.jl b/benchmark/src/planning.jl new file mode 100644 index 000000000..801cb7edc --- /dev/null +++ b/benchmark/src/planning.jl @@ -0,0 +1,140 @@ +# Shared sizing: explicit plans, no runtime probes and no mutable sizing cache. +struct PlannedRun + spec::BenchmarkSpec + backend::Symbol + N::Int + M::Int + memory::MemoryEstimate +end + +function workspace_bound(raw, name, backend) + entry = get(get(raw, "workspace", Dict()), name, Dict()) + value = get(entry, string(backend), nothing) + value === nothing && return nothing + value isa Integer && value >= 0 || error("workspace.$name.$backend must be a nonnegative byte count") + return Int(value) +end + +function contexts(spec, gs, raw) + backends = Symbol[:cunumeric] + if !endswith(spec.name, "_accelerated") + spec.cuda && spec.gpus == 1 && push!(backends, :cudajl) + gs.cupynumeric && push!(backends, :cupynumeric) + end + return [MemoryContext(; backend, fusion=spec.fusion, gpus=spec.gpus,steps=spec.n_warmup+spec.n_iter, + workspace_bytes=workspace_bound(raw, spec.name, backend)) for backend in backends] +end + +function validate_spec(s) + haskey(BENCHMARKS,s.name) || error("Unknown benchmark $(s.name)") + s.gpus > 0 && s.cpus >= 0 || error("GPU count must be positive and CPU count nonnegative") + s.n_iter > 0 && s.n_trial > 0 && s.n_warmup >= 0 || error("Invalid trial/iteration count") + for hint in (s.N_hint,s.M_hint) + hint === nothing || hint > 0 || error("Pinned dimensions must be positive") + end +end + +function dimensions_at(s, baseline) + !s.autosize && return Tuple(s.args) + n,m = baseline + b = build_benchmark(BENCHMARKS[s.name],parse_bench_type(s.T),n,m) + result = estimate_scaling(b,s.gpus) + result === nothing && error("$(s.name) cannot scale this baseline to $(s.gpus) GPUs") + return result +end + +function baseline_shape(s, k) + B = BENCHMARKS[s.name] + if B <: AbstractDMD + return (something(s.N_hint,k), something(s.M_hint,DEFAULT_DMD_M)) + elseif B <: PoissonFFT + return s.N_hint === nothing ? (k,something(s.M_hint,1)) : (s.N_hint,k) + elseif B <: MonteCarloIntegration || B <: AbstractTensorContraction + s.M_hint === nothing || s.M_hint == 1 || error("$(s.name) requires M=1") + return (something(s.N_hint,k),1) + else + return (something(s.N_hint,k),something(s.M_hint,k)) + end +end + +function candidate_runs(specs,gs,raw,baseline) + runs = PlannedRun[] + seen = Set{Any}() + for s in specs + n,m = dimensions_at(s,baseline) + b = build_benchmark(BENCHMARKS[s.name],parse_bench_type(s.T),n,m) + for c in contexts(s,gs,raw) + # Comparison backends have no fusion setting and run only once. + key = (s.name,s.T,s.gpus,s.cpus,n,m,c.backend, + c.backend == :cunumeric ? s.fusion : nothing,s.n_iter,s.n_warmup,s.n_trial) + key in seen && continue + push!(seen,key) + push!(runs,PlannedRun(s,c.backend,n,m,memory_estimate(b,c))) + end + end + # Multiple explicit blocks must not create misleading overlays. + sizes = Dict{Tuple{String,Int},Tuple{Int,Int}}() + for r in runs + key = (r.spec.T,r.spec.gpus) + previous = get!(sizes,key,(r.N,r.M)) + previous == (r.N,r.M) || error("Comparison group has incompatible pinned dimensions at $(r.spec.gpus) GPUs") + end + return runs +end + +function plan_runs(specs,gs,raw,groups,budget::Integer) + isempty(specs) && error("No benchmarks selected") + foreach(validate_spec,specs) + group_for = Dict(member=>group for (group,members) in groups for member in members) + buckets = Dict{Any,Vector{BenchmarkSpec}}() + order = Any[] + for s in specs + key = (get(group_for,s.name,s.name),s.T) + if !haskey(buckets,key) + buckets[key] = BenchmarkSpec[] + push!(order,key) + end + push!(buckets[key],s) + end + planned = PlannedRun[] + for key in order + members = buckets[key] + autos = filter(s->s.autosize,members) + if isempty(autos) + runs = candidate_runs(members,gs,raw,nothing) + all(r->peak_bytes(r.memory)<=budget,runs) || error("Pinned size exceeds memory budget in $(key[1])") + append!(planned,runs) + continue + end + length(autos)==length(members) || error("Do not mix pinned and automatic sizes in comparison group $(key[1])") + hints = unique((s.N_hint,s.M_hint) for s in members) + length(hints)==1 || error("Incompatible size constraints in comparison group $(key[1])") + s = first(members) + B = BENCHMARKS[s.name] + quantum = B <: PoissonFFT && s.N_hint !== nothing ? 1 : B <: AbstractTensorContraction || B <: PoissonFFT ? 2 : 8 + minimum_n = B <: AbstractDMD ? max(something(s.M_hint,DEFAULT_DMD_M)*DMD_TALL_RATIO,8) : quantum + lo = cld(minimum_n,quantum) + # At least one value of T must fit per candidate. Binary search uses + # BigInt memory formulas and an Int-safe bound on scaled dimensions. + hi = max(lo,Int(min(budget÷sizeof(parse_bench_type(s.T)),typemax(Int)÷(8maximum(x.gpus for x in members))))÷quantum) + make(k) = candidate_runs(members,gs,raw,baseline_shape(s,k*quantum)) + # Evaluate once before search to surface unsupported model errors. + all(r->peak_bytes(r.memory)<=budget,make(lo)) || error("Minimum problem does not fit in $(key[1])") + best = largest_feasible(lo,hi,k->all(r->peak_bytes(r.memory)<=budget,make(k))) + best === nothing && error("No feasible size for $(key[1])") + append!(planned,make(best)) + end + return planned +end + +function print_plan(runs,budget) + println("Per-GPU budget: $budget bytes; sizes are shared within each comparison group.") + for r in runs + m = r.memory + println("$(r.spec.name) / $(r.backend) / $(r.spec.T) fusion=$(r.spec.fusion) GPUs=$(r.spec.gpus) N=$(r.N) M=$(r.M)") + println(" initialization=$(m.initialization), iteration=$(m.iteration), workspace=$(m.workspace), peak=$(peak_bytes(m)) bytes") + println(" $(m.explanation)") + end + limiting = runs[argmax([peak_bytes(r.memory) for r in runs])] + println("Largest planned peak: $(limiting.spec.name) / $(limiting.backend) / $(limiting.spec.gpus) GPUs") +end diff --git a/benchmark/src/result_rows.jl b/benchmark/src/result_rows.jl new file mode 100644 index 000000000..08905de54 --- /dev/null +++ b/benchmark/src/result_rows.jl @@ -0,0 +1,47 @@ +struct Row + gpus::Int + N::Int + M::Int + time_ms::Float64 + thr::Float64 +end + +function load_runs(path) + rows = Row[] + for line in eachline(path) + isempty(strip(line)) && continue + f = split(line, ',') + length(f) == 8 || error("Invalid result row in $path") + push!(rows,Row(parse(Int,f[2]),parse(Int,f[3]),parse(Int,f[4]),parse(Float64,f[6]),parse(Float64,f[7]))) + end + isempty(rows) && return Vector{Row}[] + runs = [Row[]] + for (i,r) in enumerate(rows) + i > 1 && r.gpus < rows[i-1].gpus && push!(runs,Row[]) + push!(runs[end],r) + end + return runs +end + +function aggregate(rows) + by = Dict{Int,Vector{Row}}() + for r in rows + push!(get!(by,r.gpus,Row[]),r) + end + for (g,rs) in by + length(unique((r.N,r.M) for r in rs)) == 1 || error("Cannot combine different dimensions at $g GPUs; select one invocation") + end + sd(x) = length(x)>1 ? std(x) : 0.0 + return [(gpus=g,N=first(by[g]).N,M=first(by[g]).M, + t=mean(getfield.(by[g],:time_ms)),tsd=sd(getfield.(by[g],:time_ms)), + h=mean(getfield.(by[g],:thr)),hsd=sd(getfield.(by[g],:thr))) for g in sort(collect(keys(by)))] +end + +function validate_series_sizes(series) + sizes = Dict{Int,Tuple{Int,Int}}() + for s in series, r in s.agg + previous = get!(sizes,r.gpus,(r.N,r.M)) + previous == (r.N,r.M) || error("Comparison series use different dimensions at $(r.gpus) GPUs") + end + return nothing +end diff --git a/benchmark/src/runner.jl b/benchmark/src/runner.jl new file mode 100644 index 000000000..3b848c5c6 --- /dev/null +++ b/benchmark/src/runner.jl @@ -0,0 +1,166 @@ +using Dates, Pkg + +function cli_options(args) + config = joinpath(@__DIR__,"..","benchmarks.toml") + only = nothing + fusion = nothing + dry = false + verbose = false + positional = String[] + for arg in args + if startswith(arg,"--only=") + only = split(arg,'=';limit=2)[2] + elseif startswith(arg,"--config=") + config = abspath(split(arg,'=';limit=2)[2]) + elseif startswith(arg,"--fusion=") + value = split(arg,'=';limit=2)[2] + fusion = value == "both" ? [true,false] : [parse_fusion(value)] + elseif arg == "--dry-run" + dry = true + elseif arg in ("-v","--verbose") + verbose = true + elseif startswith(arg,"--") + error("Unknown option $arg") + else + push!(positional,arg) + end + end + return (;config,only,fusion,dry,verbose,positional) +end + +function positional_spec(p,gs) + 9 <= length(p) <= 12 || error("Expected [fusion] [check_correctness] [correctness_iter]") + n = is_auto_size(p[5]) ? nothing : parse(Int,p[5]) + m = is_auto_size(p[6]) ? nothing : parse(Int,p[6]) + auto = n === nothing || m === nothing + return BenchmarkSpec(p[3],p[4],parse(Int,p[1]),parse(Int,p[2]), + length(p)>=10 ? parse_fusion(p[10]) : true,gs.cuda, + parse(Int,p[8]),parse(Int,p[7]),parse(Int,p[9]), + auto ? [0,0] : [n,m],auto,n,m) +end + +function plan_manifest(runs,budget,raw) + versions = Dict(info.name=>string(info.version) for info in values(Pkg.dependencies()) if info.version !== nothing) + return Dict("status"=>"running","budget_bytes"=>budget,"julia"=>string(VERSION), + "versions"=>versions,"config"=>raw,"runs"=>[Dict{String,Any}( + "name"=>r.spec.name,"T"=>r.spec.T,"backend"=>string(r.backend), + "fusion"=>r.spec.fusion,"gpus"=>r.spec.gpus,"cpus"=>r.spec.cpus, + "N"=>r.N,"M"=>r.M,"n_iter"=>r.spec.n_iter,"n_warmup"=>r.spec.n_warmup, + "n_trial"=>r.spec.n_trial,"initialization_bytes"=>string(r.memory.initialization), + "iteration_bytes"=>string(r.memory.iteration),"workspace_bytes"=>string(r.memory.workspace), + "memory_explanation"=>r.memory.explanation,"status"=>"pending") for r in runs]) +end + +function prepare_backend(fusion,verbose) + println("Setting fusion=$fusion and precompiling cuNumeric...") + CNPreferences.set_broadcast_fusion!(fusion) + Pkg.precompile("cuNumeric";io=verbose ? stderr : devnull) +end + +function preflight_backends(runs;env=ENV,which=Sys.which,check=success) + any(r.backend==:cupynumeric for r in runs) || return nothing + conda = get(env,"CUNUMERIC_BENCH_CONDA",get(env,"CONDA_EXE","conda")) + executable = which(conda) + executable === nothing && error( + "cuPyNumeric is enabled, but conda is not available to the worker. " * + "Add conda to PATH or set CUNUMERIC_BENCH_CONDA to its executable path; " * + "then run bash install_cupynumeric.sh. No benchmarks have been started.", + ) + name = get(env,"CUPYNUMERIC_ENV",nothing) + name === nothing && (name = cupynumeric_env_name()) + code = "import importlib.util,sys; sys.exit(0 if importlib.util.find_spec('cupynumeric') else 1)" + check(`$executable run --no-capture-output -n $name python -c $code`) || error( + "Conda environment '$name' is unavailable or lacks cupynumeric. " * + "Run bash install_cupynumeric.sh, or set CUPYNUMERIC_ENV to an existing environment. " * + "No benchmarks have been started.", + ) + return nothing +end + +function execute_plan(runs,gs,opts,budget,raw;launch=run,prepare=prepare_backend, + results_root=normpath(joinpath(@__DIR__,"..","results")),preflight=preflight_backends) + preflight(runs) + root = normpath(joinpath(@__DIR__,"..")) + mkpath(results_root) + dir = mktempdir(results_root;prefix=Dates.format(now(),"yyyymmdd-HHMMSS")*"-",cleanup=false) + manifest = plan_manifest(runs,budget,raw) + manifest_path = joinpath(dir,"manifest.toml") + save_manifest() = open(io->TOML.print(io,manifest),manifest_path,"w") + save_manifest() + last_fusion = nothing + failed = false + for (i,r) in enumerate(runs) + s = r.spec + println("\n[$i/$(length(runs))] $(s.name) / $(r.backend), $(s.gpus) GPUs, $(r.N) × $(r.M)") + try + if r.backend == :cunumeric && last_fusion != s.fusion + prepare(s.fusion,opts.verbose) + last_fusion = s.fusion + end + b = build_benchmark(BENCHMARKS[s.name],parse_bench_type(s.T),r.N,r.M) + p = opts.positional + correctness = length(p)>=11 ? parse(Bool,p[11]) : gs.check_correctness + correct_iters = length(p)>=12 ? parse(Int,p[12]) : gs.n_correctness_iter + args = `--gpus $(s.gpus) --cpus $(s.cpus) $(s.name) $(s.T) $(r.N) $(r.M) $(s.n_iter) $(s.n_warmup) $(s.n_trial)` + corr = `$correctness $correct_iters $(total_flops(b))` + runner = joinpath(root,"run_benchmark.sh") + verbose = opts.verbose ? `--verbose` : `` + cmd = if r.backend == :cupynumeric + worker = joinpath(root,"src_py","single.py") + `bash $runner $worker $verbose --pyenv $(cupynumeric_env_name()) $args $corr` + else + worker = joinpath(root,"src","single.jl") + `bash $runner $worker $verbose $args $(string(r.backend)) $corr` + end + results = joinpath(dir,s.T) + launch(addenv(Cmd(cmd;dir=root),"CUNUMERIC_BENCH_RESULTS_DIR"=>results, + "CUNUMERIC_BENCH_JULIA"=>joinpath(Sys.BINDIR,Base.julia_exename()))) + manifest["runs"][i]["status"] = "complete" + catch e + failed = true + manifest["runs"][i]["status"] = "failed" + # ProcessFailedException prints inherited environment variables. + # Report exit codes without copying that environment into logs. + message = e isa ProcessFailedException ? + "Worker exited with code(s) " * join((p.exitcode for p in e.procs),", ") : sprint(showerror,e) + manifest["runs"][i]["error"] = message + @error "Worker failed; no size retry. Continuing independent configurations." reason=message + end + save_manifest() + end + manifest["status"] = failed ? "incomplete" : "complete" + if isempty(opts.positional) + for T in unique(r.spec.T for r in runs) + try + plotter = joinpath(root,"plot_results.jl") + results = joinpath(dir,T) + out = joinpath(root,"plots",basename(dir),T) + suffix = failed ? "_incomplete" : "" + launch(`$(Base.julia_cmd()) --project=$root $plotter $results --config=$(opts.config) --out=$out --suffix=$suffix`) + catch e + failed = true + manifest["status"] = "incomplete" + @error "Plotting failed" exception=e + end + end + end + save_manifest() + println("Results: $dir ($(manifest["status"]))") + return failed ? 1 : 0 +end + +function main(args=ARGS;budget_provider=selected_gpu_budget,executor=execute_plan) + opts = cli_options(args) + gs,specs = parse_config(opts.config;only=opts.only,fusion_override=opts.fusion) + if !isempty(opts.positional) + opts.only === nothing && opts.fusion === nothing || error("Do not mix positional runs with sweep filters") + specs = [positional_spec(opts.positional,gs)] + end + isempty(specs) && error("No benchmarks selected") + raw = TOML.parsefile(opts.config) + budget,_ = budget_provider(gs.mem_frac,maximum(s.gpus for s in specs)) + runs = plan_runs(specs,gs,raw,parse_plot_groups(opts.config),budget) + opts.dry && print_plan(runs,budget) + opts.dry && return 0 + return executor(runs,gs,opts,budget,raw) +end diff --git a/benchmark/src/single.jl b/benchmark/src/single.jl index ec672014c..ff66e2390 100644 --- a/benchmark/src/single.jl +++ b/benchmark/src/single.jl @@ -6,29 +6,53 @@ # run.jl sets the compile-time fusion pref before launch; we read it back to label results. using cuNumeric -using CUDACore using LinearAlgebra using TensorOperations -using cuTENSOR + +length(ARGS) >= 9 || error( + "single.jl args: " * + "[check_correctness] [n_correctness_iter]", +) + +const GPUS = parse(Int, ARGS[1]) +const BACKEND_NAME = ARGS[9] +const CHECK_CORRECTNESS = length(ARGS) >= 10 ? parse(Bool, ARGS[10]) : false + +# Timed CUDA.jl worker always needs CUDA. The 1-GPU cuNumeric worker loads it +# only for the tiny oracle check. +const NEED_CUDA = + BACKEND_NAME == "cudajl" || + (BACKEND_NAME == "cunumeric" && CHECK_CORRECTNESS && GPUS == 1) + +if NEED_CUDA + using CUDA + using AbstractFFTs + using cuTENSOR +end include("core.jl") -const BENCHMARK_DIR = joinpath(@__DIR__, "benchmarks") -include.(filter(contains(r".jl$"), readdir(BENCHMARK_DIR; join=true))) +include_benchmarks() # Resolve a TOML type string like "Float32" to the actual Julia type. parse_type(s) = getfield(Base, Symbol(s))::DataType -cuda_clock() = (CUDACore.synchronize(; blocking=true); time_ns() / 1e3) - -# mod runs the kernels; label tags stdout; save_as names the results CSV. -const BACKENDS = Dict( - "cunumeric" => ( - mod=cuNumeric, label="cuNumeric", save_as="cunumeric", clock=get_time_microseconds - ), - "cudajl" => ( - mod=CUDACore, label="CUDA.jl", save_as="CUDA.jl", clock=cuda_clock - ), -) +# One worker process, one backend. Clock functions have distinct types, so do +# not stash them in a Dict inferred from the cuNumeric entry. +function backend_entry(name) + name == "cunumeric" && return ( + mod=cuNumeric, label="cuNumeric", save_as="cunumeric", + clock=get_time_microseconds, + synchronize=benchmark_synchronize, + ) + if name == "cudajl" + cuda_sync() = CUDA.synchronize(; blocking=true) + cuda_clock() = (cuda_sync(); time_ns() / 1e3) + return (mod=CUDA, label="CUDA.jl", save_as="CUDA.jl", clock=cuda_clock, + synchronize=cuda_sync) + end + return error("Unknown backend '$(name)'. Known: cunumeric, cudajl") +end +const BACKEND = backend_entry(BACKEND_NAME) function run_single( gpus, name, T_str, N, M, n_iter, n_warmup, n_trial, backend; @@ -37,10 +61,7 @@ function run_single( haskey(BENCHMARKS, name) || error( "No benchmark registered for '$(name)'. Known: $(join(sort(collect(keys(BENCHMARKS))), ", "))" ) - haskey(BACKENDS, backend) || error( - "Unknown backend '$(backend)'. Known: $(join(sort(collect(keys(BACKENDS))), ", "))" - ) - bk = BACKENDS[backend] + bk = BACKEND T = parse_type(T_str) b = build_benchmark(BENCHMARKS[name], T, N, M) @@ -55,6 +76,7 @@ function run_single( n_warmup=n_warmup, n_iter=n_iter, n_trial=n_trial, + n_gpu=gpus, check_correctness=check_correctness, n_correctness_iter=n_correctness_iter, ) @@ -63,14 +85,13 @@ function run_single( "[$(label)] $(name) benchmark ($(T)) on $(N)x$(M) for $(n_iter) " * "iterations ($(n_warmup) warmup) x $(n_trial) trials", ) - br = run_benchmark(b, gs; mod=bk.mod, clock=bk.clock) + br = run_benchmark(b, gs; mod=bk.mod, clock=bk.clock, synchronize=bk.synchronize) @printf("[%s] Mean Run Time: %.5f ± %.5f ms\n", label, mean(br.times_ms), _std(br.times_ms)) @printf("[%s] FLOPS: %.5f ± %.5f GFLOPS\n", label, mean(br.gflops), _std(br.gflops)) println("[$(label)] Correctness: $(br.correctness)") return save_result(br, gpus; mod=save_as) end -gpus = parse(Int, ARGS[1]) bench_name = ARGS[2] T_str = ARGS[3] N = parse(Int, ARGS[4]) @@ -78,10 +99,8 @@ M = parse(Int, ARGS[5]) n_iter = parse(Int, ARGS[6]) n_warmup = parse(Int, ARGS[7]) n_trial = parse(Int, ARGS[8]) -backend = ARGS[9] -check_correctness = length(ARGS) >= 10 ? parse(Bool, ARGS[10]) : false n_correctness_iter = length(ARGS) >= 11 ? parse(Int, ARGS[11]) : 5 run_single( - gpus, bench_name, T_str, N, M, n_iter, n_warmup, n_trial, backend; - check_correctness=check_correctness, n_correctness_iter=n_correctness_iter, + GPUS, bench_name, T_str, N, M, n_iter, n_warmup, n_trial, BACKEND_NAME; + check_correctness=CHECK_CORRECTNESS, n_correctness_iter=n_correctness_iter, ) diff --git a/benchmark/src_py/benchmarks/dmd.py b/benchmark/src_py/benchmarks/dmd.py index e0b74180c..5091e1f1f 100644 --- a/benchmark/src_py/benchmarks/dmd.py +++ b/benchmark/src_py/benchmarks/dmd.py @@ -1,6 +1,6 @@ import cupynumeric as np -from core import register_benchmark +from core import register_benchmark, rand_array class DMD: @@ -13,23 +13,8 @@ def __init__(self, T, N, M): def dims(self): return self.N, self.M - def total_flops(self): - # Same breakdown as benchmark/src/benchmarks/dmd.jl - m = self.N - n = self.M - 1 - r = self.r - return ( - 2 * m * n * n - + 11 * n * n * n - + 2 * m * n * r - + m * r - + 2 * m * r * r - + 25 * r * r * r - + 2 * m * r * r - ) - def initialize(self): - X = np.random.rand(self.N, self.M).astype(self.T) + X = rand_array((self.N, self.M), self.T) return (X,) def run(self, state): diff --git a/benchmark/src_py/benchmarks/gemm.py b/benchmark/src_py/benchmarks/gemm.py index b5d1a4b3d..74288a9c9 100644 --- a/benchmark/src_py/benchmarks/gemm.py +++ b/benchmark/src_py/benchmarks/gemm.py @@ -1,6 +1,6 @@ import cupynumeric as np -from core import register_benchmark +from core import register_benchmark, rand_array, zeros_array class GEMM: @@ -12,13 +12,10 @@ def __init__(self, T, N, M): def dims(self): return self.N, self.M - def total_flops(self): - return self.N * self.N * (2 * self.M - 1) - def initialize(self): - A = np.random.rand(self.N, self.M).astype(self.T) - B = np.random.rand(self.M, self.N).astype(self.T) - C = np.zeros((self.N, self.N), dtype=self.T) + A = rand_array((self.N, self.M), self.T) + B = rand_array((self.M, self.N), self.T) + C = zeros_array((self.N, self.N), self.T) return (C, A, B) def run(self, state): diff --git a/benchmark/src_py/benchmarks/grayscott.py b/benchmark/src_py/benchmarks/grayscott.py index a1a89e739..931acb7a6 100644 --- a/benchmark/src_py/benchmarks/grayscott.py +++ b/benchmark/src_py/benchmarks/grayscott.py @@ -1,10 +1,11 @@ import cupynumeric as np -from core import register_benchmark +from core import register_benchmark, rand_array, zeros_array, ones_array class GrayScott: - name = "grayscott" + name = "grayscott_baseline" + fence_each_iteration = False # Timesteps belong to one trajectory. # dt = dx/5; c_u, c_v, f, k as in grayscott.jl's GSParams defaults. def __init__(self, T, N, M, dx=1.0, c_u=1.0, c_v=0.3, f=0.03, k=0.06): @@ -16,19 +17,15 @@ def __init__(self, T, N, M, dx=1.0, c_u=1.0, c_v=0.3, f=0.03, k=0.06): def dims(self): return self.N, self.M - def total_flops(self): - return self.N * self.M - def initialize(self): - d = (self.N, self.M) - u = np.ones(d, dtype=self.T) - v = np.zeros(d, dtype=self.T) - u_new = np.zeros(d, dtype=self.T) - v_new = np.zeros(d, dtype=self.T) + u = ones_array((self.N, self.M), self.T) + v = zeros_array((self.N, self.M), self.T) + u_new = zeros_array((self.N, self.M), self.T) + v_new = zeros_array((self.N, self.M), self.T) seed = min(150, self.N, self.M) - u[:seed, :seed] = np.random.rand(seed, seed).astype(self.T) - v[:seed, :seed] = np.random.rand(seed, seed).astype(self.T) + u[:seed, :seed] = rand_array((seed, seed), self.T) + v[:seed, :seed] = rand_array((seed, seed), self.T) # mutable list so run() can swap buffers in place return [u, v, u_new, v_new] diff --git a/benchmark/src_py/benchmarks/montecarlo.py b/benchmark/src_py/benchmarks/montecarlo.py index 370fc7b98..14a407cc1 100644 --- a/benchmark/src_py/benchmarks/montecarlo.py +++ b/benchmark/src_py/benchmarks/montecarlo.py @@ -1,6 +1,6 @@ import cupynumeric as np -from core import register_benchmark +from core import register_benchmark, rand_array class MonteCarlo: @@ -13,11 +13,8 @@ def __init__(self, T, N, M): def dims(self): return self.n_samples, 1 - def total_flops(self): - return self.n_samples - def initialize(self): - x = (self.T(10) * np.random.rand(self.n_samples)).astype(self.T) + x = (self.T(10) * rand_array(self.n_samples, self.T)) return (x,) def run(self, state): diff --git a/benchmark/src_py/benchmarks/poisson_fft.py b/benchmark/src_py/benchmarks/poisson_fft.py index ac952a12c..3ff23a93d 100644 --- a/benchmark/src_py/benchmarks/poisson_fft.py +++ b/benchmark/src_py/benchmarks/poisson_fft.py @@ -3,7 +3,7 @@ import cupynumeric as np import numpy as onp -from core import register_benchmark +from core import register_benchmark, rand_array def _integer_fftfreq(n): @@ -32,15 +32,8 @@ def __init__(self, T, N, M): def dims(self): return self.N, self.M - def total_flops(self): - # Same breakdown as benchmark/src/benchmarks/poisson_fft.jl - n = self.N - m = self.M - n2 = n * n - return m * (20 * n2 * math.log2(n) + 6 * n2) - def initialize(self): - f = np.random.rand(self.M, self.N, self.N).astype(self.T) + f = rand_array((self.M, self.N, self.N), self.T) kinv = np.array(_poisson_inv_laplacian(self.T, self.N)).reshape( 1, self.N, self.N ) diff --git a/benchmark/src_py/benchmarks/tensor_contractions.py b/benchmark/src_py/benchmarks/tensor_contractions.py index 6f3cd1e0d..d865ea825 100644 --- a/benchmark/src_py/benchmarks/tensor_contractions.py +++ b/benchmark/src_py/benchmarks/tensor_contractions.py @@ -1,6 +1,6 @@ import cupynumeric as np -from core import register_benchmark +from core import register_benchmark, rand_array, zeros_array def _optimal_path(expression, *operands): @@ -22,13 +22,10 @@ def __init__(self, T, N, M): def dims(self): return self.N, 1 - def total_flops(self): - return 3 * self.N**3 * (2 * self.N - 1) - def initialize(self): - A = np.random.rand(self.N, self.N, self.N).astype(self.T) - B = np.random.rand(self.N, self.N).astype(self.T) - D = np.zeros((self.N, self.N, self.N), dtype=self.T) + A = rand_array((self.N, self.N, self.N), self.T) + B = rand_array((self.N, self.N), self.T) + D = zeros_array((self.N, self.N, self.N), self.T) path = _optimal_path(self.expression, A, B, B, B) return D, A, B, path @@ -49,13 +46,10 @@ def __init__(self, T, N, M): def dims(self): return self.N, 1 - def total_flops(self): - return self.N**4 * (2 * self.N**2 - 1) - def initialize(self): - X = np.random.rand(self.N, self.N, self.N, self.N).astype(self.T) - Y = np.random.rand(self.N, self.N, self.N, self.N).astype(self.T) - C = np.zeros((self.N, self.N, self.N, self.N), dtype=self.T) + X = rand_array((self.N, self.N, self.N, self.N), self.T) + Y = rand_array((self.N, self.N, self.N, self.N), self.T) + C = zeros_array((self.N, self.N, self.N, self.N), self.T) path = _optimal_path(self.expression, X, Y) return C, X, Y, path diff --git a/benchmark/src_py/core.py b/benchmark/src_py/core.py index 1632e5999..3bb872d19 100644 --- a/benchmark/src_py/core.py +++ b/benchmark/src_py/core.py @@ -2,10 +2,11 @@ import math import cupynumeric as np +from legate.core import get_legate_runtime from legate.timing import time # blocks on preceding legate ops; returns microseconds MOD = "cupynumeric" -RESULTS_DIR = os.path.join(os.path.dirname(__file__), "..", "results") +RESULTS_DIR = os.environ.get("CUNUMERIC_BENCH_RESULTS_DIR", os.path.join(os.path.dirname(__file__), "..", "results")) DTYPES = {"Float32": np.float32, "Float64": np.float64} @@ -16,6 +17,24 @@ def parse_type(s): return DTYPES[s] +def _shape(shape): + if isinstance(shape, int): + return (shape,) + return tuple(shape) + + +def rand_array(shape, dtype): + return np.random.rand(*_shape(shape)).astype(dtype) + + +def zeros_array(shape, dtype): + return np.zeros(_shape(shape), dtype=dtype) + + +def ones_array(shape, dtype): + return np.ones(_shape(shape), dtype=dtype) + + BENCHMARKS = {} @@ -23,17 +42,21 @@ def register_benchmark(key, cls): BENCHMARKS[key] = cls -def trial(bench, n_warmup, n_iter): +def trial(bench, n_warmup, n_iter, flops): state = bench.initialize() + fence_each = getattr(bench, "fence_each_iteration", True) + synchronize = get_legate_runtime().issue_execution_fence start = None for idx in range(n_warmup + n_iter): if idx == n_warmup: start = time() bench.run(state) + if fence_each: + synchronize(block=True) total_us = time() - start mean_time_ms = total_us / (n_iter * 1e3) - gflops = bench.total_flops() / (mean_time_ms * 1e6) + gflops = flops / (mean_time_ms * 1e6) return mean_time_ms, gflops @@ -48,10 +71,10 @@ def _std(x): return math.sqrt(sum((v - m) ** 2 for v in x) / (len(x) - 1)) -def save_result(name, dims, gpus, times_ms, gflops): +def save_result(name, dims, gpus, times_ms, gflops, correctness="skipped"): os.makedirs(RESULTS_DIR, exist_ok=True) N, M = dims path = os.path.join(RESULTS_DIR, f"{name}_{MOD}.csv") with open(path, "a") as io: for i, (t, g) in enumerate(zip(times_ms, gflops), start=1): - io.write(f"{MOD},{gpus},{N},{M},{i},{t:.6f},{g:.6f},skipped\n") + io.write(f"{MOD},{gpus},{N},{M},{i},{t:.6f},{g:.6f},{correctness}\n") diff --git a/benchmark/src_py/single.py b/benchmark/src_py/single.py index 005cda31b..30a8afea4 100644 --- a/benchmark/src_py/single.py +++ b/benchmark/src_py/single.py @@ -1,5 +1,7 @@ # cupynumeric worker, run by run_benchmark.sh which sets LEGATE_CONFIG first. # Args: +# [check_correctness] [n_correctness_iter] [flops] +# flops comes from the Julia orchestrator (same total_flops as the kernel file). import os import sys @@ -19,6 +21,12 @@ def main(): n_iter = int(sys.argv[6]) n_warmup = int(sys.argv[7]) n_trial = int(sys.argv[8]) + if len(sys.argv) < 12: + raise SystemExit( + "single.py args: " + " " + ) + flops = float(sys.argv[11]) if name not in BENCHMARKS: raise ValueError( @@ -29,19 +37,22 @@ def main(): print( f"[{MOD}] {name} benchmark ({T_str}) on {N}x{M} for {n_iter} " - f"iterations ({n_warmup} warmup) x {n_trial} trials" + f"iterations ({n_warmup} warmup) x {n_trial} trials; " + + ("per-iteration synchronization" if getattr(bench, "fence_each_iteration", True) + else "batch synchronization") ) times_ms, gflops = [], [] for _ in range(n_trial): - t, g = trial(bench, n_warmup, n_iter) + t, g = trial(bench, n_warmup, n_iter, flops) times_ms.append(t) gflops.append(g) print(f"[{MOD}] Mean Run Time: {_mean(times_ms):.5f} ± {_std(times_ms):.5f} ms") print(f"[{MOD}] FLOPS: {_mean(gflops):.5f} ± {_std(gflops):.5f} GFLOPS") + print(f"[{MOD}] Correctness: skipped") - save_result(bench.name, bench.dims(), gpus, times_ms, gflops) + save_result(bench.name, bench.dims(), gpus, times_ms, gflops, "skipped") if __name__ == "__main__": diff --git a/benchmark/test/autosize.jl b/benchmark/test/autosize.jl new file mode 100644 index 000000000..ece524e0f --- /dev/null +++ b/benchmark/test/autosize.jl @@ -0,0 +1,26 @@ +using Test + +# Run with julia --project=benchmark benchmark/test/autosize.jl; no GPU needed. +include("../src/core.jl") +include("../src/benchmarks/montecarlo.jl") + +@testset "Fused Monte Carlo autosizing reserves broadcast output" begin + budget = 113_066_115_072 # Budget from the reported initialization OOM. + for T in (Float32, Float64) + N, M = fit_one_gpu(MonteCarloIntegration, T; budget) + @test M == 1 + @test N % 8 == 0 + # Check actual array requirements, independently of total_space. + @test 2 * N * sizeof(T) <= budget + @test 2 * (N + 8) * sizeof(T) > budget + for P in (1, 2, 4, 8) + n, m = estimate_scaling(MonteCarloIntegration{T}(; n_samples=N), P) + @test n == N * P + @test m == 1 + @test 2 * n * sizeof(T) <= P * budget + end + @test_throws ErrorException fit_one_gpu( + MonteCarloIntegration, T; budget=2 * 8 * sizeof(T) - 1, + ) + end +end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl new file mode 100644 index 000000000..0cae2fb91 --- /dev/null +++ b/benchmark/test/runtests.jl @@ -0,0 +1,198 @@ +using Test, Statistics, TOML +include("../src/core.jl") +include_benchmarks() +include("../src/parse_benchmarks.jl") +include("../src/memory.jl") +include("../src/planning.jl") +include("../src/runner.jl") +include("../src/result_rows.jl") +include("timing.jl") + +const CONFIG = joinpath(@__DIR__,"..","benchmarks.toml") +const RAW = TOML.parsefile(CONFIG) +const GROUPS = parse_plot_groups(CONFIG) + +# A CPU array that counts materialized arrays, exercising Julia's actual +# broadcast lowering rather than checking the kernel's source spelling. +const MATERIALIZATIONS = Ref(0) +struct CountedArray{T,N} <: AbstractArray{T,N} + data::Array{T,N} +end +struct CountedStyle{N} <: Base.Broadcast.AbstractArrayStyle{N} end +CountedStyle{N}(::Val{M}) where {N,M} = CountedStyle{M}() +Base.size(a::CountedArray) = size(a.data) +Base.getindex(a::CountedArray,I...) = getindex(a.data,I...) +Base.setindex!(a::CountedArray,v,I...) = setindex!(a.data,v,I...) +Base.BroadcastStyle(::Type{<:CountedArray{T,N}}) where {T,N} = CountedStyle{N}() +function Base.similar(bc::Base.Broadcast.Broadcasted{CountedStyle{N}},::Type{T}) where {N,T} + MATERIALIZATIONS[] += 1 + return CountedArray(Array{T}(undef,map(length,axes(bc)))) +end +function Base.similar(a::CountedArray,::Type{T},dims::Dims) where {T} + MATERIALIZATIONS[] += 1 + return CountedArray(Array{T}(undef,dims)) +end + +@testset "Monte Carlo broadcasts fuse across negation" begin + for T in (Float32,Float64) + data = T[0,0.5,1,2,5] + x = CountedArray(data) + b = MonteCarloIntegration{T}(;n_samples=length(x)) + MATERIALIZATIONS[] = 0 + got = run!(b,x) + @test MATERIALIZATIONS[] == 1 + @test got ≈ (T(10)/length(data))*sum(exp(-v^2) for v in data) + # Reproduce the old expression to prove this test detects the bug. + MATERIALIZATIONS[] = 0 + sum(exp.(-x .^ 2)) + @test MATERIALIZATIONS[] == 3 + end +end + +@testset "cuPyNumeric preflight" begin + gs = GlobalSettings(;n_warmup=1,n_iter=1,cupynumeric=true) + s = BenchmarkSpec("montecarlo","Float32",1,8,true,false,1,1,1,[0,0],true,nothing,nothing) + runs = plan_runs([s],gs,RAW,GROUPS,1000000) + env = Dict("CUNUMERIC_BENCH_CONDA"=>"/test/conda","CUPYNUMERIC_ENV"=>"testenv") + @test_throws ErrorException preflight_backends(runs;env,which=x->nothing) + @test_throws ErrorException preflight_backends(runs;env,which=identity,check=c->false) + @test preflight_backends(runs;env,which=identity,check=c->true) === nothing +end + +function spec(name;T="Float32",gpus=1,fusion=true,cuda=false,N=nothing,M=nothing,auto=true) + BenchmarkSpec(name,T,gpus,8,fusion,cuda,2,5,2, + auto ? [0,0] : [N,M],auto,N,M) +end + +@testset "Configuration and CLI" begin + gs,ss = parse_config(CONFIG;only="grayscott",fusion_override=[true,false]) + @test all(startswith(s.name,"grayscott") for s in ss) + @test Set(s.fusion for s in ss)==Set([true,false]) + @test_throws ErrorException parse_config(CONFIG;only="missing") + o = cli_options(["--only=montecarlo","--fusion=both","--dry-run"]) + @test o.only == "montecarlo" && o.dry && o.fusion == [true,false] + @test_throws ErrorException cli_options(["--typo"]) + p = positional_spec(["1","8","montecarlo","Float32","auto","1","5","2","2"],gs) + @test p.autosize && p.M_hint==1 && p.n_iter==5 + @test main(["--only=montecarlo","--dry-run"]; + budget_provider=(f,p)->(1_000_000,f),executor=(args...)->error("dry-run launched workers"))==0 +end + +@testset "Memory dispatch matrix" begin + for T in (Float32,Float64), fusion in (false,true), backend in (:cunumeric,:cudajl,:cupynumeric) + c = MemoryContext(;backend,fusion,workspace_bytes=0) + for (name,B) in BENCHMARKS + endswith(name,"_accelerated") && backend != :cunumeric && continue + m = B <: AbstractDMD ? 16 : B <: AbstractGrayScott || B <: GEMM ? 64 : 1 + b = build_benchmark(B,T,64,m) + estimate = memory_estimate(b,c) + @test estimate.initialization>0 && estimate.iteration>0 + @test !isempty(estimate.explanation) + end + end + b = MonteCarloIntegration{Float32}(;n_samples=1024) + @test peak_bytes(memory_estimate(b,MemoryContext())) == 8192 + @test peak_bytes(memory_estimate(b,MemoryContext(;backend=:cupynumeric))) == 12288 + @test peak_bytes(memory_estimate(b,MemoryContext(;fusion=false))) == 16384 + @test_throws ErrorException memory_estimate(GEMM{Float32}(;N=64,M=64),MemoryContext()) + @test_throws ErrorException memory_estimate(b,MemoryContext(;backend=:cudajl,gpus=2)) + d = DMDBaseline{Float32}(;N=1024,M=16) + @test peak_bytes(memory_estimate(d,MemoryContext(;gpus=1,workspace_bytes=0))) == + peak_bytes(memory_estimate(d,MemoryContext(;gpus=8,workspace_bytes=0))) +end + +@testset "Shared sweep planning" begin + gs = GlobalSettings(;n_warmup=2,n_iter=5,cupynumeric=true) + ss = [spec("montecarlo";gpus=p,fusion=f,cuda=true) for p in (1,2,4,8) for f in (true,false)] + runs = plan_runs(ss,gs,RAW,GROUPS,1_000_000) + @test length(runs)==8+4+1 + for p in (1,2,4,8) + @test length(unique((r.N,r.M) for r in runs if r.spec.gpus==p))==1 + end + @test all(peak_bytes(r.memory)<=1_000_000 for r in runs) + @test maximum(r.N for r in runs)==8minimum(r.N for r in runs) + fused = plan_runs([spec("montecarlo")],gs,RAW,GROUPS,1_000_000) + @test first(fused).N >= first(runs).N + onlyoff = plan_runs([spec("montecarlo";fusion=false)],gs,RAW,GROUPS,1_000_000) + @test any(r.backend==:cupynumeric for r in onlyoff) + @test_throws ErrorException plan_runs([spec("montecarlo";N=1000000,M=1,auto=false)],gs,RAW,GROUPS,100) + @test_throws ErrorException plan_runs([spec("montecarlo";N=8,M=1,auto=false),spec("montecarlo";N=16,M=1,auto=false)],gs,RAW,GROUPS,100000) + @test_throws ErrorException plan_runs([spec("montecarlo";gpus=0)],gs,RAW,GROUPS,100000) + raw = deepcopy(RAW) + raw["workspace"] = Dict("dmd_baseline"=>Dict("cunumeric"=>0)) + single = plan_runs([spec("dmd_baseline";M=16)],GlobalSettings(;n_warmup=1,n_iter=1),raw,GROUPS,10_000_000) + sweep = plan_runs([spec("dmd_baseline";M=16,gpus=p) for p in (1,8)],GlobalSettings(;n_warmup=1,n_iter=1),raw,GROUPS,10_000_000) + @test first(sweep).N < first(single).N + @test all(peak_bytes(r.memory)<=10_000_000 for r in sweep) +end + +@testset "GPU budgets" begin + inventory = "0, GPU-aaa, 1000, 900\n1, GPU-bbb, 2000, 1900\n" + kw = (;inventory,fraction="0.75",fbmem=nothing) + @test first(selected_gpu_budget(.75,1;kw...,visibility="GPU-bbb")) == 1500*1024^2 + @test first(selected_gpu_budget(.75,2;kw...,visibility="1,0")) == 750*1024^2 + @test_throws ErrorException selected_gpu_budget(.75,2;kw...,visibility="1") + @test_throws ErrorException selected_gpu_budget(.75,1;kw...,visibility="") + @test_throws ErrorException selected_gpu_budget(.75,1;kw...,visibility="2") +end + +@testset "Result dimension isolation" begin + rows = [Row(1,32,1,1.,2.),Row(1,32,1,3.,4.)] + a = aggregate(rows) + @test only(a).N==32 && only(a).t==2 + @test_throws ErrorException aggregate(vcat(rows,[Row(1,64,1,1.,2.)])) + @test_throws ErrorException validate_series_sizes([(agg=a,),(agg=aggregate([Row(1,64,1,1.,2.)]),)]) +end + +@testset "Variant lifetimes and rectangular constraints" begin + baseline = GrayScottBaseline{Float32}(;N=64,M=32) + accelerated = GrayScottFunctionAccelerated{Float32}(;N=64,M=32) + for f in (true,false) + c = MemoryContext(;fusion=f,steps=10) + @test peak_bytes(memory_estimate(accelerated,c)) < peak_bytes(memory_estimate(baseline,c)) + end + @test peak_bytes(memory_estimate(baseline,MemoryContext(;fusion=true))) < + peak_bytes(memory_estimate(baseline,MemoryContext(;fusion=false))) + @test estimate_scaling(GEMM{Float32}(;N=64,M=32),8)==(128,64) + @test estimate_scaling(baseline,4)==(128,64) + @test_throws ErrorException memory_estimate(baseline,MemoryContext(;steps=0)) + @test_throws ErrorException memory_estimate(accelerated,MemoryContext(;backend=:cupynumeric)) + gs = GlobalSettings(;n_warmup=1,n_iter=1) + ss = [spec(n;gpus=p,fusion=f) for n in ("grayscott_baseline","grayscott_function_accelerated") for p in (1,4) for f in (false,true)] + runs = plan_runs(ss,gs,RAW,GROUPS,10_000_000) + for p in (1,4) + @test length(unique((r.N,r.M) for r in runs if r.spec.gpus==p))==1 + end +end + +@testset "Execution isolation and failure status" begin + gs = GlobalSettings(;n_warmup=1,n_iter=1) + runs = plan_runs([spec("montecarlo";T=T) for T in ("Float32","Float64")],gs,RAW,GROUPS,1_000_000) + opts = cli_options(["--only=montecarlo"]) + mktempdir() do dir + calls = Cmd[] + launch(cmd) = (push!(calls,cmd); nothing) + @test execute_plan(runs,gs,opts,1_000_000,RAW;launch,prepare=(f,v)->nothing,results_root=dir)==0 + @test length(calls)==4 # two workers and one plot per dtype + run_dir = only(readdir(dir;join=true)) + manifest = TOML.parsefile(joinpath(run_dir,"manifest.toml")) + @test manifest["status"]=="complete" + @test all(r["status"]=="complete" for r in manifest["runs"]) + @test occursin("Float32",join(calls[1].env)) + @test occursin("Float64",join(calls[2].env)) + @test any(occursin("plot_results.jl",join(c.exec)) for c in calls) + end + mktempdir() do dir + calls = Ref(0) + function fail_first(cmd) + calls[] += 1 + calls[] == 1 && error("simulated worker failure") + end + @test execute_plan(runs,gs,opts,1_000_000,RAW;launch=fail_first,prepare=(f,v)->nothing,results_root=dir)==1 + @test calls[]==4 # no retry, later worker and plots still run + manifest = TOML.parsefile(joinpath(only(readdir(dir;join=true)),"manifest.toml")) + @test manifest["status"]=="incomplete" + @test manifest["runs"][1]["status"]=="failed" + @test manifest["runs"][2]["status"]=="complete" + end +end diff --git a/benchmark/test/test_timing.py b/benchmark/test/test_timing.py new file mode 100644 index 000000000..40ae4d019 --- /dev/null +++ b/benchmark/test/test_timing.py @@ -0,0 +1,63 @@ +"""CPU-only tests: python -m unittest discover -s benchmark/test -p 'test_*.py'.""" +import importlib.util +from pathlib import Path +from types import SimpleNamespace +import unittest +from unittest.mock import patch + +SOURCE = Path(__file__).resolve().parents[1] / "src_py" + + +def load(path): + spec = importlib.util.spec_from_file_location(path.stem, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +class TimingTests(unittest.TestCase): + def test_completion_order(self): + for gray_scott in (False, True): + for warmup in (0, 2): + with self.subTest(gray_scott=gray_scott, warmup=warmup): + events, ticks = [], iter((6000, 12000)) + + def clock(): + events.append("clock") + return next(ticks) + + def fence(*, block): + self.assertTrue(block) + events.append("sync") + + mocks = { + "cupynumeric": SimpleNamespace(float32=float, float64=float), + "legate.core": SimpleNamespace(get_legate_runtime=lambda: + SimpleNamespace(issue_execution_fence=fence)), + "legate.timing": SimpleNamespace(time=clock), + } + with patch.dict("sys.modules", mocks): + core = load(SOURCE / "core.py") + with patch.dict("sys.modules", {"core": core}): + gs = load(SOURCE / "benchmarks" / "grayscott.py") + + class Probe: + def initialize(self): + events.append("initialize") + return None + + def run(self, state): + events.append("run") + + # Inherit the real GrayScott policy without creating GPU arrays. + cls = type("GrayScottProbe", (Probe, gs.GrayScott), {}) if gray_scott else Probe + bench = cls.__new__(cls) + result = core.trial(bench, warmup, 3, 6000) + step = ["run"] if gray_scott else ["run", "sync"] + self.assertEqual(events, ["initialize"] + step * warmup + + ["clock"] + step * 3 + ["clock"]) + self.assertEqual(result, (2.0, 0.003)) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmark/test/timing.jl b/benchmark/test/timing.jl new file mode 100644 index 000000000..9ed66e046 --- /dev/null +++ b/benchmark/test/timing.jl @@ -0,0 +1,34 @@ +struct TimingProbe <: AbstractBenchmark{Float32} + events::Vector{Symbol} +end +struct GrayScottTimingProbe <: AbstractGrayScott{Float32} + events::Vector{Symbol} +end +const TimingProbes = Union{TimingProbe,GrayScottTimingProbe} +initialize(b::TimingProbes; mod=Base) = (push!(b.events, :initialize); ()) +run!(b::TimingProbes) = push!(b.events, :run) +total_flops(::TimingProbes) = 6000 +name(::TimingProbes) = "timing probe" + +@testset "Iteration completion policy" begin + for B in values(BENCHMARKS) + b = build_benchmark(B, Float32, 32, 32) + @test fence_each_iteration(b) == !(b isa AbstractGrayScott) + end + for B in (TimingProbe, GrayScottTimingProbe), warmup in (0, 2) + events = Symbol[] + b = B(events) + ticks = Ref(0) + clock() = (push!(events, :clock); ticks[] += 6000) + synchronize() = push!(events, :sync) + gs = GlobalSettings(; n_warmup=warmup, n_iter=3) + step = fence_each_iteration(b) ? [:run, :sync] : [:run] + expected = vcat([:initialize], repeat(step, warmup), [:clock], + repeat(step, 3), [:clock]) + # Also exercises forwarding the backend callback through run_benchmark. + result = run_benchmark(b, gs; mod=Base, clock, synchronize) + @test events == expected + @test result.times_ms == [2.0] + @test result.gflops == [0.003] + end +end diff --git a/docs/src/benchmarks/howto.md b/docs/src/benchmarks/howto.md index 4d7f540bb..22cee5258 100644 --- a/docs/src/benchmarks/howto.md +++ b/docs/src/benchmarks/howto.md @@ -24,7 +24,7 @@ With no extra args, `run.jl` reads `benchmarks.toml` and runs every expanded con 2. Calls `run_benchmark.sh`, which exports `LEGATE_CONFIG` from `--gpus` / `--cpus` **before** Julia starts 3. Launches `src/single.jl` for the cuNumeric backend (and optional comparison backends) -Add `-v` / `--verbose` for more plumbing output. +After the sweep, it launches `plot_results.jl` on the result CSVs. One-off CLI runs skip plotting. Add `-v` / `--verbose` for more plumbing output. ## `benchmarks.toml` @@ -36,30 +36,45 @@ n_warmup = 5 n_iter = 1000 n_trial = 5 cupynumeric = true # also run Python cupynumeric (needs install_cupynumeric.sh) -cuda = false # also run CUDA.jl (single-GPU configs only) +cuda = true # also run CUDA.jl (single-GPU configs only) check_correctness = true n_correctness_iter = 5 +auto_size = true +mem_frac = 0.5 # fraction of the smallest visible GPU's total RAM ``` - `n_warmup`: untimed iterations (hide compile / first-touch cost) - `n_iter`: timed iterations per trial (build task queue depth) - `n_trial`: independent trials; mean ± stddev across trials is what gets printed / saved -- `cupynumeric` / `cuda`: optional comparison backends -- `check_correctness`: one CPU-reference check per config (not per timed iter), recorded in the CSV - -Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, `dmd_baseline`, `dmd_lifetimes`, `grayscott_baseline`, `grayscott_lifetimes`, `poisson_fft`, …). Names must match what `src/benchmarks/*.jl` registers. - -DMD's `N` is the number of spatial degrees of freedom (rows of the snapshot matrix), not a grid side length. The SVD is of the tall-skinny `N × (M-1)` matrix `X1`. Thin SVD plus the rank-`r` lift is `Θ(N)` when `M` and `r` are fixed, so weak scaling is `N ∝ P` (same idea as Monte Carlo, not GEMM's `N ∝ P^{1/3}`). The flop count is in `src/benchmarks/dmd.jl`. - -`poisson_fft` solves ``M`` independent periodic Poisson problems on an ``N \times N`` grid (FFT, divide by ``-|k|^2``, inverse FFT). The transform is over the last two axes, so the leading batch axis can split across GPUs. Weak scaling is ``M \propto P`` with ``N`` fixed. A single all-axes 2-d `fft` of one grid is single-GPU and would not scale that way. +- `cupynumeric` / `cuda`: optional comparison backends. `cuda = true` still + **times** CUDA.jl on configs with `gpus == 1` (GEMM, Monte Carlo, Gray-Scott, + DMD, Poisson FFT, and the TensorOperations kernels). +- `check_correctness`: when `gpus == 1`, the cuNumeric worker also compares a + tiny problem against CUDA.jl (not CPU). The CUDA.jl timing run is unchanged + and records `skipped` for correctness. Multi-GPU and cupynumeric skip. +- `auto_size` / `mem_frac`: when `auto_size` is true and a block omits `N` + (or sets `N = "auto"`), the harness RAM-fits the 1-GPU problem then scales + with GPU count. `mem_frac` is a fraction of the *smallest* visible GPU's + total RAM; override it with `CUNUMERIC_BENCH_MEM_FRAC`. Peak bytes are + `total_space` on each Julia benchmark type (live arrays including tensor + intermediates), not the seed array. Legion / cuSOLVER scratch is the rest of + `mem_frac`. Pin `N = [20000, …]` on a block to keep paper sizes. + +Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, +`dmd_baseline`, `dmd_accelerated`, `grayscott_baseline`, the Gray-Scott +`@accelerate` forms, `poisson_fft`, `tensor_projection3`, …). Names must match +what `src/benchmarks/*.jl` registers. + +DMD's `N` is the number of spatial degrees of freedom (rows of the snapshot matrix), not a grid side length. The SVD is of the tall-skinny `N × (M-1)` matrix `X1`. Thin SVD plus the rank-`r` lift is `Θ(N)` when `M` and `r` are fixed, so weak scaling is `N ∝ P` (same idea as Monte Carlo, not GEMM's `N ∝ P^{1/3}`). Auto-sizing only scales to `P>1` when the 1-GPU `N` is at least `10 M`; otherwise the 1-GPU run still happens and larger `P` is skipped. `M` stays in the toml (intensity knob). The flop count is in `src/benchmarks/dmd.jl`. + +`poisson_fft` solves ``M`` independent periodic Poisson problems on an ``N \times N`` grid (FFT, divide by ``-|k|^2``, inverse FFT). The transform is over the last two axes, so the leading batch axis can split across GPUs. Weak scaling is ``M \propto P`` with ``N`` **held constant across the GPU sweep** — growing `N` would change the FFT size. `N` itself may be RAM-fitted (one grid filling the 1-GPU budget, then `M = P`) or pinned (`N = 1024` and `M = "auto"` to search the batch). A single all-axes 2-d `fft` of one grid is single-GPU and would not scale that way. ```toml [[gemm]] T = ["Float32"] gpus = [1, 2, 4, 8] cpus = 16 -N = [20000, 25200, 31752, 40000] -M = [20000, 25200, 31752, 40000] +# omit N → RAM-fit, then N ∝ P^{1/3} ``` ### How lists expand @@ -97,17 +112,22 @@ You can dispatch a single config without editing the TOML: julia --project=. run.jl [fusion] ``` -Example: +`N` and `M` may be integers or `auto`: ```bash julia --project=. run.jl 1 16 gemm Float32 20000 20000 1000 5 5 true +CUNUMERIC_BENCH_MEM_FRAC=0.1 julia --project=. run.jl 8 8 gemm Float32 auto auto 3 1 1 true ``` `run.jl` still goes through `run_benchmark.sh` so Legate sees the right GPU/CPU count at process start. ## Comparison backends -- **CUDA.jl:** set `cuda = true` in `[Global]`. Only runs when `gpus == 1`. +- **CUDA.jl:** set `cuda = true` in `[Global]`. Timed on configs with `gpus == 1`. + Every registered kernel has a `CuArray` path (GEMM / broadcast / FFT / + SVD via cuBLAS, cuFFT, cuSOLVER; tensor contractions use TensorOperations' + cuTENSOR extension). That timed run is separate from the tiny cuNumeric vs + CUDA.jl correctness check. - **cupynumeric (Python):** set `cupynumeric = true`, then build a matching conda env once: ```bash @@ -118,10 +138,34 @@ julia --project=. run.jl 1 16 gemm Float32 20000 20000 1000 5 5 true ## Results and timing -Each worker prints mean ± stddev run time (ms) and GFLOPS, plus a correctness tag (`pass` / `fail` / `skipped`). CSVs append under `benchmark/results/`. +Each worker prints mean ± stddev run time (ms) and GFLOPS, plus a correctness tag (`pass` / `fail` / `skipped`). On one GPU the cuNumeric tag is the tiny CUDA.jl compare; CUDA.jl and cupynumeric record `skipped`. CSVs append under `benchmark/results/`. Unfused cuNumeric runs are labeled and saved separately (for example `cunumeric_nofusion`) so they stay a distinct series from fused runs. +## Plotting + +`[plot.groups]` in `benchmarks.toml` names the figures. Related kernels share one PNG: + +```toml +[plot.groups] +grayscott = [ + "grayscott_baseline", + "grayscott_function_accelerated", + "grayscott_begin_accelerated", + "grayscott_let_accelerated", + "grayscott_expression_accelerated", +] +dmd = ["dmd_baseline", "dmd_accelerated"] +``` + +Any `[[benchmark]]` table not listed (GEMM, Poisson, Monte Carlo, tensors) is its own figure. CUDA.jl (single GPU) and cupynumeric overlay from the group's baseline CSV — the member ending in `_baseline`, or the only / first name. Accelerated variants are cuNumeric-only; their CUDA.jl and Python points still come from the baseline. + +Fused and unfused cuNumeric share a color (solid vs dashed). A full `run.jl` pass writes `plots/_weak_scaling.png` at the end. To plot existing CSVs: + +```bash +julia --project=. plot_results.jl +``` + ## Hardware notes `LEGATE_CONFIG` must be set before Julia / Legate starts. The harness does that for you via `run_benchmark.sh`. For manual REPL experiments, see [Hardware Configuration](../configuration/hardware.md). Do not expect to change GPU count mid-session without restarting Julia. diff --git a/lib/cunumeric_jl_wrapper/include/accessors.h b/lib/cunumeric_jl_wrapper/include/accessors.h index baf27c6ac..091b6cbfd 100644 --- a/lib/cunumeric_jl_wrapper/include/accessors.h +++ b/lib/cunumeric_jl_wrapper/include/accessors.h @@ -26,9 +26,11 @@ #include "cupynumeric.h" #include "jlcxx/jlcxx.hpp" +#include "legate.h" #include "legion.h" -using coord_t = long long; +// Match Legate coordinates; large scalar indices must never narrow to int. +static_assert(sizeof(legate::coord_t) >= 8, "NDArray accessors require 64-bit coordinates"); // To auto-magically generate templated classes and their // respective member functions you must define a `BuildParameterList` @@ -49,7 +51,7 @@ class NDArrayAccessor { ~NDArrayAccessor() {} // static T read(void* arr, const std::vector& dims) { - auto p = Realm::Point(0); + auto p = legate::Point(0); // Realm::Point defaults to 32-bit int. for (int i = 0; i < n_dims; ++i) { p[i] = dims[i]; } @@ -59,7 +61,7 @@ class NDArrayAccessor { // static void write(void* arr, const std::vector& dims, T val) { - auto p = Realm::Point(0); + auto p = legate::Point(0); // Preserve indices beyond INT32_MAX. for (int i = 0; i < n_dims; ++i) { p[i] = dims[i]; } diff --git a/lib/cunumeric_jl_wrapper/include/cuda_macros.h b/lib/cunumeric_jl_wrapper/include/cuda_macros.h index de9a9f2fa..35d1b331b 100644 --- a/lib/cunumeric_jl_wrapper/include/cuda_macros.h +++ b/lib/cunumeric_jl_wrapper/include/cuda_macros.h @@ -58,9 +58,10 @@ << std::endl; \ std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ << acc.accessor.strides << std::endl;); \ + /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ void *dev_ptr = const_cast(/*.lo to ensure multiple GPU support*/ \ static_cast( \ - acc.ptr(Realm::Point(shp.lo)))); \ + acc.ptr(shp.lo))); \ auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ CuDeviceArray desc; \ desc.ptr = dev_ptr; \ diff --git a/lib/cunumeric_jl_wrapper/src/cuda.cpp b/lib/cunumeric_jl_wrapper/src/cuda.cpp index 4eec2525b..2f5c01097 100644 --- a/lib/cunumeric_jl_wrapper/src/cuda.cpp +++ b/lib/cunumeric_jl_wrapper/src/cuda.cpp @@ -31,6 +31,7 @@ #include "ufi.h" // #define CUDA_DEBUG +#include "cuda_macros.h" // Shared error/debug and dense argument-packing macros. #define BLOCK_START 1 #define THREAD_START 4 @@ -39,51 +40,6 @@ // global padding for CUDA.jl kernel state std::size_t padded_bytes_kernel_state = 16; -#define ERROR_CHECK(x) \ - { \ - cudaError_t status = x; \ - if (status != cudaSuccess) { \ - fprintf(stderr, "CUDA Error at %s:%d: %s\n", __FILE__, __LINE__, \ - cudaGetErrorString(status)); \ - if (stream_) cudaStreamDestroy(stream_); \ - exit(-1); \ - } \ - } - -#define DRIVER_ERROR_CHECK(x) \ - { \ - CUresult status = x; \ - if (status != CUDA_SUCCESS) { \ - const char *err_str = nullptr; \ - cuGetErrorString(status, &err_str); \ - fprintf(stderr, "CUDA Driver Error at %s:%d: %s\n", __FILE__, __LINE__, \ - err_str); \ - if (stream_) cudaStreamDestroy(stream_); \ - exit(-1); \ - } \ - } - -#define TEST_PRINT_DEBUG(dev_ptr, N, T, format, stream, message) \ - { \ - std::vector host_arr(N); \ - ERROR_CHECK(cudaMemcpy(host_arr.data(), \ - reinterpret_cast(dev_ptr), \ - sizeof(T) * N, cudaMemcpyDeviceToHost)); \ - ERROR_CHECK(cudaStreamSynchronize(stream)); \ - fprintf(stderr, "[TEST_PRINT] %s: " format "\n", message, host_arr[0]); \ - } - -#ifdef CUDA_DEBUG -#define CUDA_DEBUG_PRINT(x) \ - do { \ - x; \ - } while (0) -#else -#define CUDA_DEBUG_PRINT(x) \ - do { \ - } while (0) -#endif - namespace ufi { using namespace Legion; // TODO CUcontext key hashing is redundant. ProcLocalStorage is local to the @@ -148,34 +104,6 @@ struct CuStridedDeviceArray { uint64_t length; }; -#define CUDA_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ - template < \ - typename T, int D, \ - typename std::enable_if<(D >= 1 && D <= REALM_MAX_DIM), int>::type = 0> \ - void cuda_device_array_arg_##MODE(char *&p, \ - const legate::PhysicalArray &rf) { \ - auto shp = rf.shape(); \ - auto acc = rf.data().ACCESSOR_CALL(); \ - CUDA_DEBUG_PRINT(std::cerr << "[RunPTXTask] " #MODE " accessor shape: " \ - << shp.lo << " - " << shp.hi << ", dim: " << D \ - << std::endl; \ - std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ - << acc.accessor.strides << std::endl;); \ - void *dev_ptr = const_cast(/*.lo to ensure multiple GPU support*/ \ - static_cast( \ - acc.ptr(Realm::Point(shp.lo)))); \ - auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ - CuDeviceArray desc; \ - desc.ptr = dev_ptr; \ - desc.maxsize = shp.volume() * sizeof(T); \ - for (size_t i = 0; i < D; ++i) { \ - desc.dims[i] = extents[i]; \ - } \ - desc.length = shp.volume(); \ - memcpy(p, &desc, sizeof(CuDeviceArray)); \ - p += sizeof(CuDeviceArray); \ - } - #define CUDA_STRIDED_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ template < \ typename T, int D, \ @@ -190,8 +118,9 @@ struct CuStridedDeviceArray { std::cerr << "[RunPTXBroadcastTask] " #MODE \ << " accessor byte strides: " << acc.accessor.strides \ << std::endl;); \ + /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ void *dev_ptr = const_cast( \ - static_cast(acc.ptr(Realm::Point(shp.lo)))); \ + static_cast(acc.ptr(shp.lo))); \ auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ CuStridedDeviceArray desc; \ desc.ptr = dev_ptr; \ @@ -380,6 +309,10 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const legate::PhysicalArray &out) { const std::uint32_t budget = std::max(lp.tx, 1u); const int dim = out.dim(); + // Cap before narrowing; Julia grid-stride loops cover the remaining elements. + const auto blocks = [](std::uint64_t n, std::uint32_t t, std::uint64_t limit) { + return static_cast(std::min((n - 1) / t + 1, limit)); + }; assert(dim > 0); @@ -395,8 +328,8 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, lp.ty = static_cast( std::min(budget / lp.tx, rows)); lp.tz = 1; - lp.bx = static_cast((cols + lp.tx - 1) / lp.tx); - lp.by = static_cast((rows + lp.ty - 1) / lp.ty); + lp.bx = blocks(cols, lp.tx, 2147483647); + lp.by = blocks(rows, lp.ty, 65535); lp.bz = 1; #ifdef CUDA_DEBUG @@ -423,9 +356,9 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const std::uint32_t z_budget = yz_budget / lp.ty; lp.tz = static_cast( std::min({z_budget, dim1, 64})); - lp.bx = static_cast((dim3 + lp.tx - 1) / lp.tx); - lp.by = static_cast((dim2 + lp.ty - 1) / lp.ty); - lp.bz = static_cast((dim1 + lp.tz - 1) / lp.tz); + lp.bx = blocks(dim3, lp.tx, 2147483647); + lp.by = blocks(dim2, lp.ty, 65535); + lp.bz = blocks(dim1, lp.tz, 65535); #ifdef CUDA_DEBUG std::cerr << "[RunPTXBroadcastTask] local shape=" << dim1 << "x" << dim2 @@ -466,10 +399,7 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const std::uint32_t threads = static_cast(std::min(budget, volume)); - const std::uint32_t blocks = - static_cast((volume + threads - 1) / threads); - - lp.bx = blocks; + lp.bx = blocks(volume, threads, 2147483647); lp.by = 1; lp.bz = 1; lp.tx = threads; diff --git a/lib/cunumeric_jl_wrapper/src/memory.cpp b/lib/cunumeric_jl_wrapper/src/memory.cpp index 9efe0fb6e..5f5bb4ef4 100644 --- a/lib/cunumeric_jl_wrapper/src/memory.cpp +++ b/lib/cunumeric_jl_wrapper/src/memory.cpp @@ -27,6 +27,7 @@ #include #include #include +#include #include "ndarray_c_api.h" @@ -40,6 +41,7 @@ static inline uint64_t query_machine_config_common( Realm::Processor::Kind proc_kind, Realm::Memory::Kind mem_kind) { Machine legion_machine{Machine::get_machine()}; uint64_t total_mem = 0; + std::set seen; // Processors on one node share memory-query results. Machine::ProcessorQuery procs = Machine::ProcessorQuery(legion_machine).only_kind(proc_kind); @@ -57,7 +59,7 @@ static inline uint64_t query_machine_config_common( ++mit) { auto mem = *mit; assert(mem.kind() == mem_kind); - total_mem += mem.capacity(); + if (seen.insert(mem).second) total_mem += mem.capacity(); } } @@ -73,6 +75,7 @@ static inline uint64_t query_allocated_bytes_common( auto ctx = Legion::Runtime::get_context(); uint64_t current_bytes = 0; + std::set seen; // Count each physical memory once, not per processor. Machine::ProcessorQuery procs = Machine::ProcessorQuery(legion_machine).only_kind(proc_kind); @@ -91,6 +94,7 @@ static inline uint64_t query_allocated_bytes_common( auto mem = *mit; assert(mem.kind() == mem_kind); + if (!seen.insert(mem).second) continue; size_t available = legion_runtime->query_available_memory(ctx, mem); size_t capacity = mem.capacity(); current_bytes += (capacity - available); diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 64149a9f7..470e589c1 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -127,16 +127,21 @@ end Int(CUDACore.threadIdx().x) end +# Widen before multiplying. C++ caps the grid; loops must still visit every element. +@inline _broadcast_grid_stride(axis) = + Int(getproperty(CUDACore.gridDim(), axis)) * Int(getproperty(CUDACore.blockDim(), axis)) + function make_linear_kernel(dest, bc::Base.Broadcast.Broadcasted, arg_plan, static_args) f = bc.f @kernel unsafe_indices = true function broadcast_kernel_linear_splat(dest, runtime_args...) I = _broadcast_linear_work_id() - if I <= length(dest) + while I <= length(dest) @inbounds args_modified = _materialize_broadcast_args( arg_plan, runtime_args, static_args, I ) @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + I += _broadcast_grid_stride(:x) end end @@ -159,12 +164,17 @@ function make_cartesian_kernel(dest, bc::Base.Broadcast.Broadcasted, arg_plan, s @kernel unsafe_indices = true function broadcast_kernel_cartesian_splat( dest, runtime_args... ) - I = _broadcast_cartesian_work_id() - if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) - @inbounds args_modified = _materialize_broadcast_args( - arg_plan, runtime_args, static_args, I - ) - @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + start = _broadcast_cartesian_work_id() + I = start + while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + @inbounds args_modified = _materialize_broadcast_args( + arg_plan, runtime_args, static_args, I + ) + @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + I += CartesianIndex(_broadcast_grid_stride(:y), 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:x)) end end @@ -192,12 +202,20 @@ function make_cartesian_kernel_3d( @kernel unsafe_indices = true function broadcast_kernel_cartesian_3d_splat( dest, runtime_args... ) - I = _broadcast_cartesian_work_id_3d() - if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) && I[3] <= size(dest, 3) - @inbounds args_modified = _materialize_broadcast_args( - arg_plan, runtime_args, static_args, I - ) - @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + start = _broadcast_cartesian_work_id_3d() + I = start + while I[3] <= size(dest, 3) + while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + @inbounds args_modified = _materialize_broadcast_args( + arg_plan, runtime_args, static_args, I + ) + @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + I += CartesianIndex(_broadcast_grid_stride(:z), 0, 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:y), I[3]) + end + I = CartesianIndex(start[1], start[2], I[3] + _broadcast_grid_stride(:x)) end end @@ -835,10 +853,15 @@ end # `args` = (outputs[1:NOUT]..., runtime_args...); bounds from the first output. function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val{2}) where {NOUT} @kernel unsafe_indices = true function broadcast_kernel_multi_2d(args...) - I = _broadcast_cartesian_work_id() + start = _broadcast_cartesian_work_id() + I = start dest = getfield(args, 1) - @inbounds if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) - _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + @inbounds while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + I += CartesianIndex(_broadcast_grid_stride(:y), 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:x)) end end return broadcast_kernel_multi_2d @@ -846,10 +869,18 @@ end function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val{3}) where {NOUT} @kernel unsafe_indices = true function broadcast_kernel_multi_3d(args...) - I = _broadcast_cartesian_work_id_3d() + start = _broadcast_cartesian_work_id_3d() + I = start dest = getfield(args, 1) - @inbounds if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) && I[3] <= size(dest, 3) - _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + @inbounds while I[3] <= size(dest, 3) + while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + I += CartesianIndex(_broadcast_grid_stride(:z), 0, 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:y), I[3]) + end + I = CartesianIndex(start[1], start[2], I[3] + _broadcast_grid_stride(:x)) end end return broadcast_kernel_multi_3d @@ -859,8 +890,9 @@ end function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val) where {NOUT} @kernel unsafe_indices = true function broadcast_kernel_multi_linear(args...) I = _broadcast_linear_work_id() - @inbounds if I <= length(getfield(args, 1)) + @inbounds while I <= length(getfield(args, 1)) _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + I += _broadcast_grid_stride(:x) end end return broadcast_kernel_multi_linear diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 66f9e4cd0..e42e33e3c 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -278,7 +278,7 @@ end nda_array_dim(arr::NDArray) = ccall((:nda_array_dim, libnda), Int32, (NDArray_t,), arr.ptr) nda_array_size(arr::NDArray) = ccall((:nda_array_size, libnda), - Int32, (NDArray_t,), arr.ptr) + UInt64, (NDArray_t,), arr.ptr) # C API returns uint64_t, not an axis/count int32_t. function nda_array_type_code(arr::NDArray) return ccall((:nda_array_type_code, libnda), Int32, (NDArray_t,), arr.ptr) diff --git a/test/regressions.jl b/test/regressions.jl new file mode 100644 index 000000000..9dcdc7b07 --- /dev/null +++ b/test/regressions.jl @@ -0,0 +1,16 @@ +using Test + +@testset "regressions" begin + @testset "array-size ABI preserves uint64_t" begin + a = cuNumeric.zeros(UInt8, 1) + try + # Check the actual C API binding without a multi-GiB allocation. + # An Int32 return truncates sizes above 2^31 - 1. + actual = cuNumeric.nda_array_size(a) + @test actual isa UInt64 + @test actual == length(a) + finally + cuNumeric.destroy!(a) + end + end +end From a7ffb3bdebf3880b7fc4f8b8cdfb3e10ce411d77 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Thu, 10 Sep 2026 22:25:15 -0400 Subject: [PATCH 25/49] =?UTF-8?q?Prototype=20owning=20Julia=20array=20conv?= =?UTF-8?q?ersions=20and=20preserve=20FFT=20submission=20=E2=80=A6=20(#193?= =?UTF-8?q?)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Make Julia array conversion owning --- src/ndarray/detail/fft.jl | 64 ++++++++++++++++-------------- src/ndarray/ndarray.jl | 27 +++++++++++-- test/analysis/lifetime.jl | 30 ++++++++------ test/array/conversion_lifetimes.jl | 30 ++++++++++++++ 4 files changed, 105 insertions(+), 46 deletions(-) create mode 100644 test/array/conversion_lifetimes.jl diff --git a/src/ndarray/detail/fft.jl b/src/ndarray/detail/fft.jl index 24046866c..b7c943915 100644 --- a/src/ndarray/detail/fft.jl +++ b/src/ndarray/detail/fft.jl @@ -203,36 +203,40 @@ function fft_task!( _bluestein_mask(axes0, size(inp), size(out)) kind = _fft_kind(T) - @task_scope _fft_scope_name(direction) begin - rt = Legate.get_runtime() - lib = cuNumeric.get_lib() - task = Legate.create_auto_task(rt, lib, cuNumeric.FFT) - cuNumeric.task_throws_exception(task, true) - - l_out = nda_to_logical_array(out) - l_in = inp === out ? l_out : nda_to_logical_array(inp) - - out_var = Legate.add_output(task, l_out) - in_var = Legate.add_input(task, l_in) - - # 26.06 fft_template.inl: kind, direction, operate_over_axes, then axes. - Legate.add_scalar(task, Legate.Scalar(kind)) - Legate.add_scalar(task, Legate.Scalar(direction)) - Legate.add_scalar(task, Legate.Scalar(operate_over)) - for ax in axes0 - Legate.add_scalar(task, Legate.Scalar(ax)) - end - - Legate.add_constraint(task, Legate.align(out_var, in_var)) - if N > length(unique_axes) - Legate.add_broadcast(task, l_in, CxxWrap.StdVector(UInt32.(unique_axes))) - else - Legate.add_broadcast(task, l_in) - end - - Legate.submit_auto_task(rt, task) - if scale - out .*= _ifft_scale(T, size(out), dims) + # Root the Julia wrappers through submission; pending tasks retain the stores. + # Internal FFT execution stays asynchronous and does not require extra copies. + GC.@preserve inp out begin + @task_scope _fft_scope_name(direction) begin + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + task = Legate.create_auto_task(rt, lib, cuNumeric.FFT) + cuNumeric.task_throws_exception(task, true) + + l_out = nda_to_logical_array(out) + l_in = inp === out ? l_out : nda_to_logical_array(inp) + + out_var = Legate.add_output(task, l_out) + in_var = Legate.add_input(task, l_in) + + # 26.06 fft_template.inl: kind, direction, operate_over_axes, then axes. + Legate.add_scalar(task, Legate.Scalar(kind)) + Legate.add_scalar(task, Legate.Scalar(direction)) + Legate.add_scalar(task, Legate.Scalar(operate_over)) + for ax in axes0 + Legate.add_scalar(task, Legate.Scalar(ax)) + end + + Legate.add_constraint(task, Legate.align(out_var, in_var)) + if N > length(unique_axes) + Legate.add_broadcast(task, l_in, CxxWrap.StdVector(UInt32.(unique_axes))) + else + Legate.add_broadcast(task, l_in) + end + + Legate.submit_auto_task(rt, task) + if scale + out .*= _ifft_scale(T, size(out), dims) + end end end return out diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 683b3f8bd..659617a1c 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -182,7 +182,7 @@ end # NDArray-specific overrides of Core's AbstractArray constructors (NDArray <: # AbstractArray): exact `Array{T}` / `Array{T,N}` / `Array` signatures so we win # over `Array{T,N}(::AbstractArray)` (which would scalar-index). Bulk path uses -# `_copy_to_julia_array`; 1-d dispatches same-type (zero-copy) vs convert. +# `_copy_to_julia_array`; 1-d has specialized same-type and converting paths. function (::Type{Array{T}})(arr::NDArray{S,0}) where {T,S} out = Array{T,0}(undef) allowscalar() do @@ -192,7 +192,15 @@ function (::Type{Array{T}})(arr::NDArray{S,0}) where {T,S} end function (::Type{Array{T}})(arr::NDArray{T,1}) where {T} - return make_array(T, Ptr{T}(get_ptr(arr)), size(arr)) + out = Vector{T}(undef, length(arr)) + isempty(out) && return out + # get_ptr waits for the source to be available in host memory. Keep its + # owner alive until the synchronous CPU copy into Julia-owned storage ends. + GC.@preserve arr out begin + src = Ptr{T}(get_ptr(arr)) + unsafe_copyto!(pointer(out), src, length(out)) + end + return out end function (::Type{Array{T}})(arr::NDArray{S,1}) where {T,S} @@ -231,7 +239,20 @@ function _nda_from_julia_array(arr::Array{T,0}) where {T} end function _nda_from_julia_array(arr::Array{T,1}) where {T} - return cuNumeric.nda_attach_external(arr) + # Prototype: the attachment borrows Julia memory only for this copy. + # Preserve the source through completion, not just task submission. + GC.@preserve arr begin + attached = cuNumeric.nda_attach_external(arr) + try + out = copy(attached) + GC.@preserve attached out begin + issue_execution_fence(; block=true) + end + return out + finally + destroy!(attached) + end + end end function _nda_from_julia_array(arr::Array{T,N}) where {T,N} diff --git a/test/analysis/lifetime.jl b/test/analysis/lifetime.jl index c6253ae8a..e05c63bbc 100644 --- a/test/analysis/lifetime.jl +++ b/test/analysis/lifetime.jl @@ -29,45 +29,49 @@ end end -@testset "Zero-Copy Verification (1D)" begin - # 1D attach remains zero-copy; N>=2 copies into a C-ordered buffer. - A = rand(Float64, 16) +@testset "Independent Storage (1D)" begin + # Conversion copies into Legate-owned storage before returning. + A = Float64.(1:16) + expected = copy(A) NA = NDArray(A) @allowscalar begin @test all(A .== Array(NA)) end - @test pointer(A) == cuNumeric.get_ptr(NA) + GC.@preserve A NA begin + @test pointer(A) != cuNumeric.get_ptr(NA) + end - # modify julia array, verify ndarray sees it + # Mutating the Julia source must not change the NDArray. A[1] = 99.0 @allowscalar begin - @test NA[1] == 99.0 + @test NA[1] == expected[1] end - # modify ndarray, verify julia array sees it + # Mutating the NDArray must not change the Julia source. @allowscalar begin NA[2] = 88.0 + @test NA[2] == 88.0 end - @test A[2] == 88.0 + @test A[2] == expected[2] end @testset "Lifetime Protection" begin - function create_attached_ndarray() + function create_owned_ndarray() local_A = rand(Float32, 100) local_A[1] = 1.23f0 return NDArray(local_A), local_A[1] end - NA, expected_val = create_attached_ndarray() + NA, expected_val = create_owned_ndarray() # force gc to try and collect the local array GC.gc(true) GC.gc(true) # do it again GC.gc(true) # and again lol - # data should still be intact via parent reference + # Data survives in Legate-owned storage after the Julia source is collected. @allowscalar begin @test NA[1] == expected_val end @@ -87,7 +91,7 @@ end NA = create_typed_ndarray() - # the temporary Float64 array should be kept alive by NA.parent + # The temporary Float64 source is no longer needed after construction. GC.gc(true) GC.gc(true) GC.gc(true) @@ -97,7 +101,7 @@ end @test NA[10] == 10.0 end - # modification should work on the attached temporary + # Modification should work on the independently owned storage. @allowscalar begin NA[1] = 42.0 @test NA[1] == 42.0 diff --git a/test/array/conversion_lifetimes.jl b/test/array/conversion_lifetimes.jl new file mode 100644 index 000000000..88e9b5bf1 --- /dev/null +++ b/test/array/conversion_lifetimes.jl @@ -0,0 +1,30 @@ +using Test + +@testset "1D conversion ownership" begin + for T in (Float32, ComplexF32) + expected = T[1, 2, 3, 4] + source = copy(expected) + a = cuNumeric.NDArray(source) + try + # Construction must finish reading source before returning. + fill!(source, T(99)) + source = nothing + GC.gc(true) + @test Array(a) == expected + + # The returned Julia vector must not alias the NDArray. + converted = Array(a) + converted[1] = T(77) + @test Array(a) == expected + + # It must also survive explicit destruction of the source owner. + survivor = Array(a) + cuNumeric.destroy!(a) + cuNumeric.issue_execution_fence(; block=true) + GC.gc(true) + @test survivor == expected + finally + cuNumeric.destroy!(a) + end + end +end From 51a2f402386eb77f0cba4711db439fa0f6a0438e Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sun, 13 Sep 2026 16:50:21 -0400 Subject: [PATCH 26/49] cuSolverMp (#192) * Enable distributed cuSolverMp solve and factorization paths --- .buildkite/run_developer_ci.sh | 1 + .github/workflows/developer.yml | 2 +- docs/src/api_preferences.md | 24 +++ docs/src/linalg.md | 53 ++++- lib/CNPreferences/src/CNPreferences.jl | 30 +++ lib/cunumeric_jl_wrapper/VERSION | 2 +- lib/cunumeric_jl_wrapper/src/types.cpp | 11 ++ lib/cunumeric_jl_wrapper/src/wrapper.cpp | 37 ++++ src/cuNumeric.jl | 22 +++ src/ndarray/detail/distributed_linalg.jl | 236 +++++++++++++++++++++++ src/ndarray/detail/linalg.jl | 112 +++++------ src/ndarray/linalg.jl | 9 + src/utilities/version.jl | 24 ++- test/analysis/lifetime.jl | 29 +++ test/analysis/type_stability.jl | 33 ++++ test/array/distributed_linalg.jl | 162 ++++++++++++++++ test/array/linalg_edge_cases.jl | 150 ++++++++++++++ test/linalg_errors.jl | 30 +++ test/linalg_preferences.jl | 34 ++++ test/runtests.jl | 2 + 20 files changed, 931 insertions(+), 72 deletions(-) create mode 100644 src/ndarray/detail/distributed_linalg.jl create mode 100644 test/array/distributed_linalg.jl create mode 100644 test/array/linalg_edge_cases.jl create mode 100644 test/linalg_errors.jl create mode 100644 test/linalg_preferences.jl diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index bba74a57c..d3ac3929e 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -86,5 +86,6 @@ cp LocalPreferences.toml test/LocalPreferences.toml julia --color=yes --project=. -e ' using Pkg + Pkg.develop(PackageSpec(path = "lib/CNPreferences")) Pkg.test("cuNumeric"; test_args = ["--jobs=8", "--verbose"]) ' diff --git a/.github/workflows/developer.yml b/.github/workflows/developer.yml index a1f9c16ce..f2d0d8565 100644 --- a/.github/workflows/developer.yml +++ b/.github/workflows/developer.yml @@ -114,4 +114,4 @@ jobs: cp LocalPreferences.toml test/LocalPreferences.toml - julia --color=yes --project=. -e 'using Pkg; Pkg.test("cuNumeric"; test_args=["--jobs=2", "--verbose"])' + julia --color=yes --project=. -e 'using Pkg; Pkg.develop(PackageSpec(path = "lib/CNPreferences")); Pkg.test("cuNumeric"; test_args=["--jobs=2", "--verbose"])' diff --git a/docs/src/api_preferences.md b/docs/src/api_preferences.md index 2973f7377..89c1f356c 100644 --- a/docs/src/api_preferences.md +++ b/docs/src/api_preferences.md @@ -10,9 +10,33 @@ Out of the box (no `LocalPreferences.toml` changes): | Broadcast fusion | **on** | | `FUSE_BROADCAST_MIN_OPS` | **2** (single-op broadcasts stay unfused) | | Task scope names | **off** | +| `MIN_SOLVE_MATRIX_SIZE` | **2048** rows | +| `MIN_SOLVE_TILE_SIZE` | **512** | +| `MIN_CHOLESKY_MATRIX_SIZE` | **8192** rows | +| `MIN_CHOLESKY_TILE_SIZE` | **2048** | +| `MIN_QR_MATRIX_SIZE` | **1048576** elements | +| `QR_TILE_SIZE` | **128** | +| `MAX_CHOLESKY_TILES_PER_PROC` | **4** | Build-mode setup (JLL / conda / developer) is documented under [Build Modes](./install.md). +## Linear algebra + +`set_linalg!` accepts the constant names above as keywords. All values must be +positive integers; unspecified settings are unchanged. Restart Julia to load +the new constants. `cuNumeric.versioninfo()` shows the effective values. + +```julia +CNPreferences.set_linalg!(; MIN_SOLVE_MATRIX_SIZE=4096, MIN_SOLVE_TILE_SIZE=512) +``` + +See [Distributed solves and factorizations](./linalg.md#distributed-solves-and-factorizations) +for selection rules and tuning details. + +```@docs +CNPreferences.set_linalg! +``` + ## Build mode ```@docs diff --git a/docs/src/linalg.md b/docs/src/linalg.md index b3a0f337a..e3f6a6efb 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -25,6 +25,56 @@ equivalent, so they get their own `batched_*` names. | `LinearAlgebra.qr(A)` | `NDArrayQR` | not supported | | `cuNumeric.solve(A, b)`, `A \ b` | `NDArray` | `cuNumeric.batched_solve(A, B)` | +## Distributed solves and factorizations + +The existing `A \ b`, `cuNumeric.solve`, `cholesky`, and `qr` APIs select +distributed tasks automatically. Selection follows cuPyNumeric 26.06: + +| Operation | cuSolverMp cutoff | Internal block size | +| --- | --- | --- | +| Square solve | dimension ≥ 2048 | 512 | +| Lower Cholesky | dimension ≥ 8192 | 2048 | +| Reduced QR | matrix contains ≥ 1048576 elements | 128 | + +These are the default values of preference-backed module constants. The loaded library +must support cuSolverMp and Legate must have more than one active GPU. The +library selects the algorithm; Legate handles placement, communication, and +redistribution. Configure resources before starting Julia, as described in +[Hardware Configuration](./configuration/hardware.md). + +Ordinary solve and QR tasks remain the fallback. Cholesky also uses the tiled +POTRF/TRSM/SYRK/GEMM algorithm for eligible multi-processor configurations; +below its partitioning cutoff this is a single tile. Batched operations still +distribute independent matrices, each of which must fit on one processor. +SVD and general eigen do not acquire distributed factorization paths. + +Results retain their existing Julia types, element promotion, and shapes. +Inputs are preserved. Kernel failures propagate when the runtime reports them; +failed collectives are not retried using another algorithm. No new factor-reuse, +triangular-solve, or CG API is introduced by this change. + +This requires the new C++ wrapper; see [Developer Mode](./developer_mode.md#cusolvermp-wrapper-update). + +### Tuning + +Use [`CNPreferences.set_linalg!`](./api_preferences.md#linear-algebra) to set any +of the seven documented constants. `MIN_*_MATRIX_SIZE` controls when an operation +can select cuSolverMp. Solve and Cholesky use the row count; QR uses the number +of matrix elements. `MIN_SOLVE_TILE_SIZE`, `MIN_CHOLESKY_TILE_SIZE`, and +`QR_TILE_SIZE` set the solver block sizes. QR uses the same block size on both axes. + +For the tiled Cholesky fallback, `MIN_CHOLESKY_MATRIX_SIZE` also sets the +single-tile cutoff. `MIN_CHOLESKY_TILE_SIZE` guides tile subdivision, and +`MAX_CHOLESKY_TILES_PER_PROC` limits the number of tiles per matrix axis relative +to the processor count. These are tuning heuristics, not memory limits. + +The constants are loaded in `cuNumeric.jl` through `load_preference(CNPreferences, ...)`. +Changing them requires a fresh Julia process. Library capability, configured +GPU/processor counts, and MP eligibility are cached once during runtime startup +in a typed `const Ref`. Solves reuse that configuration; size checks still happen +per operation. This assumes the configured machine stays fixed for the runtime's +lifetime. Scoped processor subsets would require revisiting this cache. + ## Matrix multiply For two 2D arrays, `*` performs matrix multiplication; use `.*` for an @@ -303,5 +353,4 @@ There is no public dense-matrix `lu`, matrix `inv`, or `ldiv!` yet (beyond the operations, not matrix inverse. Also missing: `eigh` / Hermitian eigen (needs `Hermitian` and `Symmetric` -support on `NDArray`), batched SVD and QR, and the multi-GPU cuSolverMp paths -for Cholesky and solve. +support on `NDArray`), batched SVD and QR, and conjugate gradient. diff --git a/lib/CNPreferences/src/CNPreferences.jl b/lib/CNPreferences/src/CNPreferences.jl index 558e47713..ba9663ee8 100644 --- a/lib/CNPreferences/src/CNPreferences.jl +++ b/lib/CNPreferences/src/CNPreferences.jl @@ -84,4 +84,34 @@ Disable named Legate task scopes. This is the default. """ disable_task_scope_names!(; kwargs...) = set_task_scope_names!(false; kwargs...) +""" + set_linalg!(; export_prefs=false, force=true, settings...) + +Set linear algebra tuning preferences using their constant names as keywords: +`MIN_SOLVE_MATRIX_SIZE` (2048), `MIN_SOLVE_TILE_SIZE` (512), +`MIN_CHOLESKY_MATRIX_SIZE` (8192), `MIN_CHOLESKY_TILE_SIZE` (2048), +`MIN_QR_MATRIX_SIZE` (1048576 elements), `QR_TILE_SIZE` (128), and +`MAX_CHOLESKY_TILES_PER_PROC` (4). All values must be positive integers. +Unspecified settings are unchanged. Restart Julia after changing preferences. + +```julia +CNPreferences.set_linalg!(; MIN_SOLVE_MATRIX_SIZE=4096, MIN_SOLVE_TILE_SIZE=512) +``` +""" +function set_linalg!(; export_prefs=false, force=true, settings...) + valid = ( + :MIN_SOLVE_MATRIX_SIZE, :MIN_SOLVE_TILE_SIZE, :MIN_CHOLESKY_MATRIX_SIZE, + :MIN_CHOLESKY_TILE_SIZE, :MIN_QR_MATRIX_SIZE, :QR_TILE_SIZE, :MAX_CHOLESKY_TILES_PER_PROC, + ) + pairs = Pair{String,Int}[] + for (key, value) in settings + key in valid || throw(ArgumentError("Unknown linear algebra setting: $key")) + value isa Integer && !(value isa Bool) && 0 < value <= typemax(Int) || + throw(ArgumentError("$key must be a positive Int")) + push!(pairs, string(key) => Int(value)) + end + isempty(pairs) && return nothing + return set_preferences!(@__MODULE__, pairs...; export_prefs, force) +end + end # module CNPreferences diff --git a/lib/cunumeric_jl_wrapper/VERSION b/lib/cunumeric_jl_wrapper/VERSION index e8c5e343a..892cb42a2 100644 --- a/lib/cunumeric_jl_wrapper/VERSION +++ b/lib/cunumeric_jl_wrapper/VERSION @@ -1 +1 @@ -26.6.1 +26.6.2 diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index df218dcc8..77ab7bc6a 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -169,6 +169,17 @@ void wrap_linalg_ops(jlcxx::Module& mod) { legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SOLVE}); mod.set_const("MP_SOLVE", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_SOLVE}); + mod.set_const("MP_POTRF", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_POTRF}); + mod.set_const("MP_QR", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_QR}); + mod.set_const("POTRS", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_POTRS}); + mod.set_const("TRSM", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRSM}); + mod.set_const("SYRK", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SYRK}); + mod.set_const("GEMM", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEMM}); + mod.set_const("TRANSPOSE_COPY_2D", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRANSPOSE_COPY_2D}); + mod.set_const("TRILU", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRILU}); mod.set_const("SVD", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SVD}); mod.set_const("CQR", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_QR}); mod.set_const("POTRF", diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 22e9540ea..68f280d87 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -21,6 +21,7 @@ #include #include #include +#include #include //needed for return type of toString methods #include #include @@ -116,6 +117,42 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { // True when the loaded cuSolver provides cusolverDnXgeev. Without it // cupynumeric has no GPU eigenvalue kernel for general matrices. mod.method("cusolver_has_geev", &cupynumeric_cusolver_has_geev); + mod.method("cusolvermp_available", &cupynumeric_has_cusolvermp); + + // Match cuPyNumeric 26.06: the MP kernels use a single NCCL communicator. + mod.method("add_nccl_communicator", [](legate::ManualTask& task) { + task.add_communicator("nccl"); + }); + mod.method("add_nccl_communicator", [](legate::AutoTask& task) { + task.add_communicator("nccl"); + }); + + // Tiled Cholesky launches subrectangles in the original partition's color + // space. Bounds are inclusive and zero-based, as in Legion::Rect. + mod.method("create_linalg_task", + [](legate::LocalTaskID id, int64_t row_lo, int64_t col_lo, + int64_t row_hi, int64_t col_hi) { + auto domain = Legion::Domain{Legion::Rect<2>{ + Legion::Point<2>{row_lo, col_lo}, + Legion::Point<2>{row_hi, col_hi}}}; + return legate::Runtime::get_runtime()->create_task( + get_lib(), id, domain); + }); + mod.method("add_input_tile", + [](legate::ManualTask& task, + std::shared_ptr part, + uint64_t row, uint64_t col) { + std::vector color{row, col}; + task.add_input(part->get_child_store(color)); + }); + mod.method("add_input_column", + [](legate::ManualTask& task, + std::shared_ptr part, + int32_t col) { + task.add_input(*part, legate::SymbolicPoint{ + std::vector{legate::dimension(0), + legate::constant(col)}}); + }); mod.method("add_input_proj", [](legate::ManualTask& task, diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 3e636a7f7..4a4432cc6 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -159,8 +159,26 @@ include("warnings.jl") # Compile-time so task scope instrumentation is fully elided when disabled. const TASK_SCOPE_NAMES = CNPreferences.TASK_SCOPE_NAMES +# Loaded at compile time; change through CNPreferences and restart Julia. +const MIN_SOLVE_MATRIX_SIZE = load_preference(CNPreferences, "MIN_SOLVE_MATRIX_SIZE", 2048) +const MIN_SOLVE_TILE_SIZE = load_preference(CNPreferences, "MIN_SOLVE_TILE_SIZE", 512) +const MIN_CHOLESKY_MATRIX_SIZE = load_preference(CNPreferences, "MIN_CHOLESKY_MATRIX_SIZE", 8192) +const MIN_CHOLESKY_TILE_SIZE = load_preference(CNPreferences, "MIN_CHOLESKY_TILE_SIZE", 2048) +const MIN_QR_MATRIX_SIZE = load_preference(CNPreferences, "MIN_QR_MATRIX_SIZE", 1048576) +const QR_TILE_SIZE = load_preference(CNPreferences, "QR_TILE_SIZE", 128) +const MAX_CHOLESKY_TILES_PER_PROC = load_preference(CNPreferences, "MAX_CHOLESKY_TILES_PER_PROC", 4) + +for key in ( + :MIN_SOLVE_MATRIX_SIZE, :MIN_SOLVE_TILE_SIZE, :MIN_CHOLESKY_MATRIX_SIZE, + :MIN_CHOLESKY_TILE_SIZE, :MIN_QR_MATRIX_SIZE, :QR_TILE_SIZE, :MAX_CHOLESKY_TILES_PER_PROC, +) + value = getfield(@__MODULE__, key) + value isa Int && value > 0 || throw(ArgumentError("$key must be a positive Int")) +end + # NDArray internal include("ndarray/detail/ndarray.jl") +include("ndarray/detail/distributed_linalg.jl") include("ndarray/detail/linalg.jl") include("ndarray/detail/fft.jl") @@ -232,6 +250,10 @@ function _start_runtime() # AA = ArgcArgv([Base.julia_cmd()[1]]) cuNumeric.initialize_cunumeric(AA.argc, getargv(AA)) + _LINALG_RUNTIME[] = _LinalgRuntime( + cusolvermp_available(), Int(Legate.num_gpus()), Int(Legate.num_procs()) + ) + _init_deferred_free!() # record launch thread for deferred frees (memory.jl) # setup /src/memory.jl diff --git a/src/ndarray/detail/distributed_linalg.jl b/src/ndarray/detail/distributed_linalg.jl new file mode 100644 index 000000000..c2115471c --- /dev/null +++ b/src/ndarray/detail/distributed_linalg.jl @@ -0,0 +1,236 @@ +# Copyright 2024 NVIDIA Corporation +# SPDX-License-Identifier: Apache-2.0 +# Task construction follows cuPyNumeric 26.06's linalg/_solve.py, _qr.py, +# and _cholesky.py. +# Keep algorithm selection here; Legate owns placement and redistribution. +struct _LinalgRuntime + available::Bool + gpus::Int + procs::Int + mp_eligible::Bool +end + +_LinalgRuntime(available::Bool, gpus::Int, procs::Int) = + _LinalgRuntime(available, gpus, procs, available && gpus > 1) + +# Populated once in _start_runtime(), including deferred initialization. +# These describe the configured machine for the lifetime of this runtime. +const _LINALG_RUNTIME = Ref(_LinalgRuntime(false, 0, 0)) + +struct _SingleProcLinalg end +struct _CuSolverMpLinalg end +struct _TiledCholesky end + +const _LINALG_CONJ_TRANSPOSE = Int32(2) + +_linalg_backend(op, a::NDArray) = _linalg_backend(op, size(a), _LINALG_RUNTIME[]) + +# Tuple length carries dimensionality in its type. Stacked systems never enter +# the MP selector; only their leading batch axes may be distributed. +_linalg_backend(::Val{:solve}, ::Tuple, ::_LinalgRuntime) = _SingleProcLinalg() + +function _linalg_backend(::Val{:solve}, shape::NTuple{2,Int}, rt::_LinalgRuntime) + use_mp = rt.mp_eligible && shape[1] >= MIN_SOLVE_MATRIX_SIZE + return use_mp ? _CuSolverMpLinalg() : _SingleProcLinalg() +end + +function _linalg_backend(::Val{:qr}, shape::NTuple{2,Int}, rt::_LinalgRuntime) + use_mp = rt.mp_eligible && prod(shape) >= MIN_QR_MATRIX_SIZE + return use_mp ? _CuSolverMpLinalg() : _SingleProcLinalg() +end + +_linalg_backend(::Val{:cholesky}, ::Tuple, ::_LinalgRuntime) = _SingleProcLinalg() + +function _linalg_backend(::Val{:cholesky}, shape::NTuple{2,Int}, rt::_LinalgRuntime) + rt.procs == 1 && return _SingleProcLinalg() + use_mp = rt.mp_eligible && shape[1] >= MIN_CHOLESKY_MATRIX_SIZE + return use_mp ? _CuSolverMpLinalg() : _TiledCholesky() +end + +function _linalg_scalars!(task, args...) + for arg in args + Legate.add_scalar(task, Legate.Scalar(arg)) + end + return nothing +end + +function _linalg_manual_task(id, lo::Tuple{Int,Int}, hi::Tuple{Int,Int}; throws=false) + task = create_linalg_task(id, Int64(lo[1]), Int64(lo[2]), Int64(hi[1]), Int64(hi[2])) + task_throws_exception(task, throws) + return task +end + +_submit_linalg_task(task) = Legate.submit_manual_task(Legate.get_runtime(), task) + +# Partitions own their store; submitted tasks retain their own references. +# Drop Julia's temporary owners promptly instead of waiting for GC. Recursion +# keeps earlier partitions protected if creating a later partition fails. +_with_linalg_partitions(f) = f() +function _with_linalg_partitions(f, spec::Tuple, specs::Tuple...) + store = nda_to_logical_store(first(spec)) + partition = try + Legate.partition_by_tiling(store, Base.tail(spec)...) + finally + finalize(store.handle) + end + try + return _with_linalg_partitions(specs...) do parts... + f(partition, parts...) + end + finally + finalize(partition.handle) + end +end + +# Partition along rows, just as Python does. Every rank uses identical color +# spaces even when a reduced QR output has fewer rows than the input. +function _mp_row_partition(n::Int, gpus::Integer) + rows = cld(n, gpus) + return rows, (cld(n, rows), 1) +end + +_solve!(::_SingleProcLinalg, x, a, b) = solve_batched(a, b, x) + +function _solve!(::_CuSolverMpLinalg, x, a, b) + n, nrhs = size(a, 1), size(b, 2) + rows, colors = _mp_row_partition(n, _LINALG_RUNTIME[].gpus) + _with_linalg_partitions( + (a, (rows, n)), (b, (rows, nrhs)), (x, (rows, nrhs)) + ) do pa, pb, px + @task_scope "mp_solve" begin + task = _linalg_manual_task(MP_SOLVE, (0, 0), (colors[1] - 1, 0); throws=true) + Legate.add_input(task, pa) + Legate.add_input(task, pb) + Legate.add_output(task, px) + _linalg_scalars!(task, Int64(n), Int64(nrhs), Int64(MIN_SOLVE_TILE_SIZE)) + add_nccl_communicator(task) + _submit_linalg_task(task) + end + end + return x +end + +function _qr!(::_CuSolverMpLinalg, q, r, a) + m, n = size(a) + rows, colors = _mp_row_partition(m, _LINALG_RUNTIME[].gpus) + tiles = (rows, n) + _with_linalg_partitions((a, tiles), (q, tiles, colors), (r, tiles, colors)) do pa, pq, pr + @task_scope "mp_qr" begin + task = _linalg_manual_task(MP_QR, (0, 0), (colors[1] - 1, 0); throws=true) + Legate.add_input(task, pa) + Legate.add_output(task, pq) + Legate.add_output(task, pr) + _linalg_scalars!(task, Int64(m), Int64(n), Int64(QR_TILE_SIZE), Int64(QR_TILE_SIZE)) + add_nccl_communicator(task) + _submit_linalg_task(task) + end + end + return nothing +end + +_cholesky!(::_SingleProcLinalg, out, a) = potrf!(out, a; lower=true, zeroout=true) + +function _cholesky!(::_CuSolverMpLinalg, out, a) + @task_scope "mp_potrf" begin + rt = Legate.get_runtime() + task = Legate.create_auto_task(rt, get_lib(), MP_POTRF) + task_throws_exception(task, true) + ai = _add_task_array!(Legate.add_input, task, a) + oi = _add_task_array!(Legate.add_output, task, out) + Legate.add_constraint(task, Legate.align(oi, ai)) + _linalg_scalars!(task, Int64(size(a, 1)), Int64(MIN_CHOLESKY_TILE_SIZE)) + add_nccl_communicator(task) + Legate.submit_auto_task(rt, task) + _cholesky_tril!(out) + end + return out +end + +function _cholesky_tril!(out::NDArray) + rt = Legate.get_runtime() + task = Legate.create_auto_task(rt, get_lib(), TRILU) + _add_task_array!(Legate.add_output, task, out) + _add_task_array!(Legate.add_input, task, out) + # The third argument identifies Cholesky to the backend/mapper. + _linalg_scalars!(task, true, Int32(0), true) + Legate.submit_auto_task(rt, task) + return nothing +end + +function _cholesky_color_shape(n::Int, procs::Int) + (procs == 1 || n <= MIN_CHOLESKY_MATRIX_SIZE) && return (1, 1) + tiles = Int(procs) + while cld(n, tiles) > MIN_CHOLESKY_TILE_SIZE && + 2 * tiles <= procs * MAX_CHOLESKY_TILES_PER_PROC + tiles *= 2 + end + return (tiles, tiles) +end + +# Each task reads only tiles ready at this stage; Legate records the DAG from +# these inputs/outputs. No execution fence or host copy is needed between steps. +function _cholesky!(::_TiledCholesky, out, a) + n = size(a, 1) + initial = _cholesky_color_shape(n, _LINALG_RUNTIME[].procs) + tile = cld(n, initial[1]) + colors = cld(n, tile) + _with_linalg_partitions((a, (tile, tile)), (out, (tile, tile))) do pa, po + @task_scope "tiled_cholesky" begin + task = _linalg_manual_task(TRANSPOSE_COPY_2D, (0, 0), (colors - 1, colors - 1)) + Legate.add_output(task, po) + Legate.add_input(task, pa) + _submit_linalg_task(task) + for i in 0:(colors - 1) + _cholesky_potrf!(po, i) + _cholesky_trsm!(po, i, colors) + for k in (i + 1):(colors - 1) + _cholesky_syrk!(po, k, i) + _cholesky_gemm!(po, k, i, colors) + end + end + task = _linalg_manual_task(TRILU, (0, 0), (colors - 1, colors - 1)) + Legate.add_output(task, po) + Legate.add_input(task, po) + _linalg_scalars!(task, true, Int32(0), true) + _submit_linalg_task(task) + end + end + return out +end + +function _cholesky_potrf!(p, i) + task = _linalg_manual_task(POTRF, (i, i), (i, i); throws=true) + Legate.add_output(task, p) + Legate.add_input(task, p) + _linalg_scalars!(task, true, false) + return _submit_linalg_task(task) +end + +function _cholesky_trsm!(p, i, colors) + i + 1 >= colors && return nothing + task = _linalg_manual_task(TRSM, (i + 1, i), (colors - 1, i); throws=true) + Legate.add_output(task, p) + add_input_tile(task, p.handle, UInt64(i), UInt64(i)) + Legate.add_input(task, p) + # Right-side solve with the conjugate transpose of the lower factor. + _linalg_scalars!(task, false, true, _LINALG_CONJ_TRANSPOSE, false) + return _submit_linalg_task(task) +end + +function _cholesky_syrk!(p, k, i) + task = _linalg_manual_task(SYRK, (k, k), (k, k)) + Legate.add_output(task, p) + add_input_tile(task, p.handle, UInt64(k), UInt64(i)) + Legate.add_input(task, p) + return _submit_linalg_task(task) +end + +function _cholesky_gemm!(p, k, i, colors) + k + 1 >= colors && return nothing + task = _linalg_manual_task(GEMM, (k + 1, k), (colors - 1, k)) + Legate.add_output(task, p) + add_input_column(task, p.handle, Int32(i)) + add_input_tile(task, p.handle, UInt64(k), UInt64(i)) + Legate.add_input(task, p) + return _submit_linalg_task(task) +end diff --git a/src/ndarray/detail/linalg.jl b/src/ndarray/detail/linalg.jl index 6722495c1..2dedd8303 100644 --- a/src/ndarray/detail/linalg.jl +++ b/src/ndarray/detail/linalg.jl @@ -1,21 +1,6 @@ -function choose_nd_color_shape(shape::NTuple{N,Int}) where {N} - color_shape = Base.ones(Int, N) - if N > 2 - color_shape[1] = Legate.num_procs() - done = false - while !done && color_shape[1] % 2 == 0 - weight_per_dim = [shape[i] / color_shape[i] for i in 1:(N - 2)] - max_weight, idx = findmax(weight_per_dim) - if weight_per_dim[idx] > 2 * weight_per_dim[1] - color_shape[1] ÷= 2 - color_shape[idx] *= 2 - else - done = true - end - end - end - return Tuple(color_shape) -end +# Only a single matrix or one batch axis is supported. Keep matrix axes whole. +choose_nd_color_shape(::NTuple{2,Int}) = (1, 1) +choose_nd_color_shape(::NTuple{3,Int}) = (_LINALG_RUNTIME[].procs, 1, 1) # One batch dimension is the ceiling for every batched op: # - the POTRF task body is only instantiated for 2 <= DIM < 4 @@ -38,26 +23,22 @@ function solve_batched(a::NDArray{T,N}, b::NDArray, x::NDArray) where {T,N} tilesize_a, color_shape = prepare_manual_task_for_batched_matrices(full_shape) tilesize_b = (tilesize_a[1:(end - 1)]..., nrhs) - store_a = nda_to_logical_store(a) - store_b = nda_to_logical_store(b) - store_x = nda_to_logical_store(x) - - tiled_a = Legate.partition_by_tiling(store_a, collect(tilesize_a)) - tiled_b = Legate.partition_by_tiling(store_b, collect(tilesize_b)) - tiled_x = Legate.partition_by_tiling(store_x, collect(tilesize_b)) - - @task_scope "solve" begin - rt = Legate.get_runtime() - domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) - lib = cuNumeric.get_lib() - task = Legate.create_manual_task(rt, lib, cuNumeric.SOLVE, domain) - cuNumeric.task_throws_exception(task, true) - - Legate.add_input(task, tiled_a) - Legate.add_input(task, tiled_b) - Legate.add_output(task, tiled_x) - - Legate.submit_manual_task(rt, task) + _with_linalg_partitions( + (a, tilesize_a), (b, tilesize_b), (x, tilesize_b) + ) do tiled_a, tiled_b, tiled_x + @task_scope "solve" begin + rt = Legate.get_runtime() + domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) + lib = cuNumeric.get_lib() + task = Legate.create_manual_task(rt, lib, cuNumeric.SOLVE, domain) + cuNumeric.task_throws_exception(task, true) + + Legate.add_input(task, tiled_a) + Legate.add_input(task, tiled_b) + Legate.add_output(task, tiled_x) + + Legate.submit_manual_task(rt, task) + end end end @@ -131,9 +112,11 @@ function _solve(a::NDArray{T,N}, b::NDArray{S,N}) where {T,S,N} " (size $(size(b)[end-1]) is different from $(size(a)[end]))", ), ) - prod(size(a)) == 0 || prod(size(b)) == 0 && return cuNumeric.zeros(T, size(b)...) + size(a)[1:(end - 2)] == size(b)[1:(end - 2)] || + throw(ArgumentError("Batched matrices must have matching batch dimensions")) x = cuNumeric.zeros(T, size(b)...) - solve_batched(a, b, x) + isempty(x) && return x + _solve!(_linalg_backend(Val(:solve), a), x, a, b) return x end @@ -161,17 +144,16 @@ function potrf!(out::NDArray{T,N}, a::NDArray{T,N}; lower::Bool, zeroout::Bool) task = Legate.create_auto_task(rt, lib, cuNumeric.POTRF) cuNumeric.task_throws_exception(task, true) - l_a = nda_to_logical_array(a) - l_out = nda_to_logical_array(out) - - in_var = Legate.add_input(task, l_a) - out_var = Legate.add_output(task, l_out) + in_var = _add_task_array!(Legate.add_input, task, a) + out_var = _add_task_array!(Legate.add_output, task, out) Legate.add_scalar(task, Legate.Scalar(lower)) Legate.add_scalar(task, Legate.Scalar(zeroout)) # Each matrix must live on one processor; only the batch axes may split. - Legate.add_broadcast(task, l_a, CxxWrap.StdVector(UInt32[N - 2, N - 1])) + Legate.add_constraint( + task, Legate.broadcast(in_var, CxxWrap.StdVector(UInt32[N - 2, N - 1])) + ) Legate.add_constraint(task, Legate.align(out_var, in_var)) Legate.submit_auto_task(rt, task) @@ -279,36 +261,32 @@ _svd_eltype(::Type{T}) where {T<:SUPPORTED_SVD_TYPES} = T # qr -function qr_single(a::NDArray{T,N}, q::NDArray, r::NDArray) where {T,N} +function _qr!(::_SingleProcLinalg, q, r, a) rt = Legate.get_runtime() lib = cuNumeric.get_lib() task = Legate.create_auto_task(rt, lib, cuNumeric.CQR) + cuNumeric.task_throws_exception(task, true) - l_a = nda_to_logical_array(a) - l_q = nda_to_logical_array(q) - l_r = nda_to_logical_array(r) - - Legate.add_input(task, l_a) - Legate.add_output(task, l_q) - Legate.add_output(task, l_r) - - Legate.add_broadcast(task, l_a) - Legate.add_broadcast(task, l_q) - Legate.add_broadcast(task, l_r) + ai = _add_task_array!(Legate.add_input, task, a) + qi = _add_task_array!(Legate.add_output, task, q) + ri = _add_task_array!(Legate.add_output, task, r) + for variable in (ai, qi, ri) + Legate.add_constraint(task, Legate.broadcast(variable)) + end - return Legate.submit_auto_task(rt, task) + Legate.submit_auto_task(rt, task) + return nothing end function _qr(a::NDArray{T,2}) where {T} m, n = size(a) k = min(m, n) - # cuSolver requires full square buffers regardless of output shape - q_buf = cuNumeric.zeros(T, m, m) - r_buf = cuNumeric.zeros(T, n, n) - qr_single(a, q_buf, r_buf) - # Host conversion assumes contiguous storage, so materialize the economy slices. - q = copy(q_buf[:, 1:k]) - r = copy(r_buf[1:k, :]) + # CQR writes dense column-major economy factors with leading dimensions + # m for Q and k for R. Square buffers give R the wrong stride when m < n. + q = cuNumeric.zeros(T, m, k) + r = cuNumeric.zeros(T, k, n) + k == 0 && return q, r + _qr!(_linalg_backend(Val(:qr), a), q, r, a) return q, r end @@ -359,7 +337,7 @@ assumed Hermitian without being checked, matching cupynumeric. function _cholesky(a::NDArray{T,N}) where {T,N} _check_square_matrices(:cholesky, a) out = cuNumeric.zeros(T, size(a)...) - potrf!(out, a; lower=true, zeroout=true) + _cholesky!(_linalg_backend(Val(:cholesky), a), out, a) return out end diff --git a/src/ndarray/linalg.jl b/src/ndarray/linalg.jl index 879edafdd..5f41eaba9 100644 --- a/src/ndarray/linalg.jl +++ b/src/ndarray/linalg.jl @@ -9,6 +9,9 @@ Solve the linear system `A * x = b`. `A` must be a square `(m, m)` matrix. `b` must have shape `(m,)` or `(m, n)`. The result has the same shape as `b`. +Large systems automatically use cuSolverMp when it is available and Legate has +multiple active GPUs. Algorithm selection follows cuPyNumeric 26.06. + Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. Integer and `Bool` inputs are converted to `Float64`. As everywhere else in the package, that conversion needs `@allowpromotion` only when it widens the @@ -51,6 +54,9 @@ being Hermitian. A non-positive-definite input raises an `ErrorException` from the task rather than `LinearAlgebra.PosDefException`, so the `check` keyword is not supported. +Large matrices use cuSolverMp when available with multiple active GPUs. +Other multi-processor configurations use cuPyNumeric's tiled Cholesky algorithm. + Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. Integer and `Bool` inputs are converted to `Float64`. As everywhere else in the package, that conversion needs `@allowpromotion` only when it widens the @@ -183,6 +189,9 @@ end Reduced QR factorization of `A`, with `A ≈ F.Q * F.R`. For an `m × n` input and `k = min(m, n)`, `F.Q` is `m × k` and `F.R` is `k × n`. +Large matrices automatically use cuSolverMp when available with multiple active +GPUs, including tall and wide inputs. + See [`NDArrayQR`](@ref) for why this is not a `LinearAlgebra.QRCompactWY`. Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. diff --git a/src/utilities/version.jl b/src/utilities/version.jl index 6f05603c6..6271e3949 100644 --- a/src/utilities/version.jl +++ b/src/utilities/version.jl @@ -28,7 +28,9 @@ end versioninfo() Prints the cuNumeric build configuration summary, including package -metadata, Julia and compiler version, and paths to core dependencies. +metadata, Julia and compiler version, paths to core dependencies, and +cuSolverMp availability and linear algebra tuning constants. Runtime GPU +eligibility is reported separately from the per-operation size/shape policy. """ function versioninfo(io::IO=stdout) name = string(Base.nameof(@__MODULE__)) @@ -54,6 +56,14 @@ function versioninfo(io::IO=stdout) is_auto_config = legate_auto_config != "0" ? true : false legate_config = is_auto_config ? "auto" : get(ENV, "LEGATE_CONFIG", "not set") + # versioninfo is also called by the test driver with LEGATE_SKIP_RUNTIME. + # Do not start the runtime or query its machine just to print diagnostics. + active = runtime_started() + not_queried = "not queried (runtime inactive)" + mp_available = active ? _LINALG_RUNTIME[].available : not_queried + active_gpus = active ? _LINALG_RUNTIME[].gpus : not_queried + mp_eligible = active ? _LINALG_RUNTIME[].mp_eligible : not_queried + str = """ ─────────────────────────────────────────────── cuNumeric Build Configuration @@ -68,6 +78,18 @@ function versioninfo(io::IO=stdout) Brodcast Fusion: $(FUSE_BROADCAST_EXPRS) Brodcast Min Ops: $(FUSE_BROADCAST_MIN_OPS) + cuSolverMp / Linear Algebra: + Library support: $mp_available + Active GPUs: $active_gpus + MP eligible before size/shape checks: $mp_eligible + MIN_SOLVE_MATRIX_SIZE: $MIN_SOLVE_MATRIX_SIZE (dimension) + MIN_SOLVE_TILE_SIZE: $MIN_SOLVE_TILE_SIZE + MIN_CHOLESKY_MATRIX_SIZE: $MIN_CHOLESKY_MATRIX_SIZE (dimension) + MIN_CHOLESKY_TILE_SIZE: $MIN_CHOLESKY_TILE_SIZE + MIN_QR_MATRIX_SIZE: $MIN_QR_MATRIX_SIZE (elements) + QR_TILE_SIZE: $QR_TILE_SIZE + MAX_CHOLESKY_TILES_PER_PROC: $MAX_CHOLESKY_TILES_PER_PROC + Hostname: $hostname Julia Version: $(VERSION) C++ Compiler: $compiler diff --git a/test/analysis/lifetime.jl b/test/analysis/lifetime.jl index e05c63bbc..45815689c 100644 --- a/test/analysis/lifetime.jl +++ b/test/analysis/lifetime.jl @@ -17,6 +17,35 @@ * Ethan Meitz =# +@testset "linear algebra partition lifetime" begin + a = cuNumeric.NDArray(reshape(Float64.(1:15), 5, 3)) + for fail in (false, true) + released = Ref(0) + function use_partitions(p, q) + for part in (p, q) + finalizer(part.handle) do _ + released[] += 1 + end + end + @test released[] == 0 + fail && error("partition callback failed") + return 7 + end + if fail + @test_throws "partition callback failed" cuNumeric._with_linalg_partitions( + use_partitions, (a, (3, 3)), (a, (3, 3), (2, 1)) + ) + else + @test (@inferred cuNumeric._with_linalg_partitions( + use_partitions, (a, (3, 3)), (a, (3, 3), (2, 1)) + )) == 7 + end + # No GC is needed to release either partition owner, even on failure. + @test released[] == 2 + end + @allowscalar @test Array(a) == reshape(Float64.(1:15), 5, 3) +end + @testset "Array ↔ NDArray value roundtrip (row-major attach)" begin A = rand(Float64, 4, 4) NA = NDArray(A) diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index 5160ade55..e8a1d3c83 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -224,6 +224,39 @@ end end end +# A synthetic configuration exercises MP selection without changing the live +# runtime cache or requiring multiple GPUs. Size remains a runtime argument. +dl_mp_backend(op, shape) = + cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(true, 4, 4)) + +@testset "distributed linear algebra inference" begin + cn = cuNumeric + mp = cn._CuSolverMpLinalg + single = cn._SingleProcLinalg + tiled = cn._TiledCholesky + @test (@inferred cn._LinalgRuntime(true, 4, 4)).mp_eligible + @test (@inferred cn._mp_row_partition(33, 4)) == (9, (4, 1)) + + # Dynamic sizes legitimately infer a small union of backend tags, never Any. + for (op, fallback, shape) in ( + (:solve, single, (cn.MIN_SOLVE_MATRIX_SIZE, cn.MIN_SOLVE_MATRIX_SIZE)), + (:qr, single, (cn.MIN_QR_MATRIX_SIZE, 1)), + (:cholesky, tiled, (cn.MIN_CHOLESKY_MATRIX_SIZE, cn.MIN_CHOLESKY_MATRIX_SIZE)), + ) + @test dl_mp_backend(Val(op), shape) isa mp + @test only(Base.return_types(dl_mp_backend, Tuple{Val{op},NTuple{2,Int}})) == + Union{mp,fallback} + end + + # Infer the real MP launchers without executing collectives on this machine. + for T in (Float32, Float64, ComplexF32, ComplexF64) + a = cn.NDArray{T,2,Nothing} + @test only(Base.return_types(cn._solve!, Tuple{mp,a,a,a})) == a + @test only(Base.return_types(cn._qr!, Tuple{mp,a,a,a})) == Nothing + @test only(Base.return_types(cn._cholesky!, Tuple{mp,a,a})) == a + end +end + @testset verbose = true "eigen" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) A = cuNumeric.NDArray(Matrix{T}(I, 2, 2)) diff --git a/test/array/distributed_linalg.jl b/test/array/distributed_linalg.jl new file mode 100644 index 000000000..3ce8a9ba5 --- /dev/null +++ b/test/array/distributed_linalg.jl @@ -0,0 +1,162 @@ +using Test, LinearAlgebra, Random +import cuNumeric + +dl_host(a) = cuNumeric.allowscalar() do + Array(a) +end +dl_tol(::Type{T}) where {T} = 200 * eps(real(T)) +dl_backend(op, shape, available, gpus, procs) = + cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(available, gpus, procs)) + +@testset "linear algebra task selection" begin + cn = cuNumeric + for (op, limit) in ( + (:solve, cn.MIN_SOLVE_MATRIX_SIZE), (:cholesky, cn.MIN_CHOLESKY_MATRIX_SIZE) + ) + for available in (false, true), gpus in (0, 1, 2, 4), n in (limit - 1, limit, limit + 1) + backend = dl_backend(Val(op), (n, n), available, gpus, max(2, gpus)) + @test (backend isa cn._CuSolverMpLinalg) == (available && gpus > 1 && n >= limit) + end + end + qr_volumes = ( + cn.MIN_QR_MATRIX_SIZE - 1, cn.MIN_QR_MATRIX_SIZE, cn.MIN_QR_MATRIX_SIZE + 1 + ) + for available in (false, true), gpus in (0, 1, 2, 4), volume in qr_volumes + backend = dl_backend(Val(:qr), (volume, 1), available, gpus, max(2, gpus)) + @test (backend isa cn._CuSolverMpLinalg) == ( + available && gpus > 1 && volume >= cn.MIN_QR_MATRIX_SIZE + ) + end + for op in (:solve, :cholesky) + @test dl_backend(Val(op), (4, 9000, 9000), true, 4, 4) isa cn._SingleProcLinalg + end + @test dl_backend(Val(:cholesky), (9000, 9000), true, 1, 1) isa cn._SingleProcLinalg + @test dl_backend(Val(:cholesky), (9000, 9000), false, 4, 4) isa cn._TiledCholesky + @test dl_backend(Val(:qr), (0, 10), true, 4, 4) isa cn._SingleProcLinalg + @test cn._mp_row_partition(33, 4) == (9, (4, 1)) + @test cn._mp_row_partition(2, 4) == (1, (2, 1)) + n = max(cn.MIN_CHOLESKY_MATRIX_SIZE + 1, 100 * cn.MIN_CHOLESKY_TILE_SIZE) + colors = cn._cholesky_color_shape(n, 4) + @test colors[1] == colors[2] + @test 4 <= colors[1] <= 4 * cn.MAX_CHOLESKY_TILES_PER_PROC + @test cn._cholesky_color_shape(cn.MIN_CHOLESKY_MATRIX_SIZE, 4) == (1, 1) +end + +@testset "cached runtime configuration" begin + rt = cuNumeric._LINALG_RUNTIME[] + @test rt.available == cuNumeric.cusolvermp_available() + @test rt.gpus == Int(cuNumeric.Legate.num_gpus()) + @test rt.procs == Int(cuNumeric.Legate.num_procs()) + @test rt.mp_eligible == (rt.available && rt.gpus > 1) + @test cuNumeric.choose_nd_color_shape((33, 33)) == (1, 1) + @test cuNumeric.choose_nd_color_shape((5, 33, 33)) == (rt.procs, 1, 1) + tiles, colors = cuNumeric.prepare_manual_task_for_batched_matrices((5, 33, 33)) + @test tiles == (cld(5, rt.procs), 33, 33) + @test colors == (cld(5, tiles[1]), 1, 1) +end + +function dl_check_solve(T) + cn = cuNumeric + rng = MersenneTwister(71) + n = 33 + a = randn(rng, T, n, n) + T(n) * I + da = cn.NDArray(a) + for nrhs in (1, 3) + b = randn(rng, T, n, nrhs) + db = cn.NDArray(b) + x = da \ db + hx = dl_host(x) + residual = norm(a * hx - b) / (norm(a) * norm(hx) + norm(b)) + @test residual <= dl_tol(T) + @test dl_host(da) == a + @test dl_host(db) == b + end + # Exercise the public vector-RHS reshape path on the selected backend. + b = randn(rng, T, n) + db = cn.NDArray(b) + x = da \ db + @test size(x) == (n,) + @test isapprox(dl_host(x), a \ b; rtol=dl_tol(T)) +end + +function dl_check_cholesky(T; backend=nothing) + cn = cuNumeric + rng = MersenneTwister(72) + n = 33 + z = randn(rng, T, n, n) + a = z * z' + T(n) * I + # Only the lower triangle is meaningful, including for complex input. + input = copy(a) + for j in 1:n, i in 1:(j - 1) + input[i, j] = T(123) + end + da = cn.NDArray(input) + factors = if backend === nothing + f = cholesky(da) + @test f isa Cholesky + f.factors + else + cn._cholesky!(backend, cn.zeros(T, n, n), da) + end + l = dl_host(factors) + residual = norm(a - l * l') / norm(a) + @test residual <= dl_tol(T) + @test istril(l) + @test dl_host(da) == input +end + +function dl_check_qr(T) + cn = cuNumeric + rng = MersenneTwister(73) + @testset "QR shape ($m, $n)" for (m, n) in ( + (33, 33), (65, 17), (17, 65), (7, 1), (1, 7) + ) + a = randn(rng, T, m, n) + da = cn.NDArray(a) + f = qr(da) + @test f isa cn.NDArrayQR + q, r = f.Q, f.R + k = min(m, n) + @test size(q) == (m, k) + @test size(r) == (k, n) + hq, hr = dl_host(q), dl_host(r) + residual = norm(a - hq * hr) / norm(a) + @test residual <= dl_tol(T) + @test norm(hq' * hq - I) / sqrt(k) <= dl_tol(T) + @test istriu(hr) + @test dl_host(da) == a + end +end + +@testset "distributed linear algebra numerics" begin + for T in (Float32, Float64, ComplexF32, ComplexF64) + @testset "$T" begin + dl_check_solve(T) + dl_check_cholesky(T) + dl_check_qr(T) + dl_check_cholesky(T; backend=cuNumeric._TiledCholesky()) + end + end +end + +@testset "empty and invalid solves" begin + cn = cuNumeric + @test size(cn.zeros(Float64, 0, 0) \ cn.zeros(Float64, 0)) == (0,) + @test size(cn.zeros(Float64, 0, 0) \ cn.zeros(Float64, 0, 3)) == (0, 3) + @test size(cn.zeros(Float64, 3, 3) \ cn.zeros(Float64, 3, 0)) == (3, 0) + @test_throws ArgumentError cn.zeros(Float64, 2, 3) \ cn.zeros(Float64, 2) + @test_throws ArgumentError cn.zeros(Float64, 3, 3) \ cn.zeros(Float64, 2) + # Reject mismatched batches before constructing partitions, including empties. + for (a_batch, b_batch) in ((2, 3), (3, 2), (0, 2), (2, 0)) + @test_throws "matching batch dimensions" cn.batched_solve( + cn.zeros(Float64, a_batch, 3, 3), cn.zeros(Float64, b_batch, 3, 1) + ) + end + @test size(cn.batched_solve(cn.zeros(Float64, 0, 3, 3), cn.zeros(Float64, 0, 3, 1))) == + (0, 3, 1) + for (m, n) in ((0, 0), (0, 3), (3, 0)) + f = qr(cn.zeros(Float64, m, n)) + @test size(f.Q) == (m, min(m, n)) + @test size(f.R) == (min(m, n), n) + end +end diff --git a/test/array/linalg_edge_cases.jl b/test/array/linalg_edge_cases.jl new file mode 100644 index 000000000..30b7ca97a --- /dev/null +++ b/test/array/linalg_edge_cases.jl @@ -0,0 +1,150 @@ +using Test, LinearAlgebra, Random +using cuNumeric: cuNumeric + +le_host(a) = cuNumeric.allowscalar() do + return Array(a) +end +le_tol(::Type{T}) where {T} = 200 * eps(real(T)) +le_residual(a, b) = norm(a - b) / max(norm(a), one(real(eltype(a)))) + +# Keep the padded parent so we can check that the operation preserves both +# its input and the elements outside the view. Do not copy the view before +# passing it to the operation: the backend must handle its layout. +function le_input(a, layout) + if layout == :slice + m, n = size(a) + parent = fill(eltype(a)(19), m + 2, n + 2) + parent[2:(m + 1), 2:(n + 1)] = a + dp = cuNumeric.NDArray(parent) + return view(dp, 2:(m + 1), 2:(n + 1)), dp, parent + elseif layout == :transpose + parent = copy(transpose(a)) + dp = cuNumeric.NDArray(parent) + return cuNumeric.transpose(dp), dp, parent + end + dp = cuNumeric.NDArray(a) + return dp, dp, a +end + +function le_qr(a, da) + m, n = size(a) + k = min(m, n) + f = qr(da) + q, r = le_host(f.Q), le_host(f.R) + @test size(q) == (m, k) + @test size(r) == (k, n) + @test le_residual(a, q * r) <= le_tol(eltype(a)) + @test norm(q' * q - I) <= le_tol(eltype(a)) * k + @test istriu(r) +end + +function le_svd(a, da) + m, n = size(a) + for full in (false, true) + f = svd(da; full) + u, s, vt = le_host(f.U), le_host(f.S), le_host(f.Vt) + @test size(u) == (m, full ? m : n) + @test size(s) == (n,) + @test size(vt) == (n, n) + @test le_residual(a, u[:, 1:n] * Diagonal(s) * vt) <= le_tol(eltype(a)) + @test norm(u' * u - I) <= le_tol(eltype(a)) * size(u, 2) + @test norm(vt * vt' - I) <= le_tol(eltype(a)) * n + @test all(s .>= 0) + @test issorted(s; rev=true) + @test isapprox(s, svdvals(a); atol=le_tol(eltype(a)), rtol=le_tol(eltype(a))) + end +end + +function le_eigen(a, da) + f = eigen(da) + w, v = le_host(f.values), le_host(f.vectors) + n = size(a, 1) + @test size(w) == (n,) + @test size(v) == (n, n) + @test norm(a * v - v * Diagonal(w)) / max(norm(a) * norm(v), 1) <= le_tol(eltype(a)) + @test all(j -> isapprox(norm(v[:, j]), 1; atol=le_tol(eltype(a))), 1:n) + # The fixtures are Hermitian, so their spectra are real, including repeated + # zero eigenvalues. Sorting by real part avoids arbitrary eigenvector order. + expected = eigvals(Hermitian(a)) + for values in (w, le_host(eigvals(da))) + @test maximum(abs, imag.(values)) <= le_tol(eltype(a)) * max(norm(a), 1) + @test isapprox( + sort(real.(values)), expected; atol=le_tol(eltype(a)), rtol=le_tol(eltype(a)) + ) + end +end + +@testset "linear algebra degenerate inputs" begin + @testset "$T" for T in (Float32, Float64, ComplexF32, ComplexF64) + @testset "QR/SVD $kind ($m, $n)" for kind in (:zero, :rank_one), + (m, n) in ((4, 4), (6, 4), (4, 6)) + + u = T.(1:m) + T <: Complex && (u .+= im .* reverse(u)) + a = kind == :zero ? zeros(T, m, n) : u * transpose(T.(1:n)) + da = cuNumeric.NDArray(a) + le_qr(a, da) + m >= n && le_svd(a, da) + @test le_host(da) == a + end + @testset "square $kind" for kind in (:zero, :rank_one) + a = zeros(T, 4, 4) + kind == :rank_one && (a[1, 1] = 3) + da = cuNumeric.NDArray(a) + le_eigen(a, da) + @test le_host(da) == a + # A zero pivot is exact here. Materialize inside @test_throws so + # asynchronous task errors are observed by the assertion. + for b in (ones(T, 4), ones(T, 4, 2)) + db = cuNumeric.NDArray(b) + @test_throws "Singular matrix" le_host(da \ db) + @test le_host(db) == b + @test le_host(da) == a + end + @test_throws "Matrix is not positive definite" le_host(cholesky(da).factors) + @test le_host(da) == a + end + end +end + +@testset "linear algebra input layouts" begin + @testset "$T $layout" for T in (Float32, Float64, ComplexF32, ComplexF64), + layout in (:slice, :transpose) + + rng = MersenneTwister(81) + @testset "QR/SVD ($m, $n)" for (m, n) in ((4, 4), (6, 4), (4, 6)) + a = randn(rng, T, m, n) + da, dp, parent = le_input(a, layout) + le_qr(a, da) + m >= n && le_svd(a, da) + @test le_host(dp) == parent + end + z = randn(rng, T, 4, 4) + a = z * z' + T(4) * I + da, dp, parent = le_input(a, layout) + l = le_host(cholesky(da).factors) + @test le_residual(a, l * l') <= le_tol(T) + @test istril(l) + @test le_host(dp) == parent + le_eigen(a, da) + @test le_host(dp) == parent + # Also exercise a non-Hermitian solve so transpose/conjugation mistakes + # cannot be hidden by the Cholesky/eigen fixture's symmetry. + a = z + T(8) * I + da, dp, parent = le_input(a, layout) + @testset "solve $nrhs RHS" for nrhs in (1, 2) + b = randn(rng, T, 4, nrhs) + db, bp, bparent = le_input(b, layout) + x = le_host(da \ db) + @test size(x) == size(b) + @test le_residual(b, a * x) <= le_tol(T) + @test le_host(bp) == bparent + @test le_host(dp) == parent + end + # A zero RHS is valid even though a zero coefficient matrix is not. + for b in (zeros(T, 4), zeros(T, 4, 2)) + @test le_host(da \ cuNumeric.NDArray(b)) == b + end + @test le_host(dp) == parent + end +end diff --git a/test/linalg_errors.jl b/test/linalg_errors.jl new file mode 100644 index 000000000..f02cb7d7d --- /dev/null +++ b/test/linalg_errors.jl @@ -0,0 +1,30 @@ +# Opt-in: run each case in a separate process under an external timeout. +# A failed collective can poison its runtime. +using cuNumeric, LinearAlgebra, Test + +length(ARGS) == 1 && only(ARGS) in ("solve", "cholesky", "tiled_cholesky") || + error("Usage: julia --project test/linalg_errors.jl solve|cholesky|tiled_cholesky") +op = only(ARGS) +a = cuNumeric.zeros(Float64, 33, 33) +b = cuNumeric.ones(Float64, 33, 1) +cuNumeric.versioninfo() +backend = op == "tiled_cholesky" ? cuNumeric._TiledCholesky() : + cuNumeric._linalg_backend(Val(Symbol(op)), a) +println("Testing numerical failure with ", typeof(backend)) + +@testset "$op numerical failure" begin + expected = op == "solve" ? r"singular"i : r"positive definite"i + @test_throws expected begin + out = if op == "solve" + a \ b + elseif op == "cholesky" + cholesky(a).factors + else + cuNumeric._cholesky!(backend, similar(a), a) + end + cuNumeric.allowscalar() do + Array(out) # Demand the result so deferred task failures surface. + end + cuNumeric.Legate.issue_execution_fence(true) + end +end diff --git a/test/linalg_preferences.jl b/test/linalg_preferences.jl new file mode 100644 index 000000000..dc5fbf752 --- /dev/null +++ b/test/linalg_preferences.jl @@ -0,0 +1,34 @@ +using Test, Pkg, CNPreferences + +@testset "linear algebra preferences" begin + mktempdir() do dir + # Preference writes must not alter the developer's active project. + write(joinpath(dir, "Project.toml"), + "[deps]\nCNPreferences = \"$(Base.PkgId(CNPreferences).uuid)\"\n") + previous = Base.active_project() + try + Pkg.activate(dir; io=devnull) + settings = ( + MIN_SOLVE_MATRIX_SIZE=32, MIN_SOLVE_TILE_SIZE=4, + MIN_CHOLESKY_MATRIX_SIZE=32, MIN_CHOLESKY_TILE_SIZE=4, + MIN_QR_MATRIX_SIZE=1, QR_TILE_SIZE=4, MAX_CHOLESKY_TILES_PER_PROC=2, + ) + CNPreferences.set_linalg!(; settings...) + for (key, value) in pairs(settings) + @test cuNumeric.load_preference(CNPreferences, string(key)) == value + end + path = joinpath(dir, "LocalPreferences.toml") + saved = read(path, String) + for bad in (0, -1, true, 1.5, "4") + @test_throws ArgumentError CNPreferences.set_linalg!(; + MIN_SOLVE_MATRIX_SIZE=100, QR_TILE_SIZE=bad + ) + @test read(path, String) == saved + end + @test_throws ArgumentError CNPreferences.set_linalg!(; QR_TIEL_SIZE=4) + @test read(path, String) == saved + finally + Pkg.activate(dirname(previous); io=devnull) + end + end +end diff --git a/test/runtests.jl b/test/runtests.jl index 1ba7b10f4..d509ee910 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -30,6 +30,8 @@ end testsuite = find_tests(@__DIR__) delete!(testsuite, "util") +# These cases deliberately fail a runtime; run individually under a timeout. +delete!(testsuite, "linalg_errors") delete!(testsuite, "array/unary/tests") delete!(testsuite, "array/binary/tests") From 43195270192b3c80098e3586bfe69582f16bd94b Mon Sep 17 00:00:00 2001 From: ejmeitz <54505069+ejmeitz@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:03:06 -0400 Subject: [PATCH 27/49] bump compat versions --- Project.toml | 4 ++-- lib/CNPreferences/Project.toml | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Project.toml b/Project.toml index 0358c7d0b..eb8961bc8 100644 --- a/Project.toml +++ b/Project.toml @@ -34,7 +34,7 @@ cuNumericTensorOperationsExt = "TensorOperations" [compat] AbstractFFTs = "1.5" -CNPreferences = "0.1.3" +CNPreferences = "0.1.4" CUDACore = "6.2" CUDATools = "6.2" CxxWrap = "0.17" @@ -51,7 +51,7 @@ Random = "1" StaticArrays = "1" StatsBase = "0.34" TensorOperations = "5.8" -cunumeric_jl_wrapper_jll = "26.6.1" +cunumeric_jl_wrapper_jll = "26.6.2" cupynumeric_jll = "26.6.0" julia = "1.10" diff --git a/lib/CNPreferences/Project.toml b/lib/CNPreferences/Project.toml index 376ace094..26ecc6905 100644 --- a/lib/CNPreferences/Project.toml +++ b/lib/CNPreferences/Project.toml @@ -1,7 +1,7 @@ name = "CNPreferences" uuid = "3e078157-ea10-49d5-bf32-908f777cd46f" authors = ["""David Krasowska and Ethan Meitz """] -version = "0.1.3" +version = "0.1.4" [deps] LegatePreferences = "8028f36a-2b64-49e9-aa04-2d0933fd2ed9" From 1277af13c1c2116aca474532207c48d542dcf4af Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sun, 13 Sep 2026 22:06:36 -0500 Subject: [PATCH 28/49] Container Opt-In CI (#194) [container] [skip dev] [skip jll] --- .github/workflows/container.yml | 73 +++++++++++++++++- .pre-commit-config.yaml | 5 ++ Dockerfile | 23 +++--- Dockerfile.developer | 127 ++++++++++++++++++++++++++++++++ 4 files changed, 214 insertions(+), 14 deletions(-) create mode 100644 Dockerfile.developer diff --git a/.github/workflows/container.yml b/.github/workflows/container.yml index 323a1e539..1d87c6d2e 100644 --- a/.github/workflows/container.yml +++ b/.github/workflows/container.yml @@ -11,6 +11,12 @@ on: type: boolean required: false default: false + # Build the image associated with a published GitHub release. + release: + types: [published] + # GitHub Actions cannot filter pushes by commit message at trigger time, so + # the job-level condition below implements the [container] opt-in. + push: workflow_run: workflows: ['CI'] types: [completed] @@ -18,7 +24,20 @@ on: - main jobs: push_to_registry: - if: ${{ !contains(toJSON(github.event), '[skip ci]') && (github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success') }} + if: >- + ${{ + !contains(toJSON(github.event), '[skip ci]') && + ( + github.event_name == 'workflow_dispatch' || + github.event_name == 'release' || + (github.event_name == 'push' && contains(github.event.head_commit.message, '[container]')) || + ( + github.event_name == 'workflow_run' && + github.event.workflow_run.conclusion == 'success' && + !contains(github.event.workflow_run.head_commit.message, '[container]') + ) + ) + }} name: Container for ${{ matrix.platform }} - Julia ${{ matrix.julia }} - CUDA ${{ matrix.cuda }} permissions: contents: read @@ -40,6 +59,42 @@ jobs: steps: - name: Check out the repo uses: actions/checkout@v4 + with: + ref: ${{ inputs.tag || github.event.release.tag_name || github.event.workflow_run.head_sha || github.sha }} + fetch-depth: 0 + + - name: Select wrapper build + id: wrapper-build + env: + EVENT_NAME: ${{ github.event_name }} + INPUT_TAG: ${{ inputs.tag }} + run: | + # Published releases and explicitly selected tags use their released + # wrapper. Branch builds use the in-tree wrapper when it differs from + # the wrapper published from main. + if [[ "$EVENT_NAME" == "release" || -n "$INPUT_TAG" ]]; then + changed=false + else + git fetch --no-tags origin +refs/heads/main:refs/remotes/origin/main + if git diff --quiet origin/main...HEAD -- lib/cunumeric_jl_wrapper; then + changed=false + else + status=$? + if [[ "$status" -eq 1 ]]; then + changed=true + else + exit "$status" + fi + fi + fi + + if [[ "$changed" == "true" ]]; then + echo "Wrapper changes detected; building Dockerfile.developer" + echo "dockerfile=Dockerfile.developer" >> "$GITHUB_OUTPUT" + else + echo "No wrapper changes detected; building Dockerfile" + echo "dockerfile=Dockerfile" >> "$GITHUB_OUTPUT" + fi - name: Get package spec id: pkg @@ -47,6 +102,12 @@ jobs: if [[ -n "${{ inputs.tag }}" ]]; then echo "ref=${{ inputs.tag }}" >> $GITHUB_OUTPUT echo "name=${{ inputs.tag }}" >> $GITHUB_OUTPUT + elif [[ "${{ github.event_name }}" == "release" ]]; then + echo "ref=${{ github.event.release.tag_name }}" >> $GITHUB_OUTPUT + echo "name=${{ github.event.release.tag_name }}" >> $GITHUB_OUTPUT + elif [[ "${{ github.event_name }}" == "workflow_run" ]]; then + echo "ref=${{ github.event.workflow_run.head_sha }}" >> $GITHUB_OUTPUT + echo "name=${{ github.event.workflow_run.head_branch }}" >> $GITHUB_OUTPUT elif [[ "${{ github.ref_type }}" == "tag" ]]; then echo "ref=${{ github.ref_name }}" >> $GITHUB_OUTPUT echo "name=${{ github.ref_name }}" >> $GITHUB_OUTPUT @@ -64,6 +125,12 @@ jobs: VERSION=$(grep "^version = " Project.toml | cut -d'"' -f2) echo "version=$VERSION" >> $GITHUB_OUTPUT + # Include the checked-out source revision in every image tag. In + # particular, github.sha identifies this workflow run for + # workflow_run events, not necessarily the commit being packaged. + COMMIT_SHA=$(git rev-parse --short=12 HEAD) + echo "commit=$COMMIT_SHA" >> $GITHUB_OUTPUT + - name: Get CUDA major version id: cuda run: | @@ -97,7 +164,7 @@ jobs: with: images: ghcr.io/${{ github.repository }} tags: | - type=raw,value=${{ steps.pkg.outputs.name }}-julia${{ matrix.julia }}-cuda${{ steps.cuda.outputs.major }}.${{ steps.cuda.outputs.minor }} + type=raw,value=${{ steps.pkg.outputs.commit }}-julia${{ matrix.julia }}-cuda${{ steps.cuda.outputs.major }}.${{ steps.cuda.outputs.minor }} type=raw,value=${{ steps.pkg.outputs.name }},enable=${{ matrix.default == true && (github.ref_type == 'tag' || inputs.tag != '') }} type=raw,value=latest,enable=${{ matrix.default == true && (github.ref_type == 'tag' || (inputs.tag != '' && inputs.mark_as_latest)) }} type=raw,value=dev,enable=${{ matrix.default == true && github.ref_type == 'branch' && inputs.tag == '' }} @@ -130,7 +197,7 @@ jobs: - name: Build image uses: docker/build-push-action@v6 with: - file: Dockerfile + file: ${{ steps.wrapper-build.outputs.dockerfile }} load: true push: false provenance: false # the build fetches the repo again, so provenance tracking is not useful diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 03d022894..938d9539b 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -14,6 +14,11 @@ repos: - id: clang-format types_or: [c++, c] + - repo: https://github.com/reteps/dockerfmt + rev: v0.5.4 + hooks: + - id: dockerfmt + - repo: local hooks: - id: julia-formatter diff --git a/Dockerfile b/Dockerfile index 00f919780..c42fcd244 100644 --- a/Dockerfile +++ b/Dockerfile @@ -27,11 +27,11 @@ ENV JULIA_NUM_THREADS=auto ARG CUNUMERIC_VERSION=25.10.00 ARG PACKAGE_SPEC_CUDA=CUDA LABEL org.opencontainers.image.authors="David Krasowska , Ethan Meitz " \ - org.opencontainers.image.description="A cuNumeric.jl container with CUDA ${CUDA_VERSION_MAJOR_MINOR}, Julia ${JULIA_VERSION}, and cuNumeric ${CUNUMERIC_VERSION}" \ - org.opencontainers.image.title="cuNumeric.jl" \ + org.opencontainers.image.description="A cuNumeric.jl container with CUDA ${CUDA_VERSION_MAJOR_MINOR}, Julia ${JULIA_VERSION}, and cuNumeric ${CUNUMERIC_VERSION}" \ + org.opencontainers.image.title="cuNumeric.jl" \ # org.opencontainers.image.url="https://juliagpu.org/cuda/" \ - org.opencontainers.image.source="https://github.com/JuliaLegate/cuNumeric.jl" \ - org.opencontainers.image.licenses="MIT" + org.opencontainers.image.source="https://github.com/JuliaLegate/cuNumeric.jl" \ + org.opencontainers.image.licenses="MIT" COPY scripts/test_container.sh /workspace/test_container.sh RUN chmod +x /workspace/test_container.sh @@ -39,8 +39,8 @@ RUN cat /workspace/test_container.sh # # system-wide packages RUN apt-get update && apt-get install -y \ - wget curl git build-essential && \ - rm -rf /var/lib/apt/lists/* + wget curl git build-essential \ + && rm -rf /var/lib/apt/lists/* ENV JULIA_DEPOT_PATH=/usr/local/share/julia @@ -65,9 +65,10 @@ RUN source /etc/.env && source /etc/.env && julia --color=yes -e ' \ using Pkg; \ Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = ENV["REF"])) \ ' -RUN #= remove useless stuff =# \ - cd /usr/local/share/julia && \ - rm -rf registries scratchspaces logs +RUN \ + #= remove useless stuff =# \ + cd /usr/local/share/julia \ + && rm -rf registries scratchspaces logs # user environment @@ -98,8 +99,8 @@ end pushfirst!(DEPOT_PATH, "/depot") EOF -RUN apt-get clean && \ - rm -rf /var/lib/apt/lists/* +RUN apt-get clean \ + && rm -rf /var/lib/apt/lists/* ENV LEGATE_AUTO_CONFIG=1 diff --git a/Dockerfile.developer b/Dockerfile.developer new file mode 100644 index 000000000..9d163d88e --- /dev/null +++ b/Dockerfile.developer @@ -0,0 +1,127 @@ +ARG JULIA_VERSION=1.11 +FROM julia:${JULIA_VERSION} + +ARG CUDA_MAJOR=13 +ARG CUDA_MINOR=0 +ENV CUDA_VERSION_MAJOR_MINOR="${CUDA_MAJOR}.${CUDA_MINOR}" + +ARG REF=main +ENV REF=${REF} +# using bash +SHELL ["/bin/bash", "-c"] +ENV DEBIAN_FRONTEND=noninteractive + +# force turn off legate auto config for precompilation. +ENV LEGATE_AUTO_CONFIG=0 + +# much of the CUDA.jl setup is from Tim Besard +# CUDA.jl Dockerfile https://github.com/JuliaGPU/CUDA.jl/blob/master/Dockerfile +# Thank you Tim for the reccomendation. + +ARG JULIA_CPU_TARGET=native +ENV JULIA_CPU_TARGET=${JULIA_CPU_TARGET} + +ENV JULIA_NUM_THREADS=auto + +ARG CUNUMERIC_VERSION=25.10.00 +ARG PACKAGE_SPEC_CUDA=CUDA +LABEL org.opencontainers.image.authors="David Krasowska , Ethan Meitz " \ + org.opencontainers.image.description="A cuNumeric.jl developer container with CUDA ${CUDA_VERSION_MAJOR_MINOR}, Julia ${JULIA_VERSION}, and cuNumeric ${CUNUMERIC_VERSION}" \ + org.opencontainers.image.title="cuNumeric.jl" \ + org.opencontainers.image.source="https://github.com/JuliaLegate/cuNumeric.jl" \ + org.opencontainers.image.licenses="MIT" + +COPY scripts/test_container.sh /workspace/test_container.sh +RUN chmod +x /workspace/test_container.sh +RUN cat /workspace/test_container.sh + +# system-wide packages +RUN apt-get update && apt-get install -y \ + wget curl git build-essential \ + && rm -rf /var/lib/apt/lists/* + +# Developer mode needs the same CMake version used by developer CI. +ARG CMAKE_VERSION=3.30.7 +RUN wget --no-verbose \ + "https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-linux-x86_64.sh" \ + -O /tmp/cmake-installer.sh \ + && sh /tmp/cmake-installer.sh --skip-license --prefix=/usr/local \ + && rm /tmp/cmake-installer.sh + +ENV JULIA_DEPOT_PATH=/usr/local/share/julia +ENV PATH="/usr/local/.juliaup/bin:/usr/local/bin:$PATH" + +# install CUDA.jl itself. +RUN julia --color=yes -e 'using Pkg; Pkg.add("CUDA"); using CUDA; CUDA.set_runtime_version!(VersionNumber(ENV["CUDA_VERSION_MAJOR_MINOR"]))' +RUN julia -e 'using Pkg; Pkg.add(name = "CUDA_Driver_jll", version = "13.0.0"); Pkg.add("CUDA_Runtime_jll")' +RUN echo "export LD_LIBRARY_PATH=\$(julia -e 'print(Sys.BINDIR * \"/../lib\")'):\$(julia -e 'using CUDA_Driver_jll; print(joinpath(CUDA_Driver_jll.artifact_dir, \"lib\"))'):\$(julia -e 'using CUDA_Runtime_jll; print(joinpath(CUDA_Runtime_jll.artifact_dir, \"lib\"))'):\$LD_LIBRARY_PATH" >> /etc/.env +RUN chmod +x /etc/.env +RUN cat /etc/.env + +RUN echo "Install Legate and cuNumeric.jl with the in-tree wrapper" +RUN source /etc/.env && julia --color=yes -e ' \ + using Pkg; \ + Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/Legate.jl", rev = "main")) \ +' + +# The workflow checks out the requested ref before starting the Docker build, +# so this source tree is the exact package revision represented by the image. +COPY Project.toml /opt/cuNumeric.jl/Project.toml +COPY deps /opt/cuNumeric.jl/deps +COPY ext /opt/cuNumeric.jl/ext +COPY lib /opt/cuNumeric.jl/lib +COPY scripts /opt/cuNumeric.jl/scripts +COPY src /opt/cuNumeric.jl/src +COPY test /opt/cuNumeric.jl/test + +RUN source /etc/.env && julia --color=yes -e ' \ + using Pkg; \ + ENV["JULIA_PKG_PRECOMPILE_AUTO"] = "0"; \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl/lib/CNPreferences")); \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl")); \ + using CNPreferences; \ + CNPreferences.use_developer_mode(); \ + Pkg.build("cuNumeric"); \ + Pkg.precompile() \ +' +RUN \ + #= remove useless stuff =# \ + cd /usr/local/share/julia \ + && rm -rf registries scratchspaces logs + +# user environment + +# we hard-code the primary depot regardless of the actual user, i.e., we do not let it +# default to `$HOME/.julia`. this is for compatibility with `docker run --user`, in which +# case there might not be a (writable) home directory. + +RUN mkdir -m 0777 /depot +# we add the user environment from a start-up script +# so that the user can mount `/depot` for persistency +COPY < Date: Thu, 17 Sep 2026 00:12:00 -0500 Subject: [PATCH 29/49] Preserve fused compound broadcasts in accelerate (#200) --- src/scoping/broadcast_lifetimes.jl | 5 +++-- src/scoping/lifetimes.jl | 5 +++-- src/scoping/util.jl | 9 ++++++--- test/analysis/accelerate.jl | 9 +++++++++ test/workflows/grayscott.jl | 28 ++++++++++++++++++++++++++++ 5 files changed, 49 insertions(+), 7 deletions(-) diff --git a/src/scoping/broadcast_lifetimes.jl b/src/scoping/broadcast_lifetimes.jl index f1228f58c..cbd763292 100644 --- a/src/scoping/broadcast_lifetimes.jl +++ b/src/scoping/broadcast_lifetimes.jl @@ -74,10 +74,11 @@ function rewrite_broadcast_lifetimes(scope) return :($lhs = $new_rhs), temps end - # A `.=` RHS is a broadcast tree: only its slices are hoisted. + # A dotted-assignment RHS is a broadcast tree: only its slices are hoisted. broadcast_assignment = _broadcast_assignment(expr) if !isnothing(broadcast_assignment) (; lhs, rhs) = broadcast_assignment + op = expr.head # NDArray slices are writable views. Hoist the destination slice so # the fused broadcast writes through it, then destroy its handle. lhs_reference = _reference(lhs) @@ -87,7 +88,7 @@ function rewrite_broadcast_lifetimes(scope) new_lhs, lhs_temps = fresh_tmp(lhs) end new_rhs, rhs_temps = rewrite_lazy_broadcast(rhs, Dict{Any,Symbol}()) - return Expr(:(.=), new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) + return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) end reference = _reference(expr) diff --git a/src/scoping/lifetimes.jl b/src/scoping/lifetimes.jl index 7f3d97db4..b13bf5d35 100644 --- a/src/scoping/lifetimes.jl +++ b/src/scoping/lifetimes.jl @@ -38,17 +38,18 @@ function rewrite_eager_lifetimes(scope) broadcast_assignment = _broadcast_assignment(expr) if !isnothing(broadcast_assignment) (; lhs, rhs) = broadcast_assignment + op = expr.head new_lhs, lhs_temps = rewrite(lhs) # Do not hoist the top-level call of the RHS to preserve fusion. call = _call(rhs) if !isnothing(call) new_rhs_args, rhs_temps = _maphoist(rewrite, call.args) new_rhs = Expr(:call, call.f, new_rhs_args...) - return Expr(:(.=), new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) + return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) end new_rhs, rhs_temps = rewrite(rhs) - return Expr(:(.=), new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) + return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) end reference = _reference(expr) diff --git a/src/scoping/util.jl b/src/scoping/util.jl index 64f68366e..985adce2c 100644 --- a/src/scoping/util.jl +++ b/src/scoping/util.jl @@ -25,9 +25,12 @@ function _assignment(expr) end function _broadcast_assignment(expr) - MacroTools.isexpr(expr, :(.=)) || return nothing - MacroTools.@capture(expr, lhs_ .= rhs_) || return nothing - return (; lhs, rhs) + expr isa Expr && length(expr.args) == 2 || return nothing + op = expr.head + op isa Symbol || return nothing + spelling = string(op) + startswith(spelling, ".") && endswith(spelling, "=") || return nothing + return (; lhs=expr.args[1], rhs=expr.args[2]) end function _call(expr) diff --git a/test/analysis/accelerate.jl b/test/analysis/accelerate.jl index 28eee8897..8a1ccc698 100644 --- a/test/analysis/accelerate.jl +++ b/test/analysis/accelerate.jl @@ -129,6 +129,15 @@ using InteractiveUtils: code_typed end )) + # Compound dotted assignments must remain one broadcast tree. Hoisting + # their RHS would add a full-size temporary and a second GPU launch. + compound = string(expand(:(function update!(x, alpha, p) + x .+= alpha .* p + x + end))) + @test occursin("x .+= alpha .* p", compound) + @test !occursin(r"tmp\d+ = alpha \.\* p", compound) + if cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric.HAS_CUDA # A same-shape chain fuses into one multi-output launch and still # frees the hoisted slice temporaries. diff --git a/test/workflows/grayscott.jl b/test/workflows/grayscott.jl index 7be2662ba..e3dbc14d7 100644 --- a/test/workflows/grayscott.jl +++ b/test/workflows/grayscott.jl @@ -282,6 +282,8 @@ function test_scoping_rewrite_pipeline() @testset "Syntax helpers" begin @test utils._assignment(:(x = y)) == (lhs=:x, rhs=:y) @test utils._broadcast_assignment(:(A[:] .= x)).rhs == :x + @test utils._broadcast_assignment(:(A[:] .+= x)).rhs == :x + @test utils._broadcast_assignment(:(A[:] .-= x)).rhs == :x @test utils._call(:(f(x, y))) == (f=:f, args=Any[:x, :y]) @test utils._dotcall(:(f.(x, y))) == (f=:f, args=Any[:x, :y]) @test utils._reference(:(A[i, j])) == (array=:A, indices=Any[:i, :j]) @@ -375,6 +377,32 @@ function test_scoping_rewrite_pipeline() end end + @testset "Compound broadcast assignments stay fused" begin + source = quote + x .+= alpha .* p + @views Ap[2:end] .-= lower[2:end] .* p[1:(end - 1)] + end + + for rewrite in ( + cuNumeric.rewrite_broadcast_lifetimes, cuNumeric.rewrite_eager_lifetimes + ) + cuNumeric.counter[] = 0 + try + rewritten, assigned = rewrite(source) + rendered = sprint(Base.show_unquoted, utils._strip_lines(rewritten)) + + @test occursin("x .+= alpha .* p", rendered) + @test occursin(".-=", rendered) + @test count(line -> occursin(".*", line), eachline(IOBuffer(rendered))) == 2 + @test !occursin(r"tmp\d+ = alpha \.\* p", rendered) + @test !occursin(r"tmp\d+ = .*lower.* \.\* .*p", rendered) + @test !isempty(assigned) # slice views are still lifetime-managed + finally + cuNumeric.counter[] = 0 + end + end + end + @testset "Scalar arithmetic stays inline" begin source = quote C .= A ./ args.dx^2 .+ (args.f + args.k) From 9f47cce3f9562218e739dcefef2ee9a2e821aed6 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Thu, 17 Sep 2026 21:45:41 -0400 Subject: [PATCH 30/49] mapreduce (#196) * mapreduce --- deps/build.jl | 10 +- deps/cxxwrap.jl | 54 ++++ deps/cxxwrap_probe/CMakeLists.txt | 42 +++ docs/make.jl | 1 + docs/src/api.md | 3 + docs/src/api_mapreduce.md | 53 ++++ docs/src/api_unary.md | 3 + docs/src/developer_mode.md | 10 + lib/cunumeric_jl_wrapper/CMakeLists.txt | 3 +- lib/cunumeric_jl_wrapper/include/mapreduce.h | 39 +++ .../include/ndarray_c_api.h | 3 + lib/cunumeric_jl_wrapper/include/ptx.h | 27 ++ lib/cunumeric_jl_wrapper/src/cuda.cpp | 49 ++-- lib/cunumeric_jl_wrapper/src/mapreduce.cpp | 222 +++++++++++++++ lib/cunumeric_jl_wrapper/src/ndarray.cpp | 6 + lib/cunumeric_jl_wrapper/src/wrapper.cpp | 84 ++++++ scripts/install_cxxwrap.sh | 54 ++-- src/cuNumeric.jl | 2 + src/cuda/README.md | 36 +++ src/cuda/mapreduce.jl | 260 ++++++++++++++++++ src/cuda/strided_device_array.jl | 4 +- src/ndarray/broadcast.jl | 17 +- src/ndarray/detail/ndarray.jl | 12 + src/ndarray/mapreduce.jl | 157 +++++++++++ src/ndarray/ndarray.jl | 24 +- test/array/conversion_lifetimes.jl | 31 +++ test/array/mapreduce_policy.jl | 77 ++++++ test/build_cxxwrap.jl | 92 +++++++ test/gpu_only/mapreduce.jl | 197 +++++++++++++ test/gpu_only/mapreduce_full.jl | 249 +++++++++++++++++ 30 files changed, 1736 insertions(+), 85 deletions(-) create mode 100644 deps/cxxwrap.jl create mode 100644 deps/cxxwrap_probe/CMakeLists.txt create mode 100644 docs/src/api_mapreduce.md create mode 100644 lib/cunumeric_jl_wrapper/include/mapreduce.h create mode 100644 lib/cunumeric_jl_wrapper/include/ptx.h create mode 100644 lib/cunumeric_jl_wrapper/src/mapreduce.cpp create mode 100644 src/cuda/README.md create mode 100644 src/cuda/mapreduce.jl create mode 100644 src/ndarray/mapreduce.jl create mode 100644 test/array/mapreduce_policy.jl create mode 100644 test/build_cxxwrap.jl create mode 100644 test/gpu_only/mapreduce.jl create mode 100644 test/gpu_only/mapreduce_full.jl diff --git a/deps/build.jl b/deps/build.jl index 68ac071ce..e9e965557 100644 --- a/deps/build.jl +++ b/deps/build.jl @@ -19,6 +19,7 @@ using Pkg using Preferences +using Libdl: dlext # The build only needs Legate's paths/tooling, not a running runtime. # Setting this env prevents a segfault on Julia 1.12 @@ -35,6 +36,7 @@ using OpenBLAS32_jll: OpenBLAS32_jll const BuildTools = Legate.BuildTools include("version.jl") +include("cxxwrap.jl") function build_cpp_wrapper( repo_root, cupynumeric_loc, legate_loc, blas_loc, install_root; @@ -60,15 +62,19 @@ function build_deps(pkg_root, cupynumeric_root, blas_root; cuda_root=nothing, cu ) end - BuildTools.build_jlcxxwrap( + ensure_cxxwrap( pkg_root, get_cupynumeric_version(cupynumeric_root); - log_dir=@__DIR__, is_compatible=is_supported_version, + log_dir=@__DIR__, ) build_cpp_wrapper( pkg_root, cupynumeric_root, up_dir(legate_lib), blas_root, install_lib; cuda_root, cuda_enabled, ) + for name in ("cunumeric_jl_wrapper", "cunumeric_c_wrapper") + library = joinpath(install_lib, "lib", "lib$name.$dlext") + isfile(library) || error("Wrapper build did not produce $library; see deps/cpp_wrapper.err. JLL override was not updated.") + end return BuildTools.set_jll_artifact_override(:cunumeric_jl_wrapper_jll, install_lib) end diff --git a/deps/cxxwrap.jl b/deps/cxxwrap.jl new file mode 100644 index 000000000..695e37a17 --- /dev/null +++ b/deps/cxxwrap.jl @@ -0,0 +1,54 @@ +# A version marker alone cannot validate a build-tree CMake export: its headers +# can live in a checkout that has since been deleted or moved. +cxxwrap_julia_identity() = string( + VERSION, '\n', realpath(joinpath(Sys.BINDIR, Base.julia_exename())), +) + +function cxxwrap_usable(override_dir; log_dir) + mktempdir() do build_dir + probe = joinpath(@__DIR__, "cxxwrap_probe") + julia = joinpath(Sys.BINDIR, Base.julia_exename()) + cmd = `cmake -S $probe -B $build_dir -DJlCxx_DIR=$override_dir -DJulia_EXECUTABLE=$julia` + open(joinpath(log_dir, "libcxxwrap_check.log"), "w") do io + return success(pipeline(cmd; stdout=io, stderr=io)) + end + end +end + +function ensure_cxxwrap(repo_root, package_version; log_dir) + override_dir = joinpath(DEPOT_PATH[1], "dev", "libcxxwrap_julia_jll", "override") + version_path = joinpath(override_dir, "LEGATE_INSTALL.txt") + julia_path = joinpath(override_dir, "JULIA_INSTALL.txt") + julia_identity = cxxwrap_julia_identity() + cached = isfile(version_path) ? tryparse(VersionNumber, strip(read(version_path, String))) : nothing + # LEGATE_INSTALL.txt records the provider version, not Julia's ABI. Legacy + # builds without a Julia identity must be rebuilt once, even if paths exist. + same_julia = isfile(julia_path) && read(julia_path, String) == julia_identity + if !isnothing(cached) && is_supported_version(cached) && + same_julia && cxxwrap_usable(override_dir; log_dir) + @info "libcxxwrap: Up to date (provider $cached, Julia $VERSION)" + return nothing + end + + @info "libcxxwrap: Missing, incompatible, or stale build. Rebuilding..." + # Invalidate before attempting the build, and propagate failures. The shared + # run_sh helper catches errors, so it cannot establish a successful build. + rm(version_path; force=true) + rm(julia_path; force=true) + script = joinpath(repo_root, "scripts", "install_cxxwrap.sh") + cmd = addenv(`bash $script $repo_root`, "JULIA" => joinpath(Sys.BINDIR, Base.julia_exename())) + open(joinpath(log_dir, "libcxxwrap.log"), "w") do out + open(joinpath(log_dir, "libcxxwrap.err"), "w") do err + try + run(pipeline(cmd; stdout=out, stderr=err)) + catch + error("libcxxwrap build failed; see $(joinpath(log_dir, "libcxxwrap.err"))") + end + end + end + cxxwrap_usable(override_dir; log_dir) || + error("libcxxwrap build produced an unusable CMake package; see $(joinpath(log_dir, "libcxxwrap_check.log"))") + write(version_path, string(package_version)) + write(julia_path, julia_identity) + return nothing +end diff --git a/deps/cxxwrap_probe/CMakeLists.txt b/deps/cxxwrap_probe/CMakeLists.txt new file mode 100644 index 000000000..054f77bec --- /dev/null +++ b/deps/cxxwrap_probe/CMakeLists.txt @@ -0,0 +1,42 @@ +cmake_minimum_required(VERSION 3.16) +project(CheckCxxWrap LANGUAGES NONE) + +# Inspect exactly the override the wrapper will use, without falling back to +# another installation or requiring a compiler/CUDA for this check. +find_package(JlCxx REQUIRED CONFIG PATHS "${JlCxx_DIR}" NO_DEFAULT_PATH) +foreach(target JlCxx::cxxwrap_julia JlCxx::cxxwrap_julia_stl) + if(NOT TARGET ${target}) + message(FATAL_ERROR "Missing target ${target}") + endif() + get_target_property(includes ${target} INTERFACE_INCLUDE_DIRECTORIES) + if(NOT includes) + message(FATAL_ERROR "Missing include directories for ${target}") + endif() + foreach(path IN LISTS includes) + if(path STREQUAL "") + continue() + endif() + if(NOT IS_DIRECTORY "${path}") + message(FATAL_ERROR "Missing include directory: ${path}") + endif() + endforeach() + get_target_property(configs ${target} IMPORTED_CONFIGURATIONS) + set(properties IMPORTED_LOCATION) + foreach(config IN LISTS configs) + string(TOUPPER "${config}" config) + list(APPEND properties "IMPORTED_LOCATION_${config}") + endforeach() + set(found_location FALSE) + foreach(property IN LISTS properties) + get_target_property(path ${target} ${property}) + if(path) + set(found_location TRUE) + if(NOT EXISTS "${path}") + message(FATAL_ERROR "Missing library: ${path}") + endif() + endif() + endforeach() + if(NOT found_location) + message(FATAL_ERROR "Missing library location for ${target}") + endif() +endforeach() diff --git a/docs/make.jl b/docs/make.jl index 7ec8a7f44..8b0b3bb2e 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -65,6 +65,7 @@ makedocs(; "Initialization" => "api_initialization.md", "Random" => "api_random.md", "Unary Operations" => "api_unary.md", + "Mapped Reductions" => "api_mapreduce.md", "Binary Operations" => "api_binary.md", "Linear Algebra" => "linalg.md", "Tensor Contractions" => "api_tensor.md", diff --git a/docs/src/api.md b/docs/src/api.md index 4e58b7d47..b998aaf10 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -2,6 +2,9 @@ Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). For `fft` / `ifft` / `fft!` / `ifft!` and `batched_fft`, see [FFT](./fft.md). There is no `plan_fft`: cupynumeric does not expose a cuFFT handle. +For `mapreduce` and the mapped forms of `sum`, `prod`, `minimum`, and `maximum`, +see [Mapped Reductions](./api_mapreduce.md). + ```@autodocs Modules = [cuNumeric] Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "ndarray/sort.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl", "scoping/accelerate.jl"] diff --git a/docs/src/api_mapreduce.md b/docs/src/api_mapreduce.md new file mode 100644 index 000000000..1e39a555b --- /dev/null +++ b/docs/src/api_mapreduce.md @@ -0,0 +1,53 @@ +# Mapped Reductions + +`mapreduce(f, op, A; dims=:, init=...)` maps and reduces one NDArray without +allocating a mapped copy. Supported operators are `+`, `*`, `min`, and `max`. +Mapped `sum`, `prod`, `minimum`, and `maximum` use the same implementation. + +```julia +A = cuNumeric.ones(Float32, 1024, 512) +energy = sum(abs2, A) # 0-d NDArray +columns = mapreduce(abs2, +, A; dims=1) # 1 × 512 +α = 0.5f0 +distance = sum(x -> abs2(x - α), A) +largest = maximum(abs, A; init=0f0) +``` + +Full reductions return a device-resident 0-d NDArray. Dimensional reductions +keep reduced axes with size one. Duplicate dimensions are ignored; positive +out-of-rank dimensions have no effect. `dims=()` still applies the mapping. + +## Types and initialization + +`sum` and `prod` use Base's integer widening rules; `mapreduce` with `+` or `*` +uses ordinary scalar arithmetic. Implicit widening, including within the mapping, +is subject to `allowpromotion`. + +Supply a neutral scalar `init`; it is applied once. Full empty reductions return +`init` when supplied. Otherwise, Base's empty-input rules apply: dimensional +sums/products return their identities, while empty extrema error except for +special cases such as `maximum(abs2, A)`. Arbitrary mappings may require `init`. + +For full reductions, combining with `init` determines the output type. For +dimensional reductions, `typeof(init)` determines it and must accommodate the +accumulator without narrowing. + +## Current limitations + +- GPU execution only, independent of broadcast-fusion settings. Existing unmapped + reductions retain CPU support. Participating GPUs must support the compilation + target; heterogeneous target selection is unsupported. +- One input NDArray and a type-stable, GPU-compilable mapping with immutable scalar + captures. Captured arrays/pointers, host allocation or side effects, and custom + reducers are unsupported. +- Results may be Bool, the supported integer widths, Float32/Float64, or + ComplexF32/ComplexF64. Complex extrema and ComplexF64 product accumulators + (including those selected by dimensional `init`) are unsupported. +- Floating-point reassociation can change rounding, overflow, and arithmetic + signed zeros. Bitwise reproducibility is not guaranteed. Floating-point extrema + preserve NaN propagation and signed-zero ordering, but not NaN payloads. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/mapreduce.jl"] +``` diff --git a/docs/src/api_unary.md b/docs/src/api_unary.md index 7582f4ce1..4adbcac70 100644 --- a/docs/src/api_unary.md +++ b/docs/src/api_unary.md @@ -18,6 +18,9 @@ The following unary operations are supported and can be broadcast over `NDArray` - `var` / `std` match Julia / StatsBase sample statistics (`corrected=true`, divisor `n-1`). Complex is not supported. - `argmax` / `argmin` are 1-d only (matching Base's `Int` return, not `CartesianIndex`). The result is a 0-d `NDArray{Int64}` of the 1-based index. Complex is not supported. +For `mapreduce` and mapped `sum`, `prod`, `minimum`, and `maximum`, see +[Mapped Reductions](./api_mapreduce.md). + ```@autodocs Modules = [cuNumeric] Pages = ["ndarray/unary.jl"] diff --git a/docs/src/developer_mode.md b/docs/src/developer_mode.md index 913977942..bf76a0821 100644 --- a/docs/src/developer_mode.md +++ b/docs/src/developer_mode.md @@ -44,6 +44,16 @@ Then restart Julia (or at least reload cuNumeric) so the new shared library is p If the build fails, check CMake / g++ (C++20) / CUDA toolkit availability as described on [Build Modes](./install.md). Build logs from the helper scripts are written under `deps/`. +Recreating or moving the repository can leave a libcxxwrap CMake export in the +Julia depot pointing at deleted headers. `Pkg.build("cuNumeric")` checks the +exported header and library paths before reusing that build and rebuilds it when +stale. It also records the Julia version and executable path: changing either +forces a rebuild. Older builds without this Julia marker are rebuilt once. +No manual checkout or depot cleanup is needed. The installer retains your +manifest and developed JLL checkout, recreating only its generated `override/` +directory. Check `deps/libcxxwrap_check.log`, `deps/libcxxwrap.log`, and +`deps/libcxxwrap.err` if this repair fails. + ### Typical edit loop ```text diff --git a/lib/cunumeric_jl_wrapper/CMakeLists.txt b/lib/cunumeric_jl_wrapper/CMakeLists.txt index b4eb3c551..104e93e96 100644 --- a/lib/cunumeric_jl_wrapper/CMakeLists.txt +++ b/lib/cunumeric_jl_wrapper/CMakeLists.txt @@ -47,7 +47,7 @@ set(SOURCES if(LEGATE_WRAPPER_ENABLE_CUDA) find_package(CUDAToolkit 13.0 REQUIRED) - list(APPEND SOURCES src/cuda.cpp) + list(APPEND SOURCES src/cuda.cpp src/mapreduce.cpp) message(STATUS "LEGATE_WRAPPER_ENABLE_CUDA=ON: adding src/cuda.cpp") else() # only disables find_package requirement for CUDAToolkit. @@ -71,6 +71,7 @@ target_link_libraries(${CXX_CUNUMERICJL_WRAPPER} PRIVATE target_include_directories(${CXX_CUNUMERICJL_WRAPPER} PRIVATE include) if(LEGATE_WRAPPER_ENABLE_CUDA) target_include_directories(${CXX_CUNUMERICJL_WRAPPER} PRIVATE ${CUDAToolkit_INCLUDE_DIRS}) + target_link_libraries(${CXX_CUNUMERICJL_WRAPPER} PRIVATE CUDA::cuda_driver CUDA::cudart) endif() install(TARGETS ${CXX_CUNUMERICJL_WRAPPER} DESTINATION lib) diff --git a/lib/cunumeric_jl_wrapper/include/mapreduce.h b/lib/cunumeric_jl_wrapper/include/mapreduce.h new file mode 100644 index 000000000..1c1f09027 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/include/mapreduce.h @@ -0,0 +1,39 @@ +#pragma once + +#include +#include + +#include "legate.h" + +namespace ufi { +// The wire values come from Legate. Julia receives this enum through CxxWrap +// and never defines a parallel table of integer operation codes. +enum class MapReduceOp : int32_t { + ADD = static_cast(legate::ReductionOpKind::ADD), + MUL = static_cast(legate::ReductionOpKind::MUL), + MIN = static_cast(legate::ReductionOpKind::MIN), + MAX = static_cast(legate::ReductionOpKind::MAX), +}; +using MapReduceOpValue = std::underlying_type_t; + +#if LEGATE_DEFINED(LEGATE_USE_CUDA) +legate::Library get_mapreduce_library(); +void register_mapreduce_tasks(); + +class RunPTXMapReduceTask : public legate::LegateTask { + public: + static inline const auto TASK_CONFIG = + legate::TaskConfig{legate::LocalTaskID{0}}.with_variant_options( + legate::VariantOptions{}.with_has_allocations(true).with_elide_device_ctx_sync(true)); + static void gpu_variant(legate::TaskContext context); +}; + +class RunPTXReduceFinishTask : public legate::LegateTask { + public: + static inline const auto TASK_CONFIG = + legate::TaskConfig{legate::LocalTaskID{1}}.with_variant_options( + legate::VariantOptions{}.with_elide_device_ctx_sync(true)); + static void gpu_variant(legate::TaskContext context); +}; +#endif +} // namespace ufi diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index 3fe6b0896..ab164b667 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -52,6 +52,9 @@ uint64_t nda_query_device_memory(); // CN_Type : Legate type of object CN_NDArray* nda_zeros_array(int32_t dim, const uint64_t* shape, CN_Type type); +// Internal allocation without a fill; every element must be written before use. +CN_NDArray* nda_empty_array(int32_t dim, const uint64_t* shape, CN_Type type); + // full(shape, value) // dim : number of dimensions // shape : pointer to array[length=dim] diff --git a/lib/cunumeric_jl_wrapper/include/ptx.h b/lib/cunumeric_jl_wrapper/include/ptx.h new file mode 100644 index 000000000..f51530dd2 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/include/ptx.h @@ -0,0 +1,27 @@ +#pragma once + +#include "legate.h" + +#if LEGATE_DEFINED(LEGATE_USE_CUDA) +#include +#include +#include +#include +#include + +namespace ufi { +// ABI matches Julia CuStridedDeviceArray; strides count elements, not bytes. +template +struct CuStridedDeviceArray { + void* ptr; + int64_t maxsize; + std::array dims; + std::array strides; + int64_t length; +}; + +// Shared by all PTX task launchers; modules remain in the existing +// processor-local cache populated by LoadPTXTask. +CUfunction lookup_ptx(const std::string& name, cudaStream_t stream); +} // namespace ufi +#endif diff --git a/lib/cunumeric_jl_wrapper/src/cuda.cpp b/lib/cunumeric_jl_wrapper/src/cuda.cpp index 2f5c01097..de9cf1522 100644 --- a/lib/cunumeric_jl_wrapper/src/cuda.cpp +++ b/lib/cunumeric_jl_wrapper/src/cuda.cpp @@ -23,12 +23,14 @@ #include #include #include +#include #include "legate.h" #include "legate/utilities/proc_local_storage.h" #include "legion.h" #include "types.h" #include "ufi.h" +#include "ptx.h" // #define CUDA_DEBUG #include "cuda_macros.h" // Shared error/debug and dense argument-packing macros. @@ -64,6 +66,19 @@ using FunctionMap = std::unordered_map cufunction_ptr{}; +CUfunction lookup_ptx(const std::string& name, cudaStream_t stream) { + CUcontext ctx; + if (cuStreamGetCtx(stream, &ctx) != CUDA_SUCCESS) + throw std::runtime_error("PTX: could not get the task CUDA context"); + if (!cufunction_ptr.has_value()) + throw std::runtime_error("PTX: no modules loaded on this processor"); + auto& functions = cufunction_ptr.get(); + auto it = functions.find({ctx, name}); + if (it == functions.end()) + throw std::runtime_error("PTX: missing kernel " + name); + return it->second; +} + #ifdef CUDA_DEBUG std::string context_to_string(CUcontext ctx) { std::ostringstream oss; @@ -92,18 +107,6 @@ struct CuDeviceArray { uint64_t length; // Number of elements (at the end) }; -// Strided — matches Julia cuNumeric.CuStridedDeviceArray (RunPTXBroadcastTask -// only). -template -struct CuStridedDeviceArray { - void *ptr; - uint64_t maxsize; - std::array dims; - std::array - strides; // element strides (byte strides / sizeof(T)) - uint64_t length; -}; - #define CUDA_STRIDED_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ template < \ typename T, int D, \ @@ -194,27 +197,7 @@ static PTXLaunchParams read_launch_params(legate::TaskContext &context) { p.ty = context.scalar(THREAD_START + 1).value(); p.tz = context.scalar(THREAD_START + 2).value(); - CUcontext ctx; - cuStreamGetCtx(p.stream, &ctx); - - FunctionKey key = {ctx, p.kernel_name}; - assert(cufunction_ptr.has_value()); - FunctionMap &fmap = cufunction_ptr.get(); - auto it = fmap.find(key); - -#ifdef CUDA_DEBUG - if (it == fmap.end()) { - std::cerr << "[RunPTXTask] Could not find key: " << key_to_string(key) - << std::endl; - for (const auto &[k, v] : fmap) { - std::cerr << "[RunPTXTask] Map key: " << key_to_string(k) << std::endl; - } - assert(0 && "[RunPTXTask] key is not found in hashmap"); - } -#endif - - assert(it != fmap.end()); - p.func = it->second; + p.func = lookup_ptx(p.kernel_name, p.stream); p.custream = reinterpret_cast(p.stream); return p; } diff --git a/lib/cunumeric_jl_wrapper/src/mapreduce.cpp b/lib/cunumeric_jl_wrapper/src/mapreduce.cpp new file mode 100644 index 000000000..afa317173 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/src/mapreduce.cpp @@ -0,0 +1,222 @@ +#include "mapreduce.h" +#include "ptx.h" + +#include +#include +#include +#include +#include +#include +#include + +#include "legate/data/buffer.h" +#include "legate/mapping/mapping.h" +#include "legate/redop/redop.h" + +extern std::size_t padded_bytes_kernel_state; + +namespace ufi { +namespace { +constexpr int THREADS = 256; +constexpr int64_t MAX_PARTIALS = 4096; + +// These tasks need their own mapper: cuPyNumeric's mapper cannot account for +// scratch allocated by tasks registered outside its built-in task table. +class MapReduceMapper : public legate::mapping::Mapper { + public: + std::vector store_mappings( + const legate::mapping::Task&, + const std::vector&) override { return {}; } + + std::optional allocation_pool_size( + const legate::mapping::Task&, legate::mapping::StoreTarget target) override { + // One partial buffer; ComplexF64 is the largest supported accumulator. + return target == legate::mapping::StoreTarget::FBMEM + ? MAX_PARTIALS * sizeof(legate::type_of) : 0; + } + + legate::Scalar tunable_value(legate::TunableID) override { + throw std::invalid_argument("mapreduce has no tunables"); + } +}; + +template +using ArrayArg = CuStridedDeviceArray; + +template +constexpr bool supported = C == legate::Type::Code::BOOL || + C == legate::Type::Code::INT8 || C == legate::Type::Code::INT16 || + C == legate::Type::Code::INT32 || C == legate::Type::Code::INT64 || + C == legate::Type::Code::UINT8 || C == legate::Type::Code::UINT16 || + C == legate::Type::Code::UINT32 || C == legate::Type::Code::UINT64 || + C == legate::Type::Code::FLOAT32 || C == legate::Type::Code::FLOAT64 || + C == legate::Type::Code::COMPLEX64 || C == legate::Type::Code::COMPLEX128; + +template +struct PackArray { + template + ArrayArg operator()(const legate::PhysicalStore& store) const { + if constexpr (supported) { + using T = legate::type_of; + auto rect = store.shape(); + auto acc = [&]() { + if constexpr (WRITE) return store.write_accessor(); + else return store.read_accessor(); + }(); + ArrayArg arg{}; + arg.ptr = const_cast(static_cast(acc.ptr(rect.lo))); + arg.length = rect.volume(); + arg.maxsize = arg.length * sizeof(T); + for (int d = 0; d < D; ++d) { + arg.dims[d] = rect.hi[d] - rect.lo[d] + 1; + arg.strides[d] = acc.accessor.strides[d] / sizeof(T); + } + return arg; + } else { + throw std::invalid_argument("mapreduce: unsupported physical element type"); + } + } +}; + +static void launch(CUfunction kernel, int blocks, cudaStream_t stream, + void** args) { + auto status = cuLaunchKernel(kernel, blocks, 1, 1, THREADS, 1, 1, 0, + reinterpret_cast(stream), args, nullptr); + if (status != CUDA_SUCCESS) { + const char* message = nullptr; + cuGetErrorString(status, &message); + throw std::runtime_error(std::string("mapreduce PTX launch: ") + (message ? message : "unknown CUDA error")); + } +} + +template +ArrayArg reduction_arg(const legate::PhysicalStore& store, + const legate::Rect& bounds, uint64_t axes, bool single) { + using T = typename OP::RHS; + size_t strides[D]; + ArrayArg arg{}; + arg.ptr = single ? store.write_accessor().ptr(bounds, strides) + : store.reduce_accessor().ptr(bounds, strides); + arg.length = 1; + for (int d = 0; d < D; ++d) { + // Promoted axes alias the same destination. Exclude those coordinates + // so exactly one Julia thread updates each physical output element. + arg.dims[d] = ((axes >> d) & 1) ? 1 : bounds.hi[d] - bounds.lo[d] + 1; + arg.strides[d] = strides[d]; // ptr(rect, strides) returns element strides + arg.length *= arg.dims[d]; + } + arg.maxsize = arg.length * sizeof(T); + return arg; +} + +template +void contribute(CUfunction kernel, cudaStream_t stream, void* state, + ArrayArg<1>& scratch, ArrayArg& dest, + bool single, int64_t start, int64_t count, int64_t chunks) { + std::array args{state, &scratch, &dest, &single, &start, &count, &chunks}; + launch(kernel, static_cast((count + THREADS - 1) / THREADS), stream, args.data()); +} + +template +void run_reduction(legate::TaskContext& context, CUfunction kernel) { + using T = typename OP::RHS; + auto input = context.input(0).data(); + auto single = context.scalar(6).value(); + auto red = single ? context.output(0).data() : context.reduction(0).data(); + const auto rect = input.shape(); + if (rect.empty()) return; + auto stream = context.get_task_stream(); + auto axes = context.scalar(1).value(); + auto full = context.scalar(2).value(); + auto combine = lookup_ptx(context.scalar(5).value(), stream); + auto src = legate::type_dispatch(input.type().code(), PackArray{}, input); + int64_t reduced = 1, retained = 1; + for (int d = 0; d < D; ++d) + (((axes >> d) & 1) ? reduced : retained) *= src.dims[d]; + + // Deferred buffers belong to the task; kernels may still be queued when + // their C++ handles leave scope. Reuse is ordered on the task's stream. + auto buffer = legate::create_buffer(MAX_PARTIALS, legate::Memory::Kind::GPU_FB_MEM); + ArrayArg<1> scratch{buffer.ptr(0), MAX_PARTIALS * int64_t(sizeof(T)), + {MAX_PARTIALS}, {1}, MAX_PARTIALS}; + std::vector state(padded_bytes_kernel_state, 0); + for (int64_t start = 0; start < retained;) { + int64_t count = std::min(MAX_PARTIALS, retained - start); + int64_t chunks = std::min(MAX_PARTIALS / count, 1 + (reduced - 1) / 1024); + std::array args{state.data(), &src, &scratch}; + std::size_t nargs = 3; + // Zero-size Julia singleton arguments are absent from the PTX signature. + if (context.scalar(4).size()) args[nargs++] = const_cast(context.scalar(4).ptr()); + args[nargs++] = &start; + args[nargs++] = &chunks; + launch(kernel, static_cast(count * chunks), stream, args.data()); + if (full) { + auto dest = reduction_arg(red, red.shape<1>(), 1, single); + contribute(combine, stream, state.data(), scratch, dest, single, start, count, chunks); + } else { + auto dest = reduction_arg(red, rect, axes, single); + contribute(combine, stream, state.data(), scratch, dest, single, start, count, chunks); + } + start += count; + } +} + +struct ReduceDispatch { + template + void operator()(legate::TaskContext& ctx, CUfunction kernel) const { + if constexpr (supported && C != legate::Type::Code::BOOL) { + using T = legate::type_of; + auto op = static_cast(ctx.scalar(3).value()); + if (op == MapReduceOp::ADD) return run_reduction, D>(ctx, kernel); + if constexpr (C != legate::Type::Code::COMPLEX128) { + if (op == MapReduceOp::MUL) return run_reduction, D>(ctx, kernel); + } + if constexpr (C != legate::Type::Code::COMPLEX64 && C != legate::Type::Code::COMPLEX128) { + if (op == MapReduceOp::MIN) return run_reduction, D>(ctx, kernel); + if (op == MapReduceOp::MAX) return run_reduction, D>(ctx, kernel); + } + } + throw std::invalid_argument("mapreduce: unsupported reduction/type combination"); + } +}; + +struct FinishDispatch { + template + void operator()(legate::TaskContext& ctx, CUfunction kernel) const { + auto input = ctx.input(0).data(); + auto output = ctx.output(0).data(); + if (output.shape().empty()) return; + auto src = legate::type_dispatch(input.type().code(), PackArray{}, input); + auto dst = legate::type_dispatch(output.type().code(), PackArray{}, output); + std::vector state(padded_bytes_kernel_state, 0); + std::array args{state.data(), &src, &dst}; + if (ctx.scalar(1).size()) args[3] = const_cast(ctx.scalar(1).ptr()); + auto blocks = static_cast(std::min(MAX_PARTIALS, 1 + (dst.length - 1) / THREADS)); + launch(kernel, blocks, ctx.get_task_stream(), args.data()); + } +}; +} // namespace + +legate::Library get_mapreduce_library() { + return legate::Runtime::get_runtime()->find_library("cuNumeric_mapreduce"); +} + +void register_mapreduce_tasks() { + auto library = legate::Runtime::get_runtime()->create_library( + "cuNumeric_mapreduce", legate::ResourceConfig{}, std::make_unique()); + RunPTXMapReduceTask::register_variants(library); + RunPTXReduceFinishTask::register_variants(library); +} + +void RunPTXMapReduceTask::gpu_variant(legate::TaskContext context) { + auto kernel = lookup_ptx(context.scalar(0).value(), context.get_task_stream()); + auto dest = context.scalar(6).value() ? context.output(0) : context.reduction(0); + legate::double_dispatch(context.input(0).dim(), dest.type().code(), + ReduceDispatch{}, context, kernel); +} + +void RunPTXReduceFinishTask::gpu_variant(legate::TaskContext context) { + auto kernel = lookup_ptx(context.scalar(0).value(), context.get_task_stream()); + legate::dim_dispatch(context.output(0).dim(), FinishDispatch{}, context, kernel); +} +} // namespace ufi diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index 91ba89617..7efac3029 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -63,6 +63,12 @@ struct CN_Store { legate::LogicalStore obj; }; +CN_NDArray* nda_empty_array(int32_t dim, const uint64_t* shape, CN_Type type) { + std::vector shp(shape, shape + dim); + auto* runtime = cupynumeric::CuPyNumericRuntime::get_runtime(); + return new CN_NDArray{runtime->create_array(shp, type.obj)}; +} + CN_NDArray* nda_zeros_array(int32_t dim, const uint64_t* shape, CN_Type type) { std::vector shp(shape, shape + dim); NDArray result = zeros(shp, type.obj); diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 68f280d87..00423dcee 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -36,6 +36,7 @@ #include "realm.h" #include "types.h" #include "ufi.h" +#include "mapreduce.h" struct WrapCppOptional { template @@ -75,6 +76,82 @@ void register_tasks() { ufi::LoadPTXTask::register_variants(library); ufi::RunPTXTask::register_variants(library); ufi::RunPTXBroadcastTask::register_variants(library); + ufi::register_mapreduce_tasks(); +} + +static legate::Scalar mapreduce_payload(const void* ptr, size_t size) { + auto bytes = static_cast(ptr); + return legate::Scalar(std::vector(bytes, bytes + size)); +} + +static void submit_mapreduce(CN_NDArray* input, CN_NDArray* accumulator, + CN_NDArray* output, uint64_t axes, bool full, bool single, + ufi::MapReduceOp op, const std::string& kernel, + const std::string& contribute_kernel, + const std::string& finish_kernel, + const void* mapper, size_t mapper_size, + const void* finish, size_t finish_size) { + auto* rt = legate::Runtime::get_runtime(); + auto library = ufi::get_mapreduce_library(); + auto src = input->obj.get_store(); + // Physical descriptors and reduction accessors use at least one dimension. + if (src.dim() == 0) src = src.promote(0, 1); + if (src.dim() > 64) throw std::invalid_argument("mapreduce rank exceeds axis mask"); + // A dimensional singleton reduction is elementwise. Reuse the finish task + // to map and seed each value without constructing reduction privileges. + if (single && !full) { + auto dst = output->obj.get_store(); + if (dst.dim() == 0) dst = dst.promote(0, 1); + auto final = rt->create_task( + library, ufi::RunPTXReduceFinishTask::TASK_CONFIG.task_id()); + auto p_src = final.add_input(src); + auto p_dst = final.add_output(dst); + final.add_constraint(legate::align(p_src, p_dst)); + final.add_scalar_arg(legate::Scalar(finish_kernel)); + final.add_scalar_arg(mapreduce_payload(finish, finish_size)); + rt->submit(std::move(final)); + return; + } + auto acc = accumulator->obj.get_store(); + auto red = acc; + if (full) { + if (red.dim() == 0) red = red.promote(0, 1); + } else { + // Project from high to low, then promote from low to high: axis numbers + // refer to the original input throughout both transformations. + for (int d = red.dim() - 1; d >= 0; --d) + if ((axes >> d) & 1) red = red.project(d, 0); + for (int d = 0; d < src.dim(); ++d) + if ((axes >> d) & 1) red = red.promote(d, src.shape()[d]); + if (red.dim() == 0) red = red.promote(0, 1); + } + auto task = rt->create_task(library, ufi::RunPTXMapReduceTask::TASK_CONFIG.task_id()); + auto p_src = task.add_input(src); + // A singleton has no reduction arithmetic: even multiplying complex Inf by + // an identity can introduce NaNs. Give it an ordinary output privilege. + auto p_red = single ? task.add_output(red) + : task.add_reduction(red, static_cast(op)); + if (!full) task.add_constraint(legate::align(p_src, p_red)); + task.add_scalar_arg(legate::Scalar(kernel)); + task.add_scalar_arg(legate::Scalar(axes)); + task.add_scalar_arg(legate::Scalar(full)); + task.add_scalar_arg(legate::Scalar(static_cast(op))); + task.add_scalar_arg(mapreduce_payload(mapper, mapper_size)); + task.add_scalar_arg(legate::Scalar(contribute_kernel)); + task.add_scalar_arg(legate::Scalar(single)); + rt->submit(std::move(task)); + + if (finish_kernel.empty()) return; + auto dst = output->obj.get_store(); + if (acc.dim() == 0) acc = acc.promote(0, 1); + if (dst.dim() == 0) dst = dst.promote(0, 1); + auto final = rt->create_task(library, ufi::RunPTXReduceFinishTask::TASK_CONFIG.task_id()); + auto p_acc = final.add_input(acc); + auto p_dst = final.add_output(dst); + final.add_constraint(legate::align(p_acc, p_dst)); + final.add_scalar_arg(legate::Scalar(finish_kernel)); + final.add_scalar_arg(mapreduce_payload(finish, finish_size)); + rt->submit(std::move(final)); } #endif @@ -86,6 +163,12 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { wrap_fft_ops(mod); wrap_bitgenerator_ops(mod); + mod.add_bits("MapReduceOp", jlcxx::julia_type("CppEnum")); + mod.set_const("MAPREDUCE_ADD", ufi::MapReduceOp::ADD); + mod.set_const("MAPREDUCE_MUL", ufi::MapReduceOp::MUL); + mod.set_const("MAPREDUCE_MIN", ufi::MapReduceOp::MIN); + mod.set_const("MAPREDUCE_MAX", ufi::MapReduceOp::MAX); + using jlcxx::ParameterList; using jlcxx::Parametric; using jlcxx::TypeVar; @@ -221,6 +304,7 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { #if LEGATE_DEFINED(LEGATE_USE_CUDA) mod.method("register_tasks", ®ister_tasks); + mod.method("submit_mapreduce", &submit_mapreduce); wrap_cuda_methods(mod); #endif } diff --git a/scripts/install_cxxwrap.sh b/scripts/install_cxxwrap.sh index 8335858cb..69eefbfe0 100755 --- a/scripts/install_cxxwrap.sh +++ b/scripts/install_cxxwrap.sh @@ -33,8 +33,8 @@ if [[ ! -d "$CUNUMERIC_ROOT_DIR" ]]; then exit 1 fi -JULIA='julia' -JULIA_PATH=$(which $JULIA) +JULIA=${JULIA:-julia} +JULIA_PATH=$(command -v "$JULIA") if [ -z "$JULIA_PATH" ]; then echo "Error: $JULIA is not installed or not in PATH." @@ -48,40 +48,38 @@ COMMIT_HASH="89e4699837bfa0929610c9e330889fb2df925b47" #(v14.2) JULIA_CXXWRAP_SRC=$CUNUMERIC_ROOT_DIR/lib/libcxxwrap-julia if [ ! -d "$JULIA_CXXWRAP_SRC" ]; then - cd $CUNUMERIC_ROOT_DIR/lib - git clone $GIT_REPO + mkdir -p "$CUNUMERIC_ROOT_DIR/lib" + git clone "$GIT_REPO" "$JULIA_CXXWRAP_SRC" fi -cd $JULIA_CXXWRAP_SRC +cd "$JULIA_CXXWRAP_SRC" git fetch --tags git checkout $COMMIT_HASH # find julia dependency path -JULIA_DEP_PATH=$($JULIA -e 'println(DEPOT_PATH[1])') +JULIA_DEP_PATH=$("$JULIA_PATH" --startup-file=no -e 'print(DEPOT_PATH[1])') # https://github.com/JuliaInterop/libcxxwrap-julia/tree/v0.13.3?tab=readme-ov-file#configuring-and-building JULIA_CXXWRAP_DEV=$JULIA_DEP_PATH/dev/libcxxwrap_julia_jll JULIA_CXXWRAP=$JULIA_CXXWRAP_DEV/override -# Clean up whatever env is there right now and -# build default version of CxxWrap / libcxxwrap_julia -#* THIS COULD BREAK SOME USERS CODE IF THEY ALREADY OVERRIDE THIS PKG -cd $CUNUMERIC_ROOT_DIR -[ -f Manifest.toml ] && rm Manifest.toml -rm -rf $JULIA_CXXWRAP_DEV -julia -e 'using Pkg; Pkg.activate("."); Pkg.add("Legate")' -julia -e 'using Pkg; Pkg.activate("."); Pkg.precompile(["CxxWrap"])' - -# https://github.com/JuliaInterop/libcxxwrap-julia/tree/v0.13.3?tab=readme-ov-file#preparing-the-install-location -# this command will download https://github.com/JuliaBinaryWrappers/libcxxwrap_julia_jll.jl and install it in JULIA_DEP_PATH -julia -e 'using Pkg; Pkg.activate("."); Pkg.develop(PackageSpec(name="libcxxwrap_julia_jll")); import libcxxwrap_julia_jll; libcxxwrap_julia_jll.dev_jll()' - - -# JULIA_CXXWRAP_OVERRIDE=$JULIA_CXXWRAP/override/ -# Delete the default JLL installation of cxxwrap_julia -rm -rf $JULIA_CXXWRAP -mkdir $JULIA_CXXWRAP - -cmake -S $JULIA_CXXWRAP_SRC -B $JULIA_CXXWRAP -DJulia_EXECUTABLE=$JULIA_PATH -DCMAKE_BUILD_TYPE=Release -cd $JULIA_CXXWRAP -make -j 16 +# Keep the resolved environment and developed JLL checkout. Only its generated +# override is disposable. Avoid importing/precompiling a possibly broken JLL +# before its replacement libraries have been built. +cd "$CUNUMERIC_ROOT_DIR" +JULIA_PKG_PRECOMPILE_AUTO=0 "$JULIA_PATH" --startup-file=no --project="$CUNUMERIC_ROOT_DIR" -e ' + using Pkg + checkout = joinpath(DEPOT_PATH[1], "dev", "libcxxwrap_julia_jll") + if isdir(checkout) + Pkg.develop(path=checkout) + else + Pkg.develop(PackageSpec(name="libcxxwrap_julia_jll"); shared=true) + end +' + +rm -rf "$JULIA_CXXWRAP" +mkdir -p "$JULIA_CXXWRAP" + +cmake -S "$JULIA_CXXWRAP_SRC" -B "$JULIA_CXXWRAP" \ + -DJulia_EXECUTABLE="$JULIA_PATH" -DCMAKE_BUILD_TYPE=Release +cmake --build "$JULIA_CXXWRAP" --parallel 16 diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 4a4432cc6..5dc9a2255 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -205,6 +205,8 @@ include("ndarray/random/bitgenerator.jl") include("ndarray/random/generator.jl") include("ndarray/random/random.jl") include("ndarray/unary.jl") +include("ndarray/mapreduce.jl") +include("cuda/mapreduce.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") include("ndarray/sort.jl") diff --git a/src/cuda/README.md b/src/cuda/README.md new file mode 100644 index 000000000..33136e060 --- /dev/null +++ b/src/cuda/README.md @@ -0,0 +1,36 @@ +# GPU mapped reductions + +The frontend in `../ndarray/mapreduce.jl` selects reduction policies, validates +shapes/types, and handles empty inputs. `mapreduce.jl` compiles and caches Julia +PTX kernels; capture values and array extents are runtime arguments. + +Each Legate GPU task: + +1. Packs the input tile's lower-bound pointer, extents, and element strides. + `_mr_offset` enumerates retained and reduced coordinates separately, so slices + need not be contiguous. +2. Maps values and reduces them in 256-thread blocks using shared memory. +3. Combines block partials into the task's reduction store, with one writer per + retained coordinate through an exclusive reduction accessor. + +**Legate combines contributions between tasks and GPUs.** The native glue in +`../../lib/cunumeric_jl_wrapper/src/mapreduce.cpp` packs descriptors and launches +PTX on the task stream. Scratch is task-local and bounded to 4,096 partials; +there is no input-sized mapped temporary. + +The reduction tasks use a separate Legate library with a mapper that reserves +64 KiB of framebuffer scratch per allocating task (4,096 × 16 bytes). The reduction +variant declares `has_allocations`; both variants declare that all device work +uses the task stream. Registering them with cuPyNumeric's mapper would omit this +scratch reservation and can abort even when device memory is available. + +Floating extrema use ordered unsigned keys to preserve NaNs and signed zeros. +Boolean product/extrema use 0/1 bytes. A finishing kernel decodes storage and +applies `init` once when needed; otherwise the accumulator is returned directly. +Singletons use output privileges to avoid identity arithmetic, while dimensional +sums/products retain Base's zero/one seed. + +Submission copies capture bytes into task-owned scalars. Temporary NDArray +handles are explicitly released after submission; native store handles use C++ +scope ownership. Warm calls neither extract host results nor insert execution +fences. Kernel registration uses the shared PTX loader on cache misses. diff --git a/src/cuda/mapreduce.jl b/src/cuda/mapreduce.jl new file mode 100644 index 000000000..25fa58b61 --- /dev/null +++ b/src/cuda/mapreduce.jl @@ -0,0 +1,260 @@ +const _MR_THREADS = 256 +const _MR_PTX_CACHE = Dict{Any,String}() +const _MR_PTX_LOCK = ReentrantLock() + +struct MapReduceMap{F,OP,R,N} + f::F + op::OP + # Axes are runtime data; changing dims does not change the mapper/kernel type. + mask::NTuple{N,Bool} +end +MapReduceMap(f::F, op::OP, ::Type{R}, mask::NTuple{N,Bool}) where {F,OP,R,N} = + MapReduceMap{F,OP,R,N}(f, op, mask) + +# All 32 lanes execute each shuffle, including lanes without input. Only valid +# sources enter the arithmetic: padding with an identity changes signed zeros +# and can turn complex infinities into NaNs. +@inline function _mr_reduce_warp(op, value, lane, active) + offset = 16 + while offset > 0 + other = CUDACore.shfl_down_sync(0xffffffff, value, offset) + if lane + offset <= active + value = op(value, other) + end + offset >>= 1 + end + return value +end + +# The result is defined on thread 1. Both callers launch exactly _MR_THREADS +# threads and supply a contiguous prefix of valid thread-local accumulators. +@inline function _mr_reduce_block(op, value::S, active) where {S} + tid = Int(CUDACore.threadIdx().x) + lane = ((tid - 1) & 31) + 1 + warp = ((tid - 1) >> 5) + 1 + value = _mr_reduce_warp(op, value, lane, min(32, active - (warp - 1) * 32)) + shared = CUDACore.CuStaticSharedArray(S, _MR_THREADS ÷ 32) + lane == 1 && (@inbounds shared[warp] = value) + CUDACore.sync_threads() + if warp == 1 + warps = (active + 31) >> 5 + if lane <= warps + @inbounds value = shared[lane] + end + value = _mr_reduce_warp(op, value, lane, warps) + end + return value +end + +# In one dimension the selected index is already bounded by the only extent. +# Preserve the physical stride without computing a remainder for every element. +@inline _mr_offset(A::CuStridedDeviceArray{T,1}, other::Int, red::Int, mask::NTuple{1,Bool}) where {T} = + (mask[1] ? red : other) * A.strides[1] + +# Indices are relative to the PhysicalStore's lower bound, already reflected in +# the descriptor pointer. Unchecked unsigned division avoids device exceptions. +@inline function _mr_offset(A::CuStridedDeviceArray{T,N}, other::Int, red::Int, mask::NTuple{N,Bool}) where {T,N} + o, r = _bitcast_uint(other), _bitcast_uint(red) + offset = 0 + @inbounds for d in 1:N + extent = _bitcast_uint(A.dims[d]) + if mask[d] + offset += _bitcast_int(Core.Intrinsics.urem_int(r, extent)) * A.strides[d] + r = Core.Intrinsics.udiv_int(r, extent) + else + offset += _bitcast_int(Core.Intrinsics.urem_int(o, extent)) * A.strides[d] + o = Core.Intrinsics.udiv_int(o, extent) + end + end + return offset +end + +function _mr_partial_kernel(A, scratch::CuStridedDeviceArray{S,1}, mapper::MapReduceMap{F,OP,R,N}, start::Int, chunks::Int) where {S,F,OP,R,N} + tid = Int(CUDACore.threadIdx().x) + block = Int(CUDACore.blockIdx().x) - 1 + other = start + _bitcast_int(Core.Intrinsics.udiv_int(_bitcast_uint(block), _bitcast_uint(chunks))) + chunk = _bitcast_int(Core.Intrinsics.urem_int(_bitcast_uint(block), _bitcast_uint(chunks))) + nred = 1 + @inbounds for d in 1:N + mapper.mask[d] && (nred *= A.dims[d]) + end + value = _mr_identity(mapper.op, S) + i = chunk * _MR_THREADS + tid - 1 + first = true + while i < nred + offset = _mr_offset(A, other, i, mapper.mask) + x = unsafe_load(pointer(A), offset + 1, Val(_strided_align(A))) + mapped = _mr_encode(mapper.op, convert(R, mapper.f(x))) + value = first ? mapped : _mr_combine(mapper.op)(value, mapped) + first = false + i += chunks * _MR_THREADS + end + active = min(_MR_THREADS, nred - chunk * _MR_THREADS) + value = _mr_reduce_block(_mr_combine(mapper.op), value, active) + tid == 1 && (@inbounds scratch[block + 1] = value) + return nothing +end + +# One thread owns each retained coordinate. The C++ launcher obtains this +# pointer from an exclusive Legate reduction accessor, as cuPyNumeric's GEMV +# task does. Legate, not this kernel, combines contributions between tasks. +function _mr_contribute_kernel(scratch, dest, op, single::Bool, start::Int, count::Int, chunks::Int) + i = (Int(CUDACore.blockIdx().x) - 1) * Int(CUDACore.blockDim().x) + Int(CUDACore.threadIdx().x) + if i <= count + @inbounds value = scratch[(i - 1) * chunks + 1] + for c in 2:chunks + @inbounds value = _mr_combine(op)(value, scratch[(i - 1) * chunks + c]) + end + @inbounds dest[start + i] = single ? value : _mr_combine(op)(dest[start + i], value) + end + return nothing +end + +# A full reduction has one output. Cooperate across a block instead of making +# a single thread serially combine every partial. Invalid lanes never enter the +# tree: an extra identity operation can change signed zeros or complex infinities. +# Keep nested GPU calls statically dispatched on Julia 1.10 as well. +function _mr_contribute_full_kernel(scratch::CuStridedDeviceArray{S,1}, dest, op::OP, + single::Bool, start::Int, count::Int, chunks::Int) where {S,OP} + tid = Int(CUDACore.threadIdx().x) + active = min(chunks, _MR_THREADS) + value = _mr_identity(op, S) + if tid <= active + @inbounds value = scratch[tid] + for c in (tid + _MR_THREADS):_MR_THREADS:chunks + @inbounds value = _mr_combine(op)(value, scratch[c]) + end + end + value = _mr_reduce_block(_mr_combine(op), value, active) + if tid == 1 + @inbounds dest[start + 1] = single ? value : _mr_combine(op)(dest[start + 1], value) + end + return nothing +end + +struct MapReduceFinish{OP,R,I} + op::OP + init::I +end +MapReduceFinish(op::OP, ::Type{R}, init::I) where {OP,R,I} = MapReduceFinish{OP,R,I}(op, init) +@inline function (finish::MapReduceFinish{OP,R})(x) where {OP,R} + return _mr_finish(_mr_combine(finish.op), _mr_decode(finish.op, R, x), finish.init) +end + +struct MapReduceSingleton{F,OP,R,O,I} + f::F + op::OP + init::I +end +function MapReduceSingleton( + f::F, op::OP, ::Type{R}, ::Type{O}, init::I, +) where {F,OP,R,O,I} + return MapReduceSingleton{F,OP,R,O,I}(f, op, init) +end +@inline function (finish::MapReduceSingleton{F,OP,R,O})(x) where {F,OP,R,O} + mapped = convert(R, finish.f(x)) + value = _mr_finish(_mr_combine(finish.op), mapped, finish.init) + return convert(O, value) +end +function _mr_finish_kernel(src, dest, finish) + i = (Int(CUDACore.blockIdx().x) - 1) * Int(CUDACore.blockDim().x) + Int(CUDACore.threadIdx().x) + step = Int(CUDACore.gridDim().x) * Int(CUDACore.blockDim().x) + while i <= length(dest) + @inbounds dest[i] = finish(src[i]) + i += step + end + return nothing +end + +function _mr_kernel_name(kernel, types) + target = (CUDACore.capability(CUDACore.device()), _COMPATIBLE_PTX_VERSION[]) + key = (kernel, types, target) + return lock(_MR_PTX_LOCK) do + get!(_MR_PTX_CACHE, key) do + buf = IOBuffer() + _emit_compatible_ptx(buf, kernel, types) + ptx = String(take!(buf)) + original = extract_kernel_name(ptx) + name = original * "_mr_" * string(hash(ptx); base=16) + ptx_task(replace(ptx, original => name), name) + name + end + end +end + +_mr_dim_seed(op::_MR_OP, ::Type{R}, init, dims) where {R} = init +_mr_dim_seed(op::_MR_ADD, ::Type{R}, ::NoReductionInit, dims::_MR_DIMS) where {R} = zero(R) +_mr_dim_seed(op::_MR_MUL, ::Type{R}, ::NoReductionInit, dims::_MR_DIMS) where {R} = one(R) + +function _mr_finish_name(finish::F, ::Type{T}, ::Type{O}, ::Val{D}) where {F,T,O,D} + return _mr_kernel_name(_mr_finish_kernel, ( + CuStridedDeviceArray{T,D,CUDACore.AS.Global}, + CuStridedDeviceArray{O,D,CUDACore.AS.Global}, F, + )) +end + +function _mr_submit(A, accumulator, result, mask, full, single, redop, + name, contribute_name, finish_name, mapper, finish) + axis_bits = sum(d -> UInt64(mask[d]) << (d - 1), 1:length(mask)) + mapper_ref, finish_ref = Ref(mapper), Ref(finish) + @task_scope "mapreduce" begin + # Submission copies these bytes into task-owned scalars. Borrowing + # Refs avoids Julia-owned CxxWrap vector handles on every call. + GC.@preserve A accumulator result mapper_ref finish_ref begin + submit_mapreduce( + CxxWrap.CxxPtr{CN_NDArray}(A.ptr), + CxxWrap.CxxPtr{CN_NDArray}(accumulator.ptr), + CxxWrap.CxxPtr{CN_NDArray}(result.ptr), + axis_bits, full, single, redop, name, contribute_name, finish_name, + Base.unsafe_convert(Ptr{Cvoid}, mapper_ref), sizeof(mapper), + Base.unsafe_convert(Ptr{Cvoid}, finish_ref), sizeof(finish), + ) + end + end + return result +end + +function _mr_launch(f, op, A::NDArray{T,N}, ::Type{R}, ::Type{O}, mask, shape, init, dims, single::Bool) where {T,N,R,O} + S = _mr_storage(op, R) + D = max(N, 1) + OD = max(length(shape), 1) + seed = _mr_dim_seed(op, R, init, dims) + if single && !(dims isa Colon) + finish = MapReduceSingleton(f, op, R, O, seed) + finish_name = _mr_finish_name(finish, T, O, Val(OD)) + result = nda_empty_array(shape, O) + try + # The singleton submission never uses the accumulator or mapper. + return _mr_submit(A, result, result, mask, false, true, _mr_redop(op, S), + "", "", finish_name, nothing, finish) + catch + destroy!(result) + rethrow() + end + end + input_type = CuStridedDeviceArray{T,D,CUDACore.AS.Global} + scratch_type = CuStridedDeviceArray{S,1,CUDACore.AS.Global} + mapper = MapReduceMap(f, op, R, mask) + name = _mr_kernel_name(_mr_partial_kernel, (input_type, scratch_type, typeof(mapper), Int, Int)) + RD = dims isa Colon ? 1 : D + contribute_kernel = dims isa Colon ? _mr_contribute_full_kernel : _mr_contribute_kernel + contribute_name = _mr_kernel_name(contribute_kernel, ( + scratch_type, CuStridedDeviceArray{S,RD,CUDACore.AS.Global}, typeof(op), Bool, Int, Int, Int, + )) + finish = MapReduceFinish(op, R, seed) + needs_finish = S !== O || !(finish.init isa NoReductionInit) + finish_name = needs_finish ? _mr_finish_name(finish, S, O, Val(OD)) : "" + accumulator = single ? nda_empty_array(shape, S) : nda_full_array(shape, _mr_identity(op, S)) + result = nothing + try + result = needs_finish ? nda_empty_array(shape, O) : accumulator + _mr_submit(A, accumulator, result, mask, dims isa Colon, single, _mr_redop(op, S), + name, contribute_name, finish_name, mapper, finish) + catch + isnothing(result) || destroy!(result) + rethrow() + finally + result === accumulator || destroy!(accumulator) + end + return result +end diff --git a/src/cuda/strided_device_array.jl b/src/cuda/strided_device_array.jl index 234044f4a..df33d258d 100644 --- a/src/cuda/strided_device_array.jl +++ b/src/cuda/strided_device_array.jl @@ -14,9 +14,9 @@ * limitations under the License. =# -# Device-side strided array packed by RunPTXBroadcastTask only. +# Device-side strided array shared by broadcast and mapped reduction tasks. # Layout must match C++ `CuStridedDeviceArray` in -# lib/cunumeric_jl_wrapper/src/cuda.cpp: +# lib/cunumeric_jl_wrapper/include/ptx.h: # ptr, maxsize, dims[N], strides[N] (element strides), length # # Dense RunPTXTask / @cuda_task still uses CUDA.jl CuDeviceArray (unchanged). diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 6ac971e4c..784e14c2f 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -67,13 +67,18 @@ function Base.similar(bc::Broadcasted{NDArrayStyle{N}}, ::Type{ElType}) where {N end function __broadcast(f::Function, _, args...) - #! WITH FUSION I THINK WE CAN SUPPORT THIS BY JUST CALLING MAP or MAP! return error( - """ - Tried to broadcast $(f). cuNumeric.jl does not support broadcasting user-defined functions yet. Please re-define \ - functions to match supported patterns. For example g(x) = x + 1 could be re-defined as \ - broadcast_g(x::NDArray) = x .+ 1. This can make the intention of code opaque to the reader, \ - but it is necessary until support is added.""", + "Broadcasting $(f) is not supported by cuNumeric's unfused broadcast path.\n" * + "Single-operation broadcasts skip fusion when FUSE_BROADCAST_MIN_OPS > 1 " * + "(current: $(FUSE_BROADCAST_MIN_OPS); fusion enabled: $(FUSE_BROADCAST_EXPRS)).\n" * + "To enable GPU fusion for eligible single-operation broadcasts, set the " * + "FUSE_BROADCAST_MIN_OPS preference to 1:\n" * + " using CNPreferences\n" * + " CNPreferences.enable_broadcast_fusion!()\n" * + " CNPreferences.set_broadcast_fusion_min_ops!(1)\n" * + "Then restart Julia and retry. This is a preference, not an environment variable.\n" * + "Fusion requires an active GPU, compatible array shapes, and a GPU-compilable function. " * + "Otherwise, rewrite the expression using supported broadcast operations.", ) end diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index e42e33e3c..2c6b9fb96 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -163,6 +163,18 @@ _scope_op(kind, op_code) = string(kind, "#", Int32(op_code)) NDArray(value::T) where {T<:SUPPORTED_TYPES} = nda_full_array((), value) # construction +# Internal outputs only: callers must overwrite every element before any read. +function nda_empty_array(dims::Dims{N}, ::Type{T}) where {T,N} + shape = collect(UInt64, dims) + legate_type = Legate.to_legate_type(T) + ptr = @task_scope "empty" begin + ccall((:nda_empty_array, libnda), + NDArray_t, (Int32, Ptr{UInt64}, Legate.LegateTypeAllocated), + Int32(N), shape, legate_type) + end + return NDArray(ptr, T, Val(N)) +end + function nda_zeros_array(dims::Dims{N}, ::Type{T}) where {T,N} shape = collect(UInt64, dims) legate_type = Legate.to_legate_type(T) diff --git a/src/ndarray/mapreduce.jl b/src/ndarray/mapreduce.jl new file mode 100644 index 000000000..6c6ec52ac --- /dev/null +++ b/src/ndarray/mapreduce.jl @@ -0,0 +1,157 @@ +# Mapping and reduction policies are separate: sum/prod widen small integers, +# whereas mapreduce with +/* uses the ordinary scalar operators. +struct NoReductionInit end +const _MR_ADD = Union{typeof(+),typeof(Base.add_sum)} +const _MR_MUL = Union{typeof(*),typeof(Base.mul_prod)} +const _MR_EXTREMA = Union{typeof(min),typeof(max)} +const _MR_OP = Union{_MR_ADD,_MR_MUL,_MR_EXTREMA} +const _MR_DIMS = Union{Integer,Tuple} + +_mr_operator(op::_MR_OP) = op +_mr_operator(op) = throw(ArgumentError("mapreduce supports only +, *, min, and max")) +_mr_combine(::_MR_ADD) = (+) +_mr_combine(::_MR_MUL) = (*) +_mr_combine(op::_MR_EXTREMA) = op +_mr_redop(::_MR_ADD, ::Type) = MAPREDUCE_ADD +_mr_redop(::_MR_MUL, ::Type) = MAPREDUCE_MUL +_mr_redop(::typeof(min), ::Type) = MAPREDUCE_MIN +_mr_redop(::typeof(max), ::Type) = MAPREDUCE_MAX + +_mr_storage(::_MR_OP, ::Type{T}) where {T} = T +# Legate's bitwise reducers do not provide Bool specializations. Product and +# extrema on 0/1 bytes preserve the Boolean operations without custom reducers. +_mr_storage(::Union{_MR_MUL,_MR_EXTREMA}, ::Type{Bool}) = UInt8 +_mr_storage(::_MR_EXTREMA, ::Type{Float32}) = UInt32 +_mr_storage(::_MR_EXTREMA, ::Type{Float64}) = UInt64 +_mr_encode(::_MR_OP, x) = x +_mr_encode(::Union{_MR_MUL,_MR_EXTREMA}, x::Bool) = UInt8(x) +_mr_nan_key(::typeof(min), ::Type{U}) where {U} = zero(U) +_mr_nan_key(::typeof(max), ::Type{U}) where {U} = typemax(U) +@inline function _mr_encode(op::_MR_EXTREMA, x::T) where {T<:Union{Float32,Float64}} + U = _mr_storage(op, T) + bits = reinterpret(U, x) + sign = one(U) << (8sizeof(U) - 1) + return isnan(x) ? _mr_nan_key(op, U) : (bits & sign == 0 ? bits ⊻ sign : ~bits) +end +_mr_decode(::_MR_OP, ::Type{T}, x) where {T} = x +_mr_decode(::Union{_MR_MUL,_MR_EXTREMA}, ::Type{Bool}, x) = !iszero(x) +@inline function _mr_decode(op::_MR_EXTREMA, ::Type{T}, x::U) where {T<:Union{Float32,Float64},U<:Unsigned} + x == _mr_nan_key(op, U) && return T(NaN) + sign = one(U) << (8sizeof(U) - 1) + return reinterpret(T, x & sign == 0 ? ~x : x ⊻ sign) +end + +_mr_identity(::_MR_ADD, ::Type{T}) where {T} = zero(T) +# -0 is neutral even when all full-reduction inputs are negative zero. +_mr_identity(::_MR_ADD, ::Type{T}) where {T<:AbstractFloat} = -zero(T) +_mr_identity(::_MR_ADD, ::Type{Complex{T}}) where {T} = Complex{T}(-zero(T), -zero(T)) +_mr_identity(::_MR_MUL, ::Type{T}) where {T} = one(T) +_mr_identity(::typeof(min), ::Type{T}) where {T} = typemax(T) +_mr_identity(::typeof(max), ::Type{T}) where {T} = typemin(T) + +function _mr_dims(shape::Dims{N}, ::Colon) where {N} + return ntuple(_ -> true, max(N, 1)), () +end +function _mr_dims(shape::Dims{N}, dims) where {N} + region = dims isa Integer ? (dims,) : dims + region isa Tuple || throw(ArgumentError("dims must be :, an integer, or a tuple of integers")) + Base.reduced_indices(map(Base.OneTo, shape), region) # Base's validation, including redundant axes + mask = ntuple(d -> d in region, max(N, 1)) + return mask, ntuple(d -> mask[d] ? 1 : shape[d], N) +end + +_mr_scalar_capture(::Type{T}) where {T<:SUPPORTED_ARRAY_TYPES} = true +_mr_scalar_capture(::Type{T}) where {T} = isbitstype(T) && all(_mr_scalar_capture, fieldtypes(T)) && !isprimitivetype(T) +struct MapReduceConvert{T} end +(::MapReduceConvert{T})(x) where {T} = T(x) +_mr_callable(f) = f +_mr_callable(::Type{T}) where {T<:SUPPORTED_ARRAY_TYPES} = MapReduceConvert{T}() +function _mr_mapped_type(f::F, ::Type{T}) where {F,T} + _mr_scalar_capture(F) || throw(ArgumentError("mapreduce requires an isbits callable with scalar captures; captured arrays and pointers are unsupported")) + M = Base.promote_op(f, T) + isconcretetype(M) && M <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError("mapreduce mapping must infer a supported scalar result type; inferred $M")) + return M +end + +_mr_supported_accumulator(op, ::Type{R}) where {R} = R +_mr_supported_accumulator(::_MR_MUL, ::Type{ComplexF64}) = + throw(ArgumentError("ComplexF64 product accumulators are unsupported: Legate has no corresponding built-in reducer")) + +function _mr_accumulator(op::_MR_OP, ::Type{M}) where {M} + op isa _MR_EXTREMA && M <: Complex && throw(ArgumentError("min/max mapreduce does not support complex mapped values")) + R = Base.promote_op(Base.reduce_first, typeof(op), M) + isconcretetype(R) && R <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError("unsupported mapreduce accumulator type $R")) + return _mr_supported_accumulator(op, R) +end +_mr_accumulator(op, M, ::NoReductionInit, dims::_MR_DIMS) = _mr_accumulator(op, M) +_mr_accumulator(op, M, init, ::Colon) = _mr_accumulator(op, M) +function _mr_accumulator(op, M, init::I, dims::_MR_DIMS) where {I} + # Narrowing after each update is not an associative distributed reduction. + R = _mr_accumulator(op, M) + op isa _MR_EXTREMA && I <: Complex && throw(ArgumentError("min/max mapreduce does not support complex init")) + Base.promote_op(_mr_combine(op), I, R) === I || throw(ArgumentError("dimensional mapreduce requires init's type to hold the accumulator without narrowing; use init::$R")) + return _mr_supported_accumulator(op, I) +end + +_mr_finish(op, x, ::NoReductionInit) = x +_mr_finish(op, x, init) = op(init, x) +_mr_output_type(op, R, ::NoReductionInit, dims::_MR_DIMS) = R +_mr_output_type(op, R, init, ::Colon) = Base.promote_op(_mr_combine(op), typeof(init), R) +_mr_output_type(op, R, init, dims::_MR_DIMS) = typeof(init) +_mr_output_type(op, R, ::NoReductionInit, ::Colon) = R + +_mr_empty(f, op, T, M, ::NoReductionInit, ::Colon) = Base.mapreduce_empty(f, op, T) +_mr_empty(f, op, T, M, init, dims) = init +_mr_empty(f, op::_MR_ADD, T, M, ::NoReductionInit, dims::_MR_DIMS) = zero(_mr_accumulator(op, M)) +_mr_empty(f, op::_MR_MUL, T, M, ::NoReductionInit, dims::_MR_DIMS) = one(_mr_accumulator(op, M)) +# Base rejects empty reduced axes before scalar reduction dispatch (which can +# throw MethodError on Julia 1.10). abs/abs2 maxima have a separate zero seed. +_mr_empty(f, op::_MR_EXTREMA, T, M, ::NoReductionInit, dims::_MR_DIMS) = + throw(ArgumentError("reducing over an empty collection is not allowed")) +_mr_empty(f::Union{typeof(abs),typeof(abs2)}, op::typeof(max), T, M, ::NoReductionInit, dims::_MR_DIMS) = + Base.mapreduce_empty(f, op, T) + +""" + mapreduce(f, op, A::NDArray; dims=:, init) + +Fuse a scalar mapping with a distributed GPU reduction. Supported operators are +`+`, `*`, `min`, and `max`. Full reductions return a 0-d `NDArray`; explicit +dimensions retain singleton axes. `sum(f, A)` and `prod(f, A)` use Base's integer +widening rules, subject to `allowpromotion`. + +Requires an active GPU target and a type-stable GPU-compilable callable with only +isbits scalar captures. One input array is supported; complex results support +only addition/product; ComplexF64 product accumulators are unsupported by Legate. +Narrowing dimensional `init` types are unsupported. +Floating-point results may differ in rounding with partitioning; extrema preserve +NaNs and signed zeros, but not NaN payloads. See the mapped-reductions documentation. +""" +function Base.mapreduce(f::F, op::OP, A::NDArray{T}; dims=:, init=NoReductionInit()) where {F,OP,T} + op = _mr_operator(op) + input_shape = size(A) + mask, shape = _mr_dims(input_shape, dims) + _has_gpu_target() || throw(ArgumentError("mapped reductions currently require a Legate GPU target")) + mapper = _mr_callable(f) + M = _mr_mapped_type(mapper, T) + init isa NoReductionInit || (isbitstype(typeof(init)) && init isa SUPPORTED_ARRAY_TYPES) || throw(ArgumentError("init must be a supported scalar number")) + R = _mr_accumulator(op, M, init, dims) + O = _mr_output_type(op, R, init, dims) + isconcretetype(O) && O <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError("unsupported mapreduce output type $O")) + is_wider_type(M, T) && assertpromotion(f, T, M) + is_wider_type(R, M) && assertpromotion(op, M, R) + is_wider_type(O, R) && assertpromotion(op, R, O) + nreduce = prod(d -> mask[d] ? input_shape[d] : 1, 1:ndims(A); init=1) + if nreduce == 0 + value = _mr_empty(f, op, T, M, init, dims) + return nda_full_array(shape, value) + end + any(iszero, input_shape) && return nda_zeros_array(shape, O) + return _mr_launch(mapper, op, A, R, O, mask, shape, init, dims, nreduce == 1) +end + +Base.mapreduce(f, op, A::NDArray, B::AbstractArray, rest::AbstractArray...; kwargs...) = + throw(ArgumentError("mapreduce currently supports one input NDArray")) + +for (name, op) in ((:sum, Base.add_sum), (:prod, Base.mul_prod), (:minimum, min), (:maximum, max)) + @eval Base.$name(f, A::NDArray; kwargs...) = mapreduce(f, $op, A; kwargs...) +end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 659617a1c..3afc1630b 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -192,32 +192,30 @@ function (::Type{Array{T}})(arr::NDArray{S,0}) where {T,S} end function (::Type{Array{T}})(arr::NDArray{T,1}) where {T} - out = Vector{T}(undef, length(arr)) - isempty(out) && return out - # get_ptr waits for the source to be available in host memory. Keep its - # owner alive until the synchronous CPU copy into Julia-owned storage ends. - GC.@preserve arr out begin - src = Ptr{T}(get_ptr(arr)) - unsafe_copyto!(pointer(out), src, length(out)) - end - return out + # Legate.get_ptr requests a write accessor. A singleton can be backed by + # a read-only future, so copy into attached host storage before mapping it. + return _copy_to_julia_array(arr) end function (::Type{Array{T}})(arr::NDArray{S,1}) where {T,S} - return T.(make_array(S, Ptr{S}(get_ptr(arr)), size(arr))) + return copyto!(Vector{T}(undef, length(arr)), _copy_to_julia_array(arr)) end # Copy logically into Julia's column-major storage. # Legate may map an NDArray in C or Fortran order. function _copy_to_julia_array(arr::NDArray{T,N}) where {T,N} out = Array{T}(undef, size(arr)) + isempty(out) && return out store = Legate.attach_external_col_major(out) ptr = cuNumeric.nda_store_to_ndarray(store.handle) finalize(store.handle) attached = NDArray(ptr, T, Val(N), out) - copyto!(attached, arr) - get_ptr(attached) # Block until the copy into `out` completes. - destroy!(attached) + try + copyto!(attached, arr) + get_ptr(attached) # Block until the copy into `out` completes. + finally + destroy!(attached) + end return out end diff --git a/test/array/conversion_lifetimes.jl b/test/array/conversion_lifetimes.jl index 88e9b5bf1..9e469d471 100644 --- a/test/array/conversion_lifetimes.jl +++ b/test/array/conversion_lifetimes.jl @@ -28,3 +28,34 @@ using Test end end end + +@testset "Singleton vector host conversion" begin + # Runtime-created singletons can use scalar futures; attached input + # vectors do not exercise the same storage representation. + for (T, S, value) in ((Float32, Float64, -0f0), + (ComplexF32, ComplexF64, ComplexF32(Inf, 0)), + (Bool, Int32, true)) + a = cuNumeric.fill(value, (1,)) + try + converted = Array(a) + widened = Array{S}(a) + @test isequal(converted, T[value]) + @test isequal(widened, S[value]) + converted[1] = zero(T) + @test isequal(Array(a), T[value]) + cuNumeric.destroy!(a) + cuNumeric.issue_execution_fence(; block=true) + GC.gc(true) + @test isequal(widened, S[value]) + finally + cuNumeric.destroy!(a) + end + end + a = cuNumeric.zeros(Float32, 0) + try + @test Array(a) == Float32[] + @test Array{Float64}(a) == Float64[] + finally + cuNumeric.destroy!(a) + end +end diff --git a/test/array/mapreduce_policy.jl b/test/array/mapreduce_policy.jl new file mode 100644 index 000000000..eaf52142e --- /dev/null +++ b/test/array/mapreduce_policy.jl @@ -0,0 +1,77 @@ +struct ReductionPointerNumber <: Number + ptr::Ptr{Float32} +end +(f::ReductionPointerNumber)(x) = x + +@testset "Mapped reduction policies" begin + CN = cuNumeric + noinit = CN.NoReductionInit() + @test CN._mr_dims((2, 3), :) == ((true, true), ()) + @test CN._mr_dims((2, 3), (1, 1, 4)) == ((true, false), (1, 3)) + @test CN._mr_dims((2, 3), ()) == ((false, false), (2, 3)) + @test_throws ArgumentError CN._mr_dims((2, 3), 0) + @test_throws ArgumentError CN._mr_dims((2, 3), (1, 1.5)) + @test_throws ArgumentError CN._mr_operator(-) + @test CN._mr_redop(+, Float32) isa CN.MapReduceOp + @test CN._mr_redop(+, Float32) == CN.MAPREDUCE_ADD + @test CN._mr_redop(Base.add_sum, Int64) == CN.MAPREDUCE_ADD + @test CN._mr_redop(*, Float32) == CN.MAPREDUCE_MUL + @test CN._mr_redop(Base.mul_prod, Int64) == CN.MAPREDUCE_MUL + @test CN._mr_redop(min, UInt32) == CN.MAPREDUCE_MIN + @test CN._mr_redop(max, UInt64) == CN.MAPREDUCE_MAX + for op in (*, Base.mul_prod, min, max), a in (false, true), b in (false, true) + @test CN._mr_storage(op, Bool) === UInt8 + encoded = CN._mr_combine(op)(CN._mr_encode(op, a), CN._mr_encode(op, b)) + @test CN._mr_decode(op, Bool, encoded) === op(a, b) + end + @test_throws ArgumentError CN._mr_accumulator(min, ComplexF32) + for op in (*, Base.mul_prod) + @test_throws ArgumentError CN._mr_accumulator(op, ComplexF64) + @test_throws ArgumentError CN._mr_accumulator(op, ComplexF32, one(ComplexF64), 1) + end + @test_throws ArgumentError CN._mr_accumulator(+, Float64, 0f0, 1) + @test_throws ArgumentError CN._mr_accumulator(min, Float32, ComplexF32(0), 1) + @test_throws ArgumentError CN._mr_mapped_type(ReductionPointerNumber(Ptr{Float32}(0)), Float32) + + for T in (Bool, Int8, UInt16, Int32, UInt64, Float32, Float64, ComplexF32), + op in (+, *, Base.add_sum, Base.mul_prod) + @test CN._mr_accumulator(op, T) === typeof(Base.reduce_first(op, one(T))) + end + @testset "Empty reduction f=$f op=$op dims=$dims" for + f in (identity, abs, abs2, x -> x*x), op in (+, *, min, max), dims in (:, 1, (1,)) + reference = try + mapreduce(f, op, Float32[]; dims) + catch e + e + end + if reference isa Exception + @test_throws typeof(reference) CN._mr_empty(f, op, Float32, Float32, noinit, dims) + else + scalar = reference isa AbstractArray ? only(reference) : reference + @test isequal(CN._mr_empty(f, op, Float32, Float32, noinit, dims), scalar) + end + end + @test CN._mr_output_type(+, Float32, 0.0, :) === Float64 + @test CN._mr_output_type(+, Float32, noinit, 1) === Float32 + + # Verify ordering against Julia, including NaNs and signed zeros. + for T in (Float32, Float64), op in (min, max) + xs = T[-Inf, -2, -0.0, 0.0, 2, Inf, NaN] + for a in xs, b in xs + encoded = op(CN._mr_encode(op, a), CN._mr_encode(op, b)) + @test isequal(CN._mr_decode(op, T, encoded), op(a, b)) + end + end + captured = let a = [1.0f0] + x -> x + a[1] + end + @test_throws ArgumentError CN._mr_mapped_type(captured, Float32) + @test CN._mr_mapped_type(CN._mr_callable(Float64), Float32) === Float64 + + if !CN._has_gpu_target() + A = CN.ones(Float32, 4) + @test_throws ArgumentError mapreduce(identity, +, A) + @test_throws ArgumentError sum(abs2, A) + CN.destroy!(A) + end +end diff --git a/test/build_cxxwrap.jl b/test/build_cxxwrap.jl new file mode 100644 index 000000000..ae978fbf5 --- /dev/null +++ b/test/build_cxxwrap.jl @@ -0,0 +1,92 @@ +# Standalone build regression tests; no CUDA, Legate, or network required. +# Run: julia --startup-file=no test/build_cxxwrap.jl +using Test +include(joinpath(@__DIR__, "..", "deps", "cxxwrap.jl")) +is_supported_version(v::VersionNumber) = v"26.6.0" <= v <= v"26.11.999" + +@testset "libcxxwrap build cache recovery" begin + mktempdir() do tmp + root = replace(tmp, '\\' => '/') + override = joinpath(root, "dev", "libcxxwrap_julia_jll", "override") + mkpath(override) + headers = "$root/headers" + mkpath(headers) + library = "$root/libcxxwrap.so" + write(library, "fixture") + config = """ + foreach(name cxxwrap_julia cxxwrap_julia_stl) + add_library(JlCxx::\${name} SHARED IMPORTED) + set_target_properties(JlCxx::\${name} PROPERTIES + INTERFACE_INCLUDE_DIRECTORIES "$headers;" + IMPORTED_CONFIGURATIONS RELEASE + IMPORTED_LOCATION_RELEASE "$library") + endforeach() + """ + config_path = joinpath(override, "JlCxxConfig.cmake") + write(config_path, config) + usable = cxxwrap_usable(override; log_dir=root) + usable || print(read(joinpath(root, "libcxxwrap_check.log"), String)) + @test usable + rm(headers; recursive=true) + @test !cxxwrap_usable(override; log_dir=root) + mkpath(headers) + rm(library) + @test !cxxwrap_usable(override; log_dir=root) + write(library, "fixture") + + marker = joinpath(override, "LEGATE_INSTALL.txt") + julia_marker = joinpath(override, "JULIA_INSTALL.txt") + write(marker, "26.6.0") + write(julia_marker, cxxwrap_julia_identity()) + original_depots = copy(DEPOT_PATH) + try + empty!(DEPOT_PATH) + push!(DEPOT_PATH, root) + # A healthy cache must work even with no installer present. + @test isnothing(ensure_cxxwrap(root, v"26.6.0"; log_dir=root)) + scripts = joinpath(root, "scripts") + mkpath(scripts) + installer = joinpath(scripts, "install_cxxwrap.sh") + write(installer, "#!/bin/bash\nexit 7\n") + write(config_path, replace(config, headers => "$root/deleted-headers")) + @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) + @test !isfile(marker) + @test !isfile(julia_marker) + + # Simulate successful rebuilding of the stale CMake export. + write(joinpath(scripts, "JlCxxConfig.cmake"), config) + write(installer, "#!/bin/bash\ncp \"\$1/scripts/JlCxxConfig.cmake\" \"\$1/dev/libcxxwrap_julia_jll/override/JlCxxConfig.cmake\"\n") + @test isnothing(ensure_cxxwrap(root, v"26.6.0"; log_dir=root)) + @test read(marker, String) == "26.6.0" + @test read(julia_marker, String) == cxxwrap_julia_identity() + + # Existing paths and a matching provider version are insufficient + # when Julia changes, or when a legacy build has no Julia marker. + write(installer, "#!/bin/bash\nexit 7\n") + old_version = VERSION.major == 1 && VERSION.minor == 11 ? "1.12.0" : "1.11.0" + write(julia_marker, old_version * "\n" * realpath(joinpath(Sys.BINDIR, Base.julia_exename()))) + @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) + @test !isfile(marker) + @test !isfile(julia_marker) + write(marker, "26.6.0") + @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) + @test !isfile(marker) + + # Moving to another Julia installation also forces a rebuild. + write(marker, "26.6.0") + write(julia_marker, string(VERSION, "\n/old/julia/bin/julia")) + @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) + @test !isfile(julia_marker) + + # A zero exit status without usable outputs is not success. + write(config_path, replace(config, headers => "$root/deleted-headers")) + write(installer, "#!/bin/bash\nexit 0\n") + @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) + @test !isfile(marker) + @test !isfile(julia_marker) + finally + empty!(DEPOT_PATH) + append!(DEPOT_PATH, original_depots) + end + end +end diff --git a/test/gpu_only/mapreduce.jl b/test/gpu_only/mapreduce.jl new file mode 100644 index 000000000..b06c25813 --- /dev/null +++ b/test/gpu_only/mapreduce.jl @@ -0,0 +1,197 @@ +_mapped_reduction_host(A) = @allowscalar ndims(A) == 0 ? cuNumeric.unwrap(A) : Array(A) + +struct ReductionAffine + scale::Float32 + offset::Float64 + sign::Int8 +end +(f::ReductionAffine)(x) = f.scale * x + f.offset + f.sign + +function _mapped_reduction_tolerances(f, input; dims=:) + T = Base.promote_op(f, eltype(input)) + region = dims isa Integer ? (dims,) : dims + n = prod((size(input, d) for d in 1:ndims(input) if dims isa Colon || d in region); init=1) + # Use mapped magnitudes so cancellation and type-changing maps are covered. + scale = maximum(x -> abs(f(x)), input; init=zero(real(T))) + return (; rtol=reduction_rtol(T, max(n, 1)), atol=reduction_atol(T, max(n, 1), scale)) +end + +function _check_mapped_reduction(f, op, input; kwargs...) + A = @allowscalar NDArray(input) + result = nothing + try + expected = mapreduce(f, op, input; kwargs...) + result = mapreduce(f, op, A; kwargs...) + @test size(result) == size(expected) + @test eltype(result) === (expected isa AbstractArray ? eltype(expected) : typeof(expected)) + actual = _mapped_reduction_host(result) + if op === min || op === max || eltype(result) <: Integer + @test isequal(actual, expected) + else + tolerances = _mapped_reduction_tolerances(f, input; dims=get(kwargs, :dims, :)) + @test isapprox(actual, expected; rtol=tolerances.rtol, atol=tolerances.atol) + end + finally + cuNumeric.destroy!(A) + isnothing(result) || cuNumeric.destroy!(result) + end +end + +@testset "Fused mapped reductions" begin + @allowpromotion begin + for T in (Bool, Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, Float32, Float64) + # A typed comprehension keeps Bool inputs byte-addressable, unlike broadcast. + input = reshape(T[mod(i, 2) for i in 0:104], 7, 3, 5) + for op in (+, *, min, max), dims in (:, 1, 2, (1, 3), (3, 1), (1, 1), (), 4) + _check_mapped_reduction(identity, op, input; dims) + end + end + for T in (ComplexF32, ComplexF64), op in (+, *) + if T === ComplexF64 && op === (*) + A = cuNumeric.ones(T, 7, 9) + try + @test_throws ArgumentError mapreduce(identity, op, A; dims=1) + @test_throws ArgumentError prod(identity, A) + finally + cuNumeric.destroy!(A) + end + continue + end + _check_mapped_reduction(identity, op, fill(T(1 + 0im), 7, 9); dims=1) + _check_mapped_reduction(abs2, +, fill(T(1 + 2im), 7, 9)) + end + for shape in ((), (1,), (1, 1)), op in (+, *, min, max) + _check_mapped_reduction(abs2, op, fill(2f0, shape)) + end + # Full singletons use reduce_first; dimensional sums/products seed + # their result with zero/one. Check exact zeros and complex infinities. + for x in (-0f0, ComplexF32(Inf, 0), ComplexF32(0, Inf)), + op in (+, *), dims in (:, 1, ()) + input = fill(x, 1) + A = @allowscalar NDArray(input) + r = nothing + try + r = mapreduce(identity, op, A; dims) + @test isequal(_mapped_reduction_host(r), mapreduce(identity, op, input; dims)) + finally + cuNumeric.destroy!(A) + isnothing(r) || cuNumeric.destroy!(r) + end + end + for T in (Float32, Float64), op in (min, max) + for pair in ((-zero(T), zero(T)), (T(NaN), T(2)), (T(-Inf), T(Inf))) + input = fill(pair[2], 131071) + input[1] = pair[1] + _check_mapped_reduction(identity, op, input) + reverse!(input) + _check_mapped_reduction(identity, op, input) + end + end + + input = reshape(Float32.(1:17017) ./ 17017f0, 7, 11, 221) + _check_mapped_reduction(ReductionAffine(2f0, 0.5, Int8(-1)), +, input) + _check_mapped_reduction(Float64, +, input) + _check_mapped_reduction(identity, +, Int8[100, 100]) + for init in (0f0, 0.0), dims in (:, 1, (1, 3)) + _check_mapped_reduction(abs2, +, input; init, dims) + end + # Non-neutral seeds expose accidental replication across partitions. + _check_mapped_reduction(identity, +, ones(Float32, 131071); init=7f0) + _check_mapped_reduction(identity, *, ones(Float32, 131071); init=3f0) + _check_mapped_reduction(identity, max, ones(Float32, 131071); init=5f0) + _check_mapped_reduction(identity, min, ones(Float32, 131071); init=-5f0) + + for dims in (:, 1), f in (identity, abs, abs2) + _check_mapped_reduction(f, +, Float32[]; dims) + _check_mapped_reduction(f, *, Float32[]; dims) + end + _check_mapped_reduction(abs2, max, Float32[]) + _check_mapped_reduction(x -> x*x, +, Float32[]; init=0f0) + _check_mapped_reduction(x -> x*x, +, zeros(Float32, 0, 3); dims=1) + _check_mapped_reduction(identity, min, zeros(Float32, 3, 0); dims=1) + + A = @allowscalar NDArray(Int8[2, 3, 4]) + try + for (fn, op) in ((sum, Base.add_sum), (prod, Base.mul_prod), (minimum, min), (maximum, max)) + r = fn(identity, A) + @test _mapped_reduction_host(r) === fn(identity, Int8[2, 3, 4]) + cuNumeric.destroy!(r) + end + finally + cuNumeric.destroy!(A) + end + + # Same closure type, different capture values must reuse PTX correctly. + A = @allowscalar NDArray(input) + try + cache_size = nothing + for alpha in (1f0, 3f0, -2f0) + f = let alpha = alpha + x -> abs2(x - alpha) + end + r = mapreduce(f, +, A) + tolerances = _mapped_reduction_tolerances(f, input) + @test isapprox( + _mapped_reduction_host(r), mapreduce(f, +, input); + rtol=tolerances.rtol, atol=tolerances.atol, + ) + cuNumeric.destroy!(r) + current_size = length(cuNumeric._MR_PTX_CACHE) + isnothing(cache_size) || (@test current_size == cache_size) + cache_size = current_size + end + @test_throws ArgumentError mapreduce(identity, -, A) + @test_throws ArgumentError mapreduce(+, +, A, A) + @test_throws ArgumentError mapreduce(identity, +, A; dims=0) + @test_throws ArgumentError mapreduce(identity, +, A; dims=1, init=Int8(0)) + finally + cuNumeric.destroy!(A) + end + + empty = cuNumeric.zeros(Float32, 0) + try + # Full empty reductions delegate to Base.mapreduce_empty. Base + # throws MethodError on Julia 1.10 and ArgumentError on 1.11+. + # Require the exact host exception rather than accepting either. + for reduce_empty in (a -> mapreduce(x -> x*x, +, a), a -> minimum(identity, a)) + expected_error = try + reduce_empty(Float32[]) + catch err + err + end + @test expected_error isa Exception + @test_throws typeof(expected_error) reduce_empty(empty) + end + @test_throws ArgumentError maximum(identity, empty; dims=1) + finally + cuNumeric.destroy!(empty) + end + + # A sliced logical store can have a nonzero origin and non-dense strides. + parent = @allowscalar NDArray(reshape(Float32.(1:323), 17, 19)) + sliced = parent[2:16, 3:18] + r = mapreduce(abs2, +, sliced; dims=1) + @test _mapped_reduction_host(r) ≈ mapreduce(abs2, +, reshape(Float32.(1:323), 17, 19)[2:16, 3:18]; dims=1) + cuNumeric.destroy!(r) + cuNumeric.destroy!(sliced) + cuNumeric.destroy!(parent) + + # Drop the source before execution completes, then consume the result + # on the device without an intervening unwrap or execution fence. + A = cuNumeric.ones(Float32, 131071) + r = mapreduce(abs2, +, A) + cuNumeric.destroy!(A) + next = r + r + cuNumeric.destroy!(r) + @test _mapped_reduction_host(next) == 2f0 * 131071 + cuNumeric.destroy!(next) + end + cuNumeric.allowpromotion(false) do + A = cuNumeric.ones(Int8, 3) + @test_throws Exception sum(identity, A) + r = mapreduce(identity, +, A) + @test eltype(r) === Int8 + cuNumeric.destroy!(r) + cuNumeric.destroy!(A) + end +end diff --git a/test/gpu_only/mapreduce_full.jl b/test/gpu_only/mapreduce_full.jl new file mode 100644 index 000000000..57d9eeb1c --- /dev/null +++ b/test/gpu_only/mapreduce_full.jl @@ -0,0 +1,249 @@ +using Test +import CUDA + +# One block for each possible nonempty prefix, so every warp boundary and tail +# is checked in a single launch. Invalid lanes deliberately contain nonidentity +# values to detect accidental participation in either level of the reduction. +function _test_block_reduction(src, dest, op) + tid = Int(CUDA.threadIdx().x) + active = Int(CUDA.blockIdx().x) + @inbounds value = src[tid, active] + value = cuNumeric._mr_reduce_block(op, value, active) + tid == 1 && (@inbounds dest[active] = value) + return nothing +end + +function _check_block_prefixes(op, values::Vector{T}) where {T} + host = fill(T(3), 256, 256) + for active in 1:256 + host[1:active, active] .= values[1:active] + end + src, dest = CUDA.CuArray(host), CUDA.zeros(T, 256) + try + CUDA.@cuda threads=256 blocks=256 _test_block_reduction(src, dest, op) + expected = [foldl(op, @view(values[1:n])) for n in 1:256] + @test isequal(Array(dest), expected) + finally + CUDA.synchronize() + CUDA.unsafe_free!(src) + CUDA.unsafe_free!(dest) + end +end + +@testset "Block reduction valid prefixes" begin + for T in (Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, + Float32, Float64, ComplexF32, ComplexF64) + ops = T <: Complex ? (T === ComplexF64 ? (+,) : (+, *)) : (+, *, min, max) + for op in ops + values = op === (*) ? fill(one(T), 256) : + op === min ? fill(T(5), 256) : T[isodd(i) for i in 1:256] + _check_block_prefixes(op, values) + end + end + for T in (Float32, Float64), op in (+, *, min, max), + value in (-zero(T), zero(T), T(NaN), T(Inf), -T(Inf)) + _check_block_prefixes(op, fill(value, 256)) + end + for value in (ComplexF32(Inf, 0), ComplexF64(0, Inf)) + _check_block_prefixes(+, fill(value, 256)) + end +end + +@testset "Runtime reduction axes" begin + for (shape, strides) in (((5,), (2,)), ((2, 3), (2, 7)), ((2, 2, 3), (2, 7, 19))) + N = length(shape) + descriptor = cuNumeric.CuStridedDeviceArray{Int32,N,CUDA.AS.Global}( + reinterpret(Core.LLVMPtr{Int32,CUDA.AS.Global}, UInt(0)), + 0, shape, strides, prod(shape), + ) + mapper_type = typeof(cuNumeric.MapReduceMap(identity, +, Int32, ntuple(_ -> true, N))) + for bits in 0:(2^N - 1) + mask = ntuple(d -> !iszero(bits & (1 << (d - 1))), N) + mapper = @inferred cuNumeric.MapReduceMap(identity, +, Int32, mask) + @test typeof(mapper) === mapper_type + @test mapper.mask === mask + reduced = CartesianIndices(ntuple(d -> mask[d] ? shape[d] : 1, N)) + retained = CartesianIndices(ntuple(d -> mask[d] ? 1 : shape[d], N)) + for (r, ri) in enumerate(reduced), (o, oi) in enumerate(retained) + expected = sum(d -> (ri[d] + oi[d] - 2) * strides[d], 1:N) + @test (@inferred cuNumeric._mr_offset(descriptor, o - 1, r - 1, mask)) == expected + end + end + end +end + +function _test_strided_partial(parent, scratch, mapper, origin, stride, n, chunks) + T = eltype(parent) + src = cuNumeric.CuStridedDeviceArray{T,1,CUDA.AS.Global}( + pointer(parent, origin + 1), (length(parent) - origin) * sizeof(T), (n,), (stride,), n, + ) + dst = cuNumeric.CuStridedDeviceArray{T,1,CUDA.AS.Global}( + pointer(scratch), length(scratch) * sizeof(T), (length(scratch),), (1,), length(scratch), + ) + cuNumeric._mr_partial_kernel(src, dst, mapper, 0, chunks) + return nothing +end + +@testset "1D reduction offsets" begin + mapper = cuNumeric.MapReduceMap(identity, +, Int32, (true,)) + for n in (1, 31, 256, 257, 1025, 4097), origin in (0, 3), stride in (1, 2, 3) + host = Int32[mod(i, 7) - 3 for i in 1:(3n + 7)] + chunks = cld(n, 1024) + parent, scratch = CUDA.CuArray(host), CUDA.zeros(Int32, chunks) + try + CUDA.@cuda threads=256 blocks=chunks _test_strided_partial( + parent, scratch, mapper, origin, stride, n, chunks, + ) + expected = sum(host[(origin + 1):stride:(origin + 1 + (n - 1)*stride)]) + @test sum(Array(scratch)) == expected + finally + CUDA.synchronize() + CUDA.unsafe_free!(parent) + CUDA.unsafe_free!(scratch) + end + end +end + +function _check_full_reduction(input, op; kwargs...) + A = @allowscalar cuNumeric.NDArray(input) + result = nothing + try + expected = mapreduce(identity, op, input; kwargs...) + result = mapreduce(identity, op, A; kwargs...) + actual = @allowscalar cuNumeric.unwrap(result) + @test typeof(actual) === typeof(expected) + @test isequal(actual, expected) + finally + isnothing(result) || cuNumeric.destroy!(result) + cuNumeric.destroy!(A) + end +end + +@testset "Full reduction launch boundaries" begin + @allowpromotion begin + for T in (Bool, Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, + Float32, Float64, ComplexF32, ComplexF64) + ops = T <: Complex ? (T === ComplexF64 ? (+,) : (+, *)) : (+, *, min, max) + for n in (1, 31, 32, 33, 255, 256, 257, 1023, 1024, 1025), op in ops + _check_full_reduction(T[isodd(i) for i in 1:n], op) + end + end + for T in (Float32, Float64), op in (+, *, min, max), n in (1, 257, 1025) + for seed in (T(3), Float64(3)) + _check_full_reduction(fill(one(T), n), op; init=seed) + end + end + for T in (Float32, Float64), op in (min, max) + for x in (-zero(T), zero(T), T(NaN), T(Inf), -T(Inf)) + _check_full_reduction(fill(x, 257), op) + end + end + for x in (-0f0, ComplexF32(Inf, 0), ComplexF32(0, Inf)), op in (+, *) + _check_full_reduction([x], op) + end + end + cuNumeric.issue_execution_fence(; block=true) +end + +@testset "Full reduction of a sliced store" begin + host = reshape(Int32.(1:323), 17, 19) + parent = @allowscalar cuNumeric.NDArray(host) + sliced = parent[2:16, 3:18] + result = nothing + try + result = mapreduce(x -> x*x, +, sliced) + cuNumeric.destroy!(sliced) + sliced = nothing + cuNumeric.destroy!(parent) + parent = nothing + @test (@allowscalar cuNumeric.unwrap(result)) == mapreduce(x -> x*x, +, host[2:16, 3:18]) + finally + isnothing(result) || cuNumeric.destroy!(result) + isnothing(sliced) || cuNumeric.destroy!(sliced) + isnothing(parent) || cuNumeric.destroy!(parent) + end +end + +# Exercise the combination independently of Legate's partitioning decisions. +function _test_full_contribution(scratch, dest, op, single, chunks) + S = eltype(scratch) + src = cuNumeric.CuStridedDeviceArray{S,1,CUDA.AS.Global}( + pointer(scratch), length(scratch) * sizeof(S), (length(scratch),), (1,), length(scratch), + ) + cuNumeric._mr_contribute_full_kernel(src, dest, op, single, 0, 1, chunks) + return nothing +end + +@testset "Cooperative full contribution" begin + for S in (Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, + Float32, Float64, ComplexF32, ComplexF64) + ops = S <: Complex ? (S === ComplexF64 ? (+,) : (+, *)) : (+, *, min, max) + for op in ops, n in (1, 31, 32, 33, 63, 65, 255, 256, 257, 511, 513, + 1023, 1024, 1025, 4095, 4096), single in (false, true) + host = S[isodd(i) for i in 1:n] + seed = S(3) + expected = foldl(op, host) + single || (expected = op(seed, expected)) + scratch, dest = CUDA.CuArray(host), CUDA.CuArray([seed]) + try + CUDA.@cuda threads=256 blocks=1 _test_full_contribution(scratch, dest, op, single, n) + @test isequal(only(Array(dest)), expected) + finally + CUDA.synchronize() + CUDA.unsafe_free!(scratch) + CUDA.unsafe_free!(dest) + end + end + end +end + +struct ReductionSingletonOnly + offset::Float32 +end +(f::ReductionSingletonOnly)(x) = x + f.offset + +@testset "Singleton preparation and asynchronous output ownership" begin + A = cuNumeric.ones(Float32, 1, 33) + results = Any[] + try + before = Set(keys(cuNumeric._MR_PTX_CACHE)) + push!(results, mapreduce(ReductionSingletonOnly(2f0), +, A; dims=1)) + added = setdiff(Set(keys(cuNumeric._MR_PTX_CACHE)), before) + @test !isempty(added) + @test all(key -> key[1] === cuNumeric._mr_finish_kernel, added) + # Queue multiple freshly allocated outputs, including decode/seed + # finishers. Their input and temporary handles may die before execution. + for _ in 1:8 + push!(results, mapreduce(identity, min, A; dims=2)) + push!(results, @allowpromotion mapreduce(abs2, +, A; init=2.0)) + push!(results, mapreduce(identity, *, A; dims=())) + end + cuNumeric.destroy!(A) + @test (@allowscalar Array(results[1])) == fill(3f0, 1, 33) + for i in 2:3:length(results) + @test (@allowscalar Array(results[i])) == fill(1f0, 1, 1) + @test (@allowscalar cuNumeric.unwrap(results[i + 1])) === 35.0 + @test (@allowscalar Array(results[i + 2])) == fill(1f0, 1, 33) + end + finally + cuNumeric.destroy!(A) + foreach(cuNumeric.destroy!, results) + cuNumeric.issue_execution_fence(; block=true) + end +end + +@testset "Full contribution special values" begin + for value in (-0f0, -0.0, ComplexF32(Inf, 0), ComplexF64(0, Inf)), + n in (1, 31, 256, 257, 4096) + host = fill(value, n) + scratch, dest = CUDA.CuArray(host), CUDA.CuArray([zero(value)]) + try + CUDA.@cuda threads=256 blocks=1 _test_full_contribution(scratch, dest, +, true, n) + @test isequal(only(Array(dest)), foldl(+, host)) + finally + CUDA.synchronize() + CUDA.unsafe_free!(scratch) + CUDA.unsafe_free!(dest) + end + end +end From 50e3f8ed4b92bb2e4b44a90d156f2852f9b086cd Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sat, 19 Sep 2026 15:29:56 -0500 Subject: [PATCH 31/49] 1.13 Julia CI Matrix Support (#197) * ci: update matrix to include Julia 1.13 * fix deprecated enum CxxWrap syntax in wrapper * bump wrapper ver --------- Co-authored-by: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Co-authored-by: ejmeitz --- .buildkite/developer.pipeline.yml | 3 +- .buildkite/jll.pipeline.yml | 1 + .github/workflows/ci.yml | 1 + .github/workflows/developer.yml | 1 + Project.toml | 18 +- lib/CNPreferences/Project.toml | 2 +- lib/cunumeric_jl_wrapper/VERSION | 2 +- lib/cunumeric_jl_wrapper/src/mapreduce.cpp | 6 +- lib/cunumeric_jl_wrapper/src/types.cpp | 274 +++++++++++---------- lib/cunumeric_jl_wrapper/src/wrapper.cpp | 19 +- scripts/install_cxxwrap.sh | 2 +- src/cuNumeric.jl | 16 +- src/cuda/cuda_ptx_task.jl | 2 +- src/memory.jl | 6 +- src/ndarray/broadcast.jl | 2 +- src/ndarray/detail/fft.jl | 2 +- src/ndarray/detail/linalg.jl | 2 +- src/scoping/accelerate.jl | 2 +- src/utilities/version.jl | 7 +- test/Project.toml | 3 + test/analysis/accelerate.jl | 2 +- test/analysis/type_stability.jl | 2 +- test/array/distributed_linalg.jl | 2 + test/array/linalg.jl | 4 - test/array/tensoroperations.jl | 2 +- test/array/unary/tests.jl | 2 +- test/gpu_only/broadcast_fusion.jl | 6 +- 27 files changed, 206 insertions(+), 185 deletions(-) diff --git a/.buildkite/developer.pipeline.yml b/.buildkite/developer.pipeline.yml index 5696d00f9..38d6429c7 100644 --- a/.buildkite/developer.pipeline.yml +++ b/.buildkite/developer.pipeline.yml @@ -16,7 +16,7 @@ steps: - JuliaCI/julia#v1: version: "{{matrix.julia}}" # Per version AND fusion: dev wrappers are ABI-specific, and same-depot - # concurrent jobs contend on Pkg locks. Separate from the JLL cache. + # concurrent jobs contend on Pkg locks. Separate from the JLL cache. cache_dir: "${HOME}/.cache/julia-buildkite-plugin-developer-{{matrix.julia}}-{{matrix.fusion}}" command: ".buildkite/run_developer_ci.sh" artifact_paths: @@ -41,6 +41,7 @@ steps: - "1.10" - "1.11" - "1.12" + - "1.13" fusion: - "on" - "off" diff --git a/.buildkite/jll.pipeline.yml b/.buildkite/jll.pipeline.yml index 7d591fb46..ffe36a3f6 100644 --- a/.buildkite/jll.pipeline.yml +++ b/.buildkite/jll.pipeline.yml @@ -50,6 +50,7 @@ steps: - "1.10" - "1.11" - "1.12" + - "1.13" fusion: - "on" - "off" diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d7b63a6e7..182fcd96a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -95,6 +95,7 @@ jobs: - '1.10' - '1.11' - '1.12' + - '1.13' os: - ubuntu-latest # include: diff --git a/.github/workflows/developer.yml b/.github/workflows/developer.yml index f2d0d8565..827e4fdae 100644 --- a/.github/workflows/developer.yml +++ b/.github/workflows/developer.yml @@ -14,6 +14,7 @@ jobs: - '1.10' - '1.11' - '1.12' + - '1.13' # runs-on: [self-hosted, linux, x64] runs-on: ubuntu-latest container: diff --git a/Project.toml b/Project.toml index eb8961bc8..ef407e436 100644 --- a/Project.toml +++ b/Project.toml @@ -2,6 +2,9 @@ name = "cuNumeric" uuid = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" version = "0.3.0" +[workspace] +projects = ["test", "dev"] + [deps] AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" @@ -35,14 +38,14 @@ cuNumericTensorOperationsExt = "TensorOperations" [compat] AbstractFFTs = "1.5" CNPreferences = "0.1.4" -CUDACore = "6.2" -CUDATools = "6.2" -CxxWrap = "0.17" +CUDACore = "6.4" +CUDATools = "6.4" +CxxWrap = "0.17.5" ExpressionExplorer = "1.1.4" JuliaFormatter = "2.3.0" KernelAbstractions = "0.9.41" -Legate = "0.2.1" -LegatePreferences = "0.1.6" +Legate = "0.2.4" +LegatePreferences = "0.1.7" MacroTools = "0.5.16" OpenBLAS32_jll = "0.3" Pkg = "1" @@ -51,9 +54,6 @@ Random = "1" StaticArrays = "1" StatsBase = "0.34" TensorOperations = "5.8" -cunumeric_jl_wrapper_jll = "26.6.2" +cunumeric_jl_wrapper_jll = "26.6.4" cupynumeric_jll = "26.6.0" julia = "1.10" - -[workspace] -projects = ["test", "dev"] diff --git a/lib/CNPreferences/Project.toml b/lib/CNPreferences/Project.toml index 26ecc6905..d1fa6276f 100644 --- a/lib/CNPreferences/Project.toml +++ b/lib/CNPreferences/Project.toml @@ -8,6 +8,6 @@ LegatePreferences = "8028f36a-2b64-49e9-aa04-2d0933fd2ed9" Preferences = "21216c6a-2e73-6563-6e65-726566657250" [compat] -LegatePreferences = "0.1.6" +LegatePreferences = "0.1.7" Preferences = "1.4.3" julia = "1.10" diff --git a/lib/cunumeric_jl_wrapper/VERSION b/lib/cunumeric_jl_wrapper/VERSION index 892cb42a2..0b6d792e1 100644 --- a/lib/cunumeric_jl_wrapper/VERSION +++ b/lib/cunumeric_jl_wrapper/VERSION @@ -1 +1 @@ -26.6.2 +26.6.4 diff --git a/lib/cunumeric_jl_wrapper/src/mapreduce.cpp b/lib/cunumeric_jl_wrapper/src/mapreduce.cpp index afa317173..808b70438 100644 --- a/lib/cunumeric_jl_wrapper/src/mapreduce.cpp +++ b/lib/cunumeric_jl_wrapper/src/mapreduce.cpp @@ -32,7 +32,7 @@ class MapReduceMapper : public legate::mapping::Mapper { const legate::mapping::Task&, legate::mapping::StoreTarget target) override { // One partial buffer; ComplexF64 is the largest supported accumulator. return target == legate::mapping::StoreTarget::FBMEM - ? MAX_PARTIALS * sizeof(legate::type_of) : 0; + ? MAX_PARTIALS * sizeof(legate::type_of_t) : 0; } legate::Scalar tunable_value(legate::TunableID) override { @@ -57,7 +57,7 @@ struct PackArray { template ArrayArg operator()(const legate::PhysicalStore& store) const { if constexpr (supported) { - using T = legate::type_of; + using T = legate::type_of_t; auto rect = store.shape(); auto acc = [&]() { if constexpr (WRITE) return store.write_accessor(); @@ -165,7 +165,7 @@ struct ReduceDispatch { template void operator()(legate::TaskContext& ctx, CUfunction kernel) const { if constexpr (supported && C != legate::Type::Code::BOOL) { - using T = legate::type_of; + using T = legate::type_of_t; auto op = static_cast(ctx.scalar(3).value()); if (op == MapReduceOp::ADD) return run_reduction, D>(ctx, kernel); if constexpr (C != legate::Type::Code::COMPLEX128) { diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index 77ab7bc6a..28ab75663 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -19,149 +19,155 @@ */ #include "types.h" - #include "cupynumeric.h" +#include + void wrap_unary_ops(jlcxx::Module& mod) { - mod.add_bits("UnaryOpCode", - jlcxx::julia_type("CppEnum")); - mod.set_const("ABSOLUTE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ABSOLUTE); - mod.set_const("ANGLE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ANGLE); - mod.set_const("ARCCOS", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOS); - mod.set_const("ARCCOSH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOSH); - mod.set_const("ARCSIN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSIN); - mod.set_const("ARCSINH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSINH); - mod.set_const("ARCTAN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTAN); - mod.set_const("ARCTANH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTANH); - mod.set_const("CBRT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CBRT); - mod.set_const("CEIL", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CEIL); - mod.set_const("CLIP", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CLIP); - mod.set_const("CONJ", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CONJ); - mod.set_const("COPY", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COPY); - mod.set_const("COS", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COS); - mod.set_const("COSH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COSH); - mod.set_const("DEG2RAD", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_DEG2RAD); - mod.set_const("EXP", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP); - mod.set_const("EXP2", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP2); - mod.set_const("EXPM1", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXPM1); - mod.set_const("FLOOR", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FLOOR); - mod.set_const("FREXP", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FREXP); - mod.set_const("GETARG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_GETARG); - mod.set_const("IMAG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_IMAG); - mod.set_const("INVERT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_INVERT); - mod.set_const("ISFINITE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISFINITE); - mod.set_const("ISINF", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISINF); - mod.set_const("ISNAN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISNAN); - mod.set_const("LOG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG); - mod.set_const("LOG10", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG10); - mod.set_const("LOG1P", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG1P); - mod.set_const("LOG2", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG2); - mod.set_const("LOGICAL_NOT", - CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOGICAL_NOT); - mod.set_const("MODF", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_MODF); - mod.set_const("NEGATIVE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_NEGATIVE); - mod.set_const("POSITIVE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_POSITIVE); - mod.set_const("RAD2DEG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RAD2DEG); - mod.set_const("REAL", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_REAL); - mod.set_const("RECIPROCAL", - CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RECIPROCAL); - mod.set_const("RINT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RINT); - mod.set_const("ROUND", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ROUND); - mod.set_const("SIGN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGN); - mod.set_const("SIGNBIT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGNBIT); - mod.set_const("SIN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIN); - mod.set_const("SINH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SINH); - mod.set_const("SQRT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQRT); - mod.set_const("SQUARE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQUARE); - mod.set_const("TAN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TAN); - mod.set_const("TANH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TANH); - mod.set_const("TRUNC", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TRUNC); + + mod.add_enum("UnaryOpCode", + std::vector( + {"ABSOLUTE", "ANGLE", "ARCCOS", "ARCCOSH", "ARCSIN", "ARCSINH", "ARCTAN", "ARCTANH", + "CBRT", "CEIL", "CLIP", "CONJ", "COPY", "COS", "COSH", "DEG2RAD", + "EXP", "EXP2", "EXPM1", "FLOOR", "FREXP", "GETARG", "IMAG", "INVERT", + "ISFINITE", "ISINF", "ISNAN", "LOG", "LOG10", "LOG1P", "LOG2", + "LOGICAL_NOT", "MODF", "NEGATIVE", "POSITIVE", "RAD2DEG", "REAL", + "RECIPROCAL", "RINT", "ROUND", "SIGN", "SIGNBIT", "SIN", "SINH", + "SQRT", "SQUARE", "TAN", "TANH", "TRUNC"}), + std::vector({ + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ABSOLUTE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ANGLE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOS), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOSH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSIN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSINH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTAN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTANH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CBRT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CEIL), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CLIP), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CONJ), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COPY), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COS), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COSH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_DEG2RAD), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP2), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXPM1), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FLOOR), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FREXP), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_GETARG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_IMAG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_INVERT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISFINITE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISINF), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISNAN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG10), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG1P), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG2), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOGICAL_NOT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_MODF), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_NEGATIVE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_POSITIVE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RAD2DEG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_REAL), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RECIPROCAL), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RINT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ROUND), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGNBIT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SINH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQRT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQUARE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TAN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TANH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TRUNC), + + }) + ); } void wrap_unary_reds(jlcxx::Module& mod) { - mod.add_bits("UnaryRedCode", - jlcxx::julia_type("CppEnum")); - mod.set_const("ALL", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ALL); - mod.set_const("ANY", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ANY); - mod.set_const("ARGMAX", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX); - mod.set_const("ARGMIN", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMIN); - mod.set_const("CONTAINS", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_CONTAINS); - mod.set_const("COUNT_NONZERO", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_COUNT_NONZERO); - mod.set_const("MAX", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MAX); - mod.set_const("MIN", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MIN); - mod.set_const("NANARGMAX", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX); - mod.set_const("NANARGMIN", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMIN); - mod.set_const("NANMAX", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMAX); - mod.set_const("NANMIN", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMIN); - mod.set_const("NANPROD", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANPROD); - mod.set_const("NANSUM", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANSUM); - mod.set_const("PROD", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_PROD); - mod.set_const("SUM", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM); - mod.set_const("SUM_SQUARES", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM_SQUARES); - mod.set_const("VARIANCE", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_VARIANCE); + + mod.add_enum("UnaryRedCode", + std::vector( + {"ALL", "ANY", "ARGMAX", "ARGMIN", "CONTAINS", "COUNT_NONZERO", "MAX", + "MIN", "NANARGMAX", "NANARGMIN", "NANMAX", "NANMIN", "NANPROD", + "NANSUM", "PROD", "SUM", "SUM_SQUARES", "VARIANCE"}), + std::vector({ + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ALL), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ANY), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_CONTAINS), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_COUNT_NONZERO), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANPROD), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANSUM), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_PROD), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM_SQUARES), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_VARIANCE) + }) + ); } void wrap_binary_ops(jlcxx::Module& mod) { - mod.add_bits("BinaryOpCode", - jlcxx::julia_type("CppEnum")); - mod.set_const("ADD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ADD); - mod.set_const("ARCTAN2", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ARCTAN2); - mod.set_const("BITWISE_AND", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_AND); - mod.set_const("BITWISE_OR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_OR); - mod.set_const("BITWISE_XOR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_XOR); - mod.set_const("COPYSIGN", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_COPYSIGN); - mod.set_const("DIVIDE", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_DIVIDE); - mod.set_const("EQUAL", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_EQUAL); - mod.set_const("FLOAT_POWER", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOAT_POWER); - mod.set_const("FLOOR_DIVIDE", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOOR_DIVIDE); - mod.set_const("FMOD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FMOD); - mod.set_const("GCD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GCD); - mod.set_const("GREATER", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER); - mod.set_const("GREATER_EQUAL", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER_EQUAL); - mod.set_const("HYPOT", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_HYPOT); - mod.set_const("ISCLOSE", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ISCLOSE); - mod.set_const("LCM", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LCM); - mod.set_const("LDEXP", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LDEXP); - mod.set_const("LEFT_SHIFT", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LEFT_SHIFT); - mod.set_const("LESS", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS); - mod.set_const("LESS_EQUAL", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS_EQUAL); - mod.set_const("LOGADDEXP", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP); - mod.set_const("LOGADDEXP2", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP2); - mod.set_const("LOGICAL_AND", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_AND); - mod.set_const("LOGICAL_OR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_OR); - mod.set_const("LOGICAL_XOR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_XOR); - mod.set_const("MAXIMUM", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MAXIMUM); - mod.set_const("MINIMUM", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MINIMUM); - mod.set_const("MOD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MOD); - mod.set_const("MULTIPLY", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MULTIPLY); - mod.set_const("NEXTAFTER", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NEXTAFTER); - mod.set_const("NOT_EQUAL", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NOT_EQUAL); - mod.set_const("POWER", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_POWER); - mod.set_const("RIGHT_SHIFT", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_RIGHT_SHIFT); - mod.set_const("SUBTRACT", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_SUBTRACT); + + mod.add_enum("BinaryOpCode", + std::vector({ + "ADD", "ARCTAN2", "BITWISE_AND", "BITWISE_OR", "BITWISE_XOR", + "COPYSIGN", "DIVIDE", "EQUAL", "FLOAT_POWER", "FLOOR_DIVIDE", + "FMOD", "GCD", "GREATER", "GREATER_EQUAL", "HYPOT", "ISCLOSE", "LCM", "LDEXP", "LEFT_SHIFT", + "LESS", "LESS_EQUAL", "LOGADDEXP", "LOGADDEXP2", "LOGICAL_AND", "LOGICAL_OR", "LOGICAL_XOR", + "MAXIMUM", "MINIMUM", "MOD", "MULTIPLY", "NEXTAFTER", + "NOT_EQUAL", "POWER", "RIGHT_SHIFT", "SUBTRACT" + }), + std::vector({ + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ADD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ARCTAN2), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_AND), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_OR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_XOR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_COPYSIGN), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_DIVIDE), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOAT_POWER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOOR_DIVIDE), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FMOD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GCD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_HYPOT), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ISCLOSE), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LCM), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LDEXP), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LEFT_SHIFT), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP2), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_AND), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_OR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_XOR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MAXIMUM), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MINIMUM), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MOD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MULTIPLY), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NEXTAFTER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NOT_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_POWER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_RIGHT_SHIFT), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_SUBTRACT) + }) + ); } void wrap_linalg_ops(jlcxx::Module& mod) { diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 00423dcee..177c7ed8c 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -163,11 +163,20 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { wrap_fft_ops(mod); wrap_bitgenerator_ops(mod); - mod.add_bits("MapReduceOp", jlcxx::julia_type("CppEnum")); - mod.set_const("MAPREDUCE_ADD", ufi::MapReduceOp::ADD); - mod.set_const("MAPREDUCE_MUL", ufi::MapReduceOp::MUL); - mod.set_const("MAPREDUCE_MIN", ufi::MapReduceOp::MIN); - mod.set_const("MAPREDUCE_MAX", ufi::MapReduceOp::MAX); + mod.add_enum("MapReduceOp", + std::vector({ + "MAPREDUCE_ADD", + "MAPREDUCE_MUL", + "MAPREDUCE_MIN", + "MAPREDUCE_MAX" + }), + std::vector({ + static_cast(ufi::MapReduceOp::ADD), + static_cast(ufi::MapReduceOp::MUL), + static_cast(ufi::MapReduceOp::MIN), + static_cast(ufi::MapReduceOp::MAX) + }) + ); using jlcxx::ParameterList; using jlcxx::Parametric; diff --git a/scripts/install_cxxwrap.sh b/scripts/install_cxxwrap.sh index 69eefbfe0..279620a83 100755 --- a/scripts/install_cxxwrap.sh +++ b/scripts/install_cxxwrap.sh @@ -44,7 +44,7 @@ fi echo "Using $JULIA at: $JULIA_PATH" GIT_REPO="https://github.com/JuliaInterop/libcxxwrap-julia.git" -COMMIT_HASH="89e4699837bfa0929610c9e330889fb2df925b47" #(v14.2) +COMMIT_HASH="ee8a49b403ced7669c8fa56cec860f567f6510aa" #(v14.11) JULIA_CXXWRAP_SRC=$CUNUMERIC_ROOT_DIR/lib/libcxxwrap-julia if [ ! -d "$JULIA_CXXWRAP_SRC" ]; then diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 5dc9a2255..8eafe040e 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -56,14 +56,10 @@ import StatsBase: var, mean, std include(joinpath(@__DIR__, "../deps/version.jl")) include("utilities/preference.jl") -const HAS_CUDA = LegatePreferences.has_cuda_gpu() -if !HAS_CUDA - @warn "We couldn't find a CUDA-enabled GPU. If you have an NVIDIA GPU something might be wrong." -end - -# `HAS_CUDA` describes the machine. A CPU-only Legate configuration on a GPU -# machine must still avoid registering or launching GPU tasks. -@inline _has_gpu_target() = HAS_CUDA && Int(Legate.num_gpus()) > 0 +# Populated after Legate starts and resolves its automatic or explicit machine +# configuration. This reflects configured GPU targets, not merely visible hardware. +const HAS_CUDA = Ref(false) +@inline _has_gpu_target() = HAS_CUDA[] const DEFAULT_FLOAT = Float32 const DEFAULT_INT = Int32 @@ -252,8 +248,10 @@ function _start_runtime() # AA = ArgcArgv([Base.julia_cmd()[1]]) cuNumeric.initialize_cunumeric(AA.argc, getargv(AA)) + num_gpus = Int(Legate.num_gpus()) + HAS_CUDA[] = num_gpus > 0 _LINALG_RUNTIME[] = _LinalgRuntime( - cusolvermp_available(), Int(Legate.num_gpus()), Int(Legate.num_procs()) + cusolvermp_available(), num_gpus, Int(Legate.num_procs()) ) _init_deferred_free!() # record launch thread for deferred frees (memory.jl) diff --git a/src/cuda/cuda_ptx_task.jl b/src/cuda/cuda_ptx_task.jl index 44213a47a..97502fda1 100644 --- a/src/cuda/cuda_ptx_task.jl +++ b/src/cuda/cuda_ptx_task.jl @@ -168,7 +168,7 @@ function ptx_task(ptx::String, kernel_name) taskid = cuNumeric.LOAD_PTX # One point task per GPU so every GPU compiles the module. - ngpus = max(Int(Legate.num_gpus()), 1) + ngpus = max(_LINALG_RUNTIME[].gpus, 1) domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector((ngpus,)))) task = Legate.create_manual_task(rt, lib, taskid, domain) Legate.add_scalar(task, Legate.string_to_scalar(ptx)) diff --git a/src/memory.jl b/src/memory.jl index 02deb81f6..50705618e 100644 --- a/src/memory.jl +++ b/src/memory.jl @@ -102,7 +102,7 @@ hard_limit(; host=true) = _limit(hard_frac, host) function register_alloc!(nbytes::Integer) # assume device allocation if we have a GPU # the recalibration phase will fix any discrepancies - if HAS_CUDA + if _has_gpu_target() atomic_add!(current_device_bytes, nbytes) else atomic_add!(current_host_bytes, nbytes) @@ -116,7 +116,7 @@ function register_alloc!(nbytes::Integer) end function register_free!(nbytes::Integer) - if HAS_CUDA + if _has_gpu_target() atomic_sub!(current_device_bytes, nbytes) else atomic_sub!(current_host_bytes, nbytes) @@ -129,7 +129,7 @@ function recalibrate_allocator!() @assert recal_host_mem >= 0 atomic_xchg!(current_host_bytes, recal_host_mem) - if HAS_CUDA + if _has_gpu_target() recal_device_mem = ccall((:nda_query_allocated_device_memory, libnda), Int64, ()) @assert recal_device_mem >= 0 atomic_xchg!(current_device_bytes, recal_device_mem) diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 784e14c2f..934b20488 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -259,7 +259,7 @@ end # Require an active GPU target so `--gpus 0` stays on the unfused path. # Fusion requires same-shaped NDArray leaves; otherwise fall back. # Single-op exprs (length < `FUSE_BROADCAST_MIN_OPS`) stay unfused by default. - @static if FUSE_BROADCAST_EXPRS && HAS_CUDA + @static if FUSE_BROADCAST_EXPRS if _has_gpu_target() && _should_attempt_broadcast_fusion(dest, bc) return fuse_broadcast_tree!(dest, bc) else diff --git a/src/ndarray/detail/fft.jl b/src/ndarray/detail/fft.jl index b7c943915..6e38f55bf 100644 --- a/src/ndarray/detail/fft.jl +++ b/src/ndarray/detail/fft.jl @@ -119,7 +119,7 @@ function _bluestein_mask( end function _assert_fft_gpu() - Legate.num_gpus() > 0 && return nothing + _has_gpu_target() && return nothing return throw( ErrorException( "FFT requires a CUDA GPU; cupynumeric's FFT task has no CPU variant" diff --git a/src/ndarray/detail/linalg.jl b/src/ndarray/detail/linalg.jl index 2dedd8303..b74e09d80 100644 --- a/src/ndarray/detail/linalg.jl +++ b/src/ndarray/detail/linalg.jl @@ -373,7 +373,7 @@ function _alloc_eigenvalues(a::NDArray{T,N}) where {T,N} end function _assert_geev_available() - (Legate.num_gpus() > 0 && !cuNumeric.cusolver_has_geev()) && error( + (_has_gpu_target() && !cuNumeric.cusolver_has_geev()) && error( "eigen requires cusolverDnXgeev, which the installed cuSolver does not " * "provide. Upgrade CUDA (12.6.2 or newer) or run without GPUs.", ) diff --git a/src/scoping/accelerate.jl b/src/scoping/accelerate.jl index 3b0268c69..c7cb37aa1 100644 --- a/src/scoping/accelerate.jl +++ b/src/scoping/accelerate.jl @@ -58,7 +58,7 @@ function _accelerate_block_soft(block, caller::Module) fallback = process_ndarray_scope( nb; on_rewrite, protected_roots=_assigned_symbols(nb) ) - @static if FUSE_BROADCAST_EXPRS && HAS_CUDA + @static if FUSE_BROADCAST_EXPRS fused = _try_fuse_block_multi(nb) if !isnothing(fused) return quote diff --git a/src/utilities/version.jl b/src/utilities/version.jl index 6271e3949..8c3f42667 100644 --- a/src/utilities/version.jl +++ b/src/utilities/version.jl @@ -50,7 +50,11 @@ function versioninfo(io::IO=stdout) dirs2 = Legate.find_dependency_paths(typeof(legate_mode)) other_dirs = merge(dirs1, dirs2) - hardware_str = HAS_CUDA ? "CPU + GPU" : "CPU Only" + active = runtime_started() + hardware_str = + active ? + (_has_gpu_target() ? "CPU + GPU" : "CPU Only") : + "not queried (runtime inactive)" legate_auto_config = get(ENV, "LEGATE_AUTO_CONFIG", "1") is_auto_config = legate_auto_config != "0" ? true : false @@ -58,7 +62,6 @@ function versioninfo(io::IO=stdout) # versioninfo is also called by the test driver with LEGATE_SKIP_RUNTIME. # Do not start the runtime or query its machine just to print diagnostics. - active = runtime_started() not_queried = "not queried (runtime inactive)" mp_available = active ? _LINALG_RUNTIME[].available : not_queried active_gpus = active ? _LINALG_RUNTIME[].gpus : not_queried diff --git a/test/Project.toml b/test/Project.toml index 6b09222cd..82d9c81cb 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -12,5 +12,8 @@ TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" +[compat] +CUDA = "6.4" + [sources] cuNumeric = {path = ".."} diff --git a/test/analysis/accelerate.jl b/test/analysis/accelerate.jl index 8a1ccc698..41f8d087f 100644 --- a/test/analysis/accelerate.jl +++ b/test/analysis/accelerate.jl @@ -138,7 +138,7 @@ using InteractiveUtils: code_typed @test occursin("x .+= alpha .* p", compound) @test !occursin(r"tmp\d+ = alpha \.\* p", compound) - if cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric.HAS_CUDA + if cuNumeric.FUSE_BROADCAST_EXPRS # A same-shape chain fuses into one multi-output launch and still # frees the hoisted slice temporaries. mo = string(expand(:( diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index e8a1d3c83..a9a8c4b49 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -276,7 +276,7 @@ end end @testset verbose = true "fft" begin - if cuNumeric.HAS_CUDA + if cuNumeric._has_gpu_target() a = cuNumeric.zeros(ComplexF32, 8) b = cuNumeric.zeros(ComplexF32, 4, 6) @test @inferred(fft(a)) !== nothing diff --git a/test/array/distributed_linalg.jl b/test/array/distributed_linalg.jl index 3ce8a9ba5..486141d40 100644 --- a/test/array/distributed_linalg.jl +++ b/test/array/distributed_linalg.jl @@ -46,6 +46,8 @@ end rt = cuNumeric._LINALG_RUNTIME[] @test rt.available == cuNumeric.cusolvermp_available() @test rt.gpus == Int(cuNumeric.Legate.num_gpus()) + @test cuNumeric.HAS_CUDA[] == (rt.gpus > 0) + @test cuNumeric._has_gpu_target() == (rt.gpus > 0) @test rt.procs == Int(cuNumeric.Legate.num_procs()) @test rt.mp_eligible == (rt.available && rt.gpus > 1) @test cuNumeric.choose_nd_color_shape((33, 33)) == (1, 1) diff --git a/test/array/linalg.jl b/test/array/linalg.jl index 55640b67f..161772402 100644 --- a/test/array/linalg.jl +++ b/test/array/linalg.jl @@ -520,10 +520,6 @@ end allowscalar() do @test isapprox(A_ref, Array(parent(L)) * Array(parent(L))'; atol=atol(T), rtol=rtol(T)) end - - # `F.U` (and hence destructuring as `L, U = F`) needs `copy(F.factors')`, - # which falls back to scalar indexing until `adjoint(::NDArray)` lands. - @test_throws "scalar-indexed" F.U end @testset "cholesky rejects bad shapes and types" begin diff --git a/test/array/tensoroperations.jl b/test/array/tensoroperations.jl index 4532ca39e..f037ae9fd 100644 --- a/test/array/tensoroperations.jl +++ b/test/array/tensoroperations.jl @@ -244,7 +244,7 @@ using TensorOperations: TensorOperations as TO GC.gc() cuNumeric.drain_pending_frees!() current_bytes = - cuNumeric.HAS_CUDA ? + cuNumeric._has_gpu_target() ? cuNumeric.current_device_bytes : cuNumeric.current_host_bytes baseline = current_bytes[] diff --git a/test/array/unary/tests.jl b/test/array/unary/tests.jl index 13f35af02..2c9e3aed7 100644 --- a/test/array/unary/tests.jl +++ b/test/array/unary/tests.jl @@ -403,7 +403,7 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) end ## TODO Int8 min/max along an axis is broken on GPU - if cuNumeric.HAS_CUDA && T == Int8 && (func == Base.minimum || func == Base.maximum) + if cuNumeric._has_gpu_target() && T == Int8 && (func == Base.minimum || func == Base.maximum) continue end diff --git a/test/gpu_only/broadcast_fusion.jl b/test/gpu_only/broadcast_fusion.jl index 8ab3afb3d..bacb59f85 100644 --- a/test/gpu_only/broadcast_fusion.jl +++ b/test/gpu_only/broadcast_fusion.jl @@ -579,7 +579,7 @@ end #= Broadcast fusion PTX compilation cache. * Verifies `_BCAST_PTX_CACHE` grows on first fused launch of a signature and * is reused (no new entry) on a second launch of the same signature. - * Gated on `FUSE_BROADCAST_EXPRS` + `HAS_CUDA`; skips otherwise. + * Gated on `FUSE_BROADCAST_EXPRS` and an active GPU target; skips otherwise. * With `FUSE_BROADCAST_MIN_OPS > 1`, single-op exprs are unfused — tests * should set min ops to 1 (LocalPreferences / ENV) to exercise the cache. =# @@ -587,8 +587,8 @@ end T=Float32 N=64 - if !(cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric.HAS_CUDA) - @info "Skipping PTX cache tests (need FUSE_BROADCAST_EXPRS && HAS_CUDA)" + if !(cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric._has_gpu_target()) + @info "Skipping PTX cache tests (need fusion and an active GPU target)" return nothing end if cuNumeric.FUSE_BROADCAST_MIN_OPS > 1 From e40358a6396784eed0553d1e07b90c79d8162fd6 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sat, 19 Sep 2026 20:27:19 -0500 Subject: [PATCH 32/49] Remove Benchmarking Module (#195) * benchmark harness changes [skip ci] * format: run all files * instantiate project script with model smoketest * smoke test for montecarlo now checks correctness for all models if correctness is enabled * reduce verbose output and provide progress meter for all models * update benchmarks_smoke.toml [skip ci] * montecarlo: default accelerate macro * montecarlo: Dagger.jl * montecarlo: map reduce cuda.jl * benchmarks: add gemm support for Dagger and JACC * benchmark: local options need to override global * benchmarks/cleanup: remove dead code * benchmarks: JACC and dagger grayscott * benchmarks: provide seperate .toml to run different forms of accelerate on grayscott * grayscott: add correctness checks * grayscott: default is accelerate * grayscott: simplfy materialization * fix rebase conflicts [container] [benchmark-container] * CG + kwarg support (#198) * CG + kwarg support * add accelerated variant * benchmarks: cg port to Dagger and Python. Change default behaviour of CG on cuNumeric * benchmarks: update Dagger CG to reduce runtime overheads --------- Co-authored-by: krasow * fake change to trigger [container] * benchmarks: JACC needs to set backend for CUDA [skip ci] * benchmarks: fix auto-size conservativeness * Benchmark Container CI (#199) [container] [benchmark container] * benchmarks: JACC CG updates. reduce copies * benchmarks: Dagger grayscott init fixes * benchmarks: Dagger CG update to support multi partitions for multi-gpu scaling. [benchmark container] * Trigger benchmark image [benchmark-container] * benchmarks: NAS FT [benchmark-container] * benchmarks: NAS/EP ported * benchmarks: montecarlo uses map_reduce; original impl is left as a comparison to see effect of map_reduce * benchmarks: NAS/FT patches * benchmarks: NAS/MG init port * benchmarks: codex audit on fairness * move benchmarking harness to seperate repo --------- Co-authored-by: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> --- .buildkite/run_developer_ci.sh | 2 +- .dockerignore | 15 + .github/workflows/container.yml | 78 ++--- .gitmodules | 3 + benchmark | 1 + benchmark/MEMORY.md | 80 ----- benchmark/Project.toml | 15 - benchmark/README.md | 196 ----------- benchmark/benchmarks.toml | 174 ---------- benchmark/debug/grayscott_accelerate.jl | 66 ---- benchmark/diagnose_montecarlo.jl | 34 -- benchmark/install_cupynumeric.sh | 81 ----- benchmark/mwe_fusion_indexing.jl | 78 ----- benchmark/plot_results.jl | 302 ----------------- benchmark/run.jl | 32 -- benchmark/run_benchmark.sh | 85 ----- benchmark/src/autosize.jl | 108 ------ benchmark/src/benchmarks/dmd.jl | 149 --------- benchmark/src/benchmarks/gemm.jl | 51 --- benchmark/src/benchmarks/grayscott.jl | 183 ----------- .../benchmarks/grayscott_accelerate_forms.jl | 103 ------ benchmark/src/benchmarks/montecarlo.jl | 56 ---- benchmark/src/benchmarks/poisson_fft.jl | 104 ------ .../src/benchmarks/tensor_contractions.jl | 123 ------- benchmark/src/core.jl | 310 ------------------ benchmark/src/memory.jl | 147 --------- benchmark/src/parse_benchmarks.jl | 213 ------------ benchmark/src/planning.jl | 140 -------- benchmark/src/result_rows.jl | 47 --- benchmark/src/runner.jl | 166 ---------- benchmark/src/single.jl | 106 ------ benchmark/src_py/benchmarks/__init__.py | 8 - benchmark/src_py/benchmarks/dmd.py | 36 -- benchmark/src_py/benchmarks/gemm.py | 26 -- benchmark/src_py/benchmarks/grayscott.py | 68 ---- benchmark/src_py/benchmarks/montecarlo.py | 25 -- benchmark/src_py/benchmarks/poisson_fft.py | 49 --- .../src_py/benchmarks/tensor_contractions.py | 62 ---- benchmark/src_py/core.py | 80 ----- benchmark/src_py/single.py | 59 ---- benchmark/test/autosize.jl | 26 -- benchmark/test/runtests.jl | 198 ----------- benchmark/test/test_timing.py | 63 ---- benchmark/test/timing.jl | 34 -- Dockerfile => docker/Dockerfile | 23 +- .../Dockerfile.developer | 8 +- docker/README.md | 26 ++ docker/entrypoint.sh | 13 + docker/profile.sh | 8 + docs/make.jl | 1 + docs/src/benchmarks/howto.md | 6 +- docs/src/examples/cg.md | 50 +++ lib/cunumeric_jl_wrapper/include/accessors.h | 3 +- .../include/cuda_macros.h | 52 +-- lib/cunumeric_jl_wrapper/src/cuda.cpp | 9 +- lib/cunumeric_jl_wrapper/src/memory.cpp | 6 +- lib/cunumeric_jl_wrapper/src/ndarray.cpp | 1 - lib/cunumeric_jl_wrapper/src/types.cpp | 20 +- lib/cunumeric_jl_wrapper/src/wrapper.cpp | 42 ++- src/ndarray/broadcast_fusion.jl | 4 +- src/ndarray/detail/distributed_linalg.jl | 9 +- src/scoping/broadcast_lifetimes.jl | 2 +- test/analysis/type_stability.jl | 5 +- test/array/distributed_linalg.jl | 9 +- test/linalg_errors.jl | 5 +- test/workflows/grayscott.jl | 4 +- 66 files changed, 279 insertions(+), 4009 deletions(-) create mode 100644 .dockerignore create mode 100644 .gitmodules create mode 160000 benchmark delete mode 100644 benchmark/MEMORY.md delete mode 100644 benchmark/Project.toml delete mode 100644 benchmark/README.md delete mode 100644 benchmark/benchmarks.toml delete mode 100644 benchmark/debug/grayscott_accelerate.jl delete mode 100644 benchmark/diagnose_montecarlo.jl delete mode 100755 benchmark/install_cupynumeric.sh delete mode 100644 benchmark/mwe_fusion_indexing.jl delete mode 100644 benchmark/plot_results.jl delete mode 100644 benchmark/run.jl delete mode 100755 benchmark/run_benchmark.sh delete mode 100644 benchmark/src/autosize.jl delete mode 100644 benchmark/src/benchmarks/dmd.jl delete mode 100644 benchmark/src/benchmarks/gemm.jl delete mode 100644 benchmark/src/benchmarks/grayscott.jl delete mode 100644 benchmark/src/benchmarks/grayscott_accelerate_forms.jl delete mode 100644 benchmark/src/benchmarks/montecarlo.jl delete mode 100644 benchmark/src/benchmarks/poisson_fft.jl delete mode 100644 benchmark/src/benchmarks/tensor_contractions.jl delete mode 100644 benchmark/src/core.jl delete mode 100644 benchmark/src/memory.jl delete mode 100644 benchmark/src/parse_benchmarks.jl delete mode 100644 benchmark/src/planning.jl delete mode 100644 benchmark/src/result_rows.jl delete mode 100644 benchmark/src/runner.jl delete mode 100644 benchmark/src/single.jl delete mode 100644 benchmark/src_py/benchmarks/__init__.py delete mode 100644 benchmark/src_py/benchmarks/dmd.py delete mode 100644 benchmark/src_py/benchmarks/gemm.py delete mode 100644 benchmark/src_py/benchmarks/grayscott.py delete mode 100644 benchmark/src_py/benchmarks/montecarlo.py delete mode 100644 benchmark/src_py/benchmarks/poisson_fft.py delete mode 100644 benchmark/src_py/benchmarks/tensor_contractions.py delete mode 100644 benchmark/src_py/core.py delete mode 100644 benchmark/src_py/single.py delete mode 100644 benchmark/test/autosize.jl delete mode 100644 benchmark/test/runtests.jl delete mode 100644 benchmark/test/test_timing.py delete mode 100644 benchmark/test/timing.jl rename Dockerfile => docker/Dockerfile (83%) rename Dockerfile.developer => docker/Dockerfile.developer (95%) create mode 100644 docker/README.md create mode 100644 docker/entrypoint.sh create mode 100644 docker/profile.sh create mode 100644 docs/src/examples/cg.md diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index d3ac3929e..574715e78 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -20,7 +20,7 @@ sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" export PATH="$CMAKE_ROOT/bin:$PATH" cmake --version -# Clean slate so cached state doesn't leak across Julia versions. +# Clean slate so cached state doesn't leak across Julia versions. rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 000000000..d658d488f --- /dev/null +++ b/.dockerignore @@ -0,0 +1,15 @@ +.git +**/.git + +**/Manifest.toml +**/LocalPreferences.toml +**/__pycache__ +**/*.pyc +**/*.log +**/*.prof + +benchmark/results +benchmark/plots +docs/node_modules +lib/cunumeric_jl_wrapper/build +deps/cupynumeric-* diff --git a/.github/workflows/container.yml b/.github/workflows/container.yml index 1d87c6d2e..787a17050 100644 --- a/.github/workflows/container.yml +++ b/.github/workflows/container.yml @@ -17,11 +17,6 @@ on: # GitHub Actions cannot filter pushes by commit message at trigger time, so # the job-level condition below implements the [container] opt-in. push: - workflow_run: - workflows: ['CI'] - types: [completed] - branches: - - main jobs: push_to_registry: if: >- @@ -30,12 +25,7 @@ jobs: ( github.event_name == 'workflow_dispatch' || github.event_name == 'release' || - (github.event_name == 'push' && contains(github.event.head_commit.message, '[container]')) || - ( - github.event_name == 'workflow_run' && - github.event.workflow_run.conclusion == 'success' && - !contains(github.event.workflow_run.head_commit.message, '[container]') - ) + (github.event_name == 'push' && contains(github.event.head_commit.message, '[container]')) ) }} name: Container for ${{ matrix.platform }} - Julia ${{ matrix.julia }} - CUDA ${{ matrix.cuda }} @@ -60,7 +50,7 @@ jobs: - name: Check out the repo uses: actions/checkout@v4 with: - ref: ${{ inputs.tag || github.event.release.tag_name || github.event.workflow_run.head_sha || github.sha }} + ref: ${{ inputs.tag || github.event.release.tag_name || github.sha }} fetch-depth: 0 - name: Select wrapper build @@ -89,11 +79,11 @@ jobs: fi if [[ "$changed" == "true" ]]; then - echo "Wrapper changes detected; building Dockerfile.developer" - echo "dockerfile=Dockerfile.developer" >> "$GITHUB_OUTPUT" + echo "Wrapper changes detected; building docker/Dockerfile.developer" + echo "dockerfile=docker/Dockerfile.developer" >> "$GITHUB_OUTPUT" else - echo "No wrapper changes detected; building Dockerfile" - echo "dockerfile=Dockerfile" >> "$GITHUB_OUTPUT" + echo "No wrapper changes detected; building docker/Dockerfile" + echo "dockerfile=docker/Dockerfile" >> "$GITHUB_OUTPUT" fi - name: Get package spec @@ -105,9 +95,6 @@ jobs: elif [[ "${{ github.event_name }}" == "release" ]]; then echo "ref=${{ github.event.release.tag_name }}" >> $GITHUB_OUTPUT echo "name=${{ github.event.release.tag_name }}" >> $GITHUB_OUTPUT - elif [[ "${{ github.event_name }}" == "workflow_run" ]]; then - echo "ref=${{ github.event.workflow_run.head_sha }}" >> $GITHUB_OUTPUT - echo "name=${{ github.event.workflow_run.head_branch }}" >> $GITHUB_OUTPUT elif [[ "${{ github.ref_type }}" == "tag" ]]; then echo "ref=${{ github.ref_name }}" >> $GITHUB_OUTPUT echo "name=${{ github.ref_name }}" >> $GITHUB_OUTPUT @@ -125,9 +112,7 @@ jobs: VERSION=$(grep "^version = " Project.toml | cut -d'"' -f2) echo "version=$VERSION" >> $GITHUB_OUTPUT - # Include the checked-out source revision in every image tag. In - # particular, github.sha identifies this workflow run for - # workflow_run events, not necessarily the commit being packaged. + # Include the checked-out source revision in every image tag. COMMIT_SHA=$(git rev-parse --short=12 HEAD) echo "commit=$COMMIT_SHA" >> $GITHUB_OUTPUT @@ -165,21 +150,29 @@ jobs: images: ghcr.io/${{ github.repository }} tags: | type=raw,value=${{ steps.pkg.outputs.commit }}-julia${{ matrix.julia }}-cuda${{ steps.cuda.outputs.major }}.${{ steps.cuda.outputs.minor }} - type=raw,value=${{ steps.pkg.outputs.name }},enable=${{ matrix.default == true && (github.ref_type == 'tag' || inputs.tag != '') }} - type=raw,value=latest,enable=${{ matrix.default == true && (github.ref_type == 'tag' || (inputs.tag != '' && inputs.mark_as_latest)) }} - type=raw,value=dev,enable=${{ matrix.default == true && github.ref_type == 'branch' && inputs.tag == '' }} + type=raw,value=${{ steps.pkg.outputs.name }},enable=${{ github.ref_type == 'tag' || inputs.tag != '' }} + type=raw,value=latest,enable=${{ github.ref_type == 'tag' || (inputs.tag != '' && inputs.mark_as_latest) }} + type=raw,value=dev,enable=${{ github.ref_type == 'branch' && inputs.tag == '' }} labels: | org.opencontainers.image.version=${{ steps.pkg.outputs.version }} - - name: Save tag to file + - name: Select immutable image tags + id: image-tags + env: + BASE_TAGS: ${{ steps.meta.outputs.tags }} + run: | + echo "base=$(printf '%s\n' "$BASE_TAGS" | head -n1)" >> "$GITHUB_OUTPUT" + + - name: Save tags to files run: | - echo "${{ steps.meta.outputs.tags }}" | cut -d',' -f1 > image_tag.txt + echo "${{ steps.image-tags.outputs.base }}" > image_tag.txt - name: Upload tag artifact uses: actions/upload-artifact@v4 with: name: docker-tag - path: image_tag.txt + path: | + image_tag.txt - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 @@ -200,16 +193,17 @@ jobs: file: ${{ steps.wrapper-build.outputs.dockerfile }} load: true push: false - provenance: false # the build fetches the repo again, so provenance tracking is not useful + provenance: false # the image is loaded and pushed explicitly below platforms: ${{ matrix.platform }} tags: ${{ steps.meta.outputs.tags }} labels: ${{ steps.meta.outputs.labels }} + cache-from: type=local,src=/tmp/.buildx-cache/base + cache-to: type=local,dest=/tmp/.buildx-cache/base-new,mode=max build-args: | JULIA_VERSION=${{ matrix.julia }} CUDA_MAJOR=${{ steps.cuda.outputs.major }} CUDA_MINOR=${{ steps.cuda.outputs.minor }} JULIA_CPU_TARGET=${{ steps.cpu_target.outputs.target }} - REF=${{ steps.pkg.outputs.ref }} # - name: Run tests in built image # run: | @@ -223,16 +217,26 @@ jobs: - name: Push image if: success() + env: + IMAGE_TAGS: ${{ steps.meta.outputs.tags }} run: | - docker push ${{ steps.meta.outputs.tags }} - docker image rm ${{ steps.meta.outputs.tags }} + while IFS= read -r image; do + docker push "$image" + done <<< "$IMAGE_TAGS" + + - name: Update base layer cache + run: | + rm -rf /tmp/.buildx-cache/base + mv /tmp/.buildx-cache/base-new /tmp/.buildx-cache/base # can happen if there is a failure to push - name: Ensure image is removed (safety) if: always() + env: + BASE_TAGS: ${{ steps.meta.outputs.tags }} run: | - if docker image inspect ${{ steps.meta.outputs.tags }} > /dev/null 2>&1; then - docker stop ${{ steps.pkg.outputs.ref }} || true - docker rm ${{ steps.pkg.outputs.ref }} || true - docker image rm ${{ steps.meta.outputs.tags }} || true - fi + while IFS= read -r image; do + if [[ -n "$image" ]] && docker image inspect "$image" > /dev/null 2>&1; then + docker image rm "$image" || true + fi + done <<< "$BASE_TAGS" diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 000000000..267133871 --- /dev/null +++ b/.gitmodules @@ -0,0 +1,3 @@ +[submodule "benchmark"] + path = benchmark + url = https://github.com/JuliaLegate/benchmarking.git diff --git a/benchmark b/benchmark new file mode 160000 index 000000000..fdc98d39d --- /dev/null +++ b/benchmark @@ -0,0 +1 @@ +Subproject commit fdc98d39d8b0a29326bdc3b8041c6fa9ed80b0b3 diff --git a/benchmark/MEMORY.md b/benchmark/MEMORY.md deleted file mode 100644 index 0f7d2d5f1..000000000 --- a/benchmark/MEMORY.md +++ /dev/null @@ -1,80 +0,0 @@ -# Memory accounting and supported runs - -`src/memory.jl` is the authoritative preflight model. The older `total_space` -helpers describe storage and are not used by the sweep planner. Byte arithmetic -uses `BigInt`; GPU count, backend, dtype, fusion, benchmark type, dimensions, -and warmup/iteration count all enter the planning calculation. - -## Lifetime policy - -Estimates are conservative upper bounds, not measured allocator peaks. They -include initialization and the entire trial. Julia tracing GC is not assumed -to run between iterations. In the baseline Monte Carlo kernel an unreferenced -broadcast output can therefore remain for every iteration. Gray-Scott baseline, -begin and expression forms can similarly retain named buffers until GC. The -function/let acceleration forms explicitly destroy last-used local arrays and -are bounded independently of iteration count. Fusion-disabled execution still -uses the accelerated lifetime rewrite. - -Fused Gray-Scott is bounded by four persistent grids plus four named interior -results and an assignment output. Unfused execution reserves three additional -interior buffers for nested operands and the outer broadcast destination. -The function/let bound conservatively retains named intermediates within an -iteration even when inter-statement fusion may eliminate them. It does not -claim an exact optimized peak. Full parent grids are counted rather than -assuming slice/halo partition placement. - -Python reference counting avoids the Julia retention allowance, but an input -and output of an unfused operation must coexist. The random helper generates -Float64 values before casting, and this conversion is counted. These models -assume the current kernels and standard supported runtime execution; physical -Legate instance lifetimes still require validation on the target machine. - -## Native workspace - -GEMM, DMD, FFT and tensor contraction scratch/packing bounds depend on native -libraries and their algorithms. The previous arbitrary extra-array allowances -are not treated as verified bounds. Supply a verified **per-GPU byte bound**: - -```toml -[workspace.gemm] -cunumeric = 268435456 -cudajl = 268435456 -cupynumeric = 268435456 -``` - -The numbers above demonstrate syntax only, not recommended bounds. Use a bound -verified for the maximum dimensions in the sweep and installed library versions. -Other keys are the registered names, e.g. `workspace.dmd_baseline` and -`workspace.tensor_contract4`. Each enabled native backend needs an entry; -missing entries fail preflight, including for pinned dimensions. Zero is valid -only when no extra workspace/packing is required. No calibration, OOM retry, -or automatic problem reduction is performed during execution. - -DMD counts the entire SVD on one GPU regardless of P. Its shared baseline is -limited by the largest requested GPU count's scaled problem. This is a memory -guard, not a distributed SVD implementation. Full factors/parent stores and -complex outputs are included in addition to the supplied workspace bound. - -GEMM and contractions conservatively count full operands on each GPU until -mapper replication bounds are verified. Poisson partitions batches and retains -the inverse Laplacian per GPU; Python FFT outputs are conservatively treated -as complex128. Native workspace remains separate. - -## Budgets and validation - -`mem_frac` applies to the smallest visible GPU capacity, capped by -`CUNUMERIC_BENCH_FBMEM_MB` when provided. That cap is also passed to Legate. -`CUDA_VISIBLE_DEVICES` numeric indices and GPU UUIDs are resolved explicitly; -unresolvable/MIG identities are rejected rather than guessed. Free memory must -cover the budget before starting. External GPU users can invalidate that check. - -Use `--dry-run` to see initialization, iteration, workspace and peak bytes for -each configuration. CPU tests verify dispatch and planner invariants; they do -not verify native allocator bounds. Before calling a configuration GPU-validated, -check its runtime allocation trace and execute the requested GPU sweep with the -installed native libraries. Record the verified workspace bounds in its config. - -```bash -julia --project=. test/runtests.jl -``` diff --git a/benchmark/Project.toml b/benchmark/Project.toml deleted file mode 100644 index 5b279307b..000000000 --- a/benchmark/Project.toml +++ /dev/null @@ -1,15 +0,0 @@ -[deps] -AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" -CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" -CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" -Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" -Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" -ProgressMeter = "92933f4c-e287-5a05-a399-4b506db050ca" -Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" -TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76" -TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" -cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" -cuTENSOR = "011b41b2-24ef-40a8-b3eb-fa098493e9e1" - -[extras] -LegatePreferences = "8028f36a-2b64-49e9-aa04-2d0933fd2ed9" diff --git a/benchmark/README.md b/benchmark/README.md deleted file mode 100644 index 95982b9b6..000000000 --- a/benchmark/README.md +++ /dev/null @@ -1,196 +0,0 @@ -# Benchmark configuration - -Benchmarks are declared in `benchmarks.toml`. `run.jl` parses it. - -Each independent iteration completes before the next is submitted, including -warmups, for cuNumeric, CUDA.jl, and cuPyNumeric. Blocking synchronization is -included in timed iterations. Gray–Scott is the exception: its timesteps form -one trajectory, so all variants retain batch synchronization at timing boundaries. -Initialization remains outside timing. Earlier non-Gray–Scott results used batch -synchronization and should be rerun for comparison. Fences do not force GC. - -## Running - -Run a complete selected sweep and plot it without editing other TOML blocks: - -```bash -julia --project=. run.jl --only=montecarlo -julia --project=. run.jl --only=grayscott -julia --project=. run.jl --only=grayscott --fusion=both -julia --project=. run.jl --only=grayscott --dry-run -``` - -`--only` accepts benchmark or plot-group names (comma-separated). `--fusion=on`, -`off`, or `both` overrides the selected blocks. `--config=path` selects another -configuration. The shipped configuration uses fusion enabled. Positional -single-run arguments remain supported and do not plot automatically. - -Automatic sizing shares one baseline per comparison group and dtype, accounting -for every selected backend, fusion setting and GPU count. Incompatible pinned -constraints fail preflight. Comparison backends run once even if only unfused -Julia configurations are selected. Read [memory accounting](MEMORY.md) before -running native-library benchmarks: verified workspace bounds are required where -the harness cannot infer them. Unexpected failures are reported, not retried. - -Each invocation writes `results///` plus a `manifest.toml` with -resolved dimensions, memory estimates, package versions and worker statuses. -Plots go to `plots///`; failed sweeps are marked incomplete. -Different sizes at the same GPU count cannot be silently merged into a plot. - -```bash -julia --project=. run.jl # runs whatever benchmarks.toml configures -``` - -`run.jl` runs each (benchmark, backend) pair in its own process via -`run_benchmark.sh`, so backends never share a GPU/runtime within a measurement. -Julia workers log correctness checking and show a trial progress meter with the -latest trial's mean time and GFLOP/s. The meter advances only after a trial -finishes, outside the timed loop; initialization and warmup can also take time -before the next update. -cuNumeric always runs; extra comparison backends are toggled in `[Global]`: - -- `cuda = true` → also run under CUDA.jl (single-GPU configs only; CUDA.jl is - single-device). Every kernel has a `CuArray` implementation. -- `cupynumeric = true` → also run under cupynumeric (see below). - -On a single GPU, the cuNumeric worker also compares a tiny problem against -CUDA.jl (`pass` / `fail` in the CSV). The timed CUDA.jl run still happens; -its correctness column is `skipped`. Multi-GPU and cupynumeric skip the check. - -Individual `[[benchmark]]` blocks may override `cuda`, `n_warmup`, `n_iter`, -and `n_trial`. Unspecified values inherit from `[Global]`. This is useful for -enabling a single-GPU CUDA comparison only for compatible benchmarks or reducing -the iteration count for expensive kernels. - -### Comparing against cupynumeric - -cupynumeric runs in a conda env whose major.minor matches this project's -resolved `cupynumeric_jll`. Build it once: - -The runner checks that conda and the requested environment are available before -starting any timed workers. If conda is not on the worker's PATH, set -`CUNUMERIC_BENCH_CONDA` to its executable path (or use `CONDA_EXE`). The installer -honors the same setting. - -```bash -./install_cupynumeric.sh # creates env cupynumeric-bench- -``` - -`run.jl` derives the env name automatically; override it with `CUPYNUMERIC_ENV`. - -## Layout - -```toml -[Global] -n_warmup = 5 -n_iter = 1000 -n_trial = 5 -auto_size = true -mem_frac = 0.5 # fraction of the smallest visible GPU's total RAM - -[[gemm]] # name registered under src/benchmarks/ -T = "Float32" # element type -gpus = 1 -cpus = 2 -N = 150 -M = 150 # optional, defaults to 1 -fusion = true # optional, defaults to true; toggles cuNumeric broadcast fusion -``` - -Repeat a `[[name]]` block to add independent configs. - -## Lists - -Any of `T`, `fusion`, `gpus`, `cpus`, `N`, `M` may be a list. They expand along -two axes: -- **`T` and `fusion` multiply.** The whole sweep runs once per type and once per - fusion setting (`fusion = [true, false]` sweeps both). -- **`gpus`, `cpus`, `N`, `M` zip** into a single lockstep sweep — element `i` - of each is paired together. - -When `[Global] auto_size = true` and a block **omits** `N` (or sets `N = "auto"`), -the harness RAM-fits the 1-GPU problem from `total_space` on that benchmark type, -then maps to `P` GPUs. Pin an explicit `N` list to keep paper sizes. `mem_frac` -is the fraction of the *smallest* visible GPU's total RAM (`CUNUMERIC_BENCH_MEM_FRAC` -overrides it). DMD keeps `M` as the intensity knob; Poisson holds grid `N` across -the GPU sweep and scales batch `M`. -Peak estimates are now backend- and variant-aware. Monte Carlo's fused -iteration needs the samples and broadcast output, but the planner also accounts -for outputs that may await Julia GC across a trial. See [memory accounting](MEMORY.md) -for the supported bounds and native-library workspace configuration. - -`fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, -default `true`); it only affects cuNumeric, so comparison backends run once, not -per variant. - -Benchmark names are defined by the registered benchmark implementations. A -benchmark may expose baseline, optimized, backend-specific, or other variants; -the harness treats each name uniformly and records each result independently. - -Each zipped field must be one of: - -- a scalar or single-element list (`cpus = 2` or `[2]`) -> broadcast to every config -- a list whose length equals the sweep length - -Any other length mismatch is an error. - -```toml -[[sgemm]] -T = ["Float64", "Float32"] # multiplies -gpus = [1, 2, 4] # -cpus = 2 # zip -> (1,2,150,150), (2,2,300,300), (4,2,600,600) -N = [150, 300, 600] # -M = [150, 300, 600] # -``` - --> 2 types * 3 sweep points = **6 runs**. - -### Gotcha - -When `T = ["Float32", "Float64"]` and a length-2 `N`/`M` sweep you get all **4** -combinations, not a paired `Float32 -> N[1], Float64 -> N[2]`. To pin a type -to a specific size, use separate `[[name]]` blocks. - -## Tensor contractions - -Two direct TensorOperations benchmarks compare the same mathematical kernel -across cuNumeric.jl, cuPyNumeric, and—when `cuda = true` on a one-GPU -entry—TensorOperations.jl's cuTENSOR backend: - -- `tensor_projection3` computes - `D[n,m,l] = A[i,j,k] * B[n,i] * B[m,j] * B[l,k]`. TensorOperations performs - three pairwise contractions with rank-3 intermediates. For equal index extent - `N`, the counted work is `3N^3(2N-1)`, asymptotically `6N^4`. -- `tensor_contract4` computes - `C[a,b,c,d] = X[a,i,c,j] * Y[i,b,j,d]`. This single contraction isolates the - primitive high-rank backend path and counts `N^4(2N^2-1)` operations. - -The Julia implementations use `@tensor opt=true` on both `NDArray` and `CuArray`; the -latter activates TensorOperations' cuTENSOR extension and is recorded as -`TensorOperations.jl / cuTENSOR`. The cuPyNumeric implementations use equivalent -`einsum` expressions with `einsum_path(optimize="optimal")`. Final outputs are -preallocated. Required intermediate allocation (projection3: two rank-3 temps; -contract4: an `N²×N²` GEMM workspace) is part of each timed iteration and of -Julia `total_space`. The orchestrator computes flop counts in Julia and passes -them to the Python worker so the formulas live in one place. - -## Plotting - -A full `run.jl` pass (no extra args) plots at the end. One-off CLI runs do -not. To plot existing CSVs: - -```bash -julia --project=. plot_results.jl -``` - -`[plot.groups]` in `benchmarks.toml` puts related kernels on one figure. -Gray-Scott's baseline and `@accelerate` forms share `grayscott`; DMD baseline -and accelerated share `dmd`. Every other `[[benchmark]]` table is its own -figure. Each figure overlays CUDA.jl (1 GPU) and cupynumeric from that -group's baseline CSV (`*_baseline`, else the only / first name). Accelerated -kernels have no CUDA.jl or Python CSVs; the overlay still comes from the -baseline. - -cuNumeric fused vs unfused uses the same color with solid vs dashed lines. -Outputs are `plots/_weak_scaling.png`. Optional flags: `--out=`, -`--suffix=`, `--config=`, or a results-directory path. diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml deleted file mode 100644 index 24f8f3590..000000000 --- a/benchmark/benchmarks.toml +++ /dev/null @@ -1,174 +0,0 @@ -[Global] -n_warmup = 2 -n_iter = 250 -n_trial = 5 -cupynumeric = true # (needs install_cupynumeric.sh) -cuda = true # also run CUDA.jl (single-GPU configs only) -# When gpus == 1, cuNumeric compares a tiny problem against CUDA.jl. -# CUDA.jl is still timed; its CSV correctness column is skipped. -check_correctness = true -n_correctness_iter = 5 -# RAM-fit the 1-GPU problem, then scale with P. Pin N (and M) on a block -# to keep paper sizes. CUNUMERIC_BENCH_MEM_FRAC overrides mem_frac. -auto_size = true -mem_frac = 0.75 - -# Names in a list share one weak-scaling figure. Unlisted [[benchmark]] -# tables each get their own. CUDA.jl and cupynumeric overlay from the -# group's baseline member (`*_baseline`, else the only / first name). -[plot.groups] -grayscott = [ - "grayscott_baseline", - "grayscott_function_accelerated", - "grayscott_begin_accelerated", - "grayscott_let_accelerated", - "grayscott_expression_accelerated", -] -dmd = ["dmd_baseline", "dmd_accelerated"] - -#################################### -# GEMM # -# Weak scaling, restricted to NxN. # -# Work ~ 2*N^2*M = 2*N^3. # -# N^3 / P --> constant. # -# N = baseline * P^(1/3). # -# Paper sizes (A100 80GB, frac~0.5): -# N = [20000, 25200, 31752, 40000] -#################################### - -[[gemm]] -T = ["Float32"] -gpus = [1, 2, 4, 8] -cpus = 8 -n_iter = 50 - -################################# -# Gray-Scott # -# Weak scaling, square NxN grid.# -# Work ~ N*M = N^2. # -# N^2 / P --> constant. # -# N = baseline * P^(1/2). # -# Paper: N = [24000, 33944, 48000, 67888] -################################# - -[[grayscott_baseline]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true - -[[grayscott_function_accelerated]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true - -[[grayscott_begin_accelerated]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true - -[[grayscott_let_accelerated]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true - -[[grayscott_expression_accelerated]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true - -################################# -# DMD # -# Snapshot matrix N × M # -# (space × time), r=min(20,M-1).# -# M is the intensity knob (fixed). -# Weak scaling: N ∝ P if N ≫ M. # -# Paper: N = [50000, 100000, 200000, 400000] -################################# - -[[dmd_baseline]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true -M = 512 -n_iter = 50 - -[[dmd_accelerated]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -fusion = true -M = 512 -n_iter = 50 - -################################# -# Spectral Poisson (FFT) # -# M independent N×N Poisson # -# solves (∇²u = f, periodic). # -# Transform is the last 2 axes; # -# batch axis M may split GPUs. # -# N is held across the P sweep # -# (changing N changes FFT size).# -# Omit N to RAM-fit one grid, # -# then M = P. Pin e.g. N = 1024 # -# and M = "auto" to fit batch. # -# Paper: N = 1024, M = 8*P # -################################# - -[[poisson_fft]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -n_iter = 50 - -################################# -# Monte-Carlo Integration # -# Work ~ N. Scale N linearly # -# Paper: N = 1e6 * P # -################################# - -[[montecarlo]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -n_iter = 50 - -######################################### -# Three-mode tensor projection # -# D[n,m,l] = A[i,j,k] B[n,i] B[m,j] # -# B[l,k] # -# Optimal pairwise work ~ 6*N^4. # -# Weak scaling: N ∝ P^{1/4}. # -# Peak RAM includes two N³ temps. # -# CUDA.jl / cuTENSOR overlays at 1 GPU. # -######################################### - -[[tensor_projection3]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -n_warmup = 2 -n_iter = 20 - -######################################### -# Rank-4 two-index tensor contraction # -# C[a,b,c,d] = X[a,i,c,j] Y[i,b,j,d] # -# Work ~ 2*N^6. # -# Weak scaling: N ∝ P^{1/6}. # -# Peak RAM includes GEMM workspace. # -# Keep n_iter small: RAM-fit N makes # -# each step much heavier than GEMM. # -######################################### - -[[tensor_contract4]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 8 -n_warmup = 2 -n_iter = 5 - diff --git a/benchmark/debug/grayscott_accelerate.jl b/benchmark/debug/grayscott_accelerate.jl deleted file mode 100644 index 53a276bb0..000000000 --- a/benchmark/debug/grayscott_accelerate.jl +++ /dev/null @@ -1,66 +0,0 @@ -#!/usr/bin/env julia - -# Print @accelerate lifetime rewrites and the fused kernels launched by each -# Gray–Scott macro form. Run with a small grid, for example: -# julia --project=benchmark benchmark/debug/grayscott_accelerate.jl 64 - -using cuNumeric - -const BENCHMARK_SRC = joinpath(@__DIR__, "..", "src") -include(joinpath(BENCHMARK_SRC, "core.jl")) -include(joinpath(BENCHMARK_SRC, "benchmarks", "grayscott.jl")) -include(joinpath(BENCHMARK_SRC, "benchmarks", "grayscott_accelerate_forms.jl")) - -const N = length(ARGS) >= 1 ? parse(Int, ARGS[1]) : 64 -N >= 4 || error("N must be at least 4") - -const FORMS = ( - (:function, "function", GrayScottFunctionAccelerated), - (:begin, "begin", GrayScottBeginAccelerated), - (:let, "let", GrayScottLetAccelerated), - (:expression, "expression", GrayScottExpressionAccelerated), -) - -function lifetime_expansion(kind) - body = deepcopy(GRAYSCOTT_STEP_BODY) - input = if kind === :function - Expr(:function, Expr(:call, :debug_step, :u, :v, :u_new, :v_new, :args), body) - elseif kind === :let - Expr(:let, body) - else - body - end - return cuNumeric._accelerate_expand(input, @__MODULE__) -end - -function print_lifetimes(kind, label) - println("\n", "="^80, "\n", uppercase(label), " — lifetime analysis\n", "="^80) - if kind === :expression - # Expression form accelerates each assignment independently. - for statement in GRAYSCOTT_STEP_BODY.args - statement isa LineNumberNode && continue - statement isa Expr && statement.head === :(=) || continue - lhs, rhs = statement.args - println("\nRHS: ", lhs) - expansion = cuNumeric._accelerate_expand(rhs, @__MODULE__) - cuNumeric.print_lifetime_analysis(expansion) - end - else - cuNumeric.print_lifetime_analysis(lifetime_expansion(kind)) - end -end - -function run_form(label, type) - println("\n", "="^80, "\n", uppercase(label), " — runtime kernels\n", "="^80) - b = build_benchmark(type, Float32, N, N) - state = only(initialize(b; deterministic=true)) - return run!(b, state) -end - -println("Gray–Scott @accelerate debug; grid=$(N)x$(N)") -cuNumeric.BCAST_FUSION_DEBUG[] = true -println("BCAST_FUSION_DEBUG = ", cuNumeric.BCAST_FUSION_DEBUG[]) -for (kind, label, type) in FORMS - print_lifetimes(kind, label) - run_form(label, type) -end diff --git a/benchmark/diagnose_montecarlo.jl b/benchmark/diagnose_montecarlo.jl deleted file mode 100644 index 8a3868665..000000000 --- a/benchmark/diagnose_montecarlo.jl +++ /dev/null @@ -1,34 +0,0 @@ -# From benchmark/: bash run_benchmark.sh diagnose_montecarlo.jl --gpus 8 --cpus 8 4266645824 -# Diagnostic only: fences deliberately change execution timing. -using cuNumeric - -length(ARGS) == 2 || error("Use run_benchmark.sh with --gpus

--cpus ") -const N = parse(Int, ARGS[2]) -N > 0 || error("N must be positive") -println("Monte Carlo diagnostic: GPUs=$(ARGS[1]), N=$N, fusion=$(cuNumeric.FUSE_BROADCAST_EXPRS)") - -function stage(f, label) - println("START: $label") - flush(stdout) - result = f() - cuNumeric.issue_execution_fence(; block=true) - println("PASS: $label") - flush(stdout) - return result -end - -x = stage("Float32 random generation") do - cuNumeric.rand(Float32, N) -end -x = stage("scale samples") do - 10.0f0 .* x -end -y = stage("fused square / negate / exponential") do - exp.(.-(x .^ 2)) -end -s = stage("sum reduction") do - sum(y) -end -stage("scale reduction result") do - (10.0f0 / N) * s -end diff --git a/benchmark/install_cupynumeric.sh b/benchmark/install_cupynumeric.sh deleted file mode 100755 index eb7d0bc9a..000000000 --- a/benchmark/install_cupynumeric.sh +++ /dev/null @@ -1,81 +0,0 @@ -#!/bin/bash -# Install a cupynumeric conda env matching the cupynumeric_jll our project resolves. -# The conda package and the JLL share the calendar-versioning scheme (e.g. 25.10), -# so we pin major.minor (patch ignored) and install from the legate channel. -# -# Usage: -# ./install_cupynumeric.sh # create a fresh env named cupynumeric-bench- -# ./install_cupynumeric.sh --name myenv # override the env name -# ./install_cupynumeric.sh --into existing # install into an existing env instead of creating one -set -euo pipefail -CONDA="${CUNUMERIC_BENCH_CONDA:-${CONDA_EXE:-conda}}" -command -v "$CONDA" >/dev/null 2>&1 || { - echo "Error: conda not found. Add it to PATH or set CUNUMERIC_BENCH_CONDA to its executable path." - exit 1 -} - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" - -ENV_NAME="" -INTO_ENV="" - -while [[ $# -gt 0 ]]; do - case $1 in - --name) - ENV_NAME=$2 - shift 2 - ;; - --into) - INTO_ENV=$2 - shift 2 - ;; - *) - echo "Unknown argument: $1" - echo "Usage: $0 [--name ] [--into ]" - exit 1 - ;; - esac -done - -# Resolve the JLL version Julia actually instantiated for this project, then keep -# major.minor only — conda packages are not published per patch. -echo "Detecting cupynumeric_jll version from the benchmark project..." -VER=$(cd "$SCRIPT_DIR" && julia --project -e ' -using Pkg -for (_, info) in Pkg.dependencies() - info.name == "cupynumeric_jll" || continue - v = info.version - isnothing(v) && continue - println("$(v.major).$(v.minor)") -end' | tail -1) - -if [[ -z "$VER" ]]; then - echo "Error: could not detect cupynumeric_jll version. Has the project been instantiated?" - exit 1 -fi - -echo "cupynumeric_jll major.minor: $VER" -SPEC="cupynumeric=$VER.*" - -# numpy 2.3 dropped the private numpy.linalg.linalg path that cupynumeric 25.10 imports. -NUMPY_SPEC="numpy<2.3" - -if [[ -n "$INTO_ENV" ]]; then - echo "Installing $SPEC into existing env '$INTO_ENV'..." - "$CONDA" install -y -n "$INTO_ENV" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" - echo "Done. Activate with: conda activate $INTO_ENV" - exit 0 -fi - -[[ -z "$ENV_NAME" ]] && ENV_NAME="cupynumeric-bench-$VER" - -if "$CONDA" env list | awk '{print $1}' | grep -qx "$ENV_NAME"; then - echo "Env '$ENV_NAME' already exists with $SPEC; nothing to do." - echo "Activate with: conda activate $ENV_NAME" - exit 0 -fi - -echo "Creating env '$ENV_NAME' with $SPEC..." -"$CONDA" create -y -n "$ENV_NAME" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" - -echo "Done. Activate with: conda activate $ENV_NAME" diff --git a/benchmark/mwe_fusion_indexing.jl b/benchmark/mwe_fusion_indexing.jl deleted file mode 100644 index db8ef7542..000000000 --- a/benchmark/mwe_fusion_indexing.jl +++ /dev/null @@ -1,78 +0,0 @@ -# From benchmark/, run each case in a fresh process: -# bash run_benchmark.sh mwe_fusion_indexing.jl --gpus 8 --cpus 8 2454267008 fused -# bash run_benchmark.sh mwe_fusion_indexing.jl --gpus 8 --cpus 8 2454267072 fused -# bash run_benchmark.sh mwe_fusion_indexing.jl --gpus 8 --cpus 8 2454267072 native -# These straddle 2^31 at the last partition's START (7*N/8), not global N. -# With N near 2^31 and eight equal partitions, all starts still fit Int32. -# Also test the original failing N=4266645824, and repeat with --gpus 1. -# No RNG, sum, benchmark harness, or preference changes. Two persistent arrays. -using cuNumeric - -function stage(f, label) - println("START: $label") - flush(stdout) - result = f() - cuNumeric.issue_execution_fence(; block=true) - println("PASS: $label") - flush(stdout) - return result -end - -function landmarks(n, p) - points = Int[1,n] - for boundary in (Int(2)^31, (cld(n,p)*k for k in 1:(p-1))...) - append!(points,filter(i->1<=i<=n,(boundary-1,boundary,boundary+1))) - end - return sort!(unique!(points)) -end - -function main(args=ARGS) - length(args)==3 || error("Use run_benchmark.sh with --gpus P --cpus C N fused|native") - p,n = parse.(Int,args[1:2]) - mode = args[3] - p>0 && n>0 || error("P and N must be positive") - mode in ("fused","native") || error("Mode must be fused or native") - println("P=$p N=$n mode=$mode; persistent data=$(2big(n)*sizeof(Float32)) bytes globally") - println("Expected last partition start (zero-based, equal tiles): $((p-1)*cld(n,p)); Int32 limit=$(typemax(Int32))") - x,y = stage("allocate and fill input/output") do - cuNumeric.ones(Float32,n),cuNumeric.zeros(Float32,n) - end - points = landmarks(n,p) - values = Dict(i=>Float32(0.125+0.025*j) for (j,i) in enumerate(points)) - stage("write boundary markers") do - for i in points - x[i:i] .= values[i] - end - end - stage("verify input markers (before fusion)") do - for i in points - @assert only(Array(x[i:i])) == values[i] "input marker mismatch at $i" - end - end - stage("$mode square / negate / exp") do - # Exact broadcast tree for exp.(.-(x .^ 2)), including literal_pow. - square = Base.Broadcast.broadcasted(Base.literal_pow,Ref(^),x,Ref(Val(2))) - tree = Base.Broadcast.instantiate(Base.Broadcast.broadcasted(exp,Base.Broadcast.broadcasted(-,square))) - if mode == "fused" - @assert cuNumeric.can_fuse_linear_broadcast(y,tree) - cuNumeric.fuse_broadcast_tree!(y,tree) - else - # Force native tasks regardless of the user's fusion preference. - tmp = cuNumeric.unravel_broadcast_tree(tree) - copyto!(y,tmp) - cuNumeric.destroy!(tmp) - end - end - stage("verify output markers (no reduction)") do - for i in points - actual = only(Array(y[i:i])) - expected = exp(-values[i]^2) - @assert isapprox(actual,expected;rtol=2f-5,atol=2f-6) "output mismatch at $i: got $actual expected $expected" - end - end - println("PASS: all $(length(points)) markers; P=$p N=$n mode=$mode") -end - -if abspath(PROGRAM_FILE)==abspath(@__FILE__) - main() -end diff --git a/benchmark/plot_results.jl b/benchmark/plot_results.jl deleted file mode 100644 index d27ae866e..000000000 --- a/benchmark/plot_results.jl +++ /dev/null @@ -1,302 +0,0 @@ -#!/usr/bin/env julia -# Generate weak-scaling plots from benchmark CSVs. -# Each figure is one `[plot.groups]` entry (or a singleton [[benchmark]]). - -using Plots -using Statistics - -include(joinpath(@__DIR__, "src", "parse_benchmarks.jl")) - -# GPU nodes often have no display; 100 = PNG. Override with GKSwstype if needed. -get!(ENV, "GKSwstype", "100") -gr() -default(; - fontfamily="sans-serif", - fg=:black, - fg_text=:black, - fg_axis=:black, - fg_border=:black, - grid=false, - gridalpha=0, - gridlinewidth=0, - minorgrid=false, - legend=false, -) - -function parse_args(args) - results_dir = "results" - out_dir = nothing - output_suffix = "" - config = joinpath(@__DIR__, "benchmarks.toml") - - for arg in args - if startswith(arg, "--out=") - out_dir = last(split(arg, "="; limit=2)) - elseif startswith(arg, "--suffix=") - output_suffix = last(split(arg, "="; limit=2)) - elseif startswith(arg, "--config=") - config = last(split(arg, "="; limit=2)) - else - results_dir = arg - end - end - results_dir = isabspath(results_dir) ? results_dir : joinpath(@__DIR__, results_dir) - config = isabspath(config) ? config : joinpath(@__DIR__, config) - if out_dir === nothing - out_dir = if basename(normpath(results_dir)) == "results" - joinpath(@__DIR__, "plots") - else - joinpath(@__DIR__, "plots", basename(normpath(results_dir))) - end - else - out_dir = isabspath(out_dir) ? out_dir : joinpath(@__DIR__, out_dir) - end - return (; results_dir, out_dir, output_suffix, config) -end - -# Fixed across every figure so GEMM / Gray-Scott / DMD read as one set. -const COLOR_CUNUMERIC = "#2a78d6" -const COLOR_CUPYNUMERIC = "#eb6834" -const COLOR_CUDA = "#1a7f37" -const COLOR_CUTENSOR = "#0d7377" -const MARKER_CUNUMERIC = :circle -const MARKER_CUPYNUMERIC = :rect -const MARKER_CUDA = :utriangle -const MARKER_CUTENSOR = :star5 - -# Extra cuNumeric variants (Gray-Scott forms, DMD accelerated). Avoid the -# reference orange/green so CUDA.jl and cuPyNumeric stay unique. -const VARIANT_COLORS = [COLOR_CUNUMERIC, "#7b2d8e", "#b8860b", "#3d5a80", "#a23b72", "#2f6f4e"] -const VARIANT_MARKERS = [:circle, :diamond, :hexagon, :dtriangle, :star4, :pentagon] - -const REF_FAMILIES = [ - ("cupynumeric", "cuPyNumeric", COLOR_CUPYNUMERIC, MARKER_CUPYNUMERIC), - ("CUDA.jl", "CUDA.jl", COLOR_CUDA, MARKER_CUDA), - ("tensoroperations_cuda", "TensorOperations.jl / cuTENSOR", COLOR_CUTENSOR, MARKER_CUTENSOR), -] - -const INK = "#111111" -const IDEALCOL = "#6e6e6e" - -const GROUP_TITLES = Dict( - "grayscott" => "Gray-Scott", - "dmd" => "DMD", - "gemm" => "GEMM", - "poisson_fft" => "Poisson FFT", - "montecarlo" => "Monte Carlo", - "tensor_projection3" => "Tensor projection (3-mode)", - "tensor_contract4" => "Tensor contraction (rank-4)", -) - -include(joinpath(@__DIR__, "src", "result_rows.jl")) - -_all_rows(runs) = reduce(vcat, runs; init=Row[]) - -function make_series(label, color, marker, ls, runs) - isempty(runs) && return nothing - return (label=label, color=color, marker=marker, ls=ls, - agg=aggregate(_all_rows(runs))) -end - -function load_csv_series(results_dir, bench, key, label, color, marker, ls) - path = joinpath(results_dir, "$(bench)_$(key).csv") - isfile(path) || return nothing - runs = load_runs(path) - isempty(runs) && return nothing - return make_series(label, color, marker, ls, runs) -end - -function group_title(group) - return get(GROUP_TITLES, group, titlecase(replace(group, '_' => ' '))) -end - -function variant_label(group, member) - member == group && return "cuNumeric.jl" - prefix = group * "_" - stem = startswith(member, prefix) ? member[(length(prefix) + 1):end] : member - return replace(stem, '_' => ' ') -end - -function cunumeric_series_label(group, member, fused) - base = variant_label(group, member) - return fused ? base : "$(base) (unfused)" -end - -function overlay_refs(results_dir, members) - series = [] - seen = Set{String}() - order = unique!(vcat([plot_baseline(members)], members)) - for (key, label, color, marker) in REF_FAMILIES - key in seen && continue - for member in order - s = load_csv_series(results_dir, member, key, label, color, marker, :solid) - s === nothing && continue - push!(series, s) - push!(seen, key) - break - end - end - return series -end - -function group_series(results_dir, group, members) - series = [] - n_members = length(members) - for (i, member) in enumerate(members) - color, marker = if n_members == 1 - COLOR_CUNUMERIC, MARKER_CUNUMERIC - else - VARIANT_COLORS[mod1(i, length(VARIANT_COLORS))], - VARIANT_MARKERS[mod1(i, length(VARIANT_MARKERS))] - end - for (key, fused, ls) in ( - ("cunumeric", true, :solid), - ("cunumeric_nofusion", false, :dash), - ) - label = cunumeric_series_label(group, member, fused) - s = load_csv_series(results_dir, member, key, label, color, marker, ls) - s === nothing || push!(series, s) - end - end - append!(series, overlay_refs(results_dir, members)) - return filter(!isnothing, series) -end - -function addline!(p, s, y; kw...) - return plot!(p, getfield.(s.agg, :gpus), y; color=s.color, - lw=2.6, ls=s.ls, marker=s.marker, ms=7, msc=s.color, markerstrokewidth=0.7, - label=s.label, kw...) -end - -function build_legend(series) - n = length(series) - cols = min(max(n, 1), 4) - rows = cld(n, cols) - slot = min(0.32, 0.92 / cols) - x0 = (1 - cols * slot) / 2 - y0 = 0.50 + 0.18 * (rows - 1) / 2 - - pl = plot(; - framestyle=:none, grid=false, ticks=false, legend=false, - xlims=(0, 1), ylims=(0, 1), widen=false, - left_margin=0Plots.mm, right_margin=0Plots.mm, - top_margin=0Plots.mm, bottom_margin=0Plots.mm, - background_color=:transparent, - ) - # Pin the coordinate system so later scatter/annotate cannot rescale it. - scatter!(pl, [0.0, 1.0], [0.0, 1.0]; ms=0, msw=0, mc=:white, label="") - - for (i, s) in enumerate(series) - r, c = divrem(i - 1, cols) - x = x0 + c * slot - y = y0 - r * 0.36 - plot!(pl, [x, x + 0.028], [y, y]; color=s.color, lw=2.8, ls=s.ls, label="") - scatter!(pl, [x + 0.014], [y]; color=s.color, marker=s.marker, - ms=7, msc=s.color, markerstrokewidth=0.6, label="") - annotate!(pl, x + 0.036, y, text(s.label, 12, :black, :left)) - end - plot!(pl; xlims=(0, 1), ylims=(0, 1), widen=false) - return pl -end - -function series_ymax(series, yfield, efield) - m = 0.0 - for s in series, x in s.agg - m = max(m, getfield(x, yfield) + getfield(x, efield)) - end - return m -end - -function positive_ylim(hi; pad=0.18) - hi > 0 || return (0, 1) - return (0, hi * (1 + pad)) -end - -function weak_scaling_figure(series; plot_title) - common = ( - xscale=:log2, xticks=([1, 2, 4, 8], ["1", "2", "4", "8"]), xlabel="GPUs", - framestyle=:box, grid=false, gridalpha=0, gridlinewidth=0, minorgrid=false, - foreground_color_grid=:white, - foreground_color_axis=:black, foreground_color_border=:black, - foreground_color_text=:black, foreground_color_guide=:black, - tickfontcolor=:black, guidefontcolor=:black, titlefontcolor=:black, - tickfontsize=14, guidefontsize=16, titlefontsize=16, - xtickfontsize=14, ytickfontsize=14, - xguidefontsize=16, yguidefontsize=16, - legend=false, xlims=(0.85, 9.4), widen=false, - left_margin=10Plots.mm, right_margin=6Plots.mm, - top_margin=5Plots.mm, bottom_margin=10Plots.mm, - ) - - p1 = plot(; ylabel="Throughput", title="Throughput", - ylims=positive_ylim(series_ymax(series, :h, :hsd); pad=0.28), common...) - for s in series - addline!(p1, s, getfield.(s.agg, :h); yerror=getfield.(s.agg, :hsd)) - end - - p2 = plot(; ylabel="Time / step (ms)", title="Time per step", - ylims=positive_ylim(series_ymax(series, :t, :tsd); pad=0.28), - common..., left_margin=28Plots.mm, yguidefontsize=15) - for s in series - addline!(p2, s, getfield.(s.agg, :t); yerror=getfield.(s.agg, :tsd)) - end - - efficiencies = Float64[] - for s in series - i1 = findfirst(x -> x.gpus == 1, s.agg) - i1 === nothing && continue - base = s.agg[i1].h - append!(efficiencies, [x.h / (x.gpus * base) for x in s.agg]) - end - p3 = plot(; ylabel="Parallel efficiency", title="Weak-scaling efficiency", - ylims=positive_ylim(max(1.0, isempty(efficiencies) ? 0.0 : maximum(efficiencies)); pad=0.12), - common..., left_margin=16Plots.mm) - hline!(p3, [1.0]; color=IDEALCOL, ls=:dashdot, lw=1.6, label="") - for s in series - i1 = findfirst(x -> x.gpus == 1, s.agg) - i1 === nothing && continue - base = s.agg[i1].h - addline!(p3, s, [x.h / (x.gpus * base) for x in s.agg]) - end - - nrows = cld(length(series), 4) - layout = if nrows > 1 - @layout([grid(1, 3); leg{0.16h}]) - else - @layout([grid(1, 3); leg{0.14h}]) - end - return plot( - p1, p2, p3, build_legend(series); - layout, - size=(1760, nrows > 1 ? 680 : 620), dpi=220, plot_title, - plot_titlefontsize=20, plot_titlefontcolor=:black, - background_color=:white, - ) -end - -function main(args=ARGS) - cfg = parse_args(args) - if !isdir(cfg.results_dir) - println("no results directory at $(cfg.results_dir)") - return nothing - end - isfile(cfg.config) || error("plot config not found: $(cfg.config)") - - mkpath(cfg.out_dir) - for (group, members) in parse_plot_groups(cfg.config) - series = group_series(cfg.results_dir, group, members) - isempty(series) && continue - validate_series_sizes(series) - fig = weak_scaling_figure( - series; plot_title=group_title(group) * " — weak scaling" - ) - out = joinpath(cfg.out_dir, "$(group)_weak_scaling$(cfg.output_suffix).png") - savefig(fig, out) - println("wrote $out") - end - return nothing -end - -if abspath(PROGRAM_FILE) == abspath(@__FILE__) - main() -end diff --git a/benchmark/run.jl b/benchmark/run.jl deleted file mode 100644 index dc6beaff1..000000000 --- a/benchmark/run.jl +++ /dev/null @@ -1,32 +0,0 @@ -# Orchestrator remains off the GPU; each timed backend uses its own process. -using Pkg - -function ensure_project_ready() - Pkg.develop([ - Pkg.PackageSpec(; path=joinpath(@__DIR__, "..", "lib", "CNPreferences")), - Pkg.PackageSpec(; path=joinpath(@__DIR__, "..")), - ]) - Pkg.instantiate() -end - -function cupynumeric_env_name() - haskey(ENV, "CUPYNUMERIC_ENV") && return ENV["CUPYNUMERIC_ENV"] - for (_, info) in Pkg.dependencies() - info.name == "cupynumeric_jll" || continue - info.version === nothing && continue - return "cupynumeric-bench-$(info.version.major).$(info.version.minor)" - end - return error("could not resolve cupynumeric_jll version; set CUPYNUMERIC_ENV explicitly") -end - -if abspath(PROGRAM_FILE) == abspath(@__FILE__) - ensure_project_ready() - include("src/core.jl") - include_benchmarks() - include("src/parse_benchmarks.jl") - include("src/memory.jl") - include("src/planning.jl") - include("src/runner.jl") - using CNPreferences: CNPreferences - exit(main()) -end diff --git a/benchmark/run_benchmark.sh b/benchmark/run_benchmark.sh deleted file mode 100755 index e9ef39307..000000000 --- a/benchmark/run_benchmark.sh +++ /dev/null @@ -1,85 +0,0 @@ -#!/bin/bash - -if [[ $# -lt 1 ]]; then - echo "Usage: $0 [--gpus ] [--cpus ] [extra_args...]" - exit 1 -fi - -# Parse arguments -FILENAME=$1 -shift - -GPUS=0 -CPUS=1 -PYENV="" -VERBOSE=0 - -while [[ $# -gt 0 ]]; do - case $1 in - --gpus) - GPUS=$2 - shift 2 - ;; - --cpus) - CPUS=$2 - shift 2 - ;; - --pyenv) - PYENV=$2 - shift 2 - ;; - --verbose) - VERBOSE=1 - shift - ;; - *) - # Collect all other arguments as extra arguments - EXTRA_ARGS+=("$1") - shift - ;; - esac -done - -# Validate the filename exists -if [[ ! -f $FILENAME ]]; then - echo "Error: File $FILENAME does not exist." - exit 1 -fi - -# Inform user of the configuration -if [[ $GPUS -lt 0 ]]; then - echo "GPUs invalid, using gpus = 0" - exit -fi - -if [[ $CPUS -lt 0 ]]; then - echo "CPUs invalid, using cpus = 1" - exit -fi - -export LEGATE_AUTO_CONFIG=1 -export LEGATE_CONFIG="--cpus=$CPUS --gpus=$GPUS" -if [[ -n ${CUNUMERIC_BENCH_FBMEM_MB:-} ]]; then - export LEGATE_CONFIG="$LEGATE_CONFIG --fbmem=$CUNUMERIC_BENCH_FBMEM_MB" -fi -export LEGATE_SHOW_CONFIG=$VERBOSE - -export LD_LIBRARY_PATH="" - -[[ $VERBOSE == 1 ]] && echo "Running $FILENAME with $CPUS CPUs and $GPUS GPUs" - -# Python (cupynumeric) workers run in the conda env built by install_cupynumeric.sh; -# Julia (cuNumeric) workers run against the local project. -if [[ $FILENAME == *.py ]]; then - if [[ -z $PYENV ]]; then - echo "Error: running a .py worker requires --pyenv (run install_cupynumeric.sh first)." - exit 1 - fi - CMD=("${CUNUMERIC_BENCH_CONDA:-${CONDA_EXE:-conda}}" run --no-capture-output -n "$PYENV" python "$FILENAME" "$GPUS" "${EXTRA_ARGS[@]}") -else - CMD=("${CUNUMERIC_BENCH_JULIA:-julia}" --project "$FILENAME" "$GPUS" "${EXTRA_ARGS[@]}") -fi - -[[ $VERBOSE == 1 ]] && printf "Running: %q " "${CMD[@]}" -[[ $VERBOSE == 1 ]] && printf '\n' -"${CMD[@]}" diff --git a/benchmark/src/autosize.jl b/benchmark/src/autosize.jl deleted file mode 100644 index 43223325e..000000000 --- a/benchmark/src/autosize.jl +++ /dev/null @@ -1,108 +0,0 @@ -# RAM query + binary search. Peak bytes and P-scaling live on each benchmark -# type (`total_space`, `estimate_scaling`, `fit_one_gpu`) so the formulas sit -# next to the kernel. Python never sees them: run.jl resolves N/M/flops and -# passes integers on the worker command line. - -align8(n::Integer) = max(8, 8 * fld(Int(n), 8)) -align2(n::Integer) = max(2, 2 * fld(Int(n), 2)) - -function scale_axis(n1::Integer, P::Integer, α::Real) - return align8(floor(Int, n1 * Float64(P)^Float64(α))) -end - -is_auto_size(x) = x isa AbstractString && lowercase(strip(string(x))) == "auto" - -function largest_feasible(lo::Int, hi::Int, pred) - hi < lo && return nothing - pred(lo) || return nothing - best = lo - while lo <= hi - mid = lo + (hi - lo) >> 1 - if pred(mid) - best = mid - lo = mid + 1 - else - hi = mid - 1 - end - end - return best -end - -function min_gpu_memory_bytes() - out = read(`nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits`, String) - mibs = Int[] - for line in split(out, '\n'; keepempty=false) - tok = strip(split(line, ',')[1]) - push!(mibs, parse(Int, tok)) - end - isempty(mibs) && error("nvidia-smi reported no GPUs") - return minimum(mibs) * 1024^2 -end - -function autosize_budget(mem_frac::Real) - frac = parse(Float64, get(ENV, "CUNUMERIC_BENCH_MEM_FRAC", string(mem_frac))) - (0 < frac <= 1) || error("mem_frac must be in (0, 1], got $frac") - bytes = Int(floor(frac * min_gpu_memory_bytes())) - bytes > 0 || error("GPU memory budget is 0") - return bytes, frac -end - -# Query only the GPUs the worker is allowed to see. Tests inject rows/env so -# device selection and budget calculation do not require a GPU. -function selected_gpu_budget(mem_frac, count; - inventory=read(`nvidia-smi --query-gpu=index,uuid,memory.total,memory.free --format=csv,noheader,nounits`,String), - visibility=get(ENV,"CUDA_VISIBLE_DEVICES",nothing), - fraction=get(ENV,"CUNUMERIC_BENCH_MEM_FRAC",string(mem_frac)), - fbmem=get(ENV,"CUNUMERIC_BENCH_FBMEM_MB",nothing)) - frac = parse(Float64,fraction) - 0 < frac <= 1 || error("mem_frac must be in (0,1]") - rows = [strip.(split(line,',')) for line in split(inventory,'\n') if !isempty(strip(line))] - devices = if visibility === nothing - rows - else - tokens = split(visibility,',') - any(t->isempty(strip(t)),tokens) && error("No usable CUDA_VISIBLE_DEVICES entries") - length(unique(tokens)) == length(tokens) || error("Duplicate CUDA_VISIBLE_DEVICES entries") - map(tokens) do token - matches = filter(r->r[1]==strip(token) || startswith(r[2],strip(token)),rows) - length(matches)==1 || error("Cannot resolve visible GPU '$token'; MIG requires a device-specific inventory") - only(matches) - end - end - 0 < count <= length(devices) || error("Requested $count GPUs, but only $(length(devices)) are visible") - # All visible devices are eligible for the runtime; use the smallest pool. - pools = [parse(Int,r[3])*1024^2 for r in devices] - free = minimum(parse(Int,r[4])*1024^2 for r in devices) - if fbmem !== nothing - cap = parse(Int,fbmem)*1024^2 - cap > 0 || error("CUNUMERIC_BENCH_FBMEM_MB must be positive") - pools = min.(pools,cap) - end - budget = floor(Int,frac*minimum(pools)) - # Do not silently shrink for transient other workloads. - free >= budget || error("Selected GPUs have only $free free bytes; planned budget is $budget. Free the devices or explicitly reduce mem_frac.") - return budget,frac -end - -function parse_bench_type(T_str::AbstractString) - return getfield(Base, Symbol(T_str))::DataType -end - -function resolve_autosize( - name::AbstractString, T_str::AbstractString, P::Integer; - mem_frac::Real, N_hint, M_hint, -) - haskey(BENCHMARKS, name) || error( - "No benchmark registered for '$(name)'. Known: $(join(sort(collect(keys(BENCHMARKS))), ", "))", - ) - B = BENCHMARKS[name] - T = parse_bench_type(T_str) - budget, frac = autosize_budget(mem_frac) - N1, M1 = fit_one_gpu(B, T; budget, N_hint, M_hint) - scaled = estimate_scaling(build_benchmark(B, T, N1, M1), P) - if scaled === nothing - return nothing - end - N, M = scaled - return (; N, M, N1, M1, budget, frac) -end diff --git a/benchmark/src/benchmarks/dmd.jl b/benchmark/src/benchmarks/dmd.jl deleted file mode 100644 index 717406269..000000000 --- a/benchmark/src/benchmarks/dmd.jl +++ /dev/null @@ -1,149 +0,0 @@ -# Exact DMD of an N×M snapshot matrix (space × time). Rank is min(20, M-1), -# matching examples/dmd.jl. -# -# N is spatial degrees of freedom (rows of X), not a grid side length. -# The SVD is tall-skinny: X1 is N × (M-1). With M (and rank r) fixed, every -# term below is Θ(N) or O(1), so weak scaling is N ∝ P — not the N³ of a -# square SVD. - -abstract type AbstractDMD{T} <: AbstractBenchmark{T} end - -Base.@kwdef struct DMDBaseline{T} <: AbstractDMD{T} - N::Int - M::Int -end - -Base.@kwdef struct DMDAccelerated{T} <: AbstractDMD{T} - N::Int - M::Int -end - -name(::DMDBaseline) = "dmd_baseline" -name(::DMDAccelerated) = "dmd_accelerated" -dims(b::AbstractDMD) = (b.N, b.M) -data(b::AbstractDMD{T}) where {T} = "DMD with T=$(T), N=$(b.N), M=$(b.M)" -allowed_types(::Type{<:AbstractDMD}) = cuNumeric.SUPPORTED_FLOAT_TYPES - -function build_benchmark(::Type{A}, ::Type{T}, N, M) where {A<:AbstractDMD,T} - return A{T}(; N=N, M=M) -end - -_dmd_rank(b::AbstractDMD) = min(20, b.M - 1) - -# X1 is m×n, m=N spatial points, n=M-1 snapshots, r = min(20, n). -# 2mn² + 11n³ thin SVD (Golub & Van Loan, Matrix Computations, 4ed, §8.6) -# 2mnr + mr B = X2 V_r Σ_r^{-1} (GEMM + column scale) -# 2mr² Ã = U_r' B -# 25r³ eigen(Ã) (LAPACK xGEEV) -# 2mr² Φ = B W -const DMD_TALL_RATIO = 10 -const DEFAULT_DMD_M = 512 - -function total_flops(b::AbstractDMD) - m = b.N - n = b.M - 1 - r = _dmd_rank(b) - return ( - 2 * m * n^2 + 11 * n^3 + - 2 * m * n * r + m * r + - 2 * m * r^2 + - 25 * r^3 + - 2 * m * r^2 - ) -end - -# X (N×M), thin-SVD U ~ N×(M-1), Vt ~ n×n, B, plus a cuSOLVER-sized fudge. -function total_space(b::AbstractDMD{T}) where {T} - n = max(b.M - 1, 1) - s = sizeof(T) - return (b.N * b.M + b.N * n + n * n + b.N * n) * s + 2 * b.N * b.M * s -end - -function estimate_scaling(b::AbstractDMD, P::Integer) - P == 1 && return (b.N, b.M) - b.N >= DMD_TALL_RATIO * b.M || return nothing - return (b.N * P, b.M) -end - -function fit_one_gpu( - ::Type{B}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {B<:AbstractDMD,T} - M = something(M_hint, DEFAULT_DMD_M) - lo = max(M, 8) - hi = max(lo, Int(fld(budget, max(6 * M * sizeof(T), 1)))) - N = largest_feasible(lo, hi, n -> total_space(B{T}(; N=n, M=M)) <= budget) - N === nothing && error("dmd M=$M does not fit in $(budget) bytes") - return (max(align8(N), M), M) -end - -function initialize(b::AbstractDMD{T}; mod=cuNumeric) where {T} - # X1 is N×(M-1); the SVD backend requires m >= n. - b.N >= b.M - 1 || throw( - ArgumentError("DMD snapshot matrix is N×M with N ≥ M-1 (got N=$(b.N), M=$(b.M))") - ) - X = rand_array(mod, T, b.N, b.M) - GC.gc() - return (X,) -end - -_dmd_T(A) = A isa NDArray ? cuNumeric.transpose(A) : transpose(A) -_dmd_row(v) = v isa NDArray ? cuNumeric.reshape(v, (1, length(v))) : reshape(v, 1, length(v)) - -# svd / eigen return factorizations whose stores the lifetime rewriter cannot -# see, so those stay outside the macro. The GEMM lift is wrapped. -# -# Do not form Diagonal(1 ./ S) inside @accelerate: the rewriter treats -# `1 ./ S` as a last-used temporary and frees it while Diagonal still holds -# that same vector. Scale columns with a broadcast instead (same math). -function _dmd_factors(X, r) - n = size(X, 2) - X1 = X[:, 1:(n - 1)] - X2 = X[:, 2:n] - F = svd(X1) - rk = min(r, length(F.S)) - return X2, F.U[:, 1:rk], F.Vt[1:rk, :], F.S[1:rk] -end - -let body = quote - Sinv = eltype(X)(1) ./ _dmd_row(S) - B = (X2 * _dmd_T(Vt)) .* Sinv - Ã = _dmd_T(U) * B - (B, Ã) - end - @eval _dmd_project(::DMDBaseline, X, X2, U, Vt, S) = $body - if CUNUMERIC_BENCH_RUNTIME - @eval @accelerate function _dmd_project( - ::DMDAccelerated, X, X2, U, Vt, S - ) - $body - end - end -end - -function _dmd_compute!(b::AbstractDMD, X, r) - X2, U, Vt, S = _dmd_factors(X, r) - B, Ã = _dmd_project(b, X, X2, U, Vt, S) - E = eigen(Ã) - CT = Complex{eltype(X)} - Bc = astype_array(B, CT) - return E.values, Bc * E.vectors -end - -run!(b::AbstractDMD, X) = _dmd_compute!(b, X, _dmd_rank(b)) - -function correctness_problem(b::AD) where {AD <: AbstractDMD} - m = min(b.M, 16) - n = max(min(b.N, 64), m - 1) - return AD(; N=n, M=m) -end -# Eigenvectors have a phase; compare sorted |λ| only. -function correctness_result(::AbstractDMD, _, out) - return sort(abs.(vec(to_host(out[1]))); rev=true) -end -function cuda_runnable(b::DMDAccelerated{T}) where {T} - return DMDBaseline{T}(; N=b.N, M=b.M) -end - -register_benchmark("dmd_baseline", DMDBaseline) -register_benchmark("dmd_accelerated", DMDAccelerated) diff --git a/benchmark/src/benchmarks/gemm.jl b/benchmark/src/benchmarks/gemm.jl deleted file mode 100644 index 886033b5e..000000000 --- a/benchmark/src/benchmarks/gemm.jl +++ /dev/null @@ -1,51 +0,0 @@ -# Interface: `name`, `dims`, `total_flops`, `total_space`, `estimate_scaling`, -# `fit_one_gpu`, `initialize`, `run!`. Peak-byte and P-scaling formulas live -# here so the orchestrator can dispatch without a name switch. - -Base.@kwdef struct GEMM{T} <: AbstractBenchmark{T} - N::Int - M::Int -end - -name(::GEMM) = "gemm" -dims(g::GEMM) = (g.N, g.M) -data(g::GEMM{T}) where {T} = "GEMM with T=$(T), N=$(g.N), M=$(g.M)" - -function allowed_types(::Type{GEMM}) - return Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_INT_TYPES} -end - -total_flops(s::GEMM) = s.N * s.N * ((2*s.M) - 1) - -# Live arrays for `mul!(C, A, B)`: A (N×M), B (M×N), C (N×N). -total_space(s::GEMM{T}) where {T} = (2 * s.N * s.M + s.N * s.N) * sizeof(T) - -function estimate_scaling(s::GEMM, P::Integer) - P == 1 && return (s.N, s.M) - return (scale_axis(s.N, P, 1//3), scale_axis(s.M, P, 1//3)) -end - -function fit_one_gpu( - ::Type{GEMM}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {T} - hi = max(8, Int(floor(sqrt(Float64(budget) / sizeof(T))))) - n = largest_feasible(8, hi, k -> total_space(GEMM{T}(; N=k, M=k)) <= budget) - n === nothing && error("gemm does not fit in $(budget) bytes") - n = align8(n) - return (n, n) -end - -function initialize(s::GEMM{T}; mod=cuNumeric) where {T} - A = rand_array(mod, T, s.N, s.M) - B = rand_array(mod, T, s.M, s.N) - C = zeros_array(mod, T, s.N, s.N) - GC.gc() - return C, A, B -end - -run!(::GEMM, C, A, B) = mul!(C, A, B) - -correctness_problem(b::GEMM{T}) where {T} = GEMM{T}(; N=min(b.N, 8), M=min(b.M, 8)) - -register_benchmark("gemm", GEMM) diff --git a/benchmark/src/benchmarks/grayscott.jl b/benchmark/src/benchmarks/grayscott.jl deleted file mode 100644 index c05066f52..000000000 --- a/benchmark/src/benchmarks/grayscott.jl +++ /dev/null @@ -1,183 +0,0 @@ -struct GSParams{T} - dx::T - dt::T - c_u::T - c_v::T - f::T - k::T -end - -function GSParams{T}(; dx=1, c_u=1.0, c_v=0.3, f=0.03, k=0.06) where {T} - return GSParams{T}(T(dx), T(dx / 5), T(c_u), T(c_v), T(f), T(k)) -end - -abstract type AbstractGrayScott{T} <: AbstractBenchmark{T} end - -# Timesteps form one trajectory; allow Legate to schedule ahead within it. -fence_each_iteration(::AbstractGrayScott) = false - -Base.@kwdef struct GrayScottBaseline{T} <: AbstractGrayScott{T} - N::Int - M::Int -end - -Base.@kwdef struct GrayScottAccelerated{T} <: AbstractGrayScott{T} - N::Int - M::Int -end - -name(::GrayScottBaseline) = "grayscott_baseline" -name(::GrayScottAccelerated) = "grayscott_accelerated" -dims(b::AbstractGrayScott) = (b.N, b.M) -data(b::AbstractGrayScott{T}) where {T} = "GrayScott with T=$(T), N=$(b.N), M=$(b.M)" -allowed_types(::Type{AbstractGrayScott}) = cuNumeric.SUPPORTED_FLOAT_TYPES -total_flops(b::AbstractGrayScott) = b.N * b.M # grid points updated per step - -# Four live grids: u, v, u_new, v_new. Legion halos are not counted. -total_space(b::AbstractGrayScott{T}) where {T} = 4 * b.N * b.M * sizeof(T) - -function estimate_scaling(b::AbstractGrayScott, P::Integer) - P == 1 && return (b.N, b.M) - return (scale_axis(b.N, P, 1//2), scale_axis(b.M, P, 1//2)) -end - -function fit_one_gpu( - ::Type{B}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {B<:AbstractGrayScott,T} - hi = max(8, Int(floor(sqrt(Float64(budget) / sizeof(T))))) - n = largest_feasible(8, hi, k -> total_space(B{T}(; N=k, M=k)) <= budget) - n === nothing && error("$B does not fit in $(budget) bytes") - n = align8(n) - return (n, n) -end - -function build_benchmark(::Type{A}, ::Type{T}, N, M) where {A<:AbstractGrayScott,T} - return A{T}(; N=N, M=M) -end - -mutable struct GrayScottState{A,P} - u::A - v::A - u_new::A - v_new::A - params::P -end - -function initialize(b::AbstractGrayScott{T}; mod=cuNumeric, deterministic::Bool=false) where {T} - u = ones_array(mod, T, b.N, b.M) - v = zeros_array(mod, T, b.N, b.M) - u_new = zeros_array(mod, T, b.N, b.M) - v_new = zeros_array(mod, T, b.N, b.M) - - seed = min(150, b.N, b.M) - if deterministic - # Shared host pattern so cuNumeric and CUDA.jl start from the same IC. - host_u = T[T(0.5) + T(0.5) * sin(T(i)) * cos(T(j)) for i in 1:seed, j in 1:seed] - host_v = T[T(0.25) + T(0.25) * cos(T(i)) * sin(T(j)) for i in 1:seed, j in 1:seed] - u[1:seed, 1:seed] = to_backend(mod, host_u) - v[1:seed, 1:seed] = to_backend(mod, host_v) - else - u[1:seed, 1:seed] = rand_array(mod, T, seed, seed) - v[1:seed, 1:seed] = rand_array(mod, T, seed, seed) - end - - return (GrayScottState(u, v, u_new, v_new, GSParams{T}()),) -end - -function to_backend_state(mod, st::GrayScottState) - return GrayScottState( - to_backend_state(mod, st.u), - to_backend_state(mod, st.v), - to_backend_state(mod, st.u_new), - to_backend_state(mod, st.v_new), - st.params, - ) -end - -function correctness_problem(b::AbstractGrayScott{T}) where {T} - return typeof(b)(; N=min(32, b.N, b.M), M=min(32, b.N, b.M)) -end -correctness_iters(::AbstractGrayScott, gs::GlobalSettings) = gs.n_correctness_iter -correctness_result(::AbstractGrayScott, state, _) = (only(state).u, only(state).v) -function cuda_runnable(b::GrayScottAccelerated{T}) where {T} - return GrayScottBaseline{T}(; N=b.N, M=b.M) -end - -# Shared syntax tree keeps every Gray-Scott variant on the exact same workload. -const GRAYSCOTT_STEP_BODY = quote - # currently we don't have NDArray^x working yet. every operator is dotted - # so each rhs fuses into a single broadcast kernel rather than shattering - # into bare +/-/* binary tasks. - F_u = ( - ( - .-u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) .+ args.f .* (1.0f0 .- u[2:(end - 1), 2:(end - 1)]) - ) - F_v = ( - ( - u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) .- (args.f + args.k) .* v[2:(end - 1), 2:(end - 1)] - ) - # 2-D Laplacian via slicing, excluding boundaries - u_lap = ( - ( - u[3:end, 2:(end - 1)] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ - u[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 .+ - ( - u[2:(end - 1), 3:end] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ - u[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - v_lap = ( - ( - v[3:end, 2:(end - 1)] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ - v[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 .+ - ( - v[2:(end - 1), 3:end] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ - v[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - - # Forward-Euler step for all interior points - u_new[2:(end - 1), 2:(end - 1)] = - ((args.c_u .* u_lap) .+ F_u) .* args.dt .+ u[2:(end - 1), 2:(end - 1)] - v_new[2:(end - 1), 2:(end - 1)] = - ((args.c_v .* v_lap) .+ F_v) .* args.dt .+ v[2:(end - 1), 2:(end - 1)] - - # Periodic boundary conditions - u_new[:, 1] = u[:, end - 1] - u_new[:, end] = u[:, 2] - u_new[1, :] = u[end - 1, :] - u_new[end, :] = u[2, :] - v_new[:, 1] = v[:, end - 1] - v_new[:, end] = v[:, 2] - v_new[1, :] = v[end - 1, :] - v_new[end, :] = v[2, :] -end - -# Original baseline and recommended function-form benchmark. -let body = deepcopy(GRAYSCOTT_STEP_BODY) - @eval _gs_step!(b::GrayScottBaseline, u, v, u_new, v_new, args::GSParams) = $body - if CUNUMERIC_BENCH_RUNTIME - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottAccelerated, u, v, u_new, v_new, args::GSParams)), body - ) - @eval $definition - end -end - -function run!(b::AbstractGrayScott, st::GrayScottState) - _gs_step!(b, st.u, st.v, st.u_new, st.v_new, st.params) - # swap references rather than copy - st.u, st.u_new = st.u_new, st.u - st.v, st.v_new = st.v_new, st.v - return nothing -end - -register_benchmark("grayscott_baseline", GrayScottBaseline) -register_benchmark("grayscott_accelerated", GrayScottAccelerated) diff --git a/benchmark/src/benchmarks/grayscott_accelerate_forms.jl b/benchmark/src/benchmarks/grayscott_accelerate_forms.jl deleted file mode 100644 index 2b83342d6..000000000 --- a/benchmark/src/benchmarks/grayscott_accelerate_forms.jl +++ /dev/null @@ -1,103 +0,0 @@ -# Compare the four scope contracts of `@accelerate` on one shared Gray-Scott step. -# Each type has a distinct result name so benchmark runs produce separate CSVs. - -abstract type AbstractGrayScottAccelerateForm{T} <: AbstractGrayScott{T} end - -Base.@kwdef struct GrayScottFunctionAccelerated{T} <: - AbstractGrayScottAccelerateForm{T} - N::Int - M::Int -end - -Base.@kwdef struct GrayScottBeginAccelerated{T} <: AbstractGrayScottAccelerateForm{T} - N::Int - M::Int -end - -Base.@kwdef struct GrayScottLetAccelerated{T} <: AbstractGrayScottAccelerateForm{T} - N::Int - M::Int -end - -Base.@kwdef struct GrayScottExpressionAccelerated{T} <: - AbstractGrayScottAccelerateForm{T} - N::Int - M::Int -end - -name(::GrayScottFunctionAccelerated) = "grayscott_function_accelerated" -name(::GrayScottBeginAccelerated) = "grayscott_begin_accelerated" -name(::GrayScottLetAccelerated) = "grayscott_let_accelerated" -name(::GrayScottExpressionAccelerated) = "grayscott_expression_accelerated" - -function cuda_runnable(b::AbstractGrayScottAccelerateForm{T}) where {T} - return GrayScottBaseline{T}(; N=b.N, M=b.M) -end - -# Function form is the reusable default: arguments and the return value survive, -# while non-returned locals may fuse across statements or die after their last use. -if CUNUMERIC_BENCH_RUNTIME - let body = deepcopy(GRAYSCOTT_STEP_BODY) - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottFunctionAccelerated, u, v, u_new, v_new, args::GSParams)), - body, - :function, - ) - @eval $definition - end - - # `begin` adds no scope. Every named local remains visible, so it measures the - # multi-output/materialized path rather than eliminating named intermediates. - let body = deepcopy(GRAYSCOTT_STEP_BODY) - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottBeginAccelerated, u, v, u_new, v_new, args::GSParams)), - body, - :begin, - ) - @eval $definition - end - - # `let` is a hard one-off scope. Only its result escapes, allowing aggressive - # inter-statement fusion and last-use cleanup for all other local temporaries. - let body = deepcopy(GRAYSCOTT_STEP_BODY) - definition = _define_accelerated_definition( - :(_gs_step!(b::GrayScottLetAccelerated, u, v, u_new, v_new, args::GSParams)), - body, - :let, - ) - @eval $definition - end -end - -# Expression form has no multi-statement scope. Accelerating each RHS preserves -# fusion inside that expression but deliberately materializes statement results, -# isolating intra-expression fusion from the inter-statement rewrite cases above. -function accelerate_grayscott_rhs(body::Expr) - statements = Any[] - for statement in body.args - if statement isa LineNumberNode - push!(statements, statement) - elseif statement isa Expr && statement.head === :(=) - lhs, rhs = statement.args - push!(statements, :($lhs = @accelerate $rhs)) - else - error("Gray-Scott expression benchmark expected assignments; got $(repr(statement))") - end - end - return Expr(:block, statements...) -end - -let body = accelerate_grayscott_rhs(deepcopy(GRAYSCOTT_STEP_BODY)) - if CUNUMERIC_BENCH_RUNTIME - @eval function _gs_step!( - b::GrayScottExpressionAccelerated, u, v, u_new, v_new, args::GSParams - ) - $body - end - end -end - -register_benchmark("grayscott_function_accelerated", GrayScottFunctionAccelerated) -register_benchmark("grayscott_begin_accelerated", GrayScottBeginAccelerated) -register_benchmark("grayscott_let_accelerated", GrayScottLetAccelerated) -register_benchmark("grayscott_expression_accelerated", GrayScottExpressionAccelerated) diff --git a/benchmark/src/benchmarks/montecarlo.jl b/benchmark/src/benchmarks/montecarlo.jl deleted file mode 100644 index c42db0066..000000000 --- a/benchmark/src/benchmarks/montecarlo.jl +++ /dev/null @@ -1,56 +0,0 @@ -Base.@kwdef struct MonteCarloIntegration{T} <: AbstractBenchmark{T} - n_samples::Int -end - -name(::MonteCarloIntegration) = "montecarlo" -dims(mci::MonteCarloIntegration) = (mci.n_samples, 1) -function data(mci::MonteCarloIntegration{T}) where {T} - return "Monte Carlo Integration with T=$(T), n_samples=$(mci.n_samples)" -end - -allowed_types(::Type{MonteCarloIntegration}) = cuNumeric.SUPPORTED_FLOAT_TYPES - -total_flops(s::MonteCarloIntegration) = s.n_samples -# Fused broadcast: samples x plus exp.(-x .^ 2), materialized before sum. -# Initialization also needs two arrays: random samples and their scaled output. -# Assumes fusion is enabled; workspace/runtime overhead uses the headroom left -# by mem_frac. This is not a memory estimate for unfused comparison backends. -total_space(s::MonteCarloIntegration{T}) where {T} = 2 * s.n_samples * sizeof(T) - -function estimate_scaling(s::MonteCarloIntegration, P::Integer) - P == 1 && return dims(s) - return (s.n_samples * P, 1) -end - -function fit_one_gpu( - ::Type{MonteCarloIntegration}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {T} - hi = max(8, Int(fld(budget, sizeof(T)))) - n = largest_feasible(8, hi, k -> total_space(MonteCarloIntegration{T}(; n_samples=k)) <= budget) - n === nothing && error("montecarlo does not fit in $(budget) bytes") - return (align8(n), 1) -end - -function initialize(mci::MonteCarloIntegration{T}; mod=cuNumeric) where {T} - # Uniform samples over the integration domain [0, 10]. - x = T(10) .* rand_array(mod, T, mci.n_samples) - GC.gc() - return (x,) -end - -_domain_volume(mci::MonteCarloIntegration{T}) where {T} = T(10) / mci.n_samples -# Dot the negation too: plain `-` materializes the squared array and prevents -# the surrounding exponential from sharing one broadcast with the square. -run!(mci::MonteCarloIntegration, x) = _domain_volume(mci) * sum(exp.(.-(x .^ 2))) - -# n_samples comes in as N; M is unused. -function build_benchmark(::Type{MonteCarloIntegration}, ::Type{T}, N, M) where {T} - return MonteCarloIntegration{T}(; n_samples=N) -end - -function correctness_problem(b::MonteCarloIntegration{T}) where {T} - return MonteCarloIntegration{T}(; n_samples=min(b.n_samples, 1024)) -end - -register_benchmark("montecarlo", MonteCarloIntegration) diff --git a/benchmark/src/benchmarks/poisson_fft.jl b/benchmark/src/benchmarks/poisson_fft.jl deleted file mode 100644 index 34eacb7f1..000000000 --- a/benchmark/src/benchmarks/poisson_fft.jl +++ /dev/null @@ -1,104 +0,0 @@ -# Spectral Poisson on a stack of periodic N×N grids. -# Each of the M right-hand sides is an independent ∇²u = f solve: -# û = fft(f), û *= −1/|k|², u = ifft(û) -# (electrostatics / gravity, not an FFT round-trip). -# -# The transform is over the last two axes, so the leading batch axis may -# partition across GPUs. A single all-axes 2-d FFT cannot. -# Weak scaling: M ∝ P, N held constant across the GPU sweep (changing N -# would change the FFT size). N itself may be RAM-fitted or pinned. - -function _integer_fftfreq(n::Int) - n2 = n ÷ 2 - return iseven(n) ? vcat(0:(n2 - 1), (-n2):-1) : vcat(0:n2, (-n2):-1) -end - -function _poisson_inv_laplacian(::Type{T}, n::Int) where {T} - freq = _integer_fftfreq(n) - invk = Matrix{T}(undef, n, n) - s = T(4 * π^2) - for j in 1:n, i in 1:n - k2 = s * T(freq[i]^2 + freq[j]^2) - invk[i, j] = k2 == 0 ? zero(T) : -inv(k2) - end - return invk -end - -Base.@kwdef struct PoissonFFT{T} <: AbstractBenchmark{T} - N::Int - M::Int -end - -name(::PoissonFFT) = "poisson_fft" -dims(b::PoissonFFT) = (b.N, b.M) -data(b::PoissonFFT{T}) where {T} = "Poisson FFT with T=$(T), N=$(b.N), M=$(b.M)" -allowed_types(::Type{PoissonFFT}) = cuNumeric.SUPPORTED_FLOAT_TYPES - -# 2-d C2C FFT: 5 n log2(n) real flops per 1-d line, 2n lines → 10 n² log2(n). -# Forward + inverse, plus n² complex muls (6 real flops each), times M batches. -function total_flops(b::PoissonFFT) - n = b.N - m = b.M - n2 = n * n - return m * (20 * n2 * log2(n) + 6 * n2) -end - -# fc and work as Complex, kinv as T, plus an FFT workspace ~ one extra complex grid. -function total_space(b::PoissonFFT{T}) where {T} - n2 = b.N * b.N - return (3 * b.M * n2) * sizeof(Complex{T}) + n2 * sizeof(T) -end - -function estimate_scaling(b::PoissonFFT, P::Integer) - P == 1 && return (b.N, b.M) - # Grid N is held across P (FFT is the last two axes). Batch M partitions. - return (b.N, b.M * P) -end - -function fit_one_gpu( - ::Type{PoissonFFT}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {T} - if N_hint !== nothing - N = N_hint - hi = max(1, Int(fld(budget, max(3 * N * N * sizeof(Complex{T}), 1)))) - M = largest_feasible(1, hi, m -> total_space(PoissonFFT{T}(; N=N, M=m)) <= budget) - M === nothing && error("poisson_fft N=$N does not fit in $(budget) bytes") - return (N, M) - else - hi = max(8, Int(floor(sqrt(Float64(budget) / (3 * sizeof(Complex{T})))))) - N = largest_feasible(8, hi, n -> total_space(PoissonFFT{T}(; N=n, M=1)) <= budget) - N === nothing && error("poisson_fft does not fit in $(budget) bytes") - return (align2(N), 1) - end -end - -function initialize(b::PoissonFFT{T}; mod=cuNumeric) where {T} - f = rand_array(mod, T, b.M, b.N, b.N) - fc = astype_array(f, Complex{T}) - work = copy(fc) - kinv = to_backend(mod, reshape(_poisson_inv_laplacian(T, b.N), 1, b.N, b.N)) - GC.gc() - return fc, work, kinv -end - -_trailing_fft_dims(A) = ntuple(i -> i + 1, ndims(A) - 1) - -if CUNUMERIC_BENCH_RUNTIME - _batched_fft!(A::NDArray) = (cuNumeric.batched_fft!(A); A) - _batched_ifft!(A::NDArray) = (cuNumeric.batched_ifft!(A); A) -end -_batched_fft!(A) = (fft!(A, _trailing_fft_dims(A)); A) -_batched_ifft!(A) = (ifft!(A, _trailing_fft_dims(A)); A) - -function run!(::PoissonFFT, fc, work, kinv) - copyto!(work, fc) - _batched_fft!(work) - work .*= kinv - _batched_ifft!(work) - return work -end - -correctness_problem(b::PoissonFFT{T}) where {T} = PoissonFFT{T}(; N=min(b.N, 32), M=1) - -register_benchmark("poisson_fft", PoissonFFT) diff --git a/benchmark/src/benchmarks/tensor_contractions.jl b/benchmark/src/benchmarks/tensor_contractions.jl deleted file mode 100644 index 08ff9a370..000000000 --- a/benchmark/src/benchmarks/tensor_contractions.jl +++ /dev/null @@ -1,123 +0,0 @@ -using TensorOperations - -abstract type AbstractTensorContraction{T} <: AbstractBenchmark{T} end - -Base.@kwdef struct TensorProjection3{T} <: AbstractTensorContraction{T} - N::Int -end - -Base.@kwdef struct TensorContract4{T} <: AbstractTensorContraction{T} - N::Int -end - -name(::TensorProjection3) = "tensor_projection3" -name(::TensorContract4) = "tensor_contract4" -dims(b::AbstractTensorContraction) = (b.N, 1) -function data(b::AbstractTensorContraction{T}) where {T} - return "$(name(b)) with T=$(T), N=$(b.N)" -end - -allowed_types(::Type{<:AbstractTensorContraction}) = cuNumeric.SUPPORTED_FLOAT_TYPES - -# Three rank-3 outputs, each containing an N-term dot product. -total_flops(b::TensorProjection3) = 3 * b.N^3 * (2 * b.N - 1) - -# N^4 output elements, each containing an N^2-term dot product. -total_flops(b::TensorContract4) = b.N^4 * (2 * b.N^2 - 1) - -# opt=true pairwise: T1[n,j,k] and T2[n,m,k] are both N³, plus A, D, B. -total_space(b::TensorProjection3{T}) where {T} = (4 * b.N^3 + b.N^2) * sizeof(T) - -# opt=true is a (N²×N²)×(N²×N²) GEMM: X, Y, C plus one workspace. -total_space(b::TensorContract4{T}) where {T} = 4 * b.N^4 * sizeof(T) - -function estimate_scaling(b::TensorProjection3, P::Integer) - P == 1 && return dims(b) - return (align2(floor(Int, b.N * Float64(P)^(1 / 4))), 1) -end - -function estimate_scaling(b::TensorContract4, P::Integer) - P == 1 && return dims(b) - return (align2(floor(Int, b.N * Float64(P)^(1 / 6))), 1) -end - -function fit_one_gpu( - ::Type{TensorProjection3}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {T} - hi = max(4, Int(floor((Float64(budget) / (4 * sizeof(T)))^(1 / 3)))) - n = largest_feasible(4, hi, k -> total_space(TensorProjection3{T}(; N=k)) <= budget) - n === nothing && error("tensor_projection3 does not fit in $(budget) bytes") - return (n, 1) -end - -function fit_one_gpu( - ::Type{TensorContract4}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {T} - hi = max(4, Int(floor((Float64(budget) / (4 * sizeof(T)))^(1 / 4)))) - n = largest_feasible(4, hi, k -> total_space(TensorContract4{T}(; N=k)) <= budget) - n === nothing && error("tensor_contract4 does not fit in $(budget) bytes") - return (n, 1) -end - -function build_benchmark( - ::Type{TensorProjection3}, ::Type{T}, N, M -) where {T} - return TensorProjection3{T}(; N=N) -end - -function build_benchmark( - ::Type{TensorContract4}, ::Type{T}, N, M -) where {T} - return TensorContract4{T}(; N=N) -end - -function initialize(b::TensorProjection3{T}; mod=cuNumeric) where {T} - A = rand_array(mod, T, b.N, b.N, b.N) - B = rand_array(mod, T, b.N, b.N) - D = zeros_array(mod, T, b.N, b.N, b.N) - GC.gc() - return D, A, B -end - -function initialize(b::TensorContract4{T}; mod=cuNumeric) where {T} - X = rand_array(mod, T, b.N, b.N, b.N, b.N) - Y = rand_array(mod, T, b.N, b.N, b.N, b.N) - C = zeros_array(mod, T, b.N, b.N, b.N, b.N) - GC.gc() - return C, X, Y -end - -function run!(::TensorProjection3, D, A, B) - @tensor opt=true D[n, m, l] = - A[i, j, k] * B[n, i] * B[m, j] * B[l, k] - return D -end - -function run!(::TensorContract4, C, X, Y) - @tensor opt=true C[a, b, c, d] = X[a, i, c, j] * Y[i, b, j, d] - return C -end - -correctness_problem(b::TensorProjection3{T}) where {T} = TensorProjection3{T}(; N=min(b.N, 4)) -correctness_problem(b::TensorContract4{T}) where {T} = TensorContract4{T}(; N=min(b.N, 4)) -function correctness_atol_rtol(::AbstractTensorContraction, ::Type{T}) where {T} - tol = T === Float32 ? 2.0f-4 : 1e-11 - return tol, tol -end - -function benchmark_backend_label( - ::AbstractTensorContraction, backend::String, default::String -) - return backend == "cudajl" ? "TensorOperations.jl / cuTENSOR" : default -end - -function benchmark_backend_save_as( - ::AbstractTensorContraction, backend::String, default::String -) - return backend == "cudajl" ? "tensoroperations_cuda" : default -end - -register_benchmark("tensor_projection3", TensorProjection3) -register_benchmark("tensor_contract4", TensorContract4) diff --git a/benchmark/src/core.jl b/benchmark/src/core.jl deleted file mode 100644 index aee95efbd..000000000 --- a/benchmark/src/core.jl +++ /dev/null @@ -1,310 +0,0 @@ -using Printf -using ProgressMeter: ProgressMeter -using Statistics - -""" -- `n_warmup::Int` : Number of warmup steps. These are not timed. Intended - to avoid pre-compilation cost being timed. -- `n_iter::Int` : Number of completed iterations per trial. Gray–Scott instead - queues timesteps and synchronizes at the trial boundaries. -- `n_trial::Int` : Number of independent trials to run. Timing is restarted and - legate in between each trial. Sets number of datapoints used to estimated - standard deviations/errors. -- `n_gpu::Int` : The number of GPUs used by legate. Set through the LEGATE_CONFIG, - this value is just bookkeeping. -- `check_correctness::Bool` : If true and `n_gpu == 1`, compare a tiny cuNumeric - result against CUDA.jl before timing. CUDA.jl / multi-GPU / Python skip. -- `n_correctness_iter::Int` : Steps to run for that single correctness check. -""" -Base.@kwdef struct GlobalSettings - n_warmup::Int # Number of warmup steps, where timing is not done. - n_iter::Int # Number of iterations to run per trial - n_trial::Int = 1 # Number of independent trials to run. Benchmark - n_gpu::Int = 0 - cupynumeric::Bool = false # also run baselines under cupynumeric for comparison - cuda::Bool = false # also run under CUDA.jl for comparison (single-GPU only) - check_correctness::Bool = false - n_correctness_iter::Int = 5 - auto_size::Bool = false - mem_frac::Float64 = 0.5 -end - -######################################### - -abstract type AbstractBenchmark{T} end - -# Independent problems must finish before the next repetition is submitted. -fence_each_iteration(::AbstractBenchmark) = true -benchmark_synchronize() = cuNumeric.issue_execution_fence(; block=true) - -# True when this file is included after `using cuNumeric` (the worker). The -# orchestrator includes the same kernel files for types / `total_space` / -# `estimate_scaling` without loading Legion; those files skip `@accelerate` -# and `::NDArray` methods in that case. -const CUNUMERIC_BENCH_RUNTIME = isdefined(@__MODULE__, :cuNumeric) - -# Interface each benchmark implements (see benchmarks/gemm.jl for a template). -function name end -function dims end -function data end -function allowed_types end -function total_flops end -function initialize end -function run! end - -include("autosize.jl") - -function include_benchmarks() - dir = joinpath(@__DIR__, "benchmarks") - for file in sort(filter(f -> endswith(f, ".jl"), readdir(dir; join=true))) - Base.include(@__MODULE__, file) - end - return nothing -end - -total_space(b::AbstractBenchmark) = - error("total_space not defined for $(typeof(b)); add a method in its benchmark file") - -function estimate_scaling(b::AbstractBenchmark, P::Integer) - P < 1 && throw(ArgumentError("P must be ≥ 1, got $P")) - P == 1 && return map(Int, dims(b)) - return error( - "estimate_scaling not defined for $(typeof(b)); add a method in its benchmark file", - ) -end - -function fit_one_gpu( - ::Type{B}, ::Type{T}; - budget::Int, N_hint=nothing, M_hint=nothing, -) where {B,T} - return error("fit_one_gpu not defined for $B; add a method in its benchmark file") -end - -# Internal adapter for benchmark generators that share a quoted step body. -function _define_accelerated_definition(signature, body, form=:function) - if form === :function - return cuNumeric._accelerate_expand(Expr(:function, signature, body), @__MODULE__) - end - scoped = form === :begin ? Expr(:block, body.args...) : Expr(:let, body) - call = Expr(:macrocall, Symbol("@accelerate"), LineNumberNode(0), scoped) - return Expr(:function, signature, Expr(:block, Base.macroexpand(@__MODULE__, call))) -end - -# Maps a benchmarks.toml table name to its benchmark type. Each benchmark file -# registers itself via `register_benchmark`. -const BENCHMARKS = Dict{String,Type}() -function register_benchmark(key::AbstractString, ::Type{B}) where {B<:AbstractBenchmark} - return BENCHMARKS[key] = B -end - -benchmark_backend_label(::AbstractBenchmark, backend::String, default::String) = default -benchmark_backend_save_as(::AbstractBenchmark, backend::String, default::String) = default - -function build_benchmark(::Type{B}, ::Type{T}, N, M) where {B<:AbstractBenchmark,T} - return B{T}(; N=N, M=M) -end - -# Optional hooks for the generic CUDA.jl check (initialize + run!). -correctness_problem(b::AbstractBenchmark) = b -correctness_iters(::AbstractBenchmark, gs::GlobalSettings) = 1 -cuda_runnable(b::AbstractBenchmark) = b -correctness_result(::AbstractBenchmark, state, out) = out === nothing ? state : out -correctness_atol_rtol(::AbstractBenchmark, ::Type{T}) where {T} = ref_atol_rtol(T) - -######################################### - -# Per-trial timings for one benchmark. `times_ms[i]`/`gflops[i]` are the mean -# over `n_iter` iterations for trial `i`; the spread across trials gives stddev. -# `correctness` is one of "pass", "fail", "skipped" — checked once per config. -struct BenchmarkResult{B<:AbstractBenchmark} - times_ms::Vector{Float64} - gflops::Vector{Float64} - benchmark::B - correctness::String -end - -# CUDA.jl 6: the worker may pass `CUDA` or `CUDACore` as `mod`. -is_cuda_backend(mod) = nameof(mod) === :CUDA || nameof(mod) === :CUDACore - -# Timed CUDA.jl is never the thing we check. Oracle compare is cuNumeric vs CUDA -# on a single GPU (the cuNumeric worker loads CUDA for the tiny problem). -function correctness_applies(gs::GlobalSettings, mod) - is_cuda_backend(mod) && return false - return gs.n_gpu == 1 -end - -function cuda_backend() - for (id, mod) in Base.loaded_modules - id.name == "CUDA" && return mod - end - return error("CUDA.jl must be loaded to check cuNumeric against CUDA.jl") -end - -# 1–4D (or more) constructors. `mod` is cuNumeric or CUDA; both expose -# rand/zeros/ones(::Type, dims...). Host `Array` is only a seed for to_backend. -rand_array(mod, ::Type{T}, dims::Integer...) where {T} = mod.rand(T, dims...) -rand_array(mod, ::Type{T}, dims::Tuple) where {T} = rand_array(mod, T, dims...) -zeros_array(mod, ::Type{T}, dims::Integer...) where {T} = mod.zeros(T, dims...) -zeros_array(mod, ::Type{T}, dims::Tuple) where {T} = zeros_array(mod, T, dims...) -ones_array(mod, ::Type{T}, dims::Integer...) where {T} = mod.ones(T, dims...) -ones_array(mod, ::Type{T}, dims::Tuple) where {T} = ones_array(mod, T, dims...) - -# Avoid `::NDArray` in the signature so the orchestrator can include this file -# without loading cuNumeric. -function astype_array(A, ::Type{T}) where {T} - nameof(typeof(A)) === :NDArray && return cuNumeric.as_type(A, T) - return T.(A) -end - -to_host(A::Array) = A -to_host(x::Number) = x -to_host(A) = Array(A) - -function to_backend(mod, A::AbstractArray) - h = A isa Array ? A : Array(A) - mod === cuNumeric && return NDArray(h) - is_cuda_backend(mod) && return mod.CuArray(h) - return h -end - -to_backend_state(mod, x::AbstractArray) = to_backend(mod, x) -to_backend_state(mod, x::Tuple) = map(s -> to_backend_state(mod, s), x) -to_backend_state(mod, x) = x - -function ref_atol_rtol(::Type{T}; atol=nothing, rtol=nothing) where {T} - default = T <: Float32 ? 1.0f-3 : 1e-10 - return something(atol, default), something(rtol, default) -end - -function isapprox_ref(actual, expected, ::Type{T}; atol=nothing, rtol=nothing) where {T} - at, rt = ref_atol_rtol(T; atol, rtol) - return isapprox(to_host(actual), to_host(expected); atol=at, rtol=rt) -end - -# Full cuNumeric reductions return 0-D arrays, while CUDA returns scalars. -# Define this only in the worker; the orchestrator does not load cuNumeric. -if CUNUMERIC_BENCH_RUNTIME - @eval function isapprox_ref( - actual::AbstractArray{<:Any,0}, expected::Number, ::Type{T}; kwargs... - ) where {T} - value = cuNumeric.@allowscalar actual[] - return isapprox_ref(value, expected, T; kwargs...) - end -end - -_all_approx(a, b, ::Type{T}; kwargs...) where {T} = isapprox_ref(a, b, T; kwargs...) -function _all_approx(a::Tuple, b::Tuple, ::Type{T}; kwargs...) where {T} - length(a) == length(b) || return false - return all(_all_approx(x, y, T; kwargs...) for (x, y) in zip(a, b)) -end - -# Host-seed once, upload to both backends, initialize/run! the same way as timing. -function check_benchmark_correctness( - b::AbstractBenchmark{T}, gs::GlobalSettings; mod=cuNumeric -) where {T} - tiny = correctness_problem(b) - seed = initialize(tiny; mod=Base) - atol, rtol = correctness_atol_rtol(b, T) - nstep = correctness_iters(tiny, gs) - return check_vs_cuda(T; atol, rtol) do backend - kernel = backend === cuNumeric ? tiny : cuda_runnable(tiny) - state = to_backend_state(backend, seed) - out = nothing - for _ in 1:nstep - out = run!(kernel, state...) - end - return correctness_result(kernel, state, out) - end -end - -# `f(mod)` runs the tiny problem on one backend and returns the value(s) to compare. -function check_vs_cuda(f, ::Type{T}; atol=nothing, rtol=nothing) where {T} - got = f(cuNumeric) - ref = f(cuda_backend()) - return _all_approx(got, ref, T; atol, rtol) ? "pass" : "fail" -end - -# One timed trial: warmup, then time `n_iter` iterations of `run!`. -function _trial( - b::AbstractBenchmark, gs::GlobalSettings; - mod=cuNumeric, clock=get_time_microseconds, synchronize=benchmark_synchronize, -) - GC.gc(true) - state = initialize(b; mod=mod) - fence_each = fence_each_iteration(b) - - start_time = nothing - for idx in 1:(gs.n_warmup + gs.n_iter) - if idx == gs.n_warmup + 1 - start_time = clock() - end - run!(b, state...) - fence_each && synchronize() - end - total_time_μs = clock() - start_time - - mean_time_ms = total_time_μs / (gs.n_iter * 1e3) - gflops = total_flops(b) / (mean_time_ms * 1e6) - return mean_time_ms, gflops -end - -# Run `n_trial` independent trials and collect their per-trial measurements. -# Correctness (if enabled) runs once before timing, not per trial/iteration. -function run_benchmark( - b::AbstractBenchmark, gs::GlobalSettings; - mod=cuNumeric, clock=get_time_microseconds, synchronize=benchmark_synchronize, -) - correctness = "skipped" - if gs.check_correctness - if correctness_applies(gs, mod) - println("Checking correctness against CUDA.jl on a small problem...") - flush(stdout) - correctness = check_benchmark_correctness(b, gs; mod=mod) - else - correctness = "skipped" - end - end - - println("Correctness: $(correctness)") - println( - "Starting $(gs.n_trial) trials; each includes initialization, " * - "$(gs.n_warmup) warmups, and $(gs.n_iter) timed iterations; " * - (fence_each_iteration(b) ? "per-iteration synchronization." : "batch synchronization."), - ) - flush(stdout) - times_ms = Float64[] - gflops = Float64[] - progress = ProgressMeter.Progress(gs.n_trial; dt=0.0, desc="$(name(b)) trials: ") - ProgressMeter.update!(progress, 0) - for trial in 1:gs.n_trial - t, g = _trial(b, gs; mod=mod, clock=clock, synchronize=synchronize) - push!(times_ms, t) - push!(gflops, g) - # Update only after _trial has stopped its clock; never inside the kernel loop. - ProgressMeter.next!(progress; showvalues=[ - ("Completed trials", "$(trial)/$(gs.n_trial)"), - ("Last trial mean (ms/iteration)", t), - ("Last trial GFLOP/s", g), - ]) - end - return BenchmarkResult(times_ms, gflops, b, correctness) -end - -_std(x) = length(x) > 1 ? std(x) : 0.0 - -function save_result(br::BenchmarkResult, gpus; mod::String="cunumeric") - N, M = dims(br.benchmark) - results = get(ENV, "CUNUMERIC_BENCH_RESULTS_DIR", joinpath(@__DIR__, "..", "results")) - path = joinpath(results, "$(name(br.benchmark))_$(mod).csv") - mkpath(dirname(path)) - open(path, "a") do io - for trial in eachindex(br.times_ms) - # correctness is per-config; repeated on each trial row for CSV joins - @printf( - io, "%s,%d,%d,%d,%d,%.6f,%.6f,%s\n", - mod, gpus, N, M, trial, - br.times_ms[trial], br.gflops[trial], br.correctness, - ) - end - end -end diff --git a/benchmark/src/memory.jl b/benchmark/src/memory.jl deleted file mode 100644 index 28eac5f56..000000000 --- a/benchmark/src/memory.jl +++ /dev/null @@ -1,147 +0,0 @@ -# Analytical estimates. This file is included after the benchmark definitions. -# All byte counts use BigInt so preflight cannot wrap on oversized dimensions. -Base.@kwdef struct MemoryContext - backend::Symbol = :cunumeric - fusion::Bool = true - gpus::Int = 1 - steps::Int = 1 - # Explicit upper bound per GPU for opaque native-library scratch/packing. - workspace_bytes::Union{Nothing,Int} = nothing -end - -struct MemoryEstimate - initialization::BigInt - iteration::BigInt - workspace::BigInt - explanation::String -end -peak_bytes(m::MemoryEstimate) = max(m.initialization, m.iteration) + m.workspace - -function library_workspace(b, c) - c.workspace_bytes === nothing && error( - "$(name(b)) / $(c.backend): native workspace bound is unknown. " * - "Set workspace_bytes to a verified per-GPU upper bound for this backend/library " * - "configuration; autosizing will not guess or probe after an OOM.", - ) - c.workspace_bytes >= 0 || error("workspace_bytes must be nonnegative") - return big(c.workspace_bytes) -end - -function validate_memory_context(b::AbstractBenchmark{T}, c) where {T} - c.gpus > 0 || error("GPU count must be positive") - c.steps > 0 || error("Trial steps must be positive") - c.backend in (:cunumeric, :cudajl, :cupynumeric) || error("Unknown backend $(c.backend)") - c.backend == :cudajl && c.gpus != 1 && error("CUDA.jl supports one GPU only") - endswith(name(b),"_accelerated") && c.backend != :cunumeric && error("$(name(b)) is cuNumeric-only") - T in (Float32, Float64) || error("Memory accounting currently supports Float32 and Float64; got $T") - all(>(0), dims(b)) || error("Problem dimensions must be positive") -end - -# A conservative slab bound: rounding a partition up cannot undercount uneven -# dimensions. Stencil halos are counted separately below. DMD is never divided. -slab(n, tail, p) = cld(big(n), p) * big(tail) -random_peak(elements, ::Type{T}, backend) where {T} = - elements * (backend == :cupynumeric ? sizeof(Float64) + sizeof(T) : sizeof(T)) - -function memory_estimate(b::MonteCarloIntegration{T}, c::MemoryContext) where {T} - validate_memory_context(b, c) - e = cld(big(b.n_samples), c.gpus) - bytes = e * sizeof(T) - init = max(2bytes, random_peak(e, T, c.backend)) - # The unfused NDArray copy path allocates an outer destination before - # recursively materializing operations; count it as well as two temporaries. - arrays = c.backend == :cupynumeric ? 3 : c.backend == :cudajl || c.fusion ? 2 : 4 - # Julia has tracing GC, not Python's reference counting. The returned - # broadcast output is not explicitly destroyed by this baseline kernel. - # This assumes the fully dotted expression in montecarlo.jl. On the unfused - # path nested broadcast temporaries are explicitly destroyed by the runtime; - # an undotted operation would instead escape that cleanup and need its own - # per-iteration retention allowance. - # Bound its retention over the complete trial instead of assuming a GC. - retained = c.backend == :cunumeric || c.backend == :cudajl ? c.steps-1 : 0 - return MemoryEstimate(init, (arrays + retained)*bytes, 0, - "samples + broadcast output; unfused temporaries; up to $retained prior Julia outputs awaiting GC; random dtype conversion") -end - -function memory_estimate(b::GEMM{T}, c::MemoryContext) where {T} - validate_memory_context(b, c) - # Without a mapper-specific replication guarantee count the complete inputs - # on each GPU. This also covers broadcast operands in distributed matmul. - a = big(b.N) * b.M * sizeof(T) - out = big(b.N)^2 * sizeof(T) - init = max(2a + out, a + random_peak(big(b.N)*b.M, T, c.backend)) - return MemoryEstimate(init, 2a + out, library_workspace(b, c), - "A, B, C; full operands per GPU (replication-safe); native packing/workspace") -end - -function memory_estimate(b::AbstractGrayScott{T}, c::MemoryContext) where {T} - validate_memory_context(b, c) - b.N >= 3 && b.M >= 3 || error("Gray-Scott requires N and M >= 3") - # Count full grids until the mapper's halo/replication contract is bounded. - # Local expressions can retain parents; never treat a slice as a free copy. - grid = big(b.N) * b.M * sizeof(T) - interior = big(max(b.N-2, 0)) * max(b.M-2, 0) * sizeof(T) - init = 4grid + random_peak(big(min(150,b.N,b.M))^2, T, c.backend) - # Four named RHS results plus an assignment output. Hard-scope acceleration - # may eliminate these, but this remains a valid upper bound for every form. - # Do not assume a lower peak solely from the @accelerate spelling. - fused = c.backend == :cudajl || (c.backend == :cunumeric && c.fusion) - # Unfused Laplacian: first branch survives evaluation of second branch; - # include outer destination and intermediate binary operands. - temps = fused ? 5 : 8 - variant = name(b) - hard_scope = b isa Union{GrayScottAccelerated,GrayScottFunctionAccelerated,GrayScottLetAccelerated} - # Hard scopes insert explicit last-use destruction whether fusion is on or - # off. Baseline/begin/expression leave the named results for tracing GC. - retained = c.backend == :cunumeric && hard_scope || c.backend == :cupynumeric ? 0 : 6*(c.steps-1) - return MemoryEstimate(init, 4grid + (temps+retained)*interior, 0, - "$variant: four persistent grids + $temps active interior buffers + $retained prior buffers awaiting GC; full-parent bound; fusion=$(c.fusion)") -end - -function memory_estimate(b::AbstractDMD{T}, c::MemoryContext) where {T} - validate_memory_context(b, c) - b.M >= 2 && b.N >= b.M-1 || error("DMD requires M >= 2 and N >= M-1") - n, m, r = big(b.N), big(b.M-1), big(_dmd_rank(b)) - # Retained X, copies/views X1/X2, full thin U/Vt/S, projected products, - # transpose copies, eigen inputs/outputs and complex lift. Count retained - # parents even when only r columns are returned from _dmd_factors. - persistent = n*big(b.M) - factors = 3n*m + m*m + m - project = 2m*r + 4n*r + 2r*r + r - complex_lift = 2*(2n*r + 2r*r + r) - # The factorization/output wrappers escape the lifetime rewriter; native - # arrays can remain until GC between trials on Julia backends. - retained_steps = c.backend == :cupynumeric ? 1 : c.steps - iteration = (persistent + retained_steps*(factors + project + complex_lift))*sizeof(T) - init = random_peak(n*big(b.M), T, c.backend) - return MemoryEstimate(init, iteration, library_workspace(b,c), - "full single-task SVD on one GPU (P does not divide memory); factors, retained parents, projections, complex lift") -end - -function memory_estimate(b::PoissonFFT{T}, c::MemoryContext) where {T} - validate_memory_context(b, c) - e = cld(big(b.M), c.gpus)*big(b.N)^2 - realbytes, complexbytes = e*sizeof(T), e*sizeof(Complex{T}) - kinv = big(b.N)^2*sizeof(T) - # Python FFT precision is conservatively bounded by complex128. - pycomplex = e*sizeof(ComplexF64) - init = max(random_peak(e,T,c.backend), realbytes+2complexbytes+kinv) - iteration = c.backend == :cupynumeric ? realbytes+2pycomplex+kinv : 2complexbytes+kinv - return MemoryEstimate(init, iteration, library_workspace(b,c), - "batched grids + replicated inverse Laplacian; Python out-of-place FFT precision; native FFT workspace") -end - -function memory_estimate(b::AbstractTensorContraction{T}, c::MemoryContext) where {T} - validate_memory_context(b,c) - # Full operands and intermediates, without assuming distributed packing. - n = big(b.N) - if b isa TensorProjection3 - live = (4n^3+n^2)*sizeof(T) - init = max((2n^3+n^2)*sizeof(T), random_peak(n^3,T,c.backend)) - else - live = 3n^4*sizeof(T) - init = max(live, n^4*sizeof(T)+random_peak(n^4,T,c.backend)) - end - return MemoryEstimate(init, live, library_workspace(b,c), - "full contraction inputs/outputs and pairwise intermediates; native packing/workspace counted separately") -end diff --git a/benchmark/src/parse_benchmarks.jl b/benchmark/src/parse_benchmarks.jl deleted file mode 100644 index 83f3d0ea4..000000000 --- a/benchmark/src/parse_benchmarks.jl +++ /dev/null @@ -1,213 +0,0 @@ -using TOML - -""" -One benchmark invocation parsed from `benchmarks.toml`. `name` selects the -benchmark type from `BENCHMARKS`; `T` is the element type (e.g. "Float32"); -`args` are the sizes (`N M`) when pinned. When `autosize` is true, `args` is -unused and `N_hint` / `M_hint` feed `fit_one_gpu`. -""" -struct BenchmarkSpec - name::String - T::String - gpus::Int - cpus::Int - fusion::Bool - cuda::Bool - n_warmup::Int - n_iter::Int - n_trial::Int - args::Vector{Int} - autosize::Bool - N_hint::Union{Int,Nothing} - M_hint::Union{Int,Nothing} -end - -# A field may be a scalar or a list. -aslist(x) = x isa AbstractVector ? collect(x) : [x] - -# `fusion` accepts a bool or "on"/"off" (or a list of these). -function parse_fusion(x) - x isa Bool && return x - s = lowercase(string(x)) - s in ("on", "true") && return true - s in ("off", "false") && return false - return error("fusion must be on/off (or true/false); got $(repr(x))") -end - -# Value of a zipped field for sweep position `i`. length==1 field broadcasts. -sweep_value(field, i) = length(field) == 1 ? field[1] : field[i] - -# Number of positions in the sweep. Every multi-element field must agree on length; -# length==1 fields broadcast and don't constrain it. -function sweep_length(name, fields) - lengths = [length(field) for (_, field) in fields if length(field) > 1] - isempty(lengths) && return 1 - allequal(lengths) || error( - "benchmark '$(name)': zipped fields gpus/cpus/N/M must share one length " * - "or be scalar; got " * join(("$k=$(length(v))" for (k, v) in fields), ", "), - ) - return first(lengths) -end - -# Names of the `[[name]]` blocks in the order they appear in the file. TOML.jl -# parses into an unordered Dict, so we scan the source to preserve run order. -function declared_order(path) - order = String[] - for line in eachline(path) - header = strip(line) - startswith(header, "[[") && endswith(header, "]]") || continue - name = strip(header[3:(end - 2)]) - name in order || push!(order, name) # if not in list, push to ordered list - end - return order -end - -function size_field(raw) - raw === nothing && return (:omitted, Int[]) - vals = collect(aslist(raw)) - autos = [is_auto_size(v) for v in vals] - if any(autos) - all(autos) || error("cannot mix auto and numeric sizes in one field; got $(repr(raw))") - return (:auto, Int[]) - end - return (:pinned, Int[Int(v) for v in vals]) -end - -function parse_config(path; only=nothing, fusion_override=nothing) - raw = TOML.parsefile(path) - - g = raw["Global"] - global_settings = GlobalSettings(; - n_warmup=g["n_warmup"], n_iter=g["n_iter"], n_trial=get(g, "n_trial", 1), - cupynumeric=get(g, "cupynumeric", false), - cuda=get(g, "cuda", false), - check_correctness=get(g, "check_correctness", false), - n_correctness_iter=get(g, "n_correctness_iter", 5), - auto_size=get(g, "auto_size", false), - mem_frac=Float64(get(g, "mem_frac", 0.5)), - ) - - specs = BenchmarkSpec[] - selected = only === nothing ? nothing : Set(split(only, ',')) - if selected !== nothing - groups = Dict(parse_plot_groups(path)) - selected = Set(vcat([get(groups, s, [s]) for s in selected]...)) - unknown = setdiff(selected, Set(declared_order(path))) - isempty(unknown) || error("Unknown benchmark selection: $(join(unknown, ", "))") - end - for name in declared_order(path) - selected !== nothing && name ∉ selected && continue - entries = raw[name] - entries isa AbstractVector || continue - for e in entries - types = aslist(get(e, "T", "Float32")) - gpus = aslist(e["gpus"]) - cpus = aslist(e["cpus"]) - fusion = aslist(fusion_override === nothing ? get(e, "fusion", true) : fusion_override) - nmode, nvals = size_field(get(e, "N", nothing)) - mmode, mvals = size_field(get(e, "M", nothing)) - cuda = get(e, "cuda", global_settings.cuda) - n_warmup = get(e, "n_warmup", global_settings.n_warmup) - n_iter = get(e, "n_iter", global_settings.n_iter) - n_trial = get(e, "n_trial", global_settings.n_trial) - block_auto = get(e, "auto_size", global_settings.auto_size) - - n_auto = nmode == :omitted || nmode == :auto - m_auto = mmode == :auto - use_auto = block_auto && (n_auto || m_auto) - if use_auto - nmode == :pinned && length(nvals) != 1 && error( - "benchmark '$(name)': autosize with pinned N requires a scalar N", - ) - mmode == :pinned && length(mvals) != 1 && error( - "benchmark '$(name)': autosize with pinned M requires a scalar M", - ) - N_hint = nmode == :pinned ? nvals[1] : nothing - M_hint = mmode == :pinned ? mvals[1] : nothing - n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus]) - else - nmode == :auto && error("benchmark '$(name)': N = \"auto\" requires auto_size = true") - mmode == :auto && error("benchmark '$(name)': M = \"auto\" requires auto_size = true") - nmode == :omitted && error( - "benchmark '$(name)' is missing N (set N or enable auto_size)", - ) - mmode == :omitted && (mvals = [1]) - N_hint = nothing - M_hint = nothing - n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus, "N" => nvals, "M" => mvals]) - end - - for T in types, fuse in fusion, i in 1:n - args = use_auto ? Int[0, 0] : Int[sweep_value(nvals, i), sweep_value(mvals, i)] - push!( - specs, - BenchmarkSpec( - name, - string(T), - Int(sweep_value(gpus, i)), - Int(sweep_value(cpus, i)), - parse_fusion(fuse), - cuda, - n_warmup, - n_iter, - n_trial, - args, - use_auto, - N_hint, - M_hint, - ), - ) - end - end - end - - return global_settings, specs -end - -# Group members that share a figure. Explicit `[plot.groups]` lists first; -# every other `[[benchmark]]` table is a singleton group named after itself. -# Group order follows the first listed member in the file. -function parse_plot_groups(path) - raw = TOML.parsefile(path) - declared = declared_order(path) - plot = get(raw, "plot", Dict{String,Any}()) - groups_tbl = get(plot, "groups", Dict{String,Any}()) - - explicit = Dict{String,Vector{String}}() - for (gname, members) in groups_tbl - explicit[string(gname)] = String[string(m) for m in aslist(members)] - end - - assigned = Set{String}() - groups = Pair{String,Vector{String}}[] - remaining = copy(explicit) - for name in declared - name in assigned && continue - gname = nothing - for (g, members) in remaining - if name in members - gname = g - break - end - end - if gname !== nothing - members = remaining[gname] - push!(groups, gname => members) - union!(assigned, members) - delete!(remaining, gname) - else - push!(groups, name => [name]) - push!(assigned, name) - end - end - for (gname, members) in remaining - push!(groups, gname => members) - end - return groups -end - -function plot_baseline(members::Vector{String}) - i = findfirst(m -> endswith(m, "_baseline"), members) - i !== nothing && return members[i] - return first(members) -end diff --git a/benchmark/src/planning.jl b/benchmark/src/planning.jl deleted file mode 100644 index 801cb7edc..000000000 --- a/benchmark/src/planning.jl +++ /dev/null @@ -1,140 +0,0 @@ -# Shared sizing: explicit plans, no runtime probes and no mutable sizing cache. -struct PlannedRun - spec::BenchmarkSpec - backend::Symbol - N::Int - M::Int - memory::MemoryEstimate -end - -function workspace_bound(raw, name, backend) - entry = get(get(raw, "workspace", Dict()), name, Dict()) - value = get(entry, string(backend), nothing) - value === nothing && return nothing - value isa Integer && value >= 0 || error("workspace.$name.$backend must be a nonnegative byte count") - return Int(value) -end - -function contexts(spec, gs, raw) - backends = Symbol[:cunumeric] - if !endswith(spec.name, "_accelerated") - spec.cuda && spec.gpus == 1 && push!(backends, :cudajl) - gs.cupynumeric && push!(backends, :cupynumeric) - end - return [MemoryContext(; backend, fusion=spec.fusion, gpus=spec.gpus,steps=spec.n_warmup+spec.n_iter, - workspace_bytes=workspace_bound(raw, spec.name, backend)) for backend in backends] -end - -function validate_spec(s) - haskey(BENCHMARKS,s.name) || error("Unknown benchmark $(s.name)") - s.gpus > 0 && s.cpus >= 0 || error("GPU count must be positive and CPU count nonnegative") - s.n_iter > 0 && s.n_trial > 0 && s.n_warmup >= 0 || error("Invalid trial/iteration count") - for hint in (s.N_hint,s.M_hint) - hint === nothing || hint > 0 || error("Pinned dimensions must be positive") - end -end - -function dimensions_at(s, baseline) - !s.autosize && return Tuple(s.args) - n,m = baseline - b = build_benchmark(BENCHMARKS[s.name],parse_bench_type(s.T),n,m) - result = estimate_scaling(b,s.gpus) - result === nothing && error("$(s.name) cannot scale this baseline to $(s.gpus) GPUs") - return result -end - -function baseline_shape(s, k) - B = BENCHMARKS[s.name] - if B <: AbstractDMD - return (something(s.N_hint,k), something(s.M_hint,DEFAULT_DMD_M)) - elseif B <: PoissonFFT - return s.N_hint === nothing ? (k,something(s.M_hint,1)) : (s.N_hint,k) - elseif B <: MonteCarloIntegration || B <: AbstractTensorContraction - s.M_hint === nothing || s.M_hint == 1 || error("$(s.name) requires M=1") - return (something(s.N_hint,k),1) - else - return (something(s.N_hint,k),something(s.M_hint,k)) - end -end - -function candidate_runs(specs,gs,raw,baseline) - runs = PlannedRun[] - seen = Set{Any}() - for s in specs - n,m = dimensions_at(s,baseline) - b = build_benchmark(BENCHMARKS[s.name],parse_bench_type(s.T),n,m) - for c in contexts(s,gs,raw) - # Comparison backends have no fusion setting and run only once. - key = (s.name,s.T,s.gpus,s.cpus,n,m,c.backend, - c.backend == :cunumeric ? s.fusion : nothing,s.n_iter,s.n_warmup,s.n_trial) - key in seen && continue - push!(seen,key) - push!(runs,PlannedRun(s,c.backend,n,m,memory_estimate(b,c))) - end - end - # Multiple explicit blocks must not create misleading overlays. - sizes = Dict{Tuple{String,Int},Tuple{Int,Int}}() - for r in runs - key = (r.spec.T,r.spec.gpus) - previous = get!(sizes,key,(r.N,r.M)) - previous == (r.N,r.M) || error("Comparison group has incompatible pinned dimensions at $(r.spec.gpus) GPUs") - end - return runs -end - -function plan_runs(specs,gs,raw,groups,budget::Integer) - isempty(specs) && error("No benchmarks selected") - foreach(validate_spec,specs) - group_for = Dict(member=>group for (group,members) in groups for member in members) - buckets = Dict{Any,Vector{BenchmarkSpec}}() - order = Any[] - for s in specs - key = (get(group_for,s.name,s.name),s.T) - if !haskey(buckets,key) - buckets[key] = BenchmarkSpec[] - push!(order,key) - end - push!(buckets[key],s) - end - planned = PlannedRun[] - for key in order - members = buckets[key] - autos = filter(s->s.autosize,members) - if isempty(autos) - runs = candidate_runs(members,gs,raw,nothing) - all(r->peak_bytes(r.memory)<=budget,runs) || error("Pinned size exceeds memory budget in $(key[1])") - append!(planned,runs) - continue - end - length(autos)==length(members) || error("Do not mix pinned and automatic sizes in comparison group $(key[1])") - hints = unique((s.N_hint,s.M_hint) for s in members) - length(hints)==1 || error("Incompatible size constraints in comparison group $(key[1])") - s = first(members) - B = BENCHMARKS[s.name] - quantum = B <: PoissonFFT && s.N_hint !== nothing ? 1 : B <: AbstractTensorContraction || B <: PoissonFFT ? 2 : 8 - minimum_n = B <: AbstractDMD ? max(something(s.M_hint,DEFAULT_DMD_M)*DMD_TALL_RATIO,8) : quantum - lo = cld(minimum_n,quantum) - # At least one value of T must fit per candidate. Binary search uses - # BigInt memory formulas and an Int-safe bound on scaled dimensions. - hi = max(lo,Int(min(budget÷sizeof(parse_bench_type(s.T)),typemax(Int)÷(8maximum(x.gpus for x in members))))÷quantum) - make(k) = candidate_runs(members,gs,raw,baseline_shape(s,k*quantum)) - # Evaluate once before search to surface unsupported model errors. - all(r->peak_bytes(r.memory)<=budget,make(lo)) || error("Minimum problem does not fit in $(key[1])") - best = largest_feasible(lo,hi,k->all(r->peak_bytes(r.memory)<=budget,make(k))) - best === nothing && error("No feasible size for $(key[1])") - append!(planned,make(best)) - end - return planned -end - -function print_plan(runs,budget) - println("Per-GPU budget: $budget bytes; sizes are shared within each comparison group.") - for r in runs - m = r.memory - println("$(r.spec.name) / $(r.backend) / $(r.spec.T) fusion=$(r.spec.fusion) GPUs=$(r.spec.gpus) N=$(r.N) M=$(r.M)") - println(" initialization=$(m.initialization), iteration=$(m.iteration), workspace=$(m.workspace), peak=$(peak_bytes(m)) bytes") - println(" $(m.explanation)") - end - limiting = runs[argmax([peak_bytes(r.memory) for r in runs])] - println("Largest planned peak: $(limiting.spec.name) / $(limiting.backend) / $(limiting.spec.gpus) GPUs") -end diff --git a/benchmark/src/result_rows.jl b/benchmark/src/result_rows.jl deleted file mode 100644 index 08905de54..000000000 --- a/benchmark/src/result_rows.jl +++ /dev/null @@ -1,47 +0,0 @@ -struct Row - gpus::Int - N::Int - M::Int - time_ms::Float64 - thr::Float64 -end - -function load_runs(path) - rows = Row[] - for line in eachline(path) - isempty(strip(line)) && continue - f = split(line, ',') - length(f) == 8 || error("Invalid result row in $path") - push!(rows,Row(parse(Int,f[2]),parse(Int,f[3]),parse(Int,f[4]),parse(Float64,f[6]),parse(Float64,f[7]))) - end - isempty(rows) && return Vector{Row}[] - runs = [Row[]] - for (i,r) in enumerate(rows) - i > 1 && r.gpus < rows[i-1].gpus && push!(runs,Row[]) - push!(runs[end],r) - end - return runs -end - -function aggregate(rows) - by = Dict{Int,Vector{Row}}() - for r in rows - push!(get!(by,r.gpus,Row[]),r) - end - for (g,rs) in by - length(unique((r.N,r.M) for r in rs)) == 1 || error("Cannot combine different dimensions at $g GPUs; select one invocation") - end - sd(x) = length(x)>1 ? std(x) : 0.0 - return [(gpus=g,N=first(by[g]).N,M=first(by[g]).M, - t=mean(getfield.(by[g],:time_ms)),tsd=sd(getfield.(by[g],:time_ms)), - h=mean(getfield.(by[g],:thr)),hsd=sd(getfield.(by[g],:thr))) for g in sort(collect(keys(by)))] -end - -function validate_series_sizes(series) - sizes = Dict{Int,Tuple{Int,Int}}() - for s in series, r in s.agg - previous = get!(sizes,r.gpus,(r.N,r.M)) - previous == (r.N,r.M) || error("Comparison series use different dimensions at $(r.gpus) GPUs") - end - return nothing -end diff --git a/benchmark/src/runner.jl b/benchmark/src/runner.jl deleted file mode 100644 index 3b848c5c6..000000000 --- a/benchmark/src/runner.jl +++ /dev/null @@ -1,166 +0,0 @@ -using Dates, Pkg - -function cli_options(args) - config = joinpath(@__DIR__,"..","benchmarks.toml") - only = nothing - fusion = nothing - dry = false - verbose = false - positional = String[] - for arg in args - if startswith(arg,"--only=") - only = split(arg,'=';limit=2)[2] - elseif startswith(arg,"--config=") - config = abspath(split(arg,'=';limit=2)[2]) - elseif startswith(arg,"--fusion=") - value = split(arg,'=';limit=2)[2] - fusion = value == "both" ? [true,false] : [parse_fusion(value)] - elseif arg == "--dry-run" - dry = true - elseif arg in ("-v","--verbose") - verbose = true - elseif startswith(arg,"--") - error("Unknown option $arg") - else - push!(positional,arg) - end - end - return (;config,only,fusion,dry,verbose,positional) -end - -function positional_spec(p,gs) - 9 <= length(p) <= 12 || error("Expected [fusion] [check_correctness] [correctness_iter]") - n = is_auto_size(p[5]) ? nothing : parse(Int,p[5]) - m = is_auto_size(p[6]) ? nothing : parse(Int,p[6]) - auto = n === nothing || m === nothing - return BenchmarkSpec(p[3],p[4],parse(Int,p[1]),parse(Int,p[2]), - length(p)>=10 ? parse_fusion(p[10]) : true,gs.cuda, - parse(Int,p[8]),parse(Int,p[7]),parse(Int,p[9]), - auto ? [0,0] : [n,m],auto,n,m) -end - -function plan_manifest(runs,budget,raw) - versions = Dict(info.name=>string(info.version) for info in values(Pkg.dependencies()) if info.version !== nothing) - return Dict("status"=>"running","budget_bytes"=>budget,"julia"=>string(VERSION), - "versions"=>versions,"config"=>raw,"runs"=>[Dict{String,Any}( - "name"=>r.spec.name,"T"=>r.spec.T,"backend"=>string(r.backend), - "fusion"=>r.spec.fusion,"gpus"=>r.spec.gpus,"cpus"=>r.spec.cpus, - "N"=>r.N,"M"=>r.M,"n_iter"=>r.spec.n_iter,"n_warmup"=>r.spec.n_warmup, - "n_trial"=>r.spec.n_trial,"initialization_bytes"=>string(r.memory.initialization), - "iteration_bytes"=>string(r.memory.iteration),"workspace_bytes"=>string(r.memory.workspace), - "memory_explanation"=>r.memory.explanation,"status"=>"pending") for r in runs]) -end - -function prepare_backend(fusion,verbose) - println("Setting fusion=$fusion and precompiling cuNumeric...") - CNPreferences.set_broadcast_fusion!(fusion) - Pkg.precompile("cuNumeric";io=verbose ? stderr : devnull) -end - -function preflight_backends(runs;env=ENV,which=Sys.which,check=success) - any(r.backend==:cupynumeric for r in runs) || return nothing - conda = get(env,"CUNUMERIC_BENCH_CONDA",get(env,"CONDA_EXE","conda")) - executable = which(conda) - executable === nothing && error( - "cuPyNumeric is enabled, but conda is not available to the worker. " * - "Add conda to PATH or set CUNUMERIC_BENCH_CONDA to its executable path; " * - "then run bash install_cupynumeric.sh. No benchmarks have been started.", - ) - name = get(env,"CUPYNUMERIC_ENV",nothing) - name === nothing && (name = cupynumeric_env_name()) - code = "import importlib.util,sys; sys.exit(0 if importlib.util.find_spec('cupynumeric') else 1)" - check(`$executable run --no-capture-output -n $name python -c $code`) || error( - "Conda environment '$name' is unavailable or lacks cupynumeric. " * - "Run bash install_cupynumeric.sh, or set CUPYNUMERIC_ENV to an existing environment. " * - "No benchmarks have been started.", - ) - return nothing -end - -function execute_plan(runs,gs,opts,budget,raw;launch=run,prepare=prepare_backend, - results_root=normpath(joinpath(@__DIR__,"..","results")),preflight=preflight_backends) - preflight(runs) - root = normpath(joinpath(@__DIR__,"..")) - mkpath(results_root) - dir = mktempdir(results_root;prefix=Dates.format(now(),"yyyymmdd-HHMMSS")*"-",cleanup=false) - manifest = plan_manifest(runs,budget,raw) - manifest_path = joinpath(dir,"manifest.toml") - save_manifest() = open(io->TOML.print(io,manifest),manifest_path,"w") - save_manifest() - last_fusion = nothing - failed = false - for (i,r) in enumerate(runs) - s = r.spec - println("\n[$i/$(length(runs))] $(s.name) / $(r.backend), $(s.gpus) GPUs, $(r.N) × $(r.M)") - try - if r.backend == :cunumeric && last_fusion != s.fusion - prepare(s.fusion,opts.verbose) - last_fusion = s.fusion - end - b = build_benchmark(BENCHMARKS[s.name],parse_bench_type(s.T),r.N,r.M) - p = opts.positional - correctness = length(p)>=11 ? parse(Bool,p[11]) : gs.check_correctness - correct_iters = length(p)>=12 ? parse(Int,p[12]) : gs.n_correctness_iter - args = `--gpus $(s.gpus) --cpus $(s.cpus) $(s.name) $(s.T) $(r.N) $(r.M) $(s.n_iter) $(s.n_warmup) $(s.n_trial)` - corr = `$correctness $correct_iters $(total_flops(b))` - runner = joinpath(root,"run_benchmark.sh") - verbose = opts.verbose ? `--verbose` : `` - cmd = if r.backend == :cupynumeric - worker = joinpath(root,"src_py","single.py") - `bash $runner $worker $verbose --pyenv $(cupynumeric_env_name()) $args $corr` - else - worker = joinpath(root,"src","single.jl") - `bash $runner $worker $verbose $args $(string(r.backend)) $corr` - end - results = joinpath(dir,s.T) - launch(addenv(Cmd(cmd;dir=root),"CUNUMERIC_BENCH_RESULTS_DIR"=>results, - "CUNUMERIC_BENCH_JULIA"=>joinpath(Sys.BINDIR,Base.julia_exename()))) - manifest["runs"][i]["status"] = "complete" - catch e - failed = true - manifest["runs"][i]["status"] = "failed" - # ProcessFailedException prints inherited environment variables. - # Report exit codes without copying that environment into logs. - message = e isa ProcessFailedException ? - "Worker exited with code(s) " * join((p.exitcode for p in e.procs),", ") : sprint(showerror,e) - manifest["runs"][i]["error"] = message - @error "Worker failed; no size retry. Continuing independent configurations." reason=message - end - save_manifest() - end - manifest["status"] = failed ? "incomplete" : "complete" - if isempty(opts.positional) - for T in unique(r.spec.T for r in runs) - try - plotter = joinpath(root,"plot_results.jl") - results = joinpath(dir,T) - out = joinpath(root,"plots",basename(dir),T) - suffix = failed ? "_incomplete" : "" - launch(`$(Base.julia_cmd()) --project=$root $plotter $results --config=$(opts.config) --out=$out --suffix=$suffix`) - catch e - failed = true - manifest["status"] = "incomplete" - @error "Plotting failed" exception=e - end - end - end - save_manifest() - println("Results: $dir ($(manifest["status"]))") - return failed ? 1 : 0 -end - -function main(args=ARGS;budget_provider=selected_gpu_budget,executor=execute_plan) - opts = cli_options(args) - gs,specs = parse_config(opts.config;only=opts.only,fusion_override=opts.fusion) - if !isempty(opts.positional) - opts.only === nothing && opts.fusion === nothing || error("Do not mix positional runs with sweep filters") - specs = [positional_spec(opts.positional,gs)] - end - isempty(specs) && error("No benchmarks selected") - raw = TOML.parsefile(opts.config) - budget,_ = budget_provider(gs.mem_frac,maximum(s.gpus for s in specs)) - runs = plan_runs(specs,gs,raw,parse_plot_groups(opts.config),budget) - opts.dry && print_plan(runs,budget) - opts.dry && return 0 - return executor(runs,gs,opts,budget,raw) -end diff --git a/benchmark/src/single.jl b/benchmark/src/single.jl deleted file mode 100644 index ff66e2390..000000000 --- a/benchmark/src/single.jl +++ /dev/null @@ -1,106 +0,0 @@ -# single.jl: worker that runs exactly one benchmark under one backend. Launched by -# run_benchmark.sh (dispatched from run.jl), which sets LEGATE_CONFIG before julia starts. -# Args: -# [check_correctness] [n_correctness_iter] -# backend is "cunumeric" or "cudajl"; run.jl launches one worker per backend. -# run.jl sets the compile-time fusion pref before launch; we read it back to label results. - -using cuNumeric -using LinearAlgebra -using TensorOperations - -length(ARGS) >= 9 || error( - "single.jl args: " * - "[check_correctness] [n_correctness_iter]", -) - -const GPUS = parse(Int, ARGS[1]) -const BACKEND_NAME = ARGS[9] -const CHECK_CORRECTNESS = length(ARGS) >= 10 ? parse(Bool, ARGS[10]) : false - -# Timed CUDA.jl worker always needs CUDA. The 1-GPU cuNumeric worker loads it -# only for the tiny oracle check. -const NEED_CUDA = - BACKEND_NAME == "cudajl" || - (BACKEND_NAME == "cunumeric" && CHECK_CORRECTNESS && GPUS == 1) - -if NEED_CUDA - using CUDA - using AbstractFFTs - using cuTENSOR -end - -include("core.jl") -include_benchmarks() - -# Resolve a TOML type string like "Float32" to the actual Julia type. -parse_type(s) = getfield(Base, Symbol(s))::DataType - -# One worker process, one backend. Clock functions have distinct types, so do -# not stash them in a Dict inferred from the cuNumeric entry. -function backend_entry(name) - name == "cunumeric" && return ( - mod=cuNumeric, label="cuNumeric", save_as="cunumeric", - clock=get_time_microseconds, - synchronize=benchmark_synchronize, - ) - if name == "cudajl" - cuda_sync() = CUDA.synchronize(; blocking=true) - cuda_clock() = (cuda_sync(); time_ns() / 1e3) - return (mod=CUDA, label="CUDA.jl", save_as="CUDA.jl", clock=cuda_clock, - synchronize=cuda_sync) - end - return error("Unknown backend '$(name)'. Known: cunumeric, cudajl") -end -const BACKEND = backend_entry(BACKEND_NAME) - -function run_single( - gpus, name, T_str, N, M, n_iter, n_warmup, n_trial, backend; - check_correctness=false, n_correctness_iter=5, -) - haskey(BENCHMARKS, name) || error( - "No benchmark registered for '$(name)'. Known: $(join(sort(collect(keys(BENCHMARKS))), ", "))" - ) - bk = BACKEND - - T = parse_type(T_str) - b = build_benchmark(BENCHMARKS[name], T, N, M) - - # unfused cuNumeric runs land in their own CSV so they stay a distinct series - fused = cuNumeric.FUSE_BROADCAST_EXPRS - default_save_as = fused ? bk.save_as : "$(bk.save_as)_nofusion" - default_label = fused ? bk.label : "$(bk.label) (no fusion)" - save_as = benchmark_backend_save_as(b, backend, default_save_as) - label = benchmark_backend_label(b, backend, default_label) - gs = GlobalSettings(; - n_warmup=n_warmup, - n_iter=n_iter, - n_trial=n_trial, - n_gpu=gpus, - check_correctness=check_correctness, - n_correctness_iter=n_correctness_iter, - ) - - println( - "[$(label)] $(name) benchmark ($(T)) on $(N)x$(M) for $(n_iter) " * - "iterations ($(n_warmup) warmup) x $(n_trial) trials", - ) - br = run_benchmark(b, gs; mod=bk.mod, clock=bk.clock, synchronize=bk.synchronize) - @printf("[%s] Mean Run Time: %.5f ± %.5f ms\n", label, mean(br.times_ms), _std(br.times_ms)) - @printf("[%s] FLOPS: %.5f ± %.5f GFLOPS\n", label, mean(br.gflops), _std(br.gflops)) - println("[$(label)] Correctness: $(br.correctness)") - return save_result(br, gpus; mod=save_as) -end - -bench_name = ARGS[2] -T_str = ARGS[3] -N = parse(Int, ARGS[4]) -M = parse(Int, ARGS[5]) -n_iter = parse(Int, ARGS[6]) -n_warmup = parse(Int, ARGS[7]) -n_trial = parse(Int, ARGS[8]) -n_correctness_iter = length(ARGS) >= 11 ? parse(Int, ARGS[11]) : 5 -run_single( - GPUS, bench_name, T_str, N, M, n_iter, n_warmup, n_trial, BACKEND_NAME; - check_correctness=CHECK_CORRECTNESS, n_correctness_iter=n_correctness_iter, -) diff --git a/benchmark/src_py/benchmarks/__init__.py b/benchmark/src_py/benchmarks/__init__.py deleted file mode 100644 index 2eee44770..000000000 --- a/benchmark/src_py/benchmarks/__init__.py +++ /dev/null @@ -1,8 +0,0 @@ -import importlib -import pkgutil - -from core import BENCHMARKS - -# Import each module so it self-registers into BENCHMARKS. -for _info in pkgutil.iter_modules(__path__): - importlib.import_module(f"{__name__}.{_info.name}") diff --git a/benchmark/src_py/benchmarks/dmd.py b/benchmark/src_py/benchmarks/dmd.py deleted file mode 100644 index 5091e1f1f..000000000 --- a/benchmark/src_py/benchmarks/dmd.py +++ /dev/null @@ -1,36 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark, rand_array - - -class DMD: - name = "dmd_baseline" - - def __init__(self, T, N, M): - self.T, self.N, self.M = T, N, M - self.r = min(20, M - 1) - - def dims(self): - return self.N, self.M - - def initialize(self): - X = rand_array((self.N, self.M), self.T) - return (X,) - - def run(self, state): - (X,) = state - n = self.M - 1 - X1 = X[:, :n] - X2 = X[:, 1:] - U, S, Vt = np.linalg.svd(X1, full_matrices=False) - r = self.r - U = U[:, :r] - Vt = Vt[:r, :] - B = (X2 @ Vt.T) * (self.T(1) / S[:r]) - At = U.T @ B - _, W = np.linalg.eig(At) - ctype = np.complex64 if self.T == np.float32 else np.complex128 - _ = B.astype(ctype) @ W - - -register_benchmark("dmd_baseline", DMD) diff --git a/benchmark/src_py/benchmarks/gemm.py b/benchmark/src_py/benchmarks/gemm.py deleted file mode 100644 index 74288a9c9..000000000 --- a/benchmark/src_py/benchmarks/gemm.py +++ /dev/null @@ -1,26 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark, rand_array, zeros_array - - -class GEMM: - name = "gemm" - - def __init__(self, T, N, M): - self.T, self.N, self.M = T, N, M - - def dims(self): - return self.N, self.M - - def initialize(self): - A = rand_array((self.N, self.M), self.T) - B = rand_array((self.M, self.N), self.T) - C = zeros_array((self.N, self.N), self.T) - return (C, A, B) - - def run(self, state): - C, A, B = state - np.matmul(A, B, out=C) - - -register_benchmark("gemm", GEMM) diff --git a/benchmark/src_py/benchmarks/grayscott.py b/benchmark/src_py/benchmarks/grayscott.py deleted file mode 100644 index 931acb7a6..000000000 --- a/benchmark/src_py/benchmarks/grayscott.py +++ /dev/null @@ -1,68 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark, rand_array, zeros_array, ones_array - - -class GrayScott: - name = "grayscott_baseline" - fence_each_iteration = False # Timesteps belong to one trajectory. - - # dt = dx/5; c_u, c_v, f, k as in grayscott.jl's GSParams defaults. - def __init__(self, T, N, M, dx=1.0, c_u=1.0, c_v=0.3, f=0.03, k=0.06): - self.T, self.N, self.M = T, N, M - self.dx = T(dx) - self.dt = T(dx / 5) - self.c_u, self.c_v, self.f, self.k = T(c_u), T(c_v), T(f), T(k) - - def dims(self): - return self.N, self.M - - def initialize(self): - u = ones_array((self.N, self.M), self.T) - v = zeros_array((self.N, self.M), self.T) - u_new = zeros_array((self.N, self.M), self.T) - v_new = zeros_array((self.N, self.M), self.T) - - seed = min(150, self.N, self.M) - u[:seed, :seed] = rand_array((seed, seed), self.T) - v[:seed, :seed] = rand_array((seed, seed), self.T) - # mutable list so run() can swap buffers in place - return [u, v, u_new, v_new] - - def run(self, state): - u, v, u_new, v_new = state - ui = u[1:-1, 1:-1] - vi = v[1:-1, 1:-1] - - F_u = (-ui * (vi * vi)) + self.f * (1 - ui) - F_v = (ui * (vi * vi)) - (self.f + self.k) * vi - - dx2 = self.dx * self.dx - u_lap = ( - (u[2:, 1:-1] - 2 * ui + u[:-2, 1:-1]) / dx2 - + (u[1:-1, 2:] - 2 * ui + u[1:-1, :-2]) / dx2 - ) - v_lap = ( - (v[2:, 1:-1] - 2 * vi + v[:-2, 1:-1]) / dx2 - + (v[1:-1, 2:] - 2 * vi + v[1:-1, :-2]) / dx2 - ) - - u_new[1:-1, 1:-1] = (self.c_u * u_lap + F_u) * self.dt + ui - v_new[1:-1, 1:-1] = (self.c_v * v_lap + F_v) * self.dt + vi - - # periodic boundary conditions - u_new[:, 0] = u[:, -2] - u_new[:, -1] = u[:, 1] - u_new[0, :] = u[-2, :] - u_new[-1, :] = u[1, :] - v_new[:, 0] = v[:, -2] - v_new[:, -1] = v[:, 1] - v_new[0, :] = v[-2, :] - v_new[-1, :] = v[1, :] - - # swap references rather than copy - state[0], state[2] = u_new, u - state[1], state[3] = v_new, v - - -register_benchmark("grayscott_baseline", GrayScott) diff --git a/benchmark/src_py/benchmarks/montecarlo.py b/benchmark/src_py/benchmarks/montecarlo.py deleted file mode 100644 index 14a407cc1..000000000 --- a/benchmark/src_py/benchmarks/montecarlo.py +++ /dev/null @@ -1,25 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark, rand_array - - -class MonteCarlo: - name = "montecarlo" - - def __init__(self, T, N, M): - self.T = T - self.n_samples = N - - def dims(self): - return self.n_samples, 1 - - def initialize(self): - x = (self.T(10) * rand_array(self.n_samples, self.T)) - return (x,) - - def run(self, state): - (x,) = state - return (self.T(10) / self.n_samples) * np.sum(np.exp(-(x * x))) - - -register_benchmark("montecarlo", MonteCarlo) diff --git a/benchmark/src_py/benchmarks/poisson_fft.py b/benchmark/src_py/benchmarks/poisson_fft.py deleted file mode 100644 index 3ff23a93d..000000000 --- a/benchmark/src_py/benchmarks/poisson_fft.py +++ /dev/null @@ -1,49 +0,0 @@ -import math - -import cupynumeric as np -import numpy as onp - -from core import register_benchmark, rand_array - - -def _integer_fftfreq(n): - n2 = n // 2 - if n % 2 == 0: - return list(range(0, n2)) + list(range(-n2, 0)) - return list(range(0, n2 + 1)) + list(range(-n2, 0)) - - -def _poisson_inv_laplacian(T, n): - freq = onp.array(_integer_fftfreq(n), dtype=T) - fx = freq[:, None] - fy = freq[None, :] - k2 = T(4.0 * math.pi**2) * (fx * fx + fy * fy) - invk = onp.zeros((n, n), dtype=T) - onp.divide(-1.0, k2, out=invk, where=k2 != 0) - return invk - - -class PoissonFFT: - name = "poisson_fft" - - def __init__(self, T, N, M): - self.T, self.N, self.M = T, N, M - - def dims(self): - return self.N, self.M - - def initialize(self): - f = rand_array((self.M, self.N, self.N), self.T) - kinv = np.array(_poisson_inv_laplacian(self.T, self.N)).reshape( - 1, self.N, self.N - ) - return (f, kinv) - - def run(self, state): - f, kinv = state - uhat = np.fft.fftn(f, axes=(1, 2)) - uhat *= kinv - return np.fft.ifftn(uhat, axes=(1, 2)) - - -register_benchmark("poisson_fft", PoissonFFT) diff --git a/benchmark/src_py/benchmarks/tensor_contractions.py b/benchmark/src_py/benchmarks/tensor_contractions.py deleted file mode 100644 index d865ea825..000000000 --- a/benchmark/src_py/benchmarks/tensor_contractions.py +++ /dev/null @@ -1,62 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark, rand_array, zeros_array - - -def _optimal_path(expression, *operands): - path, _ = np.einsum_path( - expression, *operands, optimize="optimal" - ) - # cuPyNumeric returns NumPy's sentinel-prefixed path, but its einsum forwards - # paths directly to opt_einsum, which expects only contraction tuples. - return path[1:] if path and path[0] == "einsum_path" else path - - -class TensorProjection3: - name = "tensor_projection3" - expression = "ijk,ni,mj,lk->nml" - - def __init__(self, T, N, M): - self.T, self.N = T, N - - def dims(self): - return self.N, 1 - - def initialize(self): - A = rand_array((self.N, self.N, self.N), self.T) - B = rand_array((self.N, self.N), self.T) - D = zeros_array((self.N, self.N, self.N), self.T) - path = _optimal_path(self.expression, A, B, B, B) - return D, A, B, path - - def run(self, state): - D, A, B, path = state - np.einsum( - self.expression, A, B, B, B, out=D, optimize=path - ) - - -class TensorContract4: - name = "tensor_contract4" - expression = "aicj,ibjd->abcd" - - def __init__(self, T, N, M): - self.T, self.N = T, N - - def dims(self): - return self.N, 1 - - def initialize(self): - X = rand_array((self.N, self.N, self.N, self.N), self.T) - Y = rand_array((self.N, self.N, self.N, self.N), self.T) - C = zeros_array((self.N, self.N, self.N, self.N), self.T) - path = _optimal_path(self.expression, X, Y) - return C, X, Y, path - - def run(self, state): - C, X, Y, path = state - np.einsum(self.expression, X, Y, out=C, optimize=path) - - -register_benchmark("tensor_projection3", TensorProjection3) -register_benchmark("tensor_contract4", TensorContract4) diff --git a/benchmark/src_py/core.py b/benchmark/src_py/core.py deleted file mode 100644 index 3bb872d19..000000000 --- a/benchmark/src_py/core.py +++ /dev/null @@ -1,80 +0,0 @@ -import os -import math - -import cupynumeric as np -from legate.core import get_legate_runtime -from legate.timing import time # blocks on preceding legate ops; returns microseconds - -MOD = "cupynumeric" -RESULTS_DIR = os.environ.get("CUNUMERIC_BENCH_RESULTS_DIR", os.path.join(os.path.dirname(__file__), "..", "results")) - -DTYPES = {"Float32": np.float32, "Float64": np.float64} - - -def parse_type(s): - if s not in DTYPES: - raise ValueError(f"Unsupported type '{s}'. Known: {', '.join(DTYPES)}") - return DTYPES[s] - - -def _shape(shape): - if isinstance(shape, int): - return (shape,) - return tuple(shape) - - -def rand_array(shape, dtype): - return np.random.rand(*_shape(shape)).astype(dtype) - - -def zeros_array(shape, dtype): - return np.zeros(_shape(shape), dtype=dtype) - - -def ones_array(shape, dtype): - return np.ones(_shape(shape), dtype=dtype) - - -BENCHMARKS = {} - - -def register_benchmark(key, cls): - BENCHMARKS[key] = cls - - -def trial(bench, n_warmup, n_iter, flops): - state = bench.initialize() - fence_each = getattr(bench, "fence_each_iteration", True) - synchronize = get_legate_runtime().issue_execution_fence - start = None - for idx in range(n_warmup + n_iter): - if idx == n_warmup: - start = time() - bench.run(state) - if fence_each: - synchronize(block=True) - total_us = time() - start - - mean_time_ms = total_us / (n_iter * 1e3) - gflops = flops / (mean_time_ms * 1e6) - return mean_time_ms, gflops - - -def _mean(x): - return sum(x) / len(x) - - -def _std(x): - if len(x) < 2: - return 0.0 - m = _mean(x) - return math.sqrt(sum((v - m) ** 2 for v in x) / (len(x) - 1)) - - -def save_result(name, dims, gpus, times_ms, gflops, correctness="skipped"): - os.makedirs(RESULTS_DIR, exist_ok=True) - N, M = dims - path = os.path.join(RESULTS_DIR, f"{name}_{MOD}.csv") - with open(path, "a") as io: - for i, (t, g) in enumerate(zip(times_ms, gflops), start=1): - io.write(f"{MOD},{gpus},{N},{M},{i},{t:.6f},{g:.6f},{correctness}\n") diff --git a/benchmark/src_py/single.py b/benchmark/src_py/single.py deleted file mode 100644 index 30a8afea4..000000000 --- a/benchmark/src_py/single.py +++ /dev/null @@ -1,59 +0,0 @@ -# cupynumeric worker, run by run_benchmark.sh which sets LEGATE_CONFIG first. -# Args: -# [check_correctness] [n_correctness_iter] [flops] -# flops comes from the Julia orchestrator (same total_flops as the kernel file). -import os -import sys - -# Make `core` and the `benchmarks` package importable when run as a script. -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) - -from core import MOD, parse_type, trial, save_result, _mean, _std -from benchmarks import BENCHMARKS # import populates BENCHMARKS - - -def main(): - gpus = int(sys.argv[1]) - name = sys.argv[2] - T_str = sys.argv[3] - N = int(sys.argv[4]) - M = int(sys.argv[5]) - n_iter = int(sys.argv[6]) - n_warmup = int(sys.argv[7]) - n_trial = int(sys.argv[8]) - if len(sys.argv) < 12: - raise SystemExit( - "single.py args: " - " " - ) - flops = float(sys.argv[11]) - - if name not in BENCHMARKS: - raise ValueError( - f"No benchmark registered for '{name}'. Known: {', '.join(sorted(BENCHMARKS))}" - ) - T = parse_type(T_str) - bench = BENCHMARKS[name](T, N, M) - - print( - f"[{MOD}] {name} benchmark ({T_str}) on {N}x{M} for {n_iter} " - f"iterations ({n_warmup} warmup) x {n_trial} trials; " - + ("per-iteration synchronization" if getattr(bench, "fence_each_iteration", True) - else "batch synchronization") - ) - - times_ms, gflops = [], [] - for _ in range(n_trial): - t, g = trial(bench, n_warmup, n_iter, flops) - times_ms.append(t) - gflops.append(g) - - print(f"[{MOD}] Mean Run Time: {_mean(times_ms):.5f} ± {_std(times_ms):.5f} ms") - print(f"[{MOD}] FLOPS: {_mean(gflops):.5f} ± {_std(gflops):.5f} GFLOPS") - print(f"[{MOD}] Correctness: skipped") - - save_result(bench.name, bench.dims(), gpus, times_ms, gflops, "skipped") - - -if __name__ == "__main__": - main() diff --git a/benchmark/test/autosize.jl b/benchmark/test/autosize.jl deleted file mode 100644 index ece524e0f..000000000 --- a/benchmark/test/autosize.jl +++ /dev/null @@ -1,26 +0,0 @@ -using Test - -# Run with julia --project=benchmark benchmark/test/autosize.jl; no GPU needed. -include("../src/core.jl") -include("../src/benchmarks/montecarlo.jl") - -@testset "Fused Monte Carlo autosizing reserves broadcast output" begin - budget = 113_066_115_072 # Budget from the reported initialization OOM. - for T in (Float32, Float64) - N, M = fit_one_gpu(MonteCarloIntegration, T; budget) - @test M == 1 - @test N % 8 == 0 - # Check actual array requirements, independently of total_space. - @test 2 * N * sizeof(T) <= budget - @test 2 * (N + 8) * sizeof(T) > budget - for P in (1, 2, 4, 8) - n, m = estimate_scaling(MonteCarloIntegration{T}(; n_samples=N), P) - @test n == N * P - @test m == 1 - @test 2 * n * sizeof(T) <= P * budget - end - @test_throws ErrorException fit_one_gpu( - MonteCarloIntegration, T; budget=2 * 8 * sizeof(T) - 1, - ) - end -end diff --git a/benchmark/test/runtests.jl b/benchmark/test/runtests.jl deleted file mode 100644 index 0cae2fb91..000000000 --- a/benchmark/test/runtests.jl +++ /dev/null @@ -1,198 +0,0 @@ -using Test, Statistics, TOML -include("../src/core.jl") -include_benchmarks() -include("../src/parse_benchmarks.jl") -include("../src/memory.jl") -include("../src/planning.jl") -include("../src/runner.jl") -include("../src/result_rows.jl") -include("timing.jl") - -const CONFIG = joinpath(@__DIR__,"..","benchmarks.toml") -const RAW = TOML.parsefile(CONFIG) -const GROUPS = parse_plot_groups(CONFIG) - -# A CPU array that counts materialized arrays, exercising Julia's actual -# broadcast lowering rather than checking the kernel's source spelling. -const MATERIALIZATIONS = Ref(0) -struct CountedArray{T,N} <: AbstractArray{T,N} - data::Array{T,N} -end -struct CountedStyle{N} <: Base.Broadcast.AbstractArrayStyle{N} end -CountedStyle{N}(::Val{M}) where {N,M} = CountedStyle{M}() -Base.size(a::CountedArray) = size(a.data) -Base.getindex(a::CountedArray,I...) = getindex(a.data,I...) -Base.setindex!(a::CountedArray,v,I...) = setindex!(a.data,v,I...) -Base.BroadcastStyle(::Type{<:CountedArray{T,N}}) where {T,N} = CountedStyle{N}() -function Base.similar(bc::Base.Broadcast.Broadcasted{CountedStyle{N}},::Type{T}) where {N,T} - MATERIALIZATIONS[] += 1 - return CountedArray(Array{T}(undef,map(length,axes(bc)))) -end -function Base.similar(a::CountedArray,::Type{T},dims::Dims) where {T} - MATERIALIZATIONS[] += 1 - return CountedArray(Array{T}(undef,dims)) -end - -@testset "Monte Carlo broadcasts fuse across negation" begin - for T in (Float32,Float64) - data = T[0,0.5,1,2,5] - x = CountedArray(data) - b = MonteCarloIntegration{T}(;n_samples=length(x)) - MATERIALIZATIONS[] = 0 - got = run!(b,x) - @test MATERIALIZATIONS[] == 1 - @test got ≈ (T(10)/length(data))*sum(exp(-v^2) for v in data) - # Reproduce the old expression to prove this test detects the bug. - MATERIALIZATIONS[] = 0 - sum(exp.(-x .^ 2)) - @test MATERIALIZATIONS[] == 3 - end -end - -@testset "cuPyNumeric preflight" begin - gs = GlobalSettings(;n_warmup=1,n_iter=1,cupynumeric=true) - s = BenchmarkSpec("montecarlo","Float32",1,8,true,false,1,1,1,[0,0],true,nothing,nothing) - runs = plan_runs([s],gs,RAW,GROUPS,1000000) - env = Dict("CUNUMERIC_BENCH_CONDA"=>"/test/conda","CUPYNUMERIC_ENV"=>"testenv") - @test_throws ErrorException preflight_backends(runs;env,which=x->nothing) - @test_throws ErrorException preflight_backends(runs;env,which=identity,check=c->false) - @test preflight_backends(runs;env,which=identity,check=c->true) === nothing -end - -function spec(name;T="Float32",gpus=1,fusion=true,cuda=false,N=nothing,M=nothing,auto=true) - BenchmarkSpec(name,T,gpus,8,fusion,cuda,2,5,2, - auto ? [0,0] : [N,M],auto,N,M) -end - -@testset "Configuration and CLI" begin - gs,ss = parse_config(CONFIG;only="grayscott",fusion_override=[true,false]) - @test all(startswith(s.name,"grayscott") for s in ss) - @test Set(s.fusion for s in ss)==Set([true,false]) - @test_throws ErrorException parse_config(CONFIG;only="missing") - o = cli_options(["--only=montecarlo","--fusion=both","--dry-run"]) - @test o.only == "montecarlo" && o.dry && o.fusion == [true,false] - @test_throws ErrorException cli_options(["--typo"]) - p = positional_spec(["1","8","montecarlo","Float32","auto","1","5","2","2"],gs) - @test p.autosize && p.M_hint==1 && p.n_iter==5 - @test main(["--only=montecarlo","--dry-run"]; - budget_provider=(f,p)->(1_000_000,f),executor=(args...)->error("dry-run launched workers"))==0 -end - -@testset "Memory dispatch matrix" begin - for T in (Float32,Float64), fusion in (false,true), backend in (:cunumeric,:cudajl,:cupynumeric) - c = MemoryContext(;backend,fusion,workspace_bytes=0) - for (name,B) in BENCHMARKS - endswith(name,"_accelerated") && backend != :cunumeric && continue - m = B <: AbstractDMD ? 16 : B <: AbstractGrayScott || B <: GEMM ? 64 : 1 - b = build_benchmark(B,T,64,m) - estimate = memory_estimate(b,c) - @test estimate.initialization>0 && estimate.iteration>0 - @test !isempty(estimate.explanation) - end - end - b = MonteCarloIntegration{Float32}(;n_samples=1024) - @test peak_bytes(memory_estimate(b,MemoryContext())) == 8192 - @test peak_bytes(memory_estimate(b,MemoryContext(;backend=:cupynumeric))) == 12288 - @test peak_bytes(memory_estimate(b,MemoryContext(;fusion=false))) == 16384 - @test_throws ErrorException memory_estimate(GEMM{Float32}(;N=64,M=64),MemoryContext()) - @test_throws ErrorException memory_estimate(b,MemoryContext(;backend=:cudajl,gpus=2)) - d = DMDBaseline{Float32}(;N=1024,M=16) - @test peak_bytes(memory_estimate(d,MemoryContext(;gpus=1,workspace_bytes=0))) == - peak_bytes(memory_estimate(d,MemoryContext(;gpus=8,workspace_bytes=0))) -end - -@testset "Shared sweep planning" begin - gs = GlobalSettings(;n_warmup=2,n_iter=5,cupynumeric=true) - ss = [spec("montecarlo";gpus=p,fusion=f,cuda=true) for p in (1,2,4,8) for f in (true,false)] - runs = plan_runs(ss,gs,RAW,GROUPS,1_000_000) - @test length(runs)==8+4+1 - for p in (1,2,4,8) - @test length(unique((r.N,r.M) for r in runs if r.spec.gpus==p))==1 - end - @test all(peak_bytes(r.memory)<=1_000_000 for r in runs) - @test maximum(r.N for r in runs)==8minimum(r.N for r in runs) - fused = plan_runs([spec("montecarlo")],gs,RAW,GROUPS,1_000_000) - @test first(fused).N >= first(runs).N - onlyoff = plan_runs([spec("montecarlo";fusion=false)],gs,RAW,GROUPS,1_000_000) - @test any(r.backend==:cupynumeric for r in onlyoff) - @test_throws ErrorException plan_runs([spec("montecarlo";N=1000000,M=1,auto=false)],gs,RAW,GROUPS,100) - @test_throws ErrorException plan_runs([spec("montecarlo";N=8,M=1,auto=false),spec("montecarlo";N=16,M=1,auto=false)],gs,RAW,GROUPS,100000) - @test_throws ErrorException plan_runs([spec("montecarlo";gpus=0)],gs,RAW,GROUPS,100000) - raw = deepcopy(RAW) - raw["workspace"] = Dict("dmd_baseline"=>Dict("cunumeric"=>0)) - single = plan_runs([spec("dmd_baseline";M=16)],GlobalSettings(;n_warmup=1,n_iter=1),raw,GROUPS,10_000_000) - sweep = plan_runs([spec("dmd_baseline";M=16,gpus=p) for p in (1,8)],GlobalSettings(;n_warmup=1,n_iter=1),raw,GROUPS,10_000_000) - @test first(sweep).N < first(single).N - @test all(peak_bytes(r.memory)<=10_000_000 for r in sweep) -end - -@testset "GPU budgets" begin - inventory = "0, GPU-aaa, 1000, 900\n1, GPU-bbb, 2000, 1900\n" - kw = (;inventory,fraction="0.75",fbmem=nothing) - @test first(selected_gpu_budget(.75,1;kw...,visibility="GPU-bbb")) == 1500*1024^2 - @test first(selected_gpu_budget(.75,2;kw...,visibility="1,0")) == 750*1024^2 - @test_throws ErrorException selected_gpu_budget(.75,2;kw...,visibility="1") - @test_throws ErrorException selected_gpu_budget(.75,1;kw...,visibility="") - @test_throws ErrorException selected_gpu_budget(.75,1;kw...,visibility="2") -end - -@testset "Result dimension isolation" begin - rows = [Row(1,32,1,1.,2.),Row(1,32,1,3.,4.)] - a = aggregate(rows) - @test only(a).N==32 && only(a).t==2 - @test_throws ErrorException aggregate(vcat(rows,[Row(1,64,1,1.,2.)])) - @test_throws ErrorException validate_series_sizes([(agg=a,),(agg=aggregate([Row(1,64,1,1.,2.)]),)]) -end - -@testset "Variant lifetimes and rectangular constraints" begin - baseline = GrayScottBaseline{Float32}(;N=64,M=32) - accelerated = GrayScottFunctionAccelerated{Float32}(;N=64,M=32) - for f in (true,false) - c = MemoryContext(;fusion=f,steps=10) - @test peak_bytes(memory_estimate(accelerated,c)) < peak_bytes(memory_estimate(baseline,c)) - end - @test peak_bytes(memory_estimate(baseline,MemoryContext(;fusion=true))) < - peak_bytes(memory_estimate(baseline,MemoryContext(;fusion=false))) - @test estimate_scaling(GEMM{Float32}(;N=64,M=32),8)==(128,64) - @test estimate_scaling(baseline,4)==(128,64) - @test_throws ErrorException memory_estimate(baseline,MemoryContext(;steps=0)) - @test_throws ErrorException memory_estimate(accelerated,MemoryContext(;backend=:cupynumeric)) - gs = GlobalSettings(;n_warmup=1,n_iter=1) - ss = [spec(n;gpus=p,fusion=f) for n in ("grayscott_baseline","grayscott_function_accelerated") for p in (1,4) for f in (false,true)] - runs = plan_runs(ss,gs,RAW,GROUPS,10_000_000) - for p in (1,4) - @test length(unique((r.N,r.M) for r in runs if r.spec.gpus==p))==1 - end -end - -@testset "Execution isolation and failure status" begin - gs = GlobalSettings(;n_warmup=1,n_iter=1) - runs = plan_runs([spec("montecarlo";T=T) for T in ("Float32","Float64")],gs,RAW,GROUPS,1_000_000) - opts = cli_options(["--only=montecarlo"]) - mktempdir() do dir - calls = Cmd[] - launch(cmd) = (push!(calls,cmd); nothing) - @test execute_plan(runs,gs,opts,1_000_000,RAW;launch,prepare=(f,v)->nothing,results_root=dir)==0 - @test length(calls)==4 # two workers and one plot per dtype - run_dir = only(readdir(dir;join=true)) - manifest = TOML.parsefile(joinpath(run_dir,"manifest.toml")) - @test manifest["status"]=="complete" - @test all(r["status"]=="complete" for r in manifest["runs"]) - @test occursin("Float32",join(calls[1].env)) - @test occursin("Float64",join(calls[2].env)) - @test any(occursin("plot_results.jl",join(c.exec)) for c in calls) - end - mktempdir() do dir - calls = Ref(0) - function fail_first(cmd) - calls[] += 1 - calls[] == 1 && error("simulated worker failure") - end - @test execute_plan(runs,gs,opts,1_000_000,RAW;launch=fail_first,prepare=(f,v)->nothing,results_root=dir)==1 - @test calls[]==4 # no retry, later worker and plots still run - manifest = TOML.parsefile(joinpath(only(readdir(dir;join=true)),"manifest.toml")) - @test manifest["status"]=="incomplete" - @test manifest["runs"][1]["status"]=="failed" - @test manifest["runs"][2]["status"]=="complete" - end -end diff --git a/benchmark/test/test_timing.py b/benchmark/test/test_timing.py deleted file mode 100644 index 40ae4d019..000000000 --- a/benchmark/test/test_timing.py +++ /dev/null @@ -1,63 +0,0 @@ -"""CPU-only tests: python -m unittest discover -s benchmark/test -p 'test_*.py'.""" -import importlib.util -from pathlib import Path -from types import SimpleNamespace -import unittest -from unittest.mock import patch - -SOURCE = Path(__file__).resolve().parents[1] / "src_py" - - -def load(path): - spec = importlib.util.spec_from_file_location(path.stem, path) - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - return module - - -class TimingTests(unittest.TestCase): - def test_completion_order(self): - for gray_scott in (False, True): - for warmup in (0, 2): - with self.subTest(gray_scott=gray_scott, warmup=warmup): - events, ticks = [], iter((6000, 12000)) - - def clock(): - events.append("clock") - return next(ticks) - - def fence(*, block): - self.assertTrue(block) - events.append("sync") - - mocks = { - "cupynumeric": SimpleNamespace(float32=float, float64=float), - "legate.core": SimpleNamespace(get_legate_runtime=lambda: - SimpleNamespace(issue_execution_fence=fence)), - "legate.timing": SimpleNamespace(time=clock), - } - with patch.dict("sys.modules", mocks): - core = load(SOURCE / "core.py") - with patch.dict("sys.modules", {"core": core}): - gs = load(SOURCE / "benchmarks" / "grayscott.py") - - class Probe: - def initialize(self): - events.append("initialize") - return None - - def run(self, state): - events.append("run") - - # Inherit the real GrayScott policy without creating GPU arrays. - cls = type("GrayScottProbe", (Probe, gs.GrayScott), {}) if gray_scott else Probe - bench = cls.__new__(cls) - result = core.trial(bench, warmup, 3, 6000) - step = ["run"] if gray_scott else ["run", "sync"] - self.assertEqual(events, ["initialize"] + step * warmup - + ["clock"] + step * 3 + ["clock"]) - self.assertEqual(result, (2.0, 0.003)) - - -if __name__ == "__main__": - unittest.main() diff --git a/benchmark/test/timing.jl b/benchmark/test/timing.jl deleted file mode 100644 index 9ed66e046..000000000 --- a/benchmark/test/timing.jl +++ /dev/null @@ -1,34 +0,0 @@ -struct TimingProbe <: AbstractBenchmark{Float32} - events::Vector{Symbol} -end -struct GrayScottTimingProbe <: AbstractGrayScott{Float32} - events::Vector{Symbol} -end -const TimingProbes = Union{TimingProbe,GrayScottTimingProbe} -initialize(b::TimingProbes; mod=Base) = (push!(b.events, :initialize); ()) -run!(b::TimingProbes) = push!(b.events, :run) -total_flops(::TimingProbes) = 6000 -name(::TimingProbes) = "timing probe" - -@testset "Iteration completion policy" begin - for B in values(BENCHMARKS) - b = build_benchmark(B, Float32, 32, 32) - @test fence_each_iteration(b) == !(b isa AbstractGrayScott) - end - for B in (TimingProbe, GrayScottTimingProbe), warmup in (0, 2) - events = Symbol[] - b = B(events) - ticks = Ref(0) - clock() = (push!(events, :clock); ticks[] += 6000) - synchronize() = push!(events, :sync) - gs = GlobalSettings(; n_warmup=warmup, n_iter=3) - step = fence_each_iteration(b) ? [:run, :sync] : [:run] - expected = vcat([:initialize], repeat(step, warmup), [:clock], - repeat(step, 3), [:clock]) - # Also exercises forwarding the backend callback through run_benchmark. - result = run_benchmark(b, gs; mod=Base, clock, synchronize) - @test events == expected - @test result.times_ms == [2.0] - @test result.gflops == [0.003] - end -end diff --git a/Dockerfile b/docker/Dockerfile similarity index 83% rename from Dockerfile rename to docker/Dockerfile index c42fcd244..fc2d812e8 100644 --- a/Dockerfile +++ b/docker/Dockerfile @@ -5,8 +5,6 @@ ARG CUDA_MAJOR=13 ARG CUDA_MINOR=0 ENV CUDA_VERSION_MAJOR_MINOR="${CUDA_MAJOR}.${CUDA_MINOR}" -ARG REF=main -ENV REF=${REF} # using bash SHELL ["/bin/bash", "-c"] ENV DEBIAN_FRONTEND=noninteractive @@ -55,15 +53,26 @@ RUN cat /etc/.env RUN echo "Install Legate and cuNumeric.jl" -# Install Legate.jl and cuNumeric.jl RUN source /etc/.env && source /etc/.env && julia --color=yes -e ' \ using Pkg; \ Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/Legate.jl", rev = "main")) \ ' +# The workflow checks out the requested revision before building. Copy only the +# package source so benchmark-only changes leave this image unchanged. +COPY Project.toml /opt/cuNumeric.jl/Project.toml +COPY deps /opt/cuNumeric.jl/deps +COPY ext /opt/cuNumeric.jl/ext +COPY lib /opt/cuNumeric.jl/lib +COPY scripts /opt/cuNumeric.jl/scripts +COPY src /opt/cuNumeric.jl/src +COPY test /opt/cuNumeric.jl/test + RUN source /etc/.env && source /etc/.env && julia --color=yes -e ' \ using Pkg; \ - Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = ENV["REF"])) \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl/lib/CNPreferences")); \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl")); \ + Pkg.precompile() \ ' RUN \ #= remove useless stuff =# \ @@ -104,5 +113,9 @@ RUN apt-get clean \ ENV LEGATE_AUTO_CONFIG=1 -ENTRYPOINT source /etc/.env && exec /bin/bash +COPY docker/profile.sh /etc/profile.d/cunumeric.sh +COPY docker/entrypoint.sh /usr/local/bin/cunumeric-entrypoint +RUN chmod +x /usr/local/bin/cunumeric-entrypoint +ENTRYPOINT ["/usr/local/bin/cunumeric-entrypoint"] +CMD ["/bin/bash"] WORKDIR /workspace diff --git a/Dockerfile.developer b/docker/Dockerfile.developer similarity index 95% rename from Dockerfile.developer rename to docker/Dockerfile.developer index 9d163d88e..8b82c625e 100644 --- a/Dockerfile.developer +++ b/docker/Dockerfile.developer @@ -5,8 +5,6 @@ ARG CUDA_MAJOR=13 ARG CUDA_MINOR=0 ENV CUDA_VERSION_MAJOR_MINOR="${CUDA_MAJOR}.${CUDA_MINOR}" -ARG REF=main -ENV REF=${REF} # using bash SHELL ["/bin/bash", "-c"] ENV DEBIAN_FRONTEND=noninteractive @@ -123,5 +121,9 @@ RUN apt-get clean \ ENV LEGATE_AUTO_CONFIG=1 -ENTRYPOINT source /etc/.env && exec /bin/bash +COPY docker/profile.sh /etc/profile.d/cunumeric.sh +COPY docker/entrypoint.sh /usr/local/bin/cunumeric-entrypoint +RUN chmod +x /usr/local/bin/cunumeric-entrypoint +ENTRYPOINT ["/usr/local/bin/cunumeric-entrypoint"] +CMD ["/bin/bash"] WORKDIR /workspace diff --git a/docker/README.md b/docker/README.md new file mode 100644 index 000000000..86e478ec6 --- /dev/null +++ b/docker/README.md @@ -0,0 +1,26 @@ +# Containers + +Build from the repository root: + +```bash +docker build -f docker/Dockerfile -t cunumeric:local . +``` + +The benchmark image and its opt-in CI now live in +[JuliaLegate/benchmarking](https://github.com/JuliaLegate/benchmarking). +Build it from the pinned submodule without rebuilding the base: + +```bash +git submodule update --init --recursive +docker build -f benchmark/docker/Dockerfile \ + --build-arg BASE_IMAGE=cunumeric:local \ + -t cunumeric:benchmark benchmark +``` + +Run the smoke suite with NVIDIA Container Toolkit enabled: + +```bash +docker run --rm --gpus=all \ + cunumeric:benchmark \ + julia --project=. run.jl --config=benchmarks_smoke.toml +``` diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh new file mode 100644 index 000000000..5837a10dd --- /dev/null +++ b/docker/entrypoint.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +set -e + +# shellcheck disable=SC1091 +source /etc/profile.d/cunumeric.sh + +if [[ $# -eq 0 ]]; then + set -- /bin/bash +elif [[ $1 == -* ]]; then + set -- /bin/bash "$@" +fi + +exec "$@" diff --git a/docker/profile.sh b/docker/profile.sh new file mode 100644 index 000000000..da2ef5cd6 --- /dev/null +++ b/docker/profile.sh @@ -0,0 +1,8 @@ +# shellcheck shell=sh +# Keep interactive login shells consistent with the container entrypoint. +export PATH="/opt/conda/bin:/usr/local/julia/bin:/usr/local/bin:${PATH}" + +if [ -r /etc/.env ]; then + # shellcheck disable=SC1091 + . /etc/.env +fi diff --git a/docs/make.jl b/docs/make.jl index 8b0b3bb2e..0be317b57 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -40,6 +40,7 @@ makedocs(; "Gray-Scott" => "examples/grayscott.md", "Dynamic Mode Decomposition" => "examples/dmd.md", "Periodic Poisson (FFT)" => "examples/poisson_fft.md", + "Conjugate Gradient" => "examples/cg.md", "Tensor Network Contraction" => "examples/tensor_network.md", ], "Performance Tips" => [ diff --git a/docs/src/benchmarks/howto.md b/docs/src/benchmarks/howto.md index 22cee5258..f1850c7b3 100644 --- a/docs/src/benchmarks/howto.md +++ b/docs/src/benchmarks/howto.md @@ -1,6 +1,6 @@ # How to Benchmark -The benchmark harness lives in `benchmark/` at the repo root. Configs are declared in `benchmarks.toml`. `run.jl` expands those configs and launches one worker process per run. Workers never share a GPU runtime within a measurement. +The benchmark harness lives in the `benchmark/` submodule, maintained in [JuliaLegate/benchmarking](https://github.com/JuliaLegate/benchmarking). Configs are declared in `benchmarks.toml`. `run.jl` expands those configs and launches one worker process per run. Workers never share a GPU runtime within a measurement. > [!WARNING] > We do not commit to maintaining the benchmark scripts forever. The harness evolves with the package. The ideas here (declare configs in TOML, one process per run, time with Legate fences) should still apply even if file names move. @@ -14,11 +14,13 @@ cuNumeric ops are asynchronous. A Julia call usually returns before the GPU work From the repo: ```bash +git submodule update --init --recursive cd benchmark +./instantiate_projects.sh julia --project=. run.jl ``` -With no extra args, `run.jl` reads `benchmarks.toml` and runs every expanded config. It develops `CNPreferences` and `cuNumeric` from the parent checkout, then for each config: +The setup script develops `CNPreferences` and `cuNumeric` from the parent checkout. See the harness README for cuPyNumeric's Conda setup. With no extra args, `run.jl` reads `benchmarks.toml` and runs every expanded config: 1. Sets the broadcast-fusion preference if needed and precompiles 2. Calls `run_benchmark.sh`, which exports `LEGATE_CONFIG` from `--gpus` / `--cpus` **before** Julia starts diff --git a/docs/src/examples/cg.md b/docs/src/examples/cg.md new file mode 100644 index 000000000..5cf92294f --- /dev/null +++ b/docs/src/examples/cg.md @@ -0,0 +1,50 @@ +# Conjugate gradient + +Conjugate gradient solves ``Ax=b`` for a real symmetric positive-definite +matrix using matrix-vector products, reductions and vector updates. The complete recurrence is shown below. + +```julia +using cuNumeric, LinearAlgebra + +function cg!(x, A, b; rtol=1e-8, check_every=10, max_iter=1000) + r = b - A*x + p, Ap = copy(r), similar(r) + rho = sum(r .* r) + target = rtol^2 * only(sum(b .* b)) + only(rho) <= target && return x + + for k in 1:max_iter + mul!(Ap, A, p) + # Protect zero denominators if convergence occurs between checks. + alpha = rho ./ max.(sum(p .* Ap), floatmin(eltype(x))) + x .+= alpha .* p + r .-= alpha .* Ap + next = sum(r .* r) + beta = next ./ max.(rho, floatmin(eltype(x))) + p .= r .+ beta .* p + rho = next + + if k % check_every == 0 || k == max_iter + only(rho) <= target && return x + end + end + error("CG did not converge within max_iter") +end + +A = NDArray([4.0 1.0 1.0; 1.0 3.0 0.5; 1.0 0.5 2.0]) +# cuNumeric's matrix multiplication uses single-column matrices for vectors. +b = cuNumeric.ones(Float64, 3, 1) +x = cuNumeric.zeros(Float64, 3, 1) +cg!(x, A, b; check_every=5, max_iter=100) + +# Compare the iterative result with the direct solve API. +x_direct = cuNumeric.solve(A, b) +println(Array(x)) +@assert isapprox(Array(x), Array(x_direct); rtol=1e-8) +``` + +`rho` holds the squared residual norm. Reductions and the coefficients `alpha` +and `beta` remain in cuNumeric's computation graph; `only(rho)` reads a value +back to the host at a convergence check. Increasing `check_every` lets the host +submit more iterations ahead, at the cost of potentially doing extra work +before observing convergence. A check also occurs at the iteration limit. diff --git a/lib/cunumeric_jl_wrapper/include/accessors.h b/lib/cunumeric_jl_wrapper/include/accessors.h index 091b6cbfd..1ce4c5c1d 100644 --- a/lib/cunumeric_jl_wrapper/include/accessors.h +++ b/lib/cunumeric_jl_wrapper/include/accessors.h @@ -30,7 +30,8 @@ #include "legion.h" // Match Legate coordinates; large scalar indices must never narrow to int. -static_assert(sizeof(legate::coord_t) >= 8, "NDArray accessors require 64-bit coordinates"); +static_assert(sizeof(legate::coord_t) >= 8, + "NDArray accessors require 64-bit coordinates"); // To auto-magically generate templated classes and their // respective member functions you must define a `BuildParameterList` diff --git a/lib/cunumeric_jl_wrapper/include/cuda_macros.h b/lib/cunumeric_jl_wrapper/include/cuda_macros.h index 35d1b331b..596fffb7f 100644 --- a/lib/cunumeric_jl_wrapper/include/cuda_macros.h +++ b/lib/cunumeric_jl_wrapper/include/cuda_macros.h @@ -45,31 +45,31 @@ } while (0) #endif -#define CUDA_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ - template < \ - typename T, int D, \ - typename std::enable_if<(D >= 1 && D <= REALM_MAX_DIM), int>::type = 0> \ - void cuda_device_array_arg_##MODE(char *&p, \ - const legate::PhysicalArray &rf) { \ - auto shp = rf.shape(); \ - auto acc = rf.data().ACCESSOR_CALL(); \ - CUDA_DEBUG_PRINT(std::cerr << "[RunPTXTask] " #MODE " accessor shape: " \ - << shp.lo << " - " << shp.hi << ", dim: " << D \ - << std::endl; \ - std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ - << acc.accessor.strides << std::endl;); \ +#define CUDA_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ + template < \ + typename T, int D, \ + typename std::enable_if<(D >= 1 && D <= REALM_MAX_DIM), int>::type = 0> \ + void cuda_device_array_arg_##MODE(char *&p, \ + const legate::PhysicalArray &rf) { \ + auto shp = rf.shape(); \ + auto acc = rf.data().ACCESSOR_CALL(); \ + CUDA_DEBUG_PRINT(std::cerr << "[RunPTXTask] " #MODE " accessor shape: " \ + << shp.lo << " - " << shp.hi << ", dim: " << D \ + << std::endl; \ + std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ + << acc.accessor.strides << std::endl;); \ /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ - void *dev_ptr = const_cast(/*.lo to ensure multiple GPU support*/ \ - static_cast( \ - acc.ptr(shp.lo))); \ - auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ - CuDeviceArray desc; \ - desc.ptr = dev_ptr; \ - desc.maxsize = shp.volume() * sizeof(T); \ - for (size_t i = 0; i < D; ++i) { \ - desc.dims[i] = extents[i]; \ - } \ - desc.length = shp.volume(); \ - memcpy(p, &desc, sizeof(CuDeviceArray)); \ - p += sizeof(CuDeviceArray); \ + void *dev_ptr = \ + const_cast(/*.lo to ensure multiple GPU support*/ \ + static_cast(acc.ptr(shp.lo))); \ + auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ + CuDeviceArray desc; \ + desc.ptr = dev_ptr; \ + desc.maxsize = shp.volume() * sizeof(T); \ + for (size_t i = 0; i < D; ++i) { \ + desc.dims[i] = extents[i]; \ + } \ + desc.length = shp.volume(); \ + memcpy(p, &desc, sizeof(CuDeviceArray)); \ + p += sizeof(CuDeviceArray); \ } diff --git a/lib/cunumeric_jl_wrapper/src/cuda.cpp b/lib/cunumeric_jl_wrapper/src/cuda.cpp index de9cf1522..b536598c3 100644 --- a/lib/cunumeric_jl_wrapper/src/cuda.cpp +++ b/lib/cunumeric_jl_wrapper/src/cuda.cpp @@ -121,9 +121,9 @@ struct CuDeviceArray { std::cerr << "[RunPTXBroadcastTask] " #MODE \ << " accessor byte strides: " << acc.accessor.strides \ << std::endl;); \ - /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ - void *dev_ptr = const_cast( \ - static_cast(acc.ptr(shp.lo))); \ + /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ + void *dev_ptr = \ + const_cast(static_cast(acc.ptr(shp.lo))); \ auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ CuStridedDeviceArray desc; \ desc.ptr = dev_ptr; \ @@ -293,7 +293,8 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const std::uint32_t budget = std::max(lp.tx, 1u); const int dim = out.dim(); // Cap before narrowing; Julia grid-stride loops cover the remaining elements. - const auto blocks = [](std::uint64_t n, std::uint32_t t, std::uint64_t limit) { + const auto blocks = [](std::uint64_t n, std::uint32_t t, + std::uint64_t limit) { return static_cast(std::min((n - 1) / t + 1, limit)); }; diff --git a/lib/cunumeric_jl_wrapper/src/memory.cpp b/lib/cunumeric_jl_wrapper/src/memory.cpp index 5f5bb4ef4..c00b80e68 100644 --- a/lib/cunumeric_jl_wrapper/src/memory.cpp +++ b/lib/cunumeric_jl_wrapper/src/memory.cpp @@ -41,7 +41,8 @@ static inline uint64_t query_machine_config_common( Realm::Processor::Kind proc_kind, Realm::Memory::Kind mem_kind) { Machine legion_machine{Machine::get_machine()}; uint64_t total_mem = 0; - std::set seen; // Processors on one node share memory-query results. + std::set + seen; // Processors on one node share memory-query results. Machine::ProcessorQuery procs = Machine::ProcessorQuery(legion_machine).only_kind(proc_kind); @@ -75,7 +76,8 @@ static inline uint64_t query_allocated_bytes_common( auto ctx = Legion::Runtime::get_context(); uint64_t current_bytes = 0; - std::set seen; // Count each physical memory once, not per processor. + std::set + seen; // Count each physical memory once, not per processor. Machine::ProcessorQuery procs = Machine::ProcessorQuery(legion_machine).only_kind(proc_kind); diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index 7efac3029..ba281473d 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -30,7 +30,6 @@ #include #include #include -#include #include #include #include diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index 28ab75663..f80fff258 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -179,13 +179,19 @@ void wrap_linalg_ops(jlcxx::Module& mod) { legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_POTRF}); mod.set_const("MP_QR", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_QR}); - mod.set_const("POTRS", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_POTRS}); - mod.set_const("TRSM", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRSM}); - mod.set_const("SYRK", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SYRK}); - mod.set_const("GEMM", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEMM}); - mod.set_const("TRANSPOSE_COPY_2D", - legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRANSPOSE_COPY_2D}); - mod.set_const("TRILU", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRILU}); + mod.set_const("POTRS", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_POTRS}); + mod.set_const("TRSM", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRSM}); + mod.set_const("SYRK", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SYRK}); + mod.set_const("GEMM", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEMM}); + mod.set_const( + "TRANSPOSE_COPY_2D", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRANSPOSE_COPY_2D}); + mod.set_const("TRILU", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRILU}); mod.set_const("SVD", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SVD}); mod.set_const("CQR", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_QR}); mod.set_const("POTRF", diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 177c7ed8c..4ca6e15f5 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -212,24 +212,20 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { mod.method("cusolvermp_available", &cupynumeric_has_cusolvermp); // Match cuPyNumeric 26.06: the MP kernels use a single NCCL communicator. - mod.method("add_nccl_communicator", [](legate::ManualTask& task) { - task.add_communicator("nccl"); - }); - mod.method("add_nccl_communicator", [](legate::AutoTask& task) { - task.add_communicator("nccl"); - }); + mod.method("add_nccl_communicator", + [](legate::ManualTask& task) { task.add_communicator("nccl"); }); + mod.method("add_nccl_communicator", + [](legate::AutoTask& task) { task.add_communicator("nccl"); }); // Tiled Cholesky launches subrectangles in the original partition's color // space. Bounds are inclusive and zero-based, as in Legion::Rect. - mod.method("create_linalg_task", - [](legate::LocalTaskID id, int64_t row_lo, int64_t col_lo, - int64_t row_hi, int64_t col_hi) { - auto domain = Legion::Domain{Legion::Rect<2>{ - Legion::Point<2>{row_lo, col_lo}, - Legion::Point<2>{row_hi, col_hi}}}; - return legate::Runtime::get_runtime()->create_task( - get_lib(), id, domain); - }); + mod.method("create_linalg_task", [](legate::LocalTaskID id, int64_t row_lo, + int64_t col_lo, int64_t row_hi, + int64_t col_hi) { + auto domain = Legion::Domain{Legion::Rect<2>{ + Legion::Point<2>{row_lo, col_lo}, Legion::Point<2>{row_hi, col_hi}}}; + return legate::Runtime::get_runtime()->create_task(get_lib(), id, domain); + }); mod.method("add_input_tile", [](legate::ManualTask& task, std::shared_ptr part, @@ -237,14 +233,14 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { std::vector color{row, col}; task.add_input(part->get_child_store(color)); }); - mod.method("add_input_column", - [](legate::ManualTask& task, - std::shared_ptr part, - int32_t col) { - task.add_input(*part, legate::SymbolicPoint{ - std::vector{legate::dimension(0), - legate::constant(col)}}); - }); + mod.method( + "add_input_column", + [](legate::ManualTask& task, + std::shared_ptr part, int32_t col) { + task.add_input(*part, + legate::SymbolicPoint{std::vector{ + legate::dimension(0), legate::constant(col)}}); + }); mod.method("add_input_proj", [](legate::ManualTask& task, diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 470e589c1..6a8d6eec1 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -210,7 +210,9 @@ function make_cartesian_kernel_3d( @inbounds args_modified = _materialize_broadcast_args( arg_plan, runtime_args, static_args, I ) - @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf( + f, args_modified... + ) I += CartesianIndex(_broadcast_grid_stride(:z), 0, 0) end I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:y), I[3]) diff --git a/src/ndarray/detail/distributed_linalg.jl b/src/ndarray/detail/distributed_linalg.jl index c2115471c..abfd11fc8 100644 --- a/src/ndarray/detail/distributed_linalg.jl +++ b/src/ndarray/detail/distributed_linalg.jl @@ -10,8 +10,9 @@ struct _LinalgRuntime mp_eligible::Bool end -_LinalgRuntime(available::Bool, gpus::Int, procs::Int) = - _LinalgRuntime(available, gpus, procs, available && gpus > 1) +function _LinalgRuntime(available::Bool, gpus::Int, procs::Int) + return _LinalgRuntime(available, gpus, procs, available && gpus > 1) +end # Populated once in _start_runtime(), including deferred initialization. # These describe the configured machine for the lifetime of this runtime. @@ -75,7 +76,7 @@ function _with_linalg_partitions(f, spec::Tuple, specs::Tuple...) end try return _with_linalg_partitions(specs...) do parts... - f(partition, parts...) + return f(partition, parts...) end finally finalize(partition.handle) @@ -161,7 +162,7 @@ function _cholesky_color_shape(n::Int, procs::Int) (procs == 1 || n <= MIN_CHOLESKY_MATRIX_SIZE) && return (1, 1) tiles = Int(procs) while cld(n, tiles) > MIN_CHOLESKY_TILE_SIZE && - 2 * tiles <= procs * MAX_CHOLESKY_TILES_PER_PROC + 2 * tiles <= procs * MAX_CHOLESKY_TILES_PER_PROC tiles *= 2 end return (tiles, tiles) diff --git a/src/scoping/broadcast_lifetimes.jl b/src/scoping/broadcast_lifetimes.jl index cbd763292..e5568b5a4 100644 --- a/src/scoping/broadcast_lifetimes.jl +++ b/src/scoping/broadcast_lifetimes.jl @@ -85,7 +85,7 @@ function rewrite_broadcast_lifetimes(scope) if isnothing(lhs_reference) new_lhs, lhs_temps = rewrite_materialized(lhs) else - new_lhs, lhs_temps = fresh_tmp(lhs) + new_lhs, lhs_temps = fresh_tmp(:(Base.@view $lhs)) end new_rhs, rhs_temps = rewrite_lazy_broadcast(rhs, Dict{Any,Symbol}()) return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index a9a8c4b49..86147913c 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -226,8 +226,9 @@ end # A synthetic configuration exercises MP selection without changing the live # runtime cache or requiring multiple GPUs. Size remains a runtime argument. -dl_mp_backend(op, shape) = - cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(true, 4, 4)) +function dl_mp_backend(op, shape) + return cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(true, 4, 4)) +end @testset "distributed linear algebra inference" begin cn = cuNumeric diff --git a/test/array/distributed_linalg.jl b/test/array/distributed_linalg.jl index 486141d40..11350981a 100644 --- a/test/array/distributed_linalg.jl +++ b/test/array/distributed_linalg.jl @@ -1,12 +1,13 @@ using Test, LinearAlgebra, Random -import cuNumeric +using cuNumeric: cuNumeric dl_host(a) = cuNumeric.allowscalar() do - Array(a) + return Array(a) end dl_tol(::Type{T}) where {T} = 200 * eps(real(T)) -dl_backend(op, shape, available, gpus, procs) = - cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(available, gpus, procs)) +function dl_backend(op, shape, available, gpus, procs) + return cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(available, gpus, procs)) +end @testset "linear algebra task selection" begin cn = cuNumeric diff --git a/test/linalg_errors.jl b/test/linalg_errors.jl index f02cb7d7d..cd31ad363 100644 --- a/test/linalg_errors.jl +++ b/test/linalg_errors.jl @@ -8,8 +8,11 @@ op = only(ARGS) a = cuNumeric.zeros(Float64, 33, 33) b = cuNumeric.ones(Float64, 33, 1) cuNumeric.versioninfo() -backend = op == "tiled_cholesky" ? cuNumeric._TiledCholesky() : +backend = if op == "tiled_cholesky" + cuNumeric._TiledCholesky() +else cuNumeric._linalg_backend(Val(Symbol(op)), a) +end println("Testing numerical failure with ", typeof(backend)) @testset "$op numerical failure" begin diff --git a/test/workflows/grayscott.jl b/test/workflows/grayscott.jl index e3dbc14d7..0f863dbcd 100644 --- a/test/workflows/grayscott.jl +++ b/test/workflows/grayscott.jl @@ -358,7 +358,9 @@ function test_scoping_rewrite_pipeline() stmts = utils._flatten_statements(rewritten) @test assigned == Set([:tmp1, :tmp2, :tmp3]) - @test utils._assignment(stmts[1]).lhs == :tmp1 + destination = utils._assignment(stmts[1]) + @test destination.lhs == :tmp1 + @test occursin("@view", sprint(Base.show_unquoted, destination.rhs)) @test utils._assignment(stmts[2]) == (lhs=:tmp2, rhs=:(A[2:(end - 1), 2:(end - 1)])) @test utils._assignment(stmts[3]) == From 168a0f325201a3f0fedc78cd5e3a43a876a0e9d1 Mon Sep 17 00:00:00 2001 From: krasow Date: Sat, 19 Sep 2026 20:33:50 -0500 Subject: [PATCH 33/49] run CI and build [container] From c4157d15634df8c1472faa883e33acab62ab749e Mon Sep 17 00:00:00 2001 From: krasow Date: Sun, 20 Sep 2026 01:23:47 -0500 Subject: [PATCH 34/49] CI: simplfy libcxxwrap validation tests --- .buildkite/run_developer_ci.sh | 4 ++ test/build_cxxwrap.jl | 97 +++++++--------------------------- test/runtests.jl | 2 + 3 files changed, 24 insertions(+), 79 deletions(-) diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 574715e78..f9af18261 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -20,6 +20,10 @@ sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" export PATH="$CMAKE_ROOT/bin:$PATH" cmake --version +# Exercise libcxxwrap cache validation separately from package tests, which run in +# JLL jobs where no build toolchain is installed. +julia --startup-file=no test/build_cxxwrap.jl + # Clean slate so cached state doesn't leak across Julia versions. rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml diff --git a/test/build_cxxwrap.jl b/test/build_cxxwrap.jl index ae978fbf5..38866a5d6 100644 --- a/test/build_cxxwrap.jl +++ b/test/build_cxxwrap.jl @@ -1,92 +1,31 @@ -# Standalone build regression tests; no CUDA, Legate, or network required. +# Standalone build regression test; no CUDA, Legate, or network required. # Run: julia --startup-file=no test/build_cxxwrap.jl using Test include(joinpath(@__DIR__, "..", "deps", "cxxwrap.jl")) -is_supported_version(v::VersionNumber) = v"26.6.0" <= v <= v"26.11.999" -@testset "libcxxwrap build cache recovery" begin +@testset "libcxxwrap cache validation" begin mktempdir() do tmp root = replace(tmp, '\\' => '/') - override = joinpath(root, "dev", "libcxxwrap_julia_jll", "override") - mkpath(override) headers = "$root/headers" - mkpath(headers) library = "$root/libcxxwrap.so" - write(library, "fixture") - config = """ - foreach(name cxxwrap_julia cxxwrap_julia_stl) - add_library(JlCxx::\${name} SHARED IMPORTED) - set_target_properties(JlCxx::\${name} PROPERTIES - INTERFACE_INCLUDE_DIRECTORIES "$headers;" - IMPORTED_CONFIGURATIONS RELEASE - IMPORTED_LOCATION_RELEASE "$library") - endforeach() - """ - config_path = joinpath(override, "JlCxxConfig.cmake") - write(config_path, config) - usable = cxxwrap_usable(override; log_dir=root) - usable || print(read(joinpath(root, "libcxxwrap_check.log"), String)) - @test usable - rm(headers; recursive=true) - @test !cxxwrap_usable(override; log_dir=root) + override = "$root/override" + mkpath(override) mkpath(headers) - rm(library) - @test !cxxwrap_usable(override; log_dir=root) write(library, "fixture") + write( + joinpath(override, "JlCxxConfig.cmake"), + """ +foreach(name cxxwrap_julia cxxwrap_julia_stl) + add_library(JlCxx::\${name} SHARED IMPORTED) + set_target_properties(JlCxx::\${name} PROPERTIES + INTERFACE_INCLUDE_DIRECTORIES "$headers" + IMPORTED_LOCATION "$library") +endforeach() +""", + ) - marker = joinpath(override, "LEGATE_INSTALL.txt") - julia_marker = joinpath(override, "JULIA_INSTALL.txt") - write(marker, "26.6.0") - write(julia_marker, cxxwrap_julia_identity()) - original_depots = copy(DEPOT_PATH) - try - empty!(DEPOT_PATH) - push!(DEPOT_PATH, root) - # A healthy cache must work even with no installer present. - @test isnothing(ensure_cxxwrap(root, v"26.6.0"; log_dir=root)) - scripts = joinpath(root, "scripts") - mkpath(scripts) - installer = joinpath(scripts, "install_cxxwrap.sh") - write(installer, "#!/bin/bash\nexit 7\n") - write(config_path, replace(config, headers => "$root/deleted-headers")) - @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) - @test !isfile(marker) - @test !isfile(julia_marker) - - # Simulate successful rebuilding of the stale CMake export. - write(joinpath(scripts, "JlCxxConfig.cmake"), config) - write(installer, "#!/bin/bash\ncp \"\$1/scripts/JlCxxConfig.cmake\" \"\$1/dev/libcxxwrap_julia_jll/override/JlCxxConfig.cmake\"\n") - @test isnothing(ensure_cxxwrap(root, v"26.6.0"; log_dir=root)) - @test read(marker, String) == "26.6.0" - @test read(julia_marker, String) == cxxwrap_julia_identity() - - # Existing paths and a matching provider version are insufficient - # when Julia changes, or when a legacy build has no Julia marker. - write(installer, "#!/bin/bash\nexit 7\n") - old_version = VERSION.major == 1 && VERSION.minor == 11 ? "1.12.0" : "1.11.0" - write(julia_marker, old_version * "\n" * realpath(joinpath(Sys.BINDIR, Base.julia_exename()))) - @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) - @test !isfile(marker) - @test !isfile(julia_marker) - write(marker, "26.6.0") - @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) - @test !isfile(marker) - - # Moving to another Julia installation also forces a rebuild. - write(marker, "26.6.0") - write(julia_marker, string(VERSION, "\n/old/julia/bin/julia")) - @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) - @test !isfile(julia_marker) - - # A zero exit status without usable outputs is not success. - write(config_path, replace(config, headers => "$root/deleted-headers")) - write(installer, "#!/bin/bash\nexit 0\n") - @test_throws ErrorException ensure_cxxwrap(root, v"26.6.0"; log_dir=root) - @test !isfile(marker) - @test !isfile(julia_marker) - finally - empty!(DEPOT_PATH) - append!(DEPOT_PATH, original_depots) - end + @test cxxwrap_usable(override; log_dir=root) + rm(headers; recursive=true) + @test !cxxwrap_usable(override; log_dir=root) end end diff --git a/test/runtests.jl b/test/runtests.jl index d509ee910..6d88ba144 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -30,6 +30,8 @@ end testsuite = find_tests(@__DIR__) delete!(testsuite, "util") +# This standalone build test requires CMake and runs explicitly in developer CI. +delete!(testsuite, "build_cxxwrap") # These cases deliberately fail a runtime; run individually under a timeout. delete!(testsuite, "linalg_errors") delete!(testsuite, "array/unary/tests") From 61349658bccc562ad4762c2a1bdb434328a2c97d Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Tue, 22 Sep 2026 17:06:10 -0500 Subject: [PATCH 35/49] Container builder update to Julia 1.13 (#201) * ci: update container builder to 1.13 [container] --------- Co-authored-by: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> --- .dockerignore | 44 ++++++++++++++++++++++++++++----- .github/workflows/container.yml | 6 +++-- Project.toml | 7 ++++++ docker/Dockerfile | 9 +++++-- docker/Dockerfile.developer | 9 +++++-- 5 files changed, 63 insertions(+), 12 deletions(-) diff --git a/.dockerignore b/.dockerignore index d658d488f..fd15a3363 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,15 +1,47 @@ -.git -**/.git +# Local Julia environments and generated build configuration. +**/*Manifest.toml +**/*LocalPreferences.toml +dev/Project.toml +deps/deps.jl +build_wrapper.sh +compile_wrapper.sh -**/Manifest.toml -**/LocalPreferences.toml +# Local caches, logs, and editor configuration. **/__pycache__ **/*.pyc **/*.log +**/*.err **/*.prof +**/node_modules +.vscode +.envrc +.localenv +docs/.vitepress +docs/src/.vitepress/cache +docs/src/.vitepress/dist benchmark/results benchmark/plots -docs/node_modules -lib/cunumeric_jl_wrapper/build +# Host build products must not overwrite artifacts built inside the image. +**/build +**/*.o +**/*.obj +**/*.so +**/*.so.* +**/*.dylib +**/*.dll +**/*.a +**/*.lib +**/*.exe deps/cupynumeric-* +deps/lapacke_build +libcupynumeric +install-libcupynumeric +install-cupynumeric +libcxxwrap-julia +lapacke + +# Keep root repository metadata intact, including files matching rules above. +**/.git +!.git +!.git/** diff --git a/.github/workflows/container.yml b/.github/workflows/container.yml index 787a17050..36d75fa74 100644 --- a/.github/workflows/container.yml +++ b/.github/workflows/container.yml @@ -36,8 +36,7 @@ jobs: id-token: write strategy: matrix: - # julia: ["1.10", "1.11"] - julia: ["1.11"] # 1.10 will break + julia: ["1.13"] cuda: ["13.0"] platform: ["linux/amd64"] os: ["ubuntu-22.04"] @@ -52,6 +51,8 @@ jobs: with: ref: ${{ inputs.tag || github.event.release.tag_name || github.sha }} fetch-depth: 0 + # The image includes .git; do not store the checkout token there. + persist-credentials: false - name: Select wrapper build id: wrapper-build @@ -190,6 +191,7 @@ jobs: - name: Build image uses: docker/build-push-action@v6 with: + context: . file: ${{ steps.wrapper-build.outputs.dockerfile }} load: true push: false diff --git a/Project.toml b/Project.toml index ef407e436..cdbfc2844 100644 --- a/Project.toml +++ b/Project.toml @@ -57,3 +57,10 @@ TensorOperations = "5.8" cunumeric_jl_wrapper_jll = "26.6.4" cupynumeric_jll = "26.6.0" julia = "1.10" + +[extras] +CUDA_Runtime_jll = "76a88914-d11a-5bdc-97e0-2f5a05c973a2" +CUDA_Compiler_jll = "d1e2174e-dfdc-576e-b43e-73b79eb1aca8" + +[targets] +build = ["CUDA_Runtime_jll", "CUDA_Compiler_jll"] diff --git a/docker/Dockerfile b/docker/Dockerfile index fc2d812e8..56cdebdc1 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -1,4 +1,4 @@ -ARG JULIA_VERSION=1.11 +ARG JULIA_VERSION=1.13 FROM julia:${JULIA_VERSION} ARG CUDA_MAJOR=13 @@ -46,7 +46,7 @@ ENV PATH="/usr/local/.juliaup/bin:/usr/local/bin:$PATH" # install CUDA.jl itself. RUN julia --color=yes -e 'using Pkg; Pkg.add("CUDA"); using CUDA; CUDA.set_runtime_version!(VersionNumber(ENV["CUDA_VERSION_MAJOR_MINOR"]))' -RUN julia -e 'using Pkg; Pkg.add(name = "CUDA_Driver_jll", version = "13.0.0"); Pkg.add("CUDA_Runtime_jll")' +RUN julia -e 'using Pkg; Pkg.add(["CUDA_Driver_jll", "CUDA_Runtime_jll"])' RUN echo "export LD_LIBRARY_PATH=\$(julia -e 'print(Sys.BINDIR * \"/../lib\")'):\$(julia -e 'using CUDA_Driver_jll; print(joinpath(CUDA_Driver_jll.artifact_dir, \"lib\"))'):\$(julia -e 'using CUDA_Runtime_jll; print(joinpath(CUDA_Runtime_jll.artifact_dir, \"lib\"))'):\$LD_LIBRARY_PATH" >> /etc/.env RUN chmod +x /etc/.env RUN cat /etc/.env @@ -113,6 +113,11 @@ RUN apt-get clean \ ENV LEGATE_AUTO_CONFIG=1 +# Include the complete checkout and history after package installation so +# documentation and history changes do not invalidate the expensive layers. +COPY . /opt/cuNumeric.jl/ +RUN git -C /opt/cuNumeric.jl rev-parse --verify HEAD + COPY docker/profile.sh /etc/profile.d/cunumeric.sh COPY docker/entrypoint.sh /usr/local/bin/cunumeric-entrypoint RUN chmod +x /usr/local/bin/cunumeric-entrypoint diff --git a/docker/Dockerfile.developer b/docker/Dockerfile.developer index 8b82c625e..0a4f08525 100644 --- a/docker/Dockerfile.developer +++ b/docker/Dockerfile.developer @@ -1,4 +1,4 @@ -ARG JULIA_VERSION=1.11 +ARG JULIA_VERSION=1.13 FROM julia:${JULIA_VERSION} ARG CUDA_MAJOR=13 @@ -51,7 +51,7 @@ ENV PATH="/usr/local/.juliaup/bin:/usr/local/bin:$PATH" # install CUDA.jl itself. RUN julia --color=yes -e 'using Pkg; Pkg.add("CUDA"); using CUDA; CUDA.set_runtime_version!(VersionNumber(ENV["CUDA_VERSION_MAJOR_MINOR"]))' -RUN julia -e 'using Pkg; Pkg.add(name = "CUDA_Driver_jll", version = "13.0.0"); Pkg.add("CUDA_Runtime_jll")' +RUN julia -e 'using Pkg; Pkg.add(["CUDA_Driver_jll", "CUDA_Runtime_jll"])' RUN echo "export LD_LIBRARY_PATH=\$(julia -e 'print(Sys.BINDIR * \"/../lib\")'):\$(julia -e 'using CUDA_Driver_jll; print(joinpath(CUDA_Driver_jll.artifact_dir, \"lib\"))'):\$(julia -e 'using CUDA_Runtime_jll; print(joinpath(CUDA_Runtime_jll.artifact_dir, \"lib\"))'):\$LD_LIBRARY_PATH" >> /etc/.env RUN chmod +x /etc/.env RUN cat /etc/.env @@ -121,6 +121,11 @@ RUN apt-get clean \ ENV LEGATE_AUTO_CONFIG=1 +# Include the complete checkout and history after package installation so +# documentation and history changes do not invalidate the expensive layers. +COPY . /opt/cuNumeric.jl/ +RUN git -C /opt/cuNumeric.jl rev-parse --verify HEAD + COPY docker/profile.sh /etc/profile.d/cunumeric.sh COPY docker/entrypoint.sh /usr/local/bin/cunumeric-entrypoint RUN chmod +x /usr/local/bin/cunumeric-entrypoint From 4d15553bcf9dd6b57206750fa82483084e757e0c Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:48:40 -0400 Subject: [PATCH 36/49] Add some linear algebra ops on vectors (#202) * Add matvec, dot, norm impls --- README.md | 8 ++ TODO.md | 12 +-- docs/src/linalg.md | 27 +++++ src/cuNumeric.jl | 1 + src/cuda/mapreduce.jl | 3 +- src/ndarray/binary.jl | 20 ++++ src/ndarray/detail/ndarray.jl | 9 +- src/ndarray/vector_linalg.jl | 163 ++++++++++++++++++++++++++++ test/analysis/lifetime.jl | 29 +++++ test/array/vector_linalg.jl | 194 ++++++++++++++++++++++++++++++++++ test/gpu_only/vector_norm.jl | 63 +++++++++++ 11 files changed, 516 insertions(+), 13 deletions(-) create mode 100644 src/ndarray/vector_linalg.jl create mode 100644 test/array/vector_linalg.jl create mode 100644 test/gpu_only/vector_norm.jl diff --git a/README.md b/README.md index 7942c54f6..7dc87862f 100644 --- a/README.md +++ b/README.md @@ -45,6 +45,14 @@ s = sum(A) # NDArray{T,0} x = unwrap(s) # T, e.g. Float32 ``` +`LinearAlgebra.dot` on vectors and `LinearAlgebra.norm` on dense NDArrays +return 0D NDArrays, keeping reductions asynchronous. Extract a Julia scalar explicitly with `only` or `unwrap` when needed. +See [Linear algebra](https://julialegate.github.io/cuNumeric.jl/dev/linalg). + +**0D arrays support scalar-shaped arithmetic.** Use `+`, `-`, `*`, `/`, and `^` +with two 0D NDArrays or with a 0D NDArray and a Julia number. Results stay as +0D NDArrays on the backend, following the existing promotion policy. + **The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into a task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them [for example `println`, `unwrap`, or communicating with the Julia runtime (i.e., `Array(A)`)]. Hiding latency enables performant code. For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/dev/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/dev/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/dev/perf/patterns_to_avoid). diff --git a/TODO.md b/TODO.md index 34663324d..9bbfa7bdf 100644 --- a/TODO.md +++ b/TODO.md @@ -18,9 +18,6 @@ corresponding `Base` methods are missing, so calls fall through to - `fill!` (convertible eltypes) - `collect` -**P0 done** -- `iszero` / `isone` (on-device reductions → `Bool` via scalar sync; `isone` square 2D only) - **P1** - `transpose` / `adjoint` - `unique` @@ -46,25 +43,18 @@ Starter list of easy/medium LA gaps. Prefer wiring `LinearAlgebra` entry points so they do not fall through to scalar-indexing Base paths. **BLAS-1 style (dense `NDArray`)** -- `axpy!` / `axpby!` — common scale-and-add; examples use broadcast today -- `scal!` and related in-place scale -- `LinearAlgebra.dot` if not already Base-wired to `nda_dot` +- Additional BLAS-style indexed updates. **Reductions / traces** - `LinearAlgebra.tr` for dense 2D `NDArray` — `cuNumeric.trace` exists; `tr` is already wired for `Diagonal{<:NDArray}` **Diagonal vs fallthrough (context)** -- Already on-device for `Diagonal`: `mul!` / `lmul!` / `rmul!`, `\` / `/`, - `tr`, `norm` / `opnorm`, many predicates — see `docs/src/linalg.md` - Still fallthrough / unsupported on `Diagonal` (e.g. `svd`, `pinv`, `cholesky`, host `AbstractArray` RHS): leave alone unless fixing is cheap; densify intentionally when needed **Decompositions** -- Done: `cholesky`, `eigen` / `eigvals` / `eigvecs`, `svd`, `qr`, and `\` are - wired to `LinearAlgebra` for 2D `NDArray`; `batched_solve` / `batched_cholesky` - / `batched_eigen` / `batched_eigvals` cover one batch dimension - `eigh` / Hermitian eigen — the `SYEV` task is already wrapped, but there is no entry point until `Hermitian` / `Symmetric` work on `NDArray` - `PosDefException` from `cholesky` — a non-positive-definite input now raises a diff --git a/docs/src/linalg.md b/docs/src/linalg.md index e3f6a6efb..bbcc89ff6 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -97,6 +97,33 @@ Pages = ["ndarray/binary.jl"] Filter = t -> t isa Function && nameof(t) === :mul! ``` +## Vector operations + +Numeric `NDArray`s support matrix-vector `A * x`, three- and five-argument +`mul!`, vector `dot`, `axpy!`, `axpby!`, and scalar `lmul!` / `rmul!`. +Five-argument matrix-matrix `mul!` is also available. Mixed types follow the +package's explicit promotion policy. Matrix multiplication rejects +integer-integer inputs (including Bool); mixed integer/floating-point inputs +are supported when their promoted type is supported by the matrix kernel. +Destinations of `mul!` must not alias inputs. Vector updates support exact +self-aliasing, but not partially overlapping views. + +`dot` conjugates its first argument. **`dot` and dense-array `norm` return 0D +NDArrays**, keeping their results on the backend without implicit scalar +extraction or synchronization. Use `only` or `unwrap` explicitly when a Julia +scalar is required. Bool-Bool dot products accumulate into `Int` under the +existing promotion policy. + +`norm(A, p)` is an entrywise norm, not `opnorm`. It uses mapped reductions and +currently requires a GPU target. The 2-norm fuses `abs2` into a sum reduction, +then applies a backend square root. Like cuPyNumeric, powers are accumulated +without scaling and can overflow or underflow. Integer inputs are converted +to floating point under the existing promotion policy. + +Matrix-vector contraction selects cuPyNumeric's specialized `MATVECMUL` task. +Generic solvers that require Julia scalar reductions, including IterativeSolvers +CG, need adaptation to work with these asynchronous reduction results. + ## Solve `cuNumeric.solve(A, b)` solves a linear system and returns an array with the same diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 8eafe040e..553bf086b 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -208,6 +208,7 @@ include("ndarray/linalg.jl") include("ndarray/sort.jl") include("ndarray/batched_linalg.jl") include("ndarray/contract.jl") +include("ndarray/vector_linalg.jl") include("ndarray/fft.jl") include("scoping/scoping.jl") diff --git a/src/cuda/mapreduce.jl b/src/cuda/mapreduce.jl index 25fa58b61..771abfff0 100644 --- a/src/cuda/mapreduce.jl +++ b/src/cuda/mapreduce.jl @@ -250,11 +250,12 @@ function _mr_launch(f, op, A::NDArray{T,N}, ::Type{R}, ::Type{O}, mask, shape, i result = needs_finish ? nda_empty_array(shape, O) : accumulator _mr_submit(A, accumulator, result, mask, dims isa Colon, single, _mr_redop(op, S), name, contribute_name, finish_name, mapper, finish) + # Returning here keeps the cleanup sentinel out of Julia 1.10 inference. + return result catch isnothing(result) || destroy!(result) rethrow() finally result === accumulator || destroy!(accumulator) end - return result end diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 7f4231e02..8dfb9beee 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -140,6 +140,26 @@ function Base.:(+)(rhs1::NDArray{A,N}, rhs2::NDArray{B,N}) where {A,B,N} return _nda_binary_op_promoted!(out, cuNumeric.ADD, rhs1, rhs2) end +# Scalar-shaped arithmetic stays on the backend. Array-array + and - above +# already support rank zero; higher-rank * and / keep their linear algebra meaning. +# This deliberately departs from Julia's standard array API: a 0D Array is not +# a Number, and Base does not support this full set of scalar-style operations +# on it. NDArray supports them to keep reduction arithmetic asynchronous. +for op in (:*, :/, :^) + @eval function Base.$op( + a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, b::NDArray{<:SUPPORTED_ARRAY_TYPES,0} + ) + return broadcast($op, a, b) + end +end +for op in (:+, :-, :*, :/, :^) + @eval begin + Base.$op(a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, b::Number) = broadcast($op, a, b) + Base.$op(a::Number, b::NDArray{<:SUPPORTED_ARRAY_TYPES,0}) = broadcast($op, a, b) + end +end +Base.literal_pow(::typeof(^), a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, ::Val{P}) where {P} = a ^ P + function Base.:(*)(val::V, arr::NDArray{A}) where {A,V<:Number} return _mul_scalar(__my_promote_type(A, V), val, arr) end diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 2c6b9fb96..fafb0bd87 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -658,7 +658,14 @@ function get_ptr(arr::NDArray{T,N}) where {T,N} # store with the NDArray; finalize after use (same pin class as `_add_task_array!`). st_handle = get_store(arr) # LogicalArrayImplAllocated (returned by value) la = Legate.LogicalArray{T,N}(st_handle, size(arr)) - ptr = Legate.get_ptr(la) + # Legate.get_ptr(::LogicalArray) leaves PhysicalArray/PhysicalStore handles + # to GC. Their destructors can unmap regions, which must run on the runtime + # thread, just like the logical handle below. Keep and release them here. + physical = Legate.get_physical_array(la) + data = Legate.data(physical) + ptr = Legate.get_ptr(data) + finalize(data) + finalize(physical) finalize(st_handle) return ptr end diff --git a/src/ndarray/vector_linalg.jl b/src/ndarray/vector_linalg.jl new file mode 100644 index 000000000..2855e931b --- /dev/null +++ b/src/ndarray/vector_linalg.jl @@ -0,0 +1,163 @@ +# LinearAlgebra interfaces retain asynchronous NDArray reduction results. +const _LA_FLOAT = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const _LA_INTEGER = Union{SUPPORTED_INT_TYPES,Bool} + +_matmul_eltype(::Type{T}) where {T<:_LA_FLOAT} = T +function _matmul_eltype(::Type{T}) where {T<:_LA_INTEGER} + throw(ArgumentError( + "NDArray matrix multiplication does not support integer-integer operands (including Bool); " * + "the inputs promote to $T. Convert an operand to a floating-point type explicitly.", + )) +end + +""" + mul!(y::NDArray, A::NDArray, x::NDArray[, α, β]) + +Compute `y = α * A * x + β * y` for a matrix and vector. Five-argument +matrix-matrix multiplication is also supported. Mixed integer/floating-point +inputs follow the usual promotion policy; integer-integer inputs are unsupported. +The destination must hold the promoted result and must not alias either input. +""" +function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + return mul!(y, A, x, true, false) +end + +function _linalg_mul!(C::NDArray{T}, cm, A::NDArray{TA}, am, B::NDArray{TB}, bm, α, β) where {T,TA,TB} + required = _matmul_eltype(promote_type(TA, TB)) + promote_type(required, T) === T || throw(ArgumentError("mul! output type $T cannot hold promoted input type $required")) + Ap = checked_promote_arr(mul!, A, T) + Bp = checked_promote_arr(mul!, B, T) + if iszero(α) || isempty(A) || isempty(B) || isempty(C) + _contract_prepare(C, cm, Ap, am, Bp, bm) + if !isempty(C) + if iszero(β) + fill!(C, zero(T)) + else + C .*= convert(T, β) + end + end + else + _contract_same_type!(C, cm, Ap, am, Bp, bm, α, β) + end + Ap !== A && destroy!(Ap) + Bp !== B && destroy!(Bp) + return C +end + +function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, α::Number, β::Number) + return _linalg_mul!(y, "i", A, "ij", x, "j", α, β) +end + +function LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, B::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, α::Number, β::Number) + return _linalg_mul!(C, "ij", A, "ik", B, "kj", α, β) +end + +function Base.:*(A::NDArray{TA,2}, x::NDArray{TX,1}) where {TA<:SUPPORTED_ARRAY_TYPES,TX<:SUPPORTED_ARRAY_TYPES} + T = _matmul_eltype(promote_type(TA, TX)) + size(A, 2) == length(x) || throw(DimensionMismatch("matrix-vector dimensions do not match")) + y = cuNumeric.zeros(T, size(A, 1)) + return mul!(y, A, x) +end + +_dot_eltype(::Type{T}) where {T} = T +_dot_eltype(::Type{Bool}) = Int +_dot_same_type(x::NDArray{T,1}, y::NDArray{T,1}) where {T<:Real} = nda_dot(x, y) +function _dot_same_type(x::NDArray{T,1}, y::NDArray{T,1}) where {T<:Complex} + cx = conj.(x) + result = nda_dot(cx, y) + destroy!(cx) + return result +end + +""" + dot(x::NDArray{<:Any,1}, y::NDArray{<:Any,1}) + +Hermitian inner product as a 0D NDArray, without unwrapping or synchronizing. +Conjugates the first operand for complex inputs. Numeric inputs are promoted +using the package's existing policy. Bool-Bool dot accumulates into Int. +""" +function LinearAlgebra.dot(x::NDArray{TX,1}, y::NDArray{TY,1}) where {TX<:SUPPORTED_ARRAY_TYPES,TY<:SUPPORTED_ARRAY_TYPES} + length(x) == length(y) || throw(DimensionMismatch("dot vector lengths do not match")) + T = _dot_eltype(promote_type(TX, TY)) + xp = checked_promote_arr(dot, x, T) + yp = checked_promote_arr(dot, y, T) + result = isempty(x) ? cuNumeric.zeros(T, ()) : _dot_same_type(xp, yp) + xp !== x && destroy!(xp) + yp !== y && destroy!(yp) + return result +end + +_norm_nonzero(v) = ifelse(iszero(v), zero(real(v)), one(real(v))) + +""" + norm(x::NDArray, p::Real=2) + +Entrywise p-norm as a real-valued 0D NDArray (not the matrix operator norm). +The result stays on the backend; no reduction is unwrapped. Like cuPyNumeric, +powers are accumulated without scaling and may overflow or underflow. Integer +inputs convert to floating point under the existing promotion policy. +Uses mapped reductions, which currently require a GPU target. +""" +function LinearAlgebra.norm(x::NDArray{T}, p::Real=2) where {T<:_LA_FLOAT} + R = real(T) + isempty(x) && return cuNumeric.zeros(R, ()) + p == 0 && return sum(_norm_nonzero, x) + p == 1 && return sum(abs, x) + p == Inf && return maximum(abs, x) + p == -Inf && return minimum(abs, x) + isnan(p) && return NDArray(R(NaN)) + exponent = R(p) + total = p == 2 ? sum(abs2, x) : sum(v -> abs(v)^exponent, x) + # Optimization opportunity: a specialized reduction could fuse the root into + # its final combine kernel, after all partitions' contributions are combined. + # dims=() uses the elementwise singleton path for this 0D result: one root + # kernel, without unwrapping or launching another full reduction. + result = p == 2 ? mapreduce(sqrt, +, total; dims=()) : + mapreduce(v -> v^inv(exponent), +, total; dims=()) + destroy!(total) + return result +end + +function LinearAlgebra.norm(x::NDArray{T}, p::Real=2) where {T<:_LA_INTEGER} + xp = checked_promote_arr(norm, x, float(T)) + result = norm(xp, p) + destroy!(xp) + return result +end + +""" + axpy!(α, x::NDArray{<:Any,1}, y::NDArray{<:Any,1}) + axpby!(α, x::NDArray{<:Any,1}, β, y::NDArray{<:Any,1}) + +Update numeric vectors with backend broadcasts and return `y`. Vector lengths +must match. Exact self-aliasing is supported; partially overlapping views are not. +""" +function LinearAlgebra.axpy!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + length(x) == length(y) || throw(DimensionMismatch("axpy! vector lengths do not match")) + iszero(α) && return y + y .= α .* x .+ y + return y +end + +function LinearAlgebra.axpby!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, β::Number, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + length(x) == length(y) || throw(DimensionMismatch("axpby! vector lengths do not match")) + iszero(α) && isone(β) && return y + y .= α .* x .+ β .* y + return y +end + +function LinearAlgebra.rmul!(x::NDArray{<:SUPPORTED_ARRAY_TYPES}, α::Number) + x .*= α + return x +end + +function LinearAlgebra.lmul!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES}) + x .= α .* x + return x +end + +function LinearAlgebra.ldiv!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, D::DiagonalNDArray{<:SUPPORTED_ARRAY_TYPES}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + length(x) == length(y) == size(D, 1) || throw(DimensionMismatch("diagonal solve dimensions do not match")) + y .= x ./ _diag_vec(D) + return y +end diff --git a/test/analysis/lifetime.jl b/test/analysis/lifetime.jl index 45815689c..d21a1459f 100644 --- a/test/analysis/lifetime.jl +++ b/test/analysis/lifetime.jl @@ -46,6 +46,35 @@ @allowscalar @test Array(a) == reshape(Float64.(1:15), 5, 3) end +@testset "Host-copy handles do not reach off-thread GC" begin + # Run with both thread pools enabled to exercise the original abort. All + # array operations stay on the runtime thread; only GC runs on the other pool. + if Threads.nthreads(:interactive) > 0 + runtime_thread = Threads.threadid() + function copy_and_release() + x = cuNumeric.ones(Float64, 8) + @test Array(x) == ones(8) + cuNumeric.destroy!(x) + end + function collect_elsewhere() + @test Threads.threadid() != runtime_thread + GC.gc(true) + end + for _ in 1:10 + copy_and_release() + task = if Threads.threadpool() === :interactive + Threads.@spawn :default collect_elsewhere() + else + Threads.@spawn :interactive collect_elsewhere() + end + fetch(task) + cuNumeric.drain_pending_frees!() + end + else + @test_skip false # Requires an additional thread pool. + end +end + @testset "Array ↔ NDArray value roundtrip (row-major attach)" begin A = rand(Float64, 4, 4) NA = NDArray(A) diff --git a/test/array/vector_linalg.jl b/test/array/vector_linalg.jl new file mode 100644 index 000000000..6e0b9a85a --- /dev/null +++ b/test/array/vector_linalg.jl @@ -0,0 +1,194 @@ +using Test, LinearAlgebra, Random, cuNumeric + +# Host extraction belongs to validation, not the NDArray implementation. +la_value(x::cuNumeric.NDArray{T,0}) where {T} = only(x) + +@testset "LinearAlgebra vector interface" begin + cuNumeric.allowscalar(false) + begin + Random.seed!(912) + for T in Base.uniontypes( + Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_COMPLEX_TYPES} + ) + @testset "$T" begin + R = real(T) + ah, xh, yh = randn(T, 7, 5), randn(T, 5), randn(T, 7) + a, x, y = cuNumeric.NDArray(ah), cuNumeric.NDArray(xh), cuNumeric.NDArray(yh) + @test mul!(y, a, x) === y + @test Array(y) ≈ ah * xh + @test Array(a * x) ≈ ah * xh + copyto!(y, cuNumeric.NDArray(yh)) + @test mul!(y, a, x, T(2), T(-3)) === y + @test Array(y) ≈ 2ah * xh - 3yh + fill!(y, T(NaN)) + mul!(y, a, x, T(2), zero(T)) + @test Array(y) ≈ 2ah * xh + bad = cuNumeric.NDArray(fill(T(NaN), 7, 5)) + mul!(y, bad, x, zero(T), zero(T)) + @test all(iszero, Array(y)) + @test_throws DimensionMismatch mul!(similar(x), a, x) + square = cuNumeric.NDArray(randn(T, 5, 5)) + @test_throws ArgumentError mul!(x, square, x) + bh = randn(T, 5, 3) + ch = randn(T, 7, 3) + c = cuNumeric.NDArray(ch) + mul!(c, a, cuNumeric.NDArray(bh), T(2), T(3)) + @test Array(c) ≈ 2ah * bh + 3ch + + zh = randn(T, 5) + z = cuNumeric.NDArray(zh) + @test dot(x, z) isa cuNumeric.NDArray{T,0} + @test la_value(dot(x, z)) ≈ dot(xh, zh) + @test_throws DimensionMismatch dot(x, y) + empty = cuNumeric.zeros(T, 0) + @test la_value(dot(empty, empty)) === zero(T) + @test axpy!(T(2), x, z) === z + @test Array(z) ≈ 2xh + zh + @test axpby!(T(-2), x, T(3), z) === z + @test Array(z) ≈ -2xh + 3(2xh + zh) + @test axpy!(one(T), x, x) === x + @test Array(x) ≈ 2xh + @test axpby!(T(2), x, T(3), x) === x + @test Array(x) ≈ 10xh + @test rmul!(x, T(2)) === x + @test lmul!(T(3), x) === x + @test Array(x) ≈ 60xh + @test_throws DimensionMismatch axpy!(one(T), x, y) + @test_throws DimensionMismatch axpby!(one(T), x, one(T), y) + d = Diagonal(cuNumeric.NDArray(T[1, 2, 3, 4, 5])) + ldiv!(z, d, x) + @test Array(z) ≈ 60xh ./ T[1, 2, 3, 4, 5] + + # Vector views must stay on the backend for updates and products. + storage = cuNumeric.zeros(T, 9) + dest = view(storage, 2:8) + mul!(dest, a, cuNumeric.NDArray(xh)) + @test Array(storage)[2:8] ≈ ah * xh + end + end + empty_a = cuNumeric.zeros(Float64, 3, 0) + out = cuNumeric.ones(Float64, 3) + mul!(out, empty_a, cuNumeric.zeros(Float64, 0)) + @test all(iszero, Array(out)) + cuNumeric.allowpromotion() do + a32 = cuNumeric.NDArray(Float32[2 1; 0 3]) + x64 = cuNumeric.NDArray([2.0, 4.0]) + y64 = similar(x64) + mul!(y64, a32, x64) + @test Array(y64) ≈ [8.0, 12.0] + @test la_value(dot(cuNumeric.NDArray(Float32[1, 2]), x64)) ≈ 10.0 + @test_throws ArgumentError mul!(cuNumeric.zeros(Float32, 2), a32, x64) + end + end +end + +@testset "LinearAlgebra promotion across numeric types" begin + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + types = Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + for T in types + @testset "updates $T" begin + h = ones(T, 2) + y = cuNumeric.ones(T,2) + z = cuNumeric.zeros(T,2) + @test axpy!(one(T),z,y) === y + @test Array(y) == h + @test axpby!(one(T),z,one(T),y) === y + @test Array(y) == h + @test lmul!(one(T),y) === y + @test rmul!(y,one(T)) === y + @test Array(y) == h + end + end + for T in types, S in types + @testset "$T / $S" begin + a = cuNumeric.ones(T,2,2) + b = cuNumeric.ones(S,2,2) + x, z = cuNumeric.ones(S,2), cuNumeric.ones(T,2) + expected_dot = dot(ones(T,2),ones(S,2)) + @test la_value(dot(z,x)) == expected_dot + # Only matrix kernels reject integer-integer inputs. + if !(T <: Integer && S <: Integer) + R = promote_type(T,S) + @test Array(a*x) ≈ fill(R(2),2) + y = cuNumeric.zeros(R,2) + @test mul!(y,a,x) === y + @test Array(y) ≈ fill(R(2),2) + mul!(y,a,x,R(2),R(3)) + @test Array(y) ≈ fill(R(10),2) + c = cuNumeric.zeros(R,2,2) + mul!(c,a,b,R(2),R(0)) + @test Array(c) ≈ fill(R(4),2,2) + end + end + end + end +end + +@testset "Integer-integer matrix multiplication errors" begin + cuNumeric.allowscalar(false) + ints = Base.uniontypes(Union{Bool,cuNumeric.SUPPORTED_INT_TYPES}) + for T in ints, S in ints + A, x = cuNumeric.ones(T, 2, 2), cuNumeric.ones(S, 2) + B = cuNumeric.ones(S, 2, 2) + y, C = cuNumeric.zeros(Float64, 2), cuNumeric.zeros(Float64, 2, 2) + for call in (() -> A*x, () -> mul!(y,A,x), () -> mul!(y,A,x,1,0), + () -> mul!(C,A,B,1,0), () -> mul!(y,A,x,0,0)) + err = try + call() + nothing + catch e + e + end + @test err isa ArgumentError + if err isa ArgumentError + @test occursin("integer-integer", sprint(showerror,err)) + @test occursin("Convert an operand", sprint(showerror,err)) + end + end + end + A = cuNumeric.ones(Float64,2,2) + x = cuNumeric.ones(Int32,2) + @test_throws ArgumentError mul!(cuNumeric.zeros(Int32,2), A, x) + @test_throws DimensionMismatch A * cuNumeric.ones(Float64,3) + cuNumeric.allowpromotion(false) do + @test_throws "Implicit promotion" A * cuNumeric.ones(Int8,2) + end +end + +@testset "0D coefficient arithmetic" begin + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + types = Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + pairs = [(T, T) for T in types] + append!(pairs, [(Float32, Float64), (Int32, Float32), (Float64, Int8), + (ComplexF32, Float32), (ComplexF64, Int64), (Bool, Int8), (Int8, UInt8)]) + for (T, S) in pairs + a, b = NDArray(one(T)), NDArray(one(S)) + for op in (+, -, *, /, ^) + for (x, y) in ((a, b), (a, one(S)), (one(T), b)) + result = @inferred op(x, y) + expected = op(one(T), one(S)) + @test result isa NDArray{typeof(expected),0} + @test only(result) ≈ expected + end + end + end + for T in Base.uniontypes(Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_COMPLEX_TYPES}) + a, b = NDArray(T(6)), NDArray(T(2)) + @test only(a-b) ≈ T(4) + @test only(a/b) ≈ T(3) + @test only(T(12)/a) ≈ T(2) + @test only(T(12)-a) ≈ T(6) + @test only(a^2) ≈ T(36) + exponent = 3 + @test only(a^exponent) ≈ T(216) + @test only(a^b) ≈ T(36) + @test only(T(2)^b) ≈ T(4) + @test only(a^(-1)) ≈ inv(T(6)) + end + end + # Higher-dimensional products still mean matrix multiplication. + A = NDArray([1.0 2.0; 3.0 4.0]) + @test Array(A*A) ≈ [7.0 10.0; 15.0 22.0] +end diff --git a/test/gpu_only/vector_norm.jl b/test/gpu_only/vector_norm.jl new file mode 100644 index 000000000..6743df41b --- /dev/null +++ b/test/gpu_only/vector_norm.jl @@ -0,0 +1,63 @@ +using Test, LinearAlgebra, Random, cuNumeric + +la_value(x::cuNumeric.NDArray{T,0}) where {T} = only(x) + +@testset "NDArray norms" begin + cuNumeric.allowscalar(false) + Random.seed!(912) + for T in Base.uniontypes(Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_COMPLEX_TYPES}) + @testset "$T" begin + R = real(T) + ah, xh = randn(T, 7, 5), randn(T, 5) + a, x = cuNumeric.NDArray(ah), cuNumeric.NDArray(xh) + empty = cuNumeric.zeros(T, 0) + @test la_value(norm(empty)) === zero(R) + @test la_value(norm(cuNumeric.zeros(T, 5))) === zero(R) + for p in (0, 1, 2, 3, 0.5, -1, -2, Inf, -Inf) + @test la_value(norm(x, p)) ≈ norm(xh, p) + @test la_value(norm(a, p)) ≈ norm(ah, p) + end + @test norm(x) isa cuNumeric.NDArray{R,0} + @test la_value(norm(cuNumeric.NDArray(T[0, 2, 0, 3]), 0)) == R(2) + @test la_value(norm(cuNumeric.NDArray(T[0, 2]), -1)) == zero(R) + @test isnan(la_value(norm(cuNumeric.NDArray(T[NaN, 1])))) + @test isinf(la_value(norm(cuNumeric.NDArray(T[Inf, 1])))) + # Unscaled powers follow cuPyNumeric's overflow/underflow tradeoff. + @test isinf(la_value(norm(cuNumeric.NDArray(fill(T(floatmax(R)/4), 2))))) + @test iszero(la_value(norm(cuNumeric.NDArray(fill(T(floatmin(R)), 2))))) + + end + end + cuNumeric.allowpromotion() do + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + h = ones(T, 2) + x = cuNumeric.NDArray(h) + @test la_value(norm(x)) ≈ norm(h) + @test la_value(norm(x, 0)) ≈ norm(h, 0) + end + end +end + +# Prevent constant propagation of p from hiding branch-dependent return types. +Base.@noinline function check_norm_inference(x::cuNumeric.NDArray{T}, p::Real) where {T} + R = typeof(float(real(zero(T)))) + @test (@inferred norm(x, p)) isa cuNumeric.NDArray{R,0} +end + +@testset "NDArray norm inference" begin + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + @testset "$T" begin + x = cuNumeric.ones(T, 3) + R = typeof(float(real(zero(T)))) + @test (@inferred norm(x)) isa cuNumeric.NDArray{R,0} + for p in (0, 1, 2, 3, -1, -2, 0.5, Inf, -Inf, NaN, 2.0f0) + check_norm_inference(x, p) + end + check_norm_inference(cuNumeric.zeros(T, 0), 2) + check_norm_inference(cuNumeric.ones(T, 2, 2), 2) + end + end + end +end From b56fdd9fa5feeee9ac3ec140a95eb7af3aabf36e Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:04:53 -0400 Subject: [PATCH 37/49] NDScalar Type + Opt-In Auto Unwrapping of Scalars (#203) * Add device scalar wrappers, CNScalar, which wrap 0D NDArrays * Define arithmetic on CNScalar which can be optionally converted to a Julia Number automatically with allowfetch --- README.md | 12 +- docs/make.jl | 1 + docs/src/api_cnscalar.md | 97 +++++++++ docs/src/api_mapreduce.md | 4 +- docs/src/api_tensor.md | 8 +- docs/src/api_unary.md | 4 +- docs/src/examples/montecarlo.md | 2 +- docs/src/examples/tensor_network.md | 6 +- docs/src/index.md | 8 +- docs/src/linalg.md | 45 ++--- docs/src/perf/patterns_to_avoid.md | 9 +- examples/poisson_fft.jl | 4 +- examples/tensor_network.jl | 6 +- ext/cuNumericTensorOperationsExt.jl | 13 +- src/cnscalar.jl | 300 ++++++++++++++++++++++++++++ src/cuNumeric.jl | 1 + src/ndarray/contract.jl | 9 +- src/ndarray/diagonal.jl | 81 +++++--- src/ndarray/mapreduce.jl | 8 +- src/ndarray/ndarray.jl | 16 +- src/ndarray/random/generator.jl | 20 +- src/ndarray/sort.jl | 8 +- src/ndarray/unary.jl | 26 +-- src/ndarray/vector_linalg.jl | 88 +++++--- test/array/binary/tests.jl | 26 +-- test/array/cnscalar.jl | 137 +++++++++++++ test/array/diagonal.jl | 12 +- test/array/random.jl | 2 +- test/array/sort.jl | 4 +- test/array/tensoroperations.jl | 25 ++- test/array/unary/tests.jl | 20 +- test/array/vector_linalg.jl | 4 +- test/gpu_only/mapreduce.jl | 12 +- test/gpu_only/mapreduce_full.jl | 6 +- test/gpu_only/vector_norm.jl | 8 +- test/util.jl | 2 +- 36 files changed, 810 insertions(+), 224 deletions(-) create mode 100644 docs/src/api_cnscalar.md create mode 100644 src/cnscalar.jl create mode 100644 test/array/cnscalar.jl diff --git a/README.md b/README.md index 7dc87862f..647f367b1 100644 --- a/README.md +++ b/README.md @@ -38,22 +38,20 @@ The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it **Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutating an `NDArray` slice mutates the parent and all other aliases of the underlying data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the task graph asynchronous instead of forcing synchronization to communicate with the Julia runtime. When you need a plain Julia number, call `unwrap` or `only`: +**Scalar reductions return device scalars.** Full reductions such as `sum(A)`, `dot(x,y)`, and `norm(x)` return numeric wrappers (`CNFloat`, `CNInt`, `CNUInt`, `CNBool`, or `CNComplex`) backed by 0D NDArrays. Each subtypes the corresponding abstract numeric category; `CNReal` and `CNScalar` group the wrapper families. Dimension-preserving reductions still return NDArrays. Arithmetic stays on the backend; use `fetch` for explicit host extraction: ```julia -s = sum(A) # NDArray{T,0} -x = unwrap(s) # T, e.g. Float32 +s = sum(A) # CNScalar +x = fetch(s) # native Julia scalar, e.g. Float32 ``` -`LinearAlgebra.dot` on vectors and `LinearAlgebra.norm` on dense NDArrays -return 0D NDArrays, keeping reductions asynchronous. Extract a Julia scalar explicitly with `only` or `unwrap` when needed. -See [Linear algebra](https://julialegate.github.io/cuNumeric.jl/dev/linalg). +Use `allowautofetch() do ... end` or `@allowautofetch` to permit host comparisons and numeric conversions of device scalars. Permission is task-local and disabled by default; arithmetic remains on the backend. See [Device scalars](https://julialegate.github.io/cuNumeric.jl/dev/api_cnscalar). **0D arrays support scalar-shaped arithmetic.** Use `+`, `-`, `*`, `/`, and `^` with two 0D NDArrays or with a 0D NDArray and a Julia number. Results stay as 0D NDArrays on the backend, following the existing promotion policy. -**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into a task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them [for example `println`, `unwrap`, or communicating with the Julia runtime (i.e., `Array(A)`)]. Hiding latency enables performant code. +**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into a task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them [for example `println`, `fetch`, or communicating with the Julia runtime (i.e., `Array(A)`)]. Hiding latency enables performant code. For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/dev/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/dev/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/dev/perf/patterns_to_avoid). diff --git a/docs/make.jl b/docs/make.jl index 0be317b57..c16aecae9 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -67,6 +67,7 @@ makedocs(; "Random" => "api_random.md", "Unary Operations" => "api_unary.md", "Mapped Reductions" => "api_mapreduce.md", + "Device Scalars" => "api_cnscalar.md", "Binary Operations" => "api_binary.md", "Linear Algebra" => "linalg.md", "Tensor Contractions" => "api_tensor.md", diff --git a/docs/src/api_cnscalar.md b/docs/src/api_cnscalar.md new file mode 100644 index 000000000..6ffb8d7ef --- /dev/null +++ b/docs/src/api_cnscalar.md @@ -0,0 +1,97 @@ +# Device scalars and automatic fetching + +Device scalars let you use reduction results in further calculations without +first copying their values to the host. They also participate in Julia's numeric +type hierarchy, so they can be passed to compatible numeric methods and stored +in structs with abstract numeric type constraints. Extract a host value explicitly +when you need one, or enable scoped automatic fetching for comparisons and conversions. + +Full reductions such as `sum(A)`, `dot(x, y)`, and `norm(x)` return a +`CNScalar`. Concrete wrappers mirror the numeric category of their storage: +`CNFloat <: AbstractFloat`, `CNInt <: Signed`, `CNUInt <: Unsigned`, +`CNBool <: Integer`, and `CNComplex <: Number`. `CNReal` is the union of the +four real wrapper families, and `CNScalar` also includes `CNComplex`. +Each owns a reference to a 0D NDArray, with no copy or host +extraction when it is wrapped. Reductions that retain dimensions still return +NDArrays. Explicit 0D array construction and array broadcasts remain arrays. + +`DeviceScalar{T}` is the shared dispatch alias for a raw `NDArray{T,0}` or a +`CNScalar{T}` wrapper. Use `DeviceScalar` when either representation is accepted: + +```julia +twice(x::DeviceScalar) = x .* 2 +``` + +Numeric data arguments such as search needles, fill values, and +linear algebra coefficients use this shared storage path. `Ref(s)` in broadcast +also preserves device storage for either representation. + +Host control parameters, including a norm's `p`, random distribution parameters, +and `isapprox` tolerances, require allowautofetch permission for either representation. +The `searchsorted` convenience function explicitly extracts its result indices +to construct a Julia range; +use `searchsortedfirst`/`searchsortedlast` to retain device results. + +```julia +using cuNumeric, LinearAlgebra +x = cuNumeric.NDArray([1.0, 2.0, 3.0]) +s = sum(x) # CNFloat{Float64}, accepted in <:AbstractFloat fields +t = s^2 / 2 # another device scalar +y = x .* t # backend broadcast; no implicit host extraction +value = fetch(t) # explicit synchronization, always permitted +``` + +`cnscalar(a)` wraps an existing 0D NDArray; `s.value` accesses its backend array +without synchronizing. Use `fetch(s)` for explicit host extraction. +This extends Julia's `Base.fetch`; ordinary Julia values retain their existing +`fetch(x) == x` behavior. Explicit `fetch` does not require `allowautofetch`. +Arithmetic, `min`/`max`, and promotion between numeric types keep results on the +backend. Showing a wrapper uses the backing 0D NDArray's display and extracts +its value, including in the REPL, without requiring allowautofetch permission. End +an expression with `;` to suppress REPL display and that synchronization. +The existing promotion policy still applies. + +Device scalars can also be coefficients in `contract!`, `mul!`, `axpy!`, +`axpby!`, and TensorOperations calls without enabling allowautofetch. + +Value-dependent conversions, such as floating point to integer or complex to +real, require allowautofetch permission even when the target is another CNScalar. +They preserve Julia's `InexactError` checks instead of silently truncating values. + +## Scoped permission + +Implicit host extraction is disabled by default. Comparisons, scalar predicates, +and conversion to supported host numeric types require `allowautofetch` permission: + +```julia +@allowautofetch s > 0 # Julia Bool +allowautofetch() do + Float64(s) # Julia Float64 +end +``` + +The scope returns the body's result and restores the previous permission even +when the body throws. Nested `allowautofetch(false) do ... end` disables extraction +temporarily. `allowautofetch(true)` / `allowautofetch(false)` set the calling task's +permission until changed again. Independent tasks do not inherit permission. +This permission is separate from `allowscalar` and `allowpromotion`. + +The permission also applies inside ordinary functions called within the scope; +the macro does not need to inspect their bodies: + +```julia +function host_residual(x) + r = norm(x) # device scalar + return Float64(r) # permission checked here, inside this function +end + +residual = @allowautofetch host_residual(x) +# host_residual(x) # errors outside the scope +``` + +Arithmetic remains asynchronous even inside an allowautofetch scope. There is no +general fallback that fetches arguments when Julia cannot find a method. +For example, a function accepting only `Float64` still needs `f(Float64(s))`. +Julia also requires an actual Bool in `if`: use `Bool(all(A))` within the scope, +or a comparison that returns a host Bool. Automatic fetching does not change Julia's +dispatch or condition evaluation rules. diff --git a/docs/src/api_mapreduce.md b/docs/src/api_mapreduce.md index 1e39a555b..bdb731265 100644 --- a/docs/src/api_mapreduce.md +++ b/docs/src/api_mapreduce.md @@ -6,14 +6,14 @@ Mapped `sum`, `prod`, `minimum`, and `maximum` use the same implementation. ```julia A = cuNumeric.ones(Float32, 1024, 512) -energy = sum(abs2, A) # 0-d NDArray +energy = sum(abs2, A) # CNScalar columns = mapreduce(abs2, +, A; dims=1) # 1 × 512 α = 0.5f0 distance = sum(x -> abs2(x - α), A) largest = maximum(abs, A; init=0f0) ``` -Full reductions return a device-resident 0-d NDArray. Dimensional reductions +Full reductions return a device-resident CNScalar. Dimensional reductions keep reduced axes with size one. Duplicate dimensions are ignored; positive out-of-rank dimensions have no effect. `dims=()` still applies the mapping. diff --git a/docs/src/api_tensor.md b/docs/src/api_tensor.md index 8fffa2cc6..b2b0e603e 100644 --- a/docs/src/api_tensor.md +++ b/docs/src/api_tensor.md @@ -34,8 +34,8 @@ end Prefer a Julia `Number` for scale factors when possible (i.e., `2.0f0`, `randn()`, …). This allows TensorOperations.jl to perform optimizatoins for special values like zero and one. -A 0D `NDArray` is also accepted as a scale — for example `sum(C)`, or a -fully contracted `@tensor` result. If you intend to use these results in down-stream tensor contractions keep those on device. `unwrap` copies the value to the host and **blocks the Legate runtime**, so only unwrap +A `CNScalar`, such as `sum(C)`, or a raw 0D `NDArray`, such as a +fully contracted `@tensor` result, is also accepted as a scale. Keep these results in runtime-managed storage when reusing them in tensor contractions. `fetch` retrieves a native Julia scalar and waits for the result, so only fetch when you truly need a Julia `Number` (i.e. printing of host-side if-else). ```julia @@ -48,8 +48,8 @@ B = cuNumeric.rand(Float64, 32, 16) @tensor D[i, j] := α * C[i, j] # Julia Number, no sync @tensor s = conj(C[i, j]) * C[i, j] # 0D NDArray, stays asynchronous -@tensor E[i, j] := s * C[i, j] # reuse 0D as a scale, still no unwrap -x = unwrap(s) # host Number; blocks +@tensor E[i, j] := s * C[i, j] # reuse 0D as a scale, still no fetch +x = fetch(s) # host Number; blocks ``` ## Low-level `contract` / `contract!` diff --git a/docs/src/api_unary.md b/docs/src/api_unary.md index 4adbcac70..6dcc75c35 100644 --- a/docs/src/api_unary.md +++ b/docs/src/api_unary.md @@ -14,9 +14,9 @@ The following unary operations are supported and can be broadcast over `NDArray` - `round.(A)` uses IEEE round-to-nearest-even (the `RINT` kernel), matching Julia `round(x)` / `RoundNearest`. Only that 1-arg path is wired. `digits`, `sigdigits`, and `RoundingMode` are not supported (`round.(A; digits=n)` errors). - `floor`, `ceil`, `trunc`, and `signbit` are float-only kernels. `round` also supports complex values. Bool and integer inputs are not accepted (Julia's `floor`/`ceil`/`trunc`/`round` on integers are identity). - `~` is bitwise not on integers. On `Bool` it matches `!` (the invert kernel rejects `Bool`, so we use logical not). -- Full reductions (`sum`, `mean`, `var`, `std`, `argmax`, …) return a 0-d `NDArray`, not a Julia scalar. Use `unwrap` or `A[]` (with `allowscalar`) to read a host value. +- Full reductions (`sum`, `mean`, `var`, `std`, …) return a `CNScalar` backed by a 0D NDArray. Use `fetch` to read a native Julia scalar. - `var` / `std` match Julia / StatsBase sample statistics (`corrected=true`, divisor `n-1`). Complex is not supported. -- `argmax` / `argmin` are 1-d only (matching Base's `Int` return, not `CartesianIndex`). The result is a 0-d `NDArray{Int64}` of the 1-based index. Complex is not supported. +- `argmax` / `argmin` are 1-d fetch (matching Base's `Int` return, not `CartesianIndex`). The result is a 0-d `NDArray{Int64}` of the 1-based index. Complex is not supported. For `mapreduce` and mapped `sum`, `prod`, `minimum`, and `maximum`, see [Mapped Reductions](./api_mapreduce.md). diff --git a/docs/src/examples/montecarlo.md b/docs/src/examples/montecarlo.md index eb45188a0..49c366d3f 100644 --- a/docs/src/examples/montecarlo.md +++ b/docs/src/examples/montecarlo.md @@ -26,4 +26,4 @@ println("Monte-Carlo estimate: $(estimate)") println("Analytical value: $(sqrt(pi))") ``` -The result is a 0-dimensional `NDArray`, which keeps the reduction asynchronous. Use `unwrap(estimate)` only when a Julia scalar is required. +The result is a `CNScalar` backed by a 0D NDArray, which keeps the reduction asynchronous. Use `fetch(estimate)` only when a Julia scalar is required. diff --git a/docs/src/examples/tensor_network.md b/docs/src/examples/tensor_network.md index 6caff7289..5d87358eb 100644 --- a/docs/src/examples/tensor_network.md +++ b/docs/src/examples/tensor_network.md @@ -14,7 +14,7 @@ H = S^x_1 S^x_2 + S^y_1 S^y_2 + S^z_1 S^z_2. The examples below apply it to an MPS-like state with tensors ``A^{(1)}``, ``A^{(2)}`` and boundary environments ``L``, ``R``. The first contractions keep open bond indices and stay `NDArray`s. The second contracts every index; the -ratio stays on device until `unwrap` for printing. +ratio stays on device until `fetch` for printing. ```julia # found in examples/tensor_network.jl @@ -59,13 +59,13 @@ println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) A fully contracted `@tensor` assignment is a 0D `NDArray`, like `sum`. Prefer to keep it that way and do device arithmetic (`./`) until you -need a host `Number`. `unwrap` (here, only to print) **blocks**. See +need a host `Number`. `fetch` (here, only to print) **blocks**. See [Scalars](../api_tensor.md#Scalars). ```julia @tensor energy = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] @tensor norm² = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] -println("two-site energy = ", real(unwrap(energy ./ norm²))) +println("two-site energy = ", real(fetch(energy ./ norm²))) ``` See [Tensor Contractions](../api_tensor.md) documentation for more details. diff --git a/docs/src/index.md b/docs/src/index.md index b9b693e6e..7e337dccb 100644 --- a/docs/src/index.md +++ b/docs/src/index.md @@ -40,14 +40,14 @@ The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it **Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communicate with the Julia runtime. When you need a plain Julia number, call `unwrap`: +**Scalar reductions return device scalars.** Full reductions such as `sum(A)` return a `CNScalar` backed by a 0D NDArray. Reductions that retain dimensions still return NDArrays. This keeps the Legate task graph asynchronous. When you need a native Julia scalar, call `fetch`: ```julia -s = sum(A) # NDArray{T,0} -x = unwrap(s) # T, e.g. Float32 +s = sum(A) # CNScalar +x = fetch(s) # native Julia scalar, e.g. Float32 ``` -**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. +**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `fetch`, or converting with `Array(A)`). Hiding latency enables performant code. For API details see [Initialization](./api_initialization.md), [Random](./api_random.md), [FFT](./fft.md), and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). diff --git a/docs/src/linalg.md b/docs/src/linalg.md index bbcc89ff6..ded351949 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -99,30 +99,15 @@ Filter = t -> t isa Function && nameof(t) === :mul! ## Vector operations -Numeric `NDArray`s support matrix-vector `A * x`, three- and five-argument -`mul!`, vector `dot`, `axpy!`, `axpby!`, and scalar `lmul!` / `rmul!`. -Five-argument matrix-matrix `mul!` is also available. Mixed types follow the -package's explicit promotion policy. Matrix multiplication rejects -integer-integer inputs (including Bool); mixed integer/floating-point inputs -are supported when their promoted type is supported by the matrix kernel. -Destinations of `mul!` must not alias inputs. Vector updates support exact -self-aliasing, but not partially overlapping views. - -`dot` conjugates its first argument. **`dot` and dense-array `norm` return 0D -NDArrays**, keeping their results on the backend without implicit scalar -extraction or synchronization. Use `only` or `unwrap` explicitly when a Julia -scalar is required. Bool-Bool dot products accumulate into `Int` under the -existing promotion policy. - -`norm(A, p)` is an entrywise norm, not `opnorm`. It uses mapped reductions and -currently requires a GPU target. The 2-norm fuses `abs2` into a sum reduction, -then applies a backend square root. Like cuPyNumeric, powers are accumulated -without scaling and can overflow or underflow. Integer inputs are converted -to floating point under the existing promotion policy. - -Matrix-vector contraction selects cuPyNumeric's specialized `MATVECMUL` task. -Generic solvers that require Julia scalar reductions, including IterativeSolvers -CG, need adaptation to work with these asynchronous reduction results. +Supported `LinearAlgebra` operations include: + +- `mul!(y, A, x)` and `mul!(y, A, x, α, β)` for matrix-vector products; `A * x` is also supported. +- `mul!(C, A, B, α, β)` for matrix-matrix products. +- `dot(x, y)` and entrywise `norm(A, p)` (dense-array norms currently require a GPU). +- `axpy!`, `axpby!`, `lmul!`, and `rmul!`. +- `ldiv!(y, D, x)` and `ldiv!(D, x)` for an NDArray-backed `Diagonal`. + +`dot` and dense-array `norm` return [device scalars](api_cnscalar.md). ## Solve @@ -311,7 +296,7 @@ must be `NDArray` unless noted. - `mul!`, `lmul!`, `rmul!` with `NDArray` - `D \ B`, `A / D`, `ldiv!`, `rdiv!` with `NDArray` - `inv(D)` — reciprocal on-device; zeros become Inf (no `SingularException`) -- `det(D)` — 0-dimensional `NDArray` product of the diagonal +- `det(D)` — `CNScalar` product of the diagonal **`NDArray` ± `Diagonal`** @@ -337,12 +322,12 @@ must be `NDArray` unless noted. - `eigvals(D)`, `eigen(D)`, `eigvecs(D)` — unsorted; values are a copy of the diagonal (`NDArray`), vectors are `NDArray` identity. Keyword `sortby` is not supported on this method. -- `tr`, `sum`, `prod`, `maximum`, `minimum` — 0-dimensional `NDArray` (not a Julia scalar) +- `tr`, `sum`, `prod`, `maximum`, `minimum` — `CNScalar` (not a Julia scalar) - `iszero`, `isone`, `istriu`, `istril`, `ishermitian`, `issymmetric`, `isposdef` — 0-dimensional `NDArray{Bool}` -- `opnorm(D)` / `opnorm(D, p)` for `p ∈ {1, 2, Inf}` — 0-dimensional `NDArray` -- `norm(D)` / `norm(D, p)` for finite `p` (including `±Inf`); off-diagonals are zero — 0-dimensional `NDArray` -- `cond(D)` / `cond(D, p)` for `p ∈ {1, 2, Inf}` — 0-dimensional `NDArray` -- `logdet(D)` for real `Diagonal` only — 0-dimensional `NDArray` +- `opnorm(D)` / `opnorm(D, p)` for `p ∈ {1, 2, Inf}` — `CNScalar` +- `norm(D)` / `norm(D, p)` for finite `p` (including `±Inf`); off-diagonals are zero — `CNScalar` +- `cond(D)` / `cond(D, p)` for `p ∈ {1, 2, Inf}` — `CNScalar` +- `logdet(D)` for real `Diagonal` only — `CNScalar` **Helpers on dense `NDArray`** diff --git a/docs/src/perf/patterns_to_avoid.md b/docs/src/perf/patterns_to_avoid.md index ef99566b2..28f6fb2c9 100644 --- a/docs/src/perf/patterns_to_avoid.md +++ b/docs/src/perf/patterns_to_avoid.md @@ -4,10 +4,11 @@ Accessing elements of an NDArray one at a time (e.g., `arr[5]`) is slow and should be avoided. Indexing like this requires data to be transferred between device and host and maybe even communicated across nodes. Scalar indexing will emit an error which can be opted out of with `@allowscalar` or `allowscalar() do ... end`. Several functions in the existing API invoke scalar indexing and are intended for testing (e.g., the `==` operator). -`unwrap` (and `A[]`) pulls a host `Number` out of a 0D `NDArray` and -**blocks the runtime**. Prefer a Julia `Number` when you already have one, -and leave reductions / fully contracted `@tensor` results as 0D arrays -until you actually need the host value. See [Scalars](../api_tensor.md#Scalars). +`fetch` retrieves a native Julia scalar from a `CNScalar` or a single-element +`NDArray`, waiting for the result as needed. Prefer a native Julia number when +you already have one, and leave reductions / fully contracted `@tensor` results +in runtime-managed storage until you need the host value. +See [Scalars](../api_tensor.md#Scalars). ## Implicit promotion diff --git a/examples/poisson_fft.jl b/examples/poisson_fft.jl index dc1a34809..3cdd26d7c 100644 --- a/examples/poisson_fft.jl +++ b/examples/poisson_fft.jl @@ -68,7 +68,7 @@ function main() f = NDArray(f_h) u = poisson_solve(f) err = maximum(abs.(u - NDArray(u_true))) - @printf("single grid %dx%d max |u − u_true| = %.3e\n", n, n, unwrap(err)) + @printf("single grid %dx%d max |u − u_true| = %.3e\n", n, n, fetch(err)) # Four independent charge distributions, one FFT task, batch axis first. b = 4 @@ -76,7 +76,7 @@ function main() invk = reshape(NDArray(poisson_inv_laplacian(Float32, n)), 1, n, n) u_batch = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f_batch) .* invk)) err_b = maximum(abs.(u_batch - NDArray(repeat(reshape(u_true, 1, n, n), b, 1, 1)))) - @printf("batched %d×%dx%d max |u − u_true| = %.3e\n", b, n, n, unwrap(err_b)) + @printf("batched %d×%dx%d max |u − u_true| = %.3e\n", b, n, n, fetch(err_b)) return u end diff --git a/examples/tensor_network.jl b/examples/tensor_network.jl index 2937e8025..74c731814 100644 --- a/examples/tensor_network.jl +++ b/examples/tensor_network.jl @@ -2,7 +2,7 @@ The first contractions build an MPS-like two-site state and apply a Heisenberg operator. Those results stay on device as NDArrays. The fully contracted -⟨ψ|H|ψ⟩ / ⟨ψ|ψ⟩ ratio is computed on device; unwrap only to print (it blocks). +⟨ψ|H|ψ⟩ / ⟨ψ|ψ⟩ ratio is computed on device; fetch only to print (it blocks). =# using cuNumeric @@ -35,8 +35,8 @@ println("ψ is a ", typeof(ψ), " of size ", size(ψ)) println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) # Fully contracted @tensor results are 0D NDArrays. Keep them on device -# until a host Number is required; unwrap blocks. +# until a host Number is required; fetch blocks. @tensor energy = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] @tensor norm² = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] -println("two-site energy = ", real(unwrap(energy ./ norm²))) +println("two-site energy = ", real(fetch(energy ./ norm²))) diff --git a/ext/cuNumericTensorOperationsExt.jl b/ext/cuNumericTensorOperationsExt.jl index 8fb4c0a6e..8cfef747c 100644 --- a/ext/cuNumericTensorOperationsExt.jl +++ b/ext/cuNumericTensorOperationsExt.jl @@ -38,10 +38,15 @@ function TO.select_backend( return CuNumericBackend() end +# Promotion with an CNScalar coefficient describes the scalar wrapper, but the +# tensor result stores its native backend element type. +_tensor_eltype(::Type{T}) where {T} = T +_tensor_eltype(::Type{T}) where {T<:CN.CNScalar} = CN._scalar_eltype(T) + function TO.tensoradd_type( TC, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool ) - return CN.NDArray{TC,TO.numind(pA)} + return CN.NDArray{_tensor_eltype(TC),TO.numind(pA)} end function TO.tensorcontract_type( @@ -54,7 +59,8 @@ function TO.tensorcontract_type( conjB::Bool, pAB::TO.Index2Tuple, ) - Tout = TC <: Union{Integer,Bool} ? Float64 : TC + T = _tensor_eltype(TC) + Tout = CN._contract_eltype(T) return CN.NDArray{Tout,TO.numind(pAB)} end @@ -85,6 +91,7 @@ end _as_scale(x::One, ::Type) = x _as_scale(x::Zero, ::Type) = x +_as_scale(x::CN.CNScalar, ::Type{T}) where {T} = _as_scale(x.value, T) function _as_scale(x::CN.NDArray, ::Type{T}) where {T} _require_0d_scale(x) return _convert_eltype(x, T) @@ -102,6 +109,8 @@ function _free_scale!(orig, scaled) return nothing end +_free_scale!(orig::CN.CNScalar, scaled) = _free_scale!(orig.value, scaled) + _accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, ::Zero) = (C.=A; C) _accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, ::Zero) = (C.=α .* A; C) _accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, ::One) = (C.=C .+ A; C) diff --git a/src/cnscalar.jl b/src/cnscalar.jl new file mode 100644 index 000000000..d1e657c86 --- /dev/null +++ b/src/cnscalar.jl @@ -0,0 +1,300 @@ +export CNFloat, CNInt, CNUInt, CNBool, CNReal, CNComplex, CNScalar, DeviceScalar, + cnscalar, allowautofetch, @allowautofetch + +"A floating-point device scalar; wrapping its 0D storage does not synchronize." +struct CNFloat{T<:AbstractFloat,P} <: AbstractFloat + value::NDArray{T,0,P} +end + +"A signed integer device scalar backed by a 0D NDArray." +struct CNInt{T<:Signed,P} <: Signed + value::NDArray{T,0,P} +end + +"An unsigned integer device scalar backed by a 0D NDArray." +struct CNUInt{T<:Unsigned,P} <: Unsigned + value::NDArray{T,0,P} +end + +"A Boolean device scalar. Bool is concrete, so the wrapper subtypes Integer." +struct CNBool{T<:Bool,P} <: Integer + value::NDArray{T,0,P} +end + +"A complex device scalar backed by a 0D NDArray; wrapping does not synchronize." +struct CNComplex{T<:Complex,P} <: Number + value::NDArray{T,0,P} +end + +# Bounded wrapper branches allow both DeviceScalar{Float64} and +# DeviceScalar{ComplexF64}; an exact CNComplex{Float64} would violate its bound. +const CNReal{T} = Union{CNFloat{<:T},CNInt{<:T},CNUInt{<:T},CNBool{<:T}} +const CNScalar{T} = Union{CNReal{T},CNComplex{<:T}} +const DeviceScalar{T} = Union{NDArray{T,0},CNScalar{T}} +const _REAL_SCALAR_WRAPPERS = (CNFloat, CNInt, CNUInt, CNBool) +const _SCALAR_WRAPPERS = (_REAL_SCALAR_WRAPPERS..., CNComplex) +for (W, H) in ((CNFloat, AbstractFloat), (CNInt, Signed), (CNUInt, Unsigned), + (CNBool, Bool), (CNComplex, Complex)) + @eval cnscalar(x::NDArray{T,0}) where {T<:$H} = $W(x) + @eval _scalar_type(::Type{T}) where {T<:$H} = $W{T} + @eval _scalar_eltype(::Type{<:$W{T}}) where {T} = T +end +cnscalar(x::CNScalar) = x +_scalar_result(x::NDArray{<:Number,0}) = cnscalar(x) +_scalar_result(x) = x + +_scale_storage(x::CNScalar) = x.value +# Host shortcuts must not inspect a device coefficient's value. +_host_iszero(::CNScalar) = false +_host_isone(::CNScalar) = false + +# Numeric data stays in backend storage; host control parameters are explicitly +# permission-checked for both wrapped and unwrapped device scalars. +_maybe_fetch(x) = x +function _maybe_fetch(x::DeviceScalar{<:Real}) + _assert_allowautofetch() + return only(_scale_storage(x)) +end + +_coefficient_type(x) = typeof(x) +_coefficient_type(x::DeviceScalar) = eltype(_scale_storage(x)) +_coefficient_as(::Type{T}, x::Number) where {T} = convert(T, x) +_coefficient_as(::Type{T}, x::DeviceScalar) where {T} = checked_promote_arr(_scale_storage(x), T) + +searchsortedfirst(a::NDArray{T,1}, x::CNScalar) where {T} = searchsortedfirst(a, x.value) +searchsortedlast(a::NDArray{T,1}, x::CNScalar) where {T} = searchsortedlast(a, x.value) + +function Base.fill!(a::NDArray, x::DeviceScalar) + a .= _scale_storage(x) + return a +end +function fill(x::DeviceScalar, dims::Dims) + a = cuNumeric.zeros(_coefficient_type(x), dims) + return fill!(a, x) +end +fill(x::DeviceScalar, dims::Int...) = fill(x, dims) +fill(x::DeviceScalar, dim::Int) = fill(x, (dim,)) + +""" + allowautofetch(f, allow=true) + allowautofetch(allow::Bool=true) + @allowautofetch expression + +Permit implicit host extraction of CNScalars for comparisons, predicates, and +conversion to host numeric types. Arithmetic stays on the backend. The do-block +and macro restore the calling task's previous permission, including on errors. +Permission is task-local and separate from scalar indexing and promotion. +This is not a fallback for arbitrary functions with unsupported argument types. +""" +allowautofetch(f::F, allow::Bool=true) where {F} = + task_local_storage(f, :cuNumericAllowAutoFetch, allow) +allowautofetch(allow::Bool=true) = (task_local_storage(:cuNumericAllowAutoFetch, allow); nothing) + +macro allowautofetch(ex) + quote + local previous = get(task_local_storage(), :cuNumericAllowAutoFetch, nothing) + task_local_storage(:cuNumericAllowAutoFetch, true) + @__tryfinally($(esc(ex)), + if isnothing(previous) + delete!(task_local_storage(), :cuNumericAllowAutoFetch) + else + task_local_storage(:cuNumericAllowAutoFetch, previous) + end) + end +end + +function _assert_allowautofetch() + get(task_local_storage(), :cuNumericAllowAutoFetch, false) && return nothing + throw(ArgumentError("Implicit CNScalar host extraction is disabled. Use " * + "allowautofetch() do ... end or @allowautofetch, or explicitly call fetch(x).")) +end + +""" + fetch(x::CNScalar) + fetch(x::NDArray) + +Retrieve a native Julia scalar, waiting for the result as needed. An NDArray +must contain exactly one element. Explicit fetching does not require +`allowautofetch` or `allowscalar` permission. +""" +Base.fetch(x::CNScalar) = only(x.value) +Base.only(x::CNScalar) = fetch(x) +# Preserve explicit scalar indexing of reduction results, including allowscalar. +# Number's default getindex would instead return the wrapper unchanged. +Base.getindex(x::CNScalar) = x.value[] +destroy!(x::CNScalar) = destroy!(x.value) +Base.copy(x::CNScalar) = cnscalar(copy(x.value)) +Base.broadcastable(x::CNScalar) = x.value +# Ref(device_scalar) still represents one backend value, not a host kernel arg. +Base.broadcastable(x::Base.RefValue{<:DeviceScalar}) = _scale_storage(x[]) +# Display is an intentional host extraction, just as for the backing 0D NDArray. +Base.show(io::IO, x::CNScalar) = show(io, x.value) +Base.show(io::IO, mime::MIME"text/plain", x::CNScalar) = show(io, mime, x.value) + +_scalar_operand(x::CNScalar) = x.value +_scalar_operand(x::Number) = x +_scalar_operand(x::NDArray{<:Any,0}) = x +_scalar_binary(f, x, y) = cnscalar(broadcast(f, _scalar_operand(x), _scalar_operand(y))) +function _scalar_compare(f, x, y) + _assert_allowautofetch() + result = broadcast(f, _scalar_operand(x), _scalar_operand(y)) + value = only(result) + destroy!(result) + return value +end +_scalar_host(x::CNScalar) = fetch(x) +_scalar_host(x::Number) = x +function _scalar_compare(f::Union{typeof(isless),typeof(isequal)}, x, y) + _assert_allowautofetch() + return f(_scalar_host(x), _scalar_host(y)) +end + +# Explicit intersections avoid ambiguities with Base's Real/Complex methods. +for op in (:+, :-, :*, :/, :^), W in _SCALAR_WRAPPERS + for H in (Real, Complex, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_binary($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_binary($op, x, y) + end + for V in _SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_binary($op, x, y) + end +end +for op in (:(==), :(!=), :isequal), W in _SCALAR_WRAPPERS + for H in (Real, Complex, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_compare($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_compare($op, x, y) + end + for V in _SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_compare($op, x, y) + end +end +for op in (:<, :<=, :>, :>=, :isless), W in _REAL_SCALAR_WRAPPERS + for H in (Real, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_compare($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_compare($op, x, y) + end + for V in _REAL_SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_compare($op, x, y) + end +end +for op in (:min, :max), W in _REAL_SCALAR_WRAPPERS + for H in (Real, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_binary($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_binary($op, x, y) + end + for V in _REAL_SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_binary($op, x, y) + end +end +Base.literal_pow(::typeof(^), x::CNScalar, ::Val{P}) where {P} = x ^ P + + +Base.:*(x::CNScalar, a::NDArray) = broadcast(*, x, a) +Base.:*(a::NDArray, x::CNScalar) = broadcast(*, a, x) +for op in (:+, :-, :*, :/, :^) + @eval begin + Base.$op(x::CNScalar, a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}) = _scalar_binary($op, x, a) + Base.$op(a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, x::CNScalar) = _scalar_binary($op, a, x) + end +end + +function _scalar_unary(f, x::CNScalar) + # Some backend unary kernels reject rank zero. A size-one view avoids host + # extraction and works on both CPU and GPU targets. + input = cuNumeric.reshape(x.value, 1) + output = broadcast(f, input) + result = cuNumeric.reshape(output, ()) + destroy!(input) + destroy!(output) + return cnscalar(result) +end +for op in (:-, :abs, :sqrt), W in _SCALAR_WRAPPERS + @eval Base.$op(x::$W) = _scalar_unary($op, x) +end +for op in (:conj, :real, :imag) + @eval Base.$op(x::CNComplex) = _scalar_unary($op, x) +end +for W in _SCALAR_WRAPPERS + @eval Base.:+(x::$W) = x + @eval Base.inv(x::$W) = one(eltype(x.value)) / x +end +for W in _REAL_SCALAR_WRAPPERS + @eval Base.real(x::$W) = x + @eval Base.conj(x::$W) = x + @eval Base.imag(x::$W) = zero(x) +end +Base.abs2(x::CNReal) = x * x +Base.abs2(x::CNComplex) = real(x * conj(x)) +Base.:!(x::CNReal{Bool}) = _scalar_unary(!, x) +for op in (:iszero, :isone, :isfinite, :isinf, :isnan), W in _SCALAR_WRAPPERS + @eval function Base.$op(x::$W) + _assert_allowautofetch() + return $op(fetch(x)) + end +end + +_scalar_convert(::Type{T}, x::CNScalar) where {T} = cnscalar(checked_promote_arr(x.value, T)) +_scalar_convert(::Type{T}, x::Number) where {T} = cnscalar(NDArray(convert(T, x))) +function _checked_scalar_convert(::Type{T}, x::CNScalar) where {T} + # These conversions can throw InexactError based on the value. A backend + # dtype cast would silently truncate or discard an imaginary component. + _assert_allowautofetch() + return cnscalar(NDArray(convert(T, fetch(x)))) +end +_scalar_convert(::Type{T}, x::CNComplex) where {T<:Real} = _checked_scalar_convert(T, x) +_scalar_convert(::Type{T}, x::CNFloat) where {T<:Integer} = + _checked_scalar_convert(T, x) +function _scalar_convert(::Type{T}, x::Union{CNInt,CNUInt,CNBool}) where {T<:Integer} + S = _scalar_eltype(typeof(x)) + if typemin(T) <= typemin(S) && typemax(T) >= typemax(S) + return cnscalar(checked_promote_arr(x.value, T)) + end + return _checked_scalar_convert(T, x) +end +for W in _SCALAR_WRAPPERS + @eval begin + $W{T}(x::Number) where {T} = _scalar_convert(T, x) + Base.convert(::Type{S}, x::Number) where {T,S<:$W{T}} = _scalar_convert(T, x) + Base.convert(::Type{S}, x::S) where {S<:$W} = x + Base.zero(::Type{S}) where {S<:$W} = cnscalar(NDArray(zero(_scalar_eltype(S)))) + Base.one(::Type{S}) where {S<:$W} = cnscalar(NDArray(one(_scalar_eltype(S)))) + end +end +Base.zero(x::CNScalar) = zero(typeof(x)) +Base.one(x::CNScalar) = one(typeof(x)) +Base.real(::Type{S}) where {S<:CNScalar} = _scalar_type(real(_scalar_eltype(S))) +Base.float(x::CNScalar) = _scalar_convert(float(_scalar_eltype(typeof(x))), x) +Base.float(x::CNFloat) = x +Base.float(::Type{S}) where {S<:CNScalar} = _scalar_type(float(_scalar_eltype(S))) + +for T in Base.uniontypes(SUPPORTED_ARRAY_TYPES), W in _SCALAR_WRAPPERS + @eval function Base.convert(::Type{$T}, x::$W) + _assert_allowautofetch() + return convert($T, fetch(x)) + end + @eval (::Type{$T})(x::$W) = convert($T, x) +end + +# Generic promotion keeps values device-backed; host conversion is a separate, +# permission-checked operation. Real and complex wrappers remain distinct. +for W in _SCALAR_WRAPPERS + @eval Base.promote_rule(::Type{S}, ::Type{T}) where {S<:$W,T<:Union{Real,Complex}} = + _scalar_type(promote_type(_scalar_eltype(S), T)) + for V in _SCALAR_WRAPPERS + @eval Base.promote_rule(::Type{S}, ::Type{T}) where {S<:$W,T<:$V} = + _scalar_type(promote_type(_scalar_eltype(S), _scalar_eltype(T))) + end +end +for T in Base.uniontypes(SUPPORTED_ARRAY_TYPES), W in _SCALAR_WRAPPERS + @eval Base.promote_rule(::Type{$T}, ::Type{S}) where {S<:$W} = + _scalar_type(promote_type($T, _scalar_eltype(S))) +end + +# Internal composition of reductions consumes storage, never a host scalar. +Base.mapreduce(f, op, x::CNScalar; kwargs...) = mapreduce(f, op, x.value; kwargs...) +_div_nelem(x::CNScalar, n::Integer) = x / n +function _sqrt_ndarray(x::CNScalar) + result = sqrt(x) + destroy!(x) + return result +end diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 553bf086b..c3fe9b29d 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -191,6 +191,7 @@ const FUSE_BROADCAST_EXPRS = CNPreferences.FUSE_BROADCAST const FUSE_BROADCAST_MIN_OPS = CNPreferences.FUSE_BROADCAST_MIN_OPS # Functionality +include("cnscalar.jl") include("ndarray/diagonal.jl") include("ndarray/promotion.jl") include("cuda/cuda_ptx_task.jl") diff --git a/src/ndarray/contract.jl b/src/ndarray/contract.jl index 2cf7a9ed5..374277b1d 100644 --- a/src/ndarray/contract.jl +++ b/src/ndarray/contract.jl @@ -212,7 +212,7 @@ function contract!( Ap = checked_promote_arr(contract!, A, T) Bp = checked_promote_arr(contract!, B, T) try - return _contract_same_type!(C, Cmodes, Ap, Amodes, Bp, Bmodes, α, β) + return _contract_same_type!(C, Cmodes, Ap, Amodes, Bp, Bmodes, _scale_storage(α), _scale_storage(β)) finally Ap !== A && destroy!(Ap) Bp !== B && destroy!(Bp) @@ -228,6 +228,13 @@ function contract!(C::NDArray, Cmodes, A::NDArray, Amodes, B::NDArray, Bmodes; return throw(ArgumentError("array type $bad is unsupported in contract!")) end +# Device scalar wrappers specialize this to expose storage without host extraction. +_scale_storage(x) = x +_host_iszero(x::Number) = iszero(x) +_host_iszero(::NDArray) = false +_host_isone(x::Number) = isone(x) +_host_isone(::NDArray) = false + function _require_0d_scale(x::NDArray) ndims(x) == 0 || throw( ArgumentError("contract! scale factor must be a Number or a 0-d NDArray") diff --git a/src/ndarray/diagonal.jl b/src/ndarray/diagonal.jl index 97d34ea57..27a06b47a 100644 --- a/src/ndarray/diagonal.jl +++ b/src/ndarray/diagonal.jl @@ -52,7 +52,7 @@ other reductions like `sum`. function trace(arr::NDArray{T,2}; offset::Int=0, a1::Int=0, a2::Int=1) where {T} LinearAlgebra.checksquare(arr) T_OUT = Base.promote_op(Base.sum, Vector{T}) - return nda_trace(arr, Int32(offset), Int32(a1), Int32(a2), T_OUT) + return cnscalar(nda_trace(arr, Int32(offset), Int32(a1), Int32(a2), T_OUT)) end function LinearAlgebra.tr(arr::NDArray{<:Any,2}) @@ -192,20 +192,20 @@ Base.sum(D::DiagonalNDArray) = sum(_diag_vec(D)) # Diagonal has off-diagonal zeros, so the product is zero — match Base, as 0D. function Base.prod(D::DiagonalNDArray{T}) where {T<:Number} n = size(D, 1) - n == 0 && return NDArray(one(T)) + n == 0 && return cnscalar(NDArray(one(T))) n == 1 && return prod(_diag_vec(D)) - return NDArray(zero(T)) + return cnscalar(NDArray(zero(T))) end function Base.maximum(D::DiagonalNDArray{T}) where {T<:Number} maxdiag = maximum(_diag_vec(D)) - size(D, 1) > 1 && return max.(zero(T), maxdiag) + size(D, 1) > 1 && return max(zero(T), maxdiag) return maxdiag end function Base.minimum(D::DiagonalNDArray{T}) where {T<:Number} mindiag = minimum(_diag_vec(D)) - size(D, 1) > 1 && return min.(zero(T), mindiag) + size(D, 1) > 1 && return min(zero(T), mindiag) return mindiag end @@ -216,22 +216,22 @@ end # Base walks `iszero(D.diag)` by scalar iteration; keep on-device via `iszero(D)`. function LinearAlgebra.istriu(D::DiagonalNDArray, k::Integer=0) - return k <= 0 ? NDArray(true) : iszero(D) + return k <= 0 ? cnscalar(NDArray(true)) : iszero(D) end function LinearAlgebra.istril(D::DiagonalNDArray, k::Integer=0) - return k >= 0 ? NDArray(true) : iszero(D) + return k >= 0 ? cnscalar(NDArray(true)) : iszero(D) end # Real Diagonal is always Hermitian/symmetric in Base; Complex Hermitian needs isreal(diag). -LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Real}) = NDArray(true) +LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Real}) = cnscalar(NDArray(true)) function LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Complex}) return all(imag(_diag_vec(D)) .== zero(real(eltype(D)))) end -LinearAlgebra.issymmetric(D::DiagonalNDArray{<:Number}) = NDArray(true) +LinearAlgebra.issymmetric(D::DiagonalNDArray{<:Number}) = cnscalar(NDArray(true)) # Base `isposdef(D) = all(isposdef, D.diag)` scalar-iterates. function LinearAlgebra.isposdef(D::DiagonalNDArray{T}) where {T<:Real} - isempty(D) && return NDArray(true) + isempty(D) && return cnscalar(NDArray(true)) return all(_diag_vec(D) .> zero(T)) end function LinearAlgebra.isposdef(D::DiagonalNDArray{T}) where {T<:Complex} @@ -271,37 +271,44 @@ end LinearAlgebra.logdet(D::DiagonalNDArray{<:Real}) = sum(log.(_diag_vec(D))) # Operator / entrywise norms from the diagonal only (no host densify). +for op in (:norm, :opnorm, :cond) + @eval LinearAlgebra.$op(D::DiagonalNDArray, p::NDArray{<:Real,0}) = $op(D, _maybe_fetch(p)) +end + function LinearAlgebra.opnorm(D::DiagonalNDArray, p::Real=2) + p = _maybe_fetch(p) if !(p == 1 || p == 2 || p == Inf) throw(ArgumentError(lazy"invalid p-norm p=$p. Valid: 1, 2, Inf")) end - isempty(D) && return NDArray(float(real(zero(eltype(D))))) + isempty(D) && return cnscalar(NDArray(float(real(zero(eltype(D)))))) return maximum(abs.(_diag_vec(D))) end function LinearAlgebra.norm(D::DiagonalNDArray, p::Real=2) + p = _maybe_fetch(p) # Off-diagonals are zero, so the matrix vec-norm equals the diag vec-norm. d = abs.(_diag_vec(D)) if p == 2 - return sqrt.(sum(d .^ 2)) + return sqrt(sum(d .^ 2)) elseif p == 1 return sum(d) elseif p == Inf - return isempty(D) ? NDArray(float(real(zero(eltype(D))))) : maximum(d) + return isempty(D) ? cnscalar(NDArray(float(real(zero(eltype(D)))))) : maximum(d) elseif p == -Inf - return isempty(D) ? NDArray(float(real(zero(eltype(D))))) : minimum(d) + return isempty(D) ? cnscalar(NDArray(float(real(zero(eltype(D)))))) : minimum(d) else - return sum(d .^ p) .^ (one(p) / p) + return sum(d .^ p) ^ (one(p) / p) end end function LinearAlgebra.cond(D::DiagonalNDArray, p::Real=2) + p = _maybe_fetch(p) if !(p == 1 || p == 2 || p == Inf) throw(ArgumentError(lazy"invalid p-norm p=$p. Valid: 1, 2, Inf")) end - isempty(D) && return NDArray(float(one(real(eltype(D))))) + isempty(D) && return cnscalar(NDArray(float(one(real(eltype(D)))))) dabs = abs.(_diag_vec(D)) - return maximum(dabs) ./ minimum(dabs) + return maximum(dabs) / minimum(dabs) end function Base.:+(A::NDArray{T,2}, D::DiagonalNDArray) where {T} @@ -324,6 +331,16 @@ Base.:-(D::DiagonalNDArray, A::NDArray{<:Any,2}) = D + (-A) return isone(λ) ? E : nda_multiply_scalar(E, R(λ)) end +function _uniformscaling_eye(::Type{R}, n::Integer, λ::NDArray{<:Any,0}) where {R} + E = _eye(R, Int(n)) + E .*= _coefficient_as(R, λ) + return E +end + +_uniformscale_mul(::Type{T}, λ::Number, a::NDArray) where {T} = _mul_scalar(T, λ, a) +_uniformscale_mul(::Type{T}, λ::NDArray{<:Any,0}, a::NDArray) where {T} = + a .* _coefficient_as(T, λ) + function NDArray{T}(J::LinearAlgebra.UniformScaling, dims::Dims{2}) where {T} A = zeros(T, dims) copyto!(A, J) @@ -332,29 +349,29 @@ end function NDArray{T}(J::LinearAlgebra.UniformScaling, m::Integer, n::Integer) where {T} return NDArray{T}(J, Dims((Int(m), Int(n)))) end -NDArray(J::LinearAlgebra.UniformScaling{T}, dims::Dims{2}) where {T} = NDArray{T}(J, dims) +NDArray(J::LinearAlgebra.UniformScaling{T}, dims::Dims{2}) where {T} = NDArray{_coefficient_type(J.λ)}(J, dims) function NDArray(J::LinearAlgebra.UniformScaling{T}, m::Integer, n::Integer) where {T} - return NDArray{T}(J, Dims((Int(m), Int(n)))) + return NDArray(J, Dims((Int(m), Int(n)))) end function Base.copyto!(A::NDArray{T,2}, J::LinearAlgebra.UniformScaling) where {T} m, n = size(A) - if iszero(J.λ) + if _host_iszero(J.λ) return fill!(A, zero(T)) elseif m == n - return copyto!(A, _uniformscaling_eye(T, m, J.λ)) + return copyto!(A, _uniformscaling_eye(T, m, _scale_storage(J.λ))) else fill!(A, zero(T)) k = min(m, n) - A[1:k, 1:k] = _uniformscaling_eye(T, k, J.λ) + A[1:k, 1:k] = _uniformscaling_eye(T, k, _scale_storage(J.λ)) return A end end function Base.:+(A::NDArray{T,2}, J::LinearAlgebra.UniformScaling) where {T} LinearAlgebra.checksquare(A) - R = Base.promote_op(+, T, typeof(J.λ)) - return A + _uniformscaling_eye(R, size(A, 1), J.λ) + R = Base.promote_op(+, T, _coefficient_type(J.λ)) + return A + _uniformscaling_eye(R, size(A, 1), _scale_storage(J.λ)) end Base.:+(J::LinearAlgebra.UniformScaling, A::NDArray{<:Any,2}) = A + J @@ -365,10 +382,10 @@ end # Scale by λ without promoting the array (A * I must not Bool→Float32 promote). function Base.:*(A::NDArray{T}, J::LinearAlgebra.UniformScaling) where {T} - return _mul_scalar(T, J.λ, A) + return _uniformscale_mul(T, _scale_storage(J.λ), A) end function Base.:*(J::LinearAlgebra.UniformScaling, A::NDArray{T}) where {T} - return _mul_scalar(T, J.λ, A) + return _uniformscale_mul(T, _scale_storage(J.λ), A) end function Base.one(A::NDArray{T,2}) where {T} @@ -384,24 +401,24 @@ end # Keep Diagonal structure: D + λI == Diagonal(d .+ λ), not a dense matrix. function Base.:+(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} - R = Base.promote_op(+, T, typeof(J.λ)) - return Diagonal(_diag_vec(D) .+ convert(R, J.λ)) + R = Base.promote_op(+, T, _coefficient_type(J.λ)) + return Diagonal(_diag_vec(D) .+ _coefficient_as(R, J.λ)) end Base.:+(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray) = D + J Base.:-(D::DiagonalNDArray, J::LinearAlgebra.UniformScaling) = D + (-J) function Base.:-(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray{T}) where {T} - R = Base.promote_op(-, typeof(J.λ), T) - return Diagonal(convert(R, J.λ) .- _diag_vec(D)) + R = Base.promote_op(-, _coefficient_type(J.λ), T) + return Diagonal(_coefficient_as(R, J.λ) .- _diag_vec(D)) end function Base.:*(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} - return Diagonal(_mul_scalar(T, J.λ, _diag_vec(D))) + return Diagonal(_uniformscale_mul(T, _scale_storage(J.λ), _diag_vec(D))) end Base.:*(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray) = D * J function Base.copyto!(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} - fill!(_diag_vec(D), convert(T, J.λ)) + fill!(_diag_vec(D), _coefficient_as(T, J.λ)) return D end diff --git a/src/ndarray/mapreduce.jl b/src/ndarray/mapreduce.jl index 6c6ec52ac..409e11f09 100644 --- a/src/ndarray/mapreduce.jl +++ b/src/ndarray/mapreduce.jl @@ -115,7 +115,7 @@ _mr_empty(f::Union{typeof(abs),typeof(abs2)}, op::typeof(max), T, M, ::NoReducti mapreduce(f, op, A::NDArray; dims=:, init) Fuse a scalar mapping with a distributed GPU reduction. Supported operators are -`+`, `*`, `min`, and `max`. Full reductions return a 0-d `NDArray`; explicit +`+`, `*`, `min`, and `max`. Full reductions return an `CNScalar`; explicit dimensions retain singleton axes. `sum(f, A)` and `prod(f, A)` use Base's integer widening rules, subject to `allowpromotion`. @@ -143,10 +143,10 @@ function Base.mapreduce(f::F, op::OP, A::NDArray{T}; dims=:, init=NoReductionIni nreduce = prod(d -> mask[d] ? input_shape[d] : 1, 1:ndims(A); init=1) if nreduce == 0 value = _mr_empty(f, op, T, M, init, dims) - return nda_full_array(shape, value) + return _scalar_result(nda_full_array(shape, value)) end - any(iszero, input_shape) && return nda_zeros_array(shape, O) - return _mr_launch(mapper, op, A, R, O, mask, shape, init, dims, nreduce == 1) + any(iszero, input_shape) && return _scalar_result(nda_zeros_array(shape, O)) + return _scalar_result(_mr_launch(mapper, op, A, R, O, mask, shape, init, dims, nreduce == 1)) end Base.mapreduce(f, op, A::NDArray, B::AbstractArray, rest::AbstractArray...; kwargs...) = diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 3afc1630b..a59b7c43d 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -18,7 +18,7 @@ * Nader Rahhal =# -export unwrap, squeeze +export squeeze # See TODO.md (Base / LinearAlgebra sections) for AbstractArray and LA gaps. @@ -855,7 +855,7 @@ function Base.only(x::NDArray{T,N}) where {T,N} return @allowscalar x[firstindex(x)] end -unwrap(x::NDArray) = only(x) +Base.fetch(x::NDArray) = only(x) @doc""" ==(arr1::NDArray, arr2::NDArray) @@ -863,7 +863,7 @@ unwrap(x::NDArray) = only(x) Element-wise equality reduced to a 0-d `NDArray{Bool}` (not a Julia `Bool`). Same shape and values yields true; mismatched shape or rank yields false. -Mixed dtypes follow Julia `==` (promote, then compare). Use `unwrap` or `A[]` +Mixed dtypes follow Julia `==` (promote, then compare). Use `fetch` or `A[]` (with `allowscalar`) for a host value. Broadcast `.==` / `.!=` stay elementwise. # Examples @@ -876,8 +876,8 @@ a == c ``` """ function Base.:(==)(a::NDArray, b::NDArray) - size(a) == size(b) || return NDArray(false) - return _array_equal_impl(a, b) + size(a) == size(b) || return cnscalar(NDArray(false)) + return cnscalar(_array_equal_impl(a, b)) end function Base.:(!=)(a::NDArray, b::NDArray) @@ -978,15 +978,15 @@ isapprox(julia_arr, arr2) """ function Base.isapprox(julia_array::AbstractArray{T}, arr::NDArray{T}; atol=0, rtol=0) where {T} #! REPLCE THIS WITH BIN_OP isapprox - return compare(julia_array, arr, atol, rtol) + return compare(julia_array, arr, _maybe_fetch(atol), _maybe_fetch(rtol)) end function Base.isapprox(arr::NDArray{T}, julia_array::AbstractArray{T}; atol=0, rtol=0) where {T} - return compare(julia_array, arr, atol, rtol) + return compare(julia_array, arr, _maybe_fetch(atol), _maybe_fetch(rtol)) end function Base.isapprox(arr::NDArray{T}, arr2::NDArray{T}; atol=0, rtol=0) where {T} - return compare(arr, arr2, atol, rtol) + return compare(arr, arr2, _maybe_fetch(atol), _maybe_fetch(rtol)) end # HDF5 signature. A leftover empty/truncated file from a crashed write has no diff --git a/src/ndarray/random/generator.jl b/src/ndarray/random/generator.jl index f05dd06a3..af489568a 100644 --- a/src/ndarray/random/generator.jl +++ b/src/ndarray/random/generator.jl @@ -149,18 +149,18 @@ function random(g::Generator, ::Type{Bool}, dims::Dims) return integers(g, Int16, dims; low=0, high=2) .!= Int16(0) end -function _randn!(g::Generator, arr::NDArray{Float32}; loc::Real=0, scale::Real=1) +function _randn!(g::Generator, arr::NDArray{Float32}; loc::Union{Real,DeviceScalar{<:Real}}=0, scale::Union{Real,DeviceScalar{<:Real}}=1) _bitgenerator_distribution!( arr, g.bit_generator, cuNumeric.BITGENDIST_NORMAL_32, - _EMPTY_INT64, SVector{2,Float32}(Float32(loc), Float32(scale)), _EMPTY_FLOAT64, + _EMPTY_INT64, SVector{2,Float32}(Float32(_maybe_fetch(loc)), Float32(_maybe_fetch(scale))), _EMPTY_FLOAT64, ) return arr end -function _randn!(g::Generator, arr::NDArray{Float64}; loc::Real=0, scale::Real=1) +function _randn!(g::Generator, arr::NDArray{Float64}; loc::Union{Real,DeviceScalar{<:Real}}=0, scale::Union{Real,DeviceScalar{<:Real}}=1) _bitgenerator_distribution!( arr, g.bit_generator, cuNumeric.BITGENDIST_NORMAL_64, - _EMPTY_INT64, _EMPTY_FLOAT32, SVector{2,Float64}(Float64(loc), Float64(scale)), + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{2,Float64}(Float64(_maybe_fetch(loc)), Float64(_maybe_fetch(scale))), ) return arr end @@ -205,8 +205,8 @@ function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_ return arr end -function randexp!(g::Generator, arr::NDArray{Float32}; scale::Real=1) - s = Float32(scale) +function randexp!(g::Generator, arr::NDArray{Float32}; scale::Union{Real,DeviceScalar{<:Real}}=1) + s = Float32(_maybe_fetch(scale)) s > 0 || throw(ArgumentError("scale must be positive, got $scale")) _bitgenerator_distribution!( arr, g.bit_generator, cuNumeric.BITGENDIST_EXPONENTIAL_32, @@ -215,8 +215,8 @@ function randexp!(g::Generator, arr::NDArray{Float32}; scale::Real=1) return arr end -function randexp!(g::Generator, arr::NDArray{Float64}; scale::Real=1) - s = Float64(scale) +function randexp!(g::Generator, arr::NDArray{Float64}; scale::Union{Real,DeviceScalar{<:Real}}=1) + s = Float64(_maybe_fetch(scale)) s > 0 || throw(ArgumentError("scale must be positive, got $scale")) _bitgenerator_distribution!( arr, g.bit_generator, cuNumeric.BITGENDIST_EXPONENTIAL_64, @@ -225,12 +225,12 @@ function randexp!(g::Generator, arr::NDArray{Float64}; scale::Real=1) return arr end -function randexp!(g::Generator, arr::NDArray{T}; scale::Real=1) where {T} +function randexp!(g::Generator, arr::NDArray{T}; scale::Union{Real,DeviceScalar{<:Real}}=1) where {T} return error("randexp! only supports Float32 and Float64 NDArray storage") end function randexp( - g::Generator, ::Type{T}, dims::Dims; scale::Real=1 + g::Generator, ::Type{T}, dims::Dims; scale::Union{Real,DeviceScalar{<:Real}}=1 ) where {T<:SUPPORTED_FLOAT_TYPES} arr = zeros(T, dims) randexp!(g, arr; scale=scale) diff --git a/src/ndarray/sort.jl b/src/ndarray/sort.jl index 72164e9d0..c1c91fe62 100644 --- a/src/ndarray/sort.jl +++ b/src/ndarray/sort.jl @@ -80,7 +80,7 @@ Insertion indices into a 1-d sorted `a`. `x` may be a `Number` or an `NDArray` of needles. Scalar queries return a 0-d `NDArray{Int64}` (not a Julia `Int`); use -`unwrap` for an `Int`. Array queries return an `NDArray{Int64}` with the +`fetch` for an `Int`. Array queries return an `NDArray{Int64}` with the shape of the needles. Indices are 1-based. Not `Base.searchsortedfirst` / `searchsortedlast`. `a` must already be sorted @@ -113,13 +113,13 @@ end cuNumeric.searchsorted(a::NDArray{T,1}, x::Number) `searchsortedfirst(a, x):searchsortedlast(a, x)` as a `UnitRange`, matching -Base's scalar search. Materializes two 0-d index arrays via `unwrap`. +Base's scalar search. Materializes two 0-d index arrays via `fetch`. Not `Base.searchsorted`. """ -function searchsorted(a::NDArray{T,1}, x::Number) where {T} +function searchsorted(a::NDArray{T,1}, x::Union{Number,DeviceScalar}) where {T} lo = searchsortedfirst(a, x) hi = searchsortedlast(a, x) - return unwrap(lo):unwrap(hi) + return fetch(lo):fetch(hi) end @doc""" diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 79f7a2edd..d56150581 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -261,8 +261,8 @@ The following unary reduction operations are supported and can be applied direct • `var` / `std` (sample / `corrected=true`; real types only) • `argmax` / `argmin` (1-d only) -Full reductions return a **0-d `NDArray`**, not a Julia scalar. Use `unwrap` or -`A[]` (with `allowscalar`) when you need a host value. +Full reductions return an **`CNScalar`** backed by a 0D NDArray. Use `fetch` or +`only` when you need a host value. Reduction over specific dimensions is supported via the `dims` keyword argument, following the same keepdims semantics as Julia's base reduction functions. @@ -304,7 +304,7 @@ const unary_reduction_map = Dict{Function,UnaryRedCode}( # VARIANCE opcode is unused: compose sample var from mean / sum instead. ) -# Full reductions return 0-d NDArrays (not Julia scalars). That is intentional. +# Public full reductions wrap backend 0D NDArrays as CNScalars. function _unary_reduction_apply(out, op_code, input::NDArray{T}, ::Type{T}) where {T} return nda_unary_reduction(out, op_code, input) @@ -363,7 +363,7 @@ end for (base_func, op_code) in unary_reduction_map @eval begin function $(Symbol(base_func))(input::NDArray{T,N}; dims=Colon()) where {T,N} - return _unary_reduction_impl($base_func, $(op_code), input, dims) + return _scalar_result(_unary_reduction_impl($base_func, $(op_code), input, dims)) end end end @@ -397,11 +397,11 @@ function _bool_reduction_impl(op_code, input::NDArray{Bool}, dims) end function Base.all(input::NDArray{Bool}; dims=Colon()) - return _bool_reduction_impl(cuNumeric.ALL, input, dims) + return _scalar_result(_bool_reduction_impl(cuNumeric.ALL, input, dims)) end function Base.any(input::NDArray{Bool}; dims=Colon()) - return _bool_reduction_impl(cuNumeric.ANY, input, dims) + return _scalar_result(_bool_reduction_impl(cuNumeric.ANY, input, dims)) end # Compare on-device against `zero(T)` / `_eye(T, n)` (identity filled with `one(T)`). @@ -411,14 +411,14 @@ function Base.iszero(A::NDArray{T}) where {T} end function Base.isone(A::NDArray{T,2}) where {T} m, n = size(A) - m != n && return NDArray(false) # LinearAlgebra.isone: only square matrices + m != n && return cnscalar(NDArray(false)) # LinearAlgebra.isone: only square matrices return all(A .== _eye(T, m)) end # Boolean multiplication is logical conjunction. cuPyNumeric's PROD reduction # uses a numeric fill identity, which Legate rejects for a Boolean target. function Base.prod(input::NDArray{Bool}; dims=Colon()) - return _unary_reduction_impl(Base.prod, cuNumeric.ALL, input, dims) + return _scalar_result(_unary_reduction_impl(Base.prod, cuNumeric.ALL, input, dims)) end # Number of elements a reduction with `dims` collapses. Used by mean/var/std. @@ -442,7 +442,7 @@ end """ mean(A::NDArray; dims=:) -Arithmetic mean of `A`. Full reduction returns a 0-d `NDArray`, not a Julia +Arithmetic mean of `A`. Full reduction returns an `CNScalar`, not a host scalar. With `dims`, the reduced axes are kept as size 1, matching Base. """ function mean(arr::NDArray; dims=Colon()) @@ -457,13 +457,13 @@ end std(A::NDArray; corrected=true, mean=nothing, dims=:) Sample variance and standard deviation (`corrected=true`, divisor `n-1`), -matching Julia / StatsBase. Real types only. Returns a 0-d or reduced +matching Julia / StatsBase. Real types only. Returns an `CNScalar` or dimension-preserving `NDArray`, not a Julia scalar. """ function var(arr::NDArray{T}; corrected::Bool=true, mean=nothing, dims=Colon()) where {T<:Real} μ = isnothing(mean) ? cuNumeric.mean(arr; dims=dims) : mean centered = arr .- μ - isnothing(mean) && μ isa NDArray && destroy!(μ) + isnothing(mean) && μ isa Union{NDArray,CNScalar} && destroy!(μ) sq = centered .^ 2 destroy!(centered) s = sum(sq; dims=dims) @@ -481,7 +481,7 @@ end # SQRT kernel rejects 0-d (shape [] vs [1]). Wrap the host sqrt back into a 0-d array. function _sqrt_ndarray(v::NDArray{T,0}) where {T} - s = T(sqrt(unwrap(v))) + s = T(sqrt(fetch(v))) destroy!(v) return NDArray(s) end @@ -505,7 +505,7 @@ end # count(!iszero, A::NDArray; dims=:) # # Count `true` values in a `Bool` array, or nonzeros in a numeric array. -# Returns a 0-d or reduced `NDArray` of integers, not a Julia `Int`. +# Returns an `CNScalar` or dimension-preserving `NDArray` of integers, not a Julia `Int`. # """ # function count(arr::NDArray{Bool}; dims=Colon()) # return _count_nonzero(arr, dims) diff --git a/src/ndarray/vector_linalg.jl b/src/ndarray/vector_linalg.jl index 2855e931b..7fbbb5449 100644 --- a/src/ndarray/vector_linalg.jl +++ b/src/ndarray/vector_linalg.jl @@ -11,29 +11,31 @@ function _matmul_eltype(::Type{T}) where {T<:_LA_INTEGER} end """ - mul!(y::NDArray, A::NDArray, x::NDArray[, α, β]) + mul!(y::NDArray, A::NDArray, x::NDArray) -Compute `y = α * A * x + β * y` for a matrix and vector. Five-argument -matrix-matrix multiplication is also supported. Mixed integer/floating-point -inputs follow the usual promotion policy; integer-integer inputs are unsupported. -The destination must hold the promoted result and must not alias either input. +Store the matrix-vector product `A * x` in `y`. """ function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) return mul!(y, A, x, true, false) end +_mul_output_scale(x::Number, ::Type{T}) where {T} = convert(T, x) +_mul_output_scale(x::NDArray, ::Type{T}) where {T} = x + function _linalg_mul!(C::NDArray{T}, cm, A::NDArray{TA}, am, B::NDArray{TB}, bm, α, β) where {T,TA,TB} required = _matmul_eltype(promote_type(TA, TB)) promote_type(required, T) === T || throw(ArgumentError("mul! output type $T cannot hold promoted input type $required")) Ap = checked_promote_arr(mul!, A, T) Bp = checked_promote_arr(mul!, B, T) - if iszero(α) || isempty(A) || isempty(B) || isempty(C) + α = _scale_storage(α) + β = _scale_storage(β) + if _host_iszero(α) || isempty(A) || isempty(B) || isempty(C) _contract_prepare(C, cm, Ap, am, Bp, bm) if !isempty(C) - if iszero(β) + if _host_iszero(β) fill!(C, zero(T)) else - C .*= convert(T, β) + C .*= _mul_output_scale(β, T) end end else @@ -44,14 +46,32 @@ function _linalg_mul!(C::NDArray{T}, cm, A::NDArray{TA}, am, B::NDArray{TB}, bm, return C end -function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, α::Number, β::Number) +""" + mul!(y::NDArray, A::NDArray, x::NDArray, α, β) + +Store `α * A * x + β * y` in `y`. The destination must not alias an input. +`α` and `β` accept host numbers, 0D `NDArray{T,0}` values, or `CNScalar` wrappers. +""" +function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, α::Union{Number,DeviceScalar}, β::Union{Number,DeviceScalar}) return _linalg_mul!(y, "i", A, "ij", x, "j", α, β) end -function LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, B::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, α::Number, β::Number) +""" + mul!(C::NDArray, A::NDArray, B::NDArray, α, β) + +Store `α * A * B + β * C` in `C`. The destination must not alias an input. +`α` and `β` accept host numbers, 0D `NDArray{T,0}` values, or `CNScalar` wrappers. +""" +function LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, B::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, α::Union{Number,DeviceScalar}, β::Union{Number,DeviceScalar}) return _linalg_mul!(C, "ij", A, "ik", B, "kj", α, β) end +# Resolve intersections with LinearAlgebra's host Number signatures. +LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, α::Number, β::Number) = + _linalg_mul!(y, "i", A, "ij", x, "j", α, β) +LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, B::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, α::Number, β::Number) = + _linalg_mul!(C, "ij", A, "ik", B, "kj", α, β) + function Base.:*(A::NDArray{TA,2}, x::NDArray{TX,1}) where {TA<:SUPPORTED_ARRAY_TYPES,TX<:SUPPORTED_ARRAY_TYPES} T = _matmul_eltype(promote_type(TA, TX)) size(A, 2) == length(x) || throw(DimensionMismatch("matrix-vector dimensions do not match")) @@ -70,11 +90,9 @@ function _dot_same_type(x::NDArray{T,1}, y::NDArray{T,1}) where {T<:Complex} end """ - dot(x::NDArray{<:Any,1}, y::NDArray{<:Any,1}) + dot(x::NDArray, y::NDArray) -Hermitian inner product as a 0D NDArray, without unwrapping or synchronizing. -Conjugates the first operand for complex inputs. Numeric inputs are promoted -using the package's existing policy. Bool-Bool dot accumulates into Int. +Return the vector inner product as a `CNScalar`. Complex inputs conjugate `x`. """ function LinearAlgebra.dot(x::NDArray{TX,1}, y::NDArray{TY,1}) where {TX<:SUPPORTED_ARRAY_TYPES,TY<:SUPPORTED_ARRAY_TYPES} length(x) == length(y) || throw(DimensionMismatch("dot vector lengths do not match")) @@ -84,28 +102,30 @@ function LinearAlgebra.dot(x::NDArray{TX,1}, y::NDArray{TY,1}) where {TX<:SUPPOR result = isempty(x) ? cuNumeric.zeros(T, ()) : _dot_same_type(xp, yp) xp !== x && destroy!(xp) yp !== y && destroy!(yp) - return result + return cnscalar(result) end _norm_nonzero(v) = ifelse(iszero(v), zero(real(v)), one(real(v))) +LinearAlgebra.norm(x::NDArray{<:SUPPORTED_ARRAY_TYPES}, p::NDArray{<:Real,0}) = + norm(x, _maybe_fetch(p)) + """ norm(x::NDArray, p::Real=2) -Entrywise p-norm as a real-valued 0D NDArray (not the matrix operator norm). -The result stays on the backend; no reduction is unwrapped. Like cuPyNumeric, -powers are accumulated without scaling and may overflow or underflow. Integer -inputs convert to floating point under the existing promotion policy. -Uses mapped reductions, which currently require a GPU target. +Return the entrywise `p`-norm as a real `CNScalar`, not a matrix operator norm. +Dense-array norms currently require a GPU. Unscaled accumulation can overflow +or underflow. """ function LinearAlgebra.norm(x::NDArray{T}, p::Real=2) where {T<:_LA_FLOAT} + p = _maybe_fetch(p) R = real(T) - isempty(x) && return cuNumeric.zeros(R, ()) + isempty(x) && return cnscalar(cuNumeric.zeros(R, ())) p == 0 && return sum(_norm_nonzero, x) p == 1 && return sum(abs, x) p == Inf && return maximum(abs, x) p == -Inf && return minimum(abs, x) - isnan(p) && return NDArray(R(NaN)) + isnan(p) && return cnscalar(NDArray(R(NaN))) exponent = R(p) total = p == 2 ? sum(abs2, x) : sum(v -> abs(v)^exponent, x) # Optimization opportunity: a specialized reduction could fuse the root into @@ -126,22 +146,25 @@ function LinearAlgebra.norm(x::NDArray{T}, p::Real=2) where {T<:_LA_INTEGER} end """ - axpy!(α, x::NDArray{<:Any,1}, y::NDArray{<:Any,1}) - axpby!(α, x::NDArray{<:Any,1}, β, y::NDArray{<:Any,1}) + axpy!(α, x::NDArray, y::NDArray) -Update numeric vectors with backend broadcasts and return `y`. Vector lengths -must match. Exact self-aliasing is supported; partially overlapping views are not. +Update and return `y` with `y = α * x + y`. """ -function LinearAlgebra.axpy!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) +function LinearAlgebra.axpy!(α::Union{Number,DeviceScalar}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) length(x) == length(y) || throw(DimensionMismatch("axpy! vector lengths do not match")) - iszero(α) && return y + _host_iszero(α) && return y y .= α .* x .+ y return y end -function LinearAlgebra.axpby!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, β::Number, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) +""" + axpby!(α, x::NDArray, β, y::NDArray) + +Update and return `y` with `y = α * x + β * y`. +""" +function LinearAlgebra.axpby!(α::Union{Number,DeviceScalar}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, β::Union{Number,DeviceScalar}, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) length(x) == length(y) || throw(DimensionMismatch("axpby! vector lengths do not match")) - iszero(α) && isone(β) && return y + _host_iszero(α) && _host_isone(β) && return y y .= α .* x .+ β .* y return y end @@ -156,6 +179,11 @@ function LinearAlgebra.lmul!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES}) return x end +# Keep the Number signatures above to resolve Base's AbstractArray/Number +# intersections. Raw device scalars share their implementation via the wrapper. +LinearAlgebra.rmul!(x::NDArray{<:SUPPORTED_ARRAY_TYPES}, α::NDArray{<:Any,0}) = rmul!(x, cnscalar(α)) +LinearAlgebra.lmul!(α::NDArray{<:Any,0}, x::NDArray{<:SUPPORTED_ARRAY_TYPES}) = lmul!(cnscalar(α), x) + function LinearAlgebra.ldiv!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, D::DiagonalNDArray{<:SUPPORTED_ARRAY_TYPES}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) length(x) == length(y) == size(D, 1) || throw(DimensionMismatch("diagonal solve dimensions do not match")) y .= x ./ _diag_vec(D) diff --git a/test/array/binary/tests.jl b/test/array/binary/tests.jl index 87887b86a..b43ee5966 100644 --- a/test/array/binary/tests.jl +++ b/test/array/binary/tests.jl @@ -182,11 +182,11 @@ function run_binary_ops_tests(types) neq = arr_cn != arr_cn2 @test ndims(eq) == 0 @test ndims(neq) == 0 - @test unwrap(eq) - @test !unwrap(arr_cn == arr_cn2) - @test unwrap(neq) - @test !unwrap(arr_cn != arr_cn) - @test unwrap(all(arr_cn .== arr_cn)) + @test fetch(eq) + @test !fetch(arr_cn == arr_cn2) + @test fetch(neq) + @test !fetch(arr_cn != arr_cn) + @test fetch(all(arr_cn .== arr_cn)) end end end @@ -210,13 +210,13 @@ function run_array_equal_tests() allowscalar() do @test ndims(n32 == n64) == 0 - @test unwrap(n32 == n64) == (a32 == a64) - @test unwrap(n32 != n64) == (a32 != a64) + @test fetch(n32 == n64) == (a32 == a64) + @test fetch(n32 != n64) == (a32 != a64) b64 = Float64[1, 2, 4] m64 = NDArray(b64) - @test unwrap(n32 == m64) == (a32 == b64) - @test unwrap(n32 != m64) == (a32 != b64) + @test fetch(n32 == m64) == (a32 == b64) + @test fetch(n32 != m64) == (a32 != b64) end same = cuNumeric.ones(2, 2) @@ -224,10 +224,10 @@ function run_array_equal_tests() other_rank = cuNumeric.ones(4) allowscalar() do @test ndims(same == other_shape) == 0 - @test !unwrap(same == other_shape) - @test unwrap(same != other_shape) - @test !unwrap(same == other_rank) - @test unwrap(same != other_rank) + @test !fetch(same == other_shape) + @test fetch(same != other_shape) + @test !fetch(same == other_rank) + @test fetch(same != other_rank) end end end diff --git a/test/array/cnscalar.jl b/test/array/cnscalar.jl new file mode 100644 index 000000000..3ad7d38d0 --- /dev/null +++ b/test/array/cnscalar.jl @@ -0,0 +1,137 @@ +using Test, LinearAlgebra, cuNumeric + +struct CNScalarRealSlot{T<:Real} + value::T +end +struct CNScalarNumberSlot{T<:Number} + value::T +end + +@testset "CNScalar storage and arithmetic" begin + allowautofetch(false) + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + a = NDArray(one(T)) + x = cnscalar(a) + @test x.value === a + @test x isa Number + @test x isa supertype(T) + @test isconcretetype(fieldtype(typeof(x), :value)) + @test x isa CNScalar{T} + @test CNScalarNumberSlot(x).value === x + if T <: Real + @test x isa Real + @test CNScalarRealSlot(x).value === x + end + @test fetch(x) == one(T) + @test fetch(a) == one(T) + @test only(x) == fetch(x) + @test_throws ErrorException x[] + @test_throws ErrorException (@allowautofetch x[]) + cuNumeric.allowscalar() do + @test x[] === a[] + @test x[] == one(T) + @test_throws ArgumentError x == one(T) + end + for op in (+, -, *, /, ^) + for (l, r) in ((x, x), (x, one(T)), (one(T), x)) + result = @inferred op(l, r) + @test result isa CNScalar + @test fetch(result) ≈ op(one(T), one(T)) + end + end + @test fetch(zero(x)) == zero(T) + @test fetch(one(x)) == one(T) + for op in (abs, abs2, sqrt, conj, real, imag, inv) + @test fetch(op(x)) ≈ op(one(T)) + end + @test_throws ArgumentError convert(T, x) + @test allowautofetch(() -> convert(T, x)) == one(T) + @test_throws ArgumentError x == one(T) + @test (@allowautofetch x == one(T)) + @test_throws ArgumentError x == one(T) + end + x = sum(NDArray([1.0, 2.0, 3.0])) + @test x isa CNReal{Float64} + @test fetch(x^2) == 36.0 + @test fetch(sqrt(x)) ≈ sqrt(6.0) + @test Array(NDArray([1.0, 2.0]) .* x) == [6.0, 12.0] + @test fetch(max(x, 2.0)) == 6.0 + @test (@allowautofetch x > 2.0) + @test (@allowautofetch isless(2.0, x)) + @test (@allowautofetch 2x) isa CNScalar + p, q = promote(x, 2.0) + @test p isa CNScalar && q isa CNScalar + @test fetch(p) == 6.0 && fetch(q) == 2.0 + for host in (3.0, 1.0 + 2.0im) + p, q = promote(cnscalar(NDArray(2.0f0)), host) + @test p isa CNScalar && q isa CNScalar + @test fetch(p) == 2.0 && fetch(q) == host + end + fractional = cnscalar(NDArray(1.5)) + @test_throws ArgumentError CNInt{Int64}(fractional) + @test_throws InexactError @allowautofetch CNInt{Int64}(fractional) + imaginary = cnscalar(NDArray(1.0 + 2.0im)) + @test_throws ArgumentError CNFloat{Float64}(imaginary) + @test_throws InexactError @allowautofetch CNFloat{Float64}(imaginary) + end +end + +@testset "Device scalar parent storage" begin + for T in (Float64, Int64, UInt64, Bool, ComplexF64) + host = fill(one(T)) + a = NDArray(host) + x = @inferred cnscalar(a) + @test x.value === a + @test x.value.parent === host + @test isconcretetype(fieldtype(typeof(x), :value)) + @test fetch(x) == one(T) + end +end + +@testset "Scoped allowautofetch" begin + x = cnscalar(NDArray(2.0)) + @test allowautofetch(() -> 42) == 42 + @test (@allowautofetch 43) == 43 + @test_throws ErrorException allowautofetch() do + error("scope test") + end + @test_throws ArgumentError Float64(x) + @allowautofetch begin + @test Float64(x) == 2.0 + allowautofetch(false) do + @test_throws ArgumentError Float64(x) + end + @test Float64(x) == 2.0 + # Independent tasks do not inherit this permission. + @test fetch(@async get(task_local_storage(), :cuNumericAllowAutoFetch, false)) == false + end + @test_throws ArgumentError Float64(x) + @test_throws ErrorException @allowautofetch error("macro scope test") + @test_throws ArgumentError Float64(x) + allowautofetch(true) + @test Float64(x) == 2.0 + allowautofetch(false) + @test_throws ArgumentError Float64(x) +end + +@testset "Reduction return boundaries" begin + x = NDArray([1.0, 2.0, 3.0]) + for f in (sum, prod, minimum, maximum, cuNumeric.mean, cuNumeric.var, cuNumeric.std) + result = f(x) + @test result isa CNReal + @test fetch(result) ≈ f([1.0, 2.0, 3.0]) + @test f(x; dims=1) isa NDArray{<:Any,1} + end + @test all(NDArray([true, true])) isa CNReal{Bool} + @test fetch(any(NDArray([false, true]))) + @test dot(x,x) isa CNReal + @test fetch(dot(x,x)) == 14.0 + z = NDArray(ComplexF64[1+2im, 3-im]) + @test dot(z,z) isa CNComplex + @test fetch(dot(z,z)) ≈ dot(ComplexF64[1+2im, 3-im], ComplexF64[1+2im, 3-im]) + @test (x == x) isa CNReal{Bool} + @test fetch(x != NDArray([3.0, 2.0, 1.0])) + @test cuNumeric.trace(NDArray([1.0 2.0; 3.0 4.0])) isa CNReal +end diff --git a/test/array/diagonal.jl b/test/array/diagonal.jl index 52f2ee21d..bcc02f303 100644 --- a/test/array/diagonal.jl +++ b/test/array/diagonal.jl @@ -37,15 +37,15 @@ function _host_diag_compare(ref, out, ::Type{T}) where {T} end end -function _host_scalar_compare(ref, out::NDArray{<:Any,0}, ::Type{T}) where {T} +function _host_scalar_compare(ref, out::CNScalar, ::Type{T}) where {T} allowscalar() do - @test ref ≈ out[] atol=atol(T) rtol=rtol(T) + @test ref ≈ fetch(out) atol=atol(T) rtol=rtol(T) end end -function _host_bool_compare(ref::Bool, out::NDArray{Bool,0}) +function _host_bool_compare(ref::Bool, out::CNReal{Bool}) allowscalar() do - @test out[] == ref + @test fetch(out) == ref end end @@ -109,7 +109,7 @@ end @testset "offset=$k" for k in (-2, -1, 0, 1, 2) ref = sum(diag(A, k)) out = cuNumeric.trace(nda; offset=k) - @test out isa NDArray{<:Any,0} + @test out isa CNScalar _host_scalar_compare(ref, out, eltype(ref)) end end @@ -467,7 +467,7 @@ end d = my_rand(T, 3) Dh = Diagonal(d) D = Diagonal(NDArray(d)) - @test det(D) isa NDArray{<:Any,0} + @test det(D) isa CNScalar _host_scalar_compare(det(Dh), det(D), T) _host_scalar_compare(det(Matrix(D)), det(D), T) diff --git a/test/array/random.jl b/test/array/random.jl index 9013534d0..9049c3307 100644 --- a/test/array/random.jl +++ b/test/array/random.jl @@ -196,7 +196,7 @@ end Ω = T(2) * xmax samples = Ω .* cuNumeric.rand(T, n) .- xmax integrand = (x) -> @. exp(-x^2) - estimate = unwrap((Ω / n) * sum(integrand(samples))) + estimate = fetch((Ω / n) * sum(integrand(samples))) @test isapprox(estimate, T(sqrt(π)); atol=T(0.08)) end diff --git a/test/array/sort.jl b/test/array/sort.jl index ee96378ff..7a61a227b 100644 --- a/test/array/sort.jl +++ b/test/array/sort.jl @@ -116,9 +116,9 @@ end continue end for x in _search_needles(T) - @test cuNumeric.unwrap(cuNumeric.searchsortedfirst(nda, x)) == + @test cuNumeric.fetch(cuNumeric.searchsortedfirst(nda, x)) == Base.searchsortedfirst(A, x) - @test cuNumeric.unwrap(cuNumeric.searchsortedlast(nda, x)) == + @test cuNumeric.fetch(cuNumeric.searchsortedlast(nda, x)) == Base.searchsortedlast(A, x) @test cuNumeric.searchsorted(nda, x) == Base.searchsorted(A, x) end diff --git a/test/array/tensoroperations.jl b/test/array/tensoroperations.jl index f037ae9fd..ab75475e5 100644 --- a/test/array/tensoroperations.jl +++ b/test/array/tensoroperations.jl @@ -213,25 +213,28 @@ using TensorOperations: TensorOperations as TO @test only(Array(scalar_product)) ≈ sum(uhost .* vhost) end - @testset "0D NDArray scale factors" begin + @testset "Device scalar scale factors" begin hostC = reshape(collect(Float64, 1:6), 2, 3) C = NDArray(hostC) - α = sum(C) - @test α isa NDArray - @test ndims(α) == 0 - - @tensor scaled[i, j] := α * C[i, j] - @test Array(scaled) ≈ only(Array(α)) .* hostC + reduced = sum(C) + @test reduced isa CNScalar + @test ndims(reduced) == 0 + allowautofetch(false) do + for α in (NDArray(sum(hostC)), reduced) + @tensor scaled[i, j] := α * C[i, j] + @test Array(scaled) ≈ sum(hostC) .* hostC + + dest = NDArray(fill(2.0, 2, 3)) + @tensor dest[i, j] += α * C[i, j] + @test Array(dest) ≈ fill(2.0, 2, 3) .+ sum(hostC) .* hostC + end + end @tensor s = C[i, j] * C[i, j] @test s isa NDArray @test ndims(s) == 0 @tensor scaled2[i, j] := s * C[i, j] @test Array(scaled2) ≈ only(Array(s)) .* hostC - - dest = NDArray(fill(2.0, 2, 3)) - @tensor dest[i, j] += α * C[i, j] - @test Array(dest) ≈ fill(2.0, 2, 3) .+ only(Array(α)) .* hostC end @testset "temporary destruction" begin diff --git a/test/array/unary/tests.jl b/test/array/unary/tests.jl index 2c9e3aed7..44dfd9d29 100644 --- a/test/array/unary/tests.jl +++ b/test/array/unary/tests.jl @@ -144,10 +144,10 @@ function test_mean_var_std( rtolv = reduction_rtol(T, n) allowpromotion(true) do allowscalar() do - @test isapprox(mean(julia_arr), unwrap(mean(cunumeric_arr)); atol=atolv, rtol=rtolv) + @test isapprox(mean(julia_arr), fetch(mean(cunumeric_arr)); atol=atolv, rtol=rtolv) if T <: Real - @test isapprox(var(julia_arr), unwrap(var(cunumeric_arr)); atol=atolv, rtol=rtolv) - @test isapprox(std(julia_arr), unwrap(std(cunumeric_arr)); atol=atolv, rtol=rtolv) + @test isapprox(var(julia_arr), fetch(var(cunumeric_arr)); atol=atolv, rtol=rtolv) + @test isapprox(std(julia_arr), fetch(std(cunumeric_arr)); atol=atolv, rtol=rtolv) end end for d in 1:N @@ -199,9 +199,9 @@ end # function test_count(julia_arr::AbstractArray{T,N}, cunumeric_arr::NDArray{T,N}) where {T,N} # allowpromotion(true) do # allowscalar() do -# @test count(!iszero, julia_arr) == unwrap(count(!iszero, cunumeric_arr)) +# @test count(!iszero, julia_arr) == fetch(count(!iszero, cunumeric_arr)) # if T == Bool -# @test count(julia_arr) == unwrap(count(cunumeric_arr)) +# @test count(julia_arr) == fetch(count(cunumeric_arr)) # end # end # for d in 1:N @@ -233,10 +233,10 @@ function test_argmax_argmin() allowpromotion(true) do allowscalar() do vn = NDArray(v) - @test unwrap(argmax(vn)) == 2 - @test unwrap(argmin(vn)) == 1 - @test unwrap(argmax(vn)) == argmax(v) - @test unwrap(argmin(vn)) == argmin(v) + @test fetch(argmax(vn)) == 2 + @test fetch(argmin(vn)) == 1 + @test fetch(argmax(vn)) == argmax(v) + @test fetch(argmin(vn)) == argmin(v) c = NDArray(ComplexF32[1, 2]) @test_throws ArgumentError argmax(c) @@ -344,7 +344,7 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) allowscalar() do @test isapprox( sum(x), - unwrap(sum(ndx)); + fetch(sum(ndx)); atol=atol(T) * N, rtol=reduction_rtol(T, N), ) diff --git a/test/array/vector_linalg.jl b/test/array/vector_linalg.jl index 6e0b9a85a..1e422d11b 100644 --- a/test/array/vector_linalg.jl +++ b/test/array/vector_linalg.jl @@ -1,7 +1,7 @@ using Test, LinearAlgebra, Random, cuNumeric # Host extraction belongs to validation, not the NDArray implementation. -la_value(x::cuNumeric.NDArray{T,0}) where {T} = only(x) +la_value(x::cuNumeric.CNScalar) = fetch(x) @testset "LinearAlgebra vector interface" begin cuNumeric.allowscalar(false) @@ -37,7 +37,7 @@ la_value(x::cuNumeric.NDArray{T,0}) where {T} = only(x) zh = randn(T, 5) z = cuNumeric.NDArray(zh) - @test dot(x, z) isa cuNumeric.NDArray{T,0} + @test dot(x, z) isa (T <: Real ? CNReal{T} : CNComplex{T}) @test la_value(dot(x, z)) ≈ dot(xh, zh) @test_throws DimensionMismatch dot(x, y) empty = cuNumeric.zeros(T, 0) diff --git a/test/gpu_only/mapreduce.jl b/test/gpu_only/mapreduce.jl index b06c25813..9ab8fe14b 100644 --- a/test/gpu_only/mapreduce.jl +++ b/test/gpu_only/mapreduce.jl @@ -1,4 +1,6 @@ -_mapped_reduction_host(A) = @allowscalar ndims(A) == 0 ? cuNumeric.unwrap(A) : Array(A) +_mapped_reduction_eltype(x::cuNumeric.CNScalar) = eltype(x.value) +_mapped_reduction_eltype(x) = eltype(x) +_mapped_reduction_host(A) = @allowscalar ndims(A) == 0 ? cuNumeric.fetch(A) : Array(A) struct ReductionAffine scale::Float32 @@ -23,9 +25,9 @@ function _check_mapped_reduction(f, op, input; kwargs...) expected = mapreduce(f, op, input; kwargs...) result = mapreduce(f, op, A; kwargs...) @test size(result) == size(expected) - @test eltype(result) === (expected isa AbstractArray ? eltype(expected) : typeof(expected)) + @test _mapped_reduction_eltype(result) === (expected isa AbstractArray ? eltype(expected) : typeof(expected)) actual = _mapped_reduction_host(result) - if op === min || op === max || eltype(result) <: Integer + if op === min || op === max || _mapped_reduction_eltype(result) <: Integer @test isequal(actual, expected) else tolerances = _mapped_reduction_tolerances(f, input; dims=get(kwargs, :dims, :)) @@ -177,7 +179,7 @@ end cuNumeric.destroy!(parent) # Drop the source before execution completes, then consume the result - # on the device without an intervening unwrap or execution fence. + # on the device without an intervening fetch or execution fence. A = cuNumeric.ones(Float32, 131071) r = mapreduce(abs2, +, A) cuNumeric.destroy!(A) @@ -190,7 +192,7 @@ end A = cuNumeric.ones(Int8, 3) @test_throws Exception sum(identity, A) r = mapreduce(identity, +, A) - @test eltype(r) === Int8 + @test _mapped_reduction_eltype(r) === Int8 cuNumeric.destroy!(r) cuNumeric.destroy!(A) end diff --git a/test/gpu_only/mapreduce_full.jl b/test/gpu_only/mapreduce_full.jl index 57d9eeb1c..3c0008c8f 100644 --- a/test/gpu_only/mapreduce_full.jl +++ b/test/gpu_only/mapreduce_full.jl @@ -110,7 +110,7 @@ function _check_full_reduction(input, op; kwargs...) try expected = mapreduce(identity, op, input; kwargs...) result = mapreduce(identity, op, A; kwargs...) - actual = @allowscalar cuNumeric.unwrap(result) + actual = @allowscalar cuNumeric.fetch(result) @test typeof(actual) === typeof(expected) @test isequal(actual, expected) finally @@ -156,7 +156,7 @@ end sliced = nothing cuNumeric.destroy!(parent) parent = nothing - @test (@allowscalar cuNumeric.unwrap(result)) == mapreduce(x -> x*x, +, host[2:16, 3:18]) + @test (@allowscalar cuNumeric.fetch(result)) == mapreduce(x -> x*x, +, host[2:16, 3:18]) finally isnothing(result) || cuNumeric.destroy!(result) isnothing(sliced) || cuNumeric.destroy!(sliced) @@ -222,7 +222,7 @@ end @test (@allowscalar Array(results[1])) == fill(3f0, 1, 33) for i in 2:3:length(results) @test (@allowscalar Array(results[i])) == fill(1f0, 1, 1) - @test (@allowscalar cuNumeric.unwrap(results[i + 1])) === 35.0 + @test (@allowscalar cuNumeric.fetch(results[i + 1])) === 35.0 @test (@allowscalar Array(results[i + 2])) == fill(1f0, 1, 33) end finally diff --git a/test/gpu_only/vector_norm.jl b/test/gpu_only/vector_norm.jl index 6743df41b..34a21745a 100644 --- a/test/gpu_only/vector_norm.jl +++ b/test/gpu_only/vector_norm.jl @@ -1,6 +1,6 @@ using Test, LinearAlgebra, Random, cuNumeric -la_value(x::cuNumeric.NDArray{T,0}) where {T} = only(x) +la_value(x::cuNumeric.CNScalar) = fetch(x) @testset "NDArray norms" begin cuNumeric.allowscalar(false) @@ -17,7 +17,7 @@ la_value(x::cuNumeric.NDArray{T,0}) where {T} = only(x) @test la_value(norm(x, p)) ≈ norm(xh, p) @test la_value(norm(a, p)) ≈ norm(ah, p) end - @test norm(x) isa cuNumeric.NDArray{R,0} + @test norm(x) isa cuNumeric.CNReal{R} @test la_value(norm(cuNumeric.NDArray(T[0, 2, 0, 3]), 0)) == R(2) @test la_value(norm(cuNumeric.NDArray(T[0, 2]), -1)) == zero(R) @test isnan(la_value(norm(cuNumeric.NDArray(T[NaN, 1])))) @@ -41,7 +41,7 @@ end # Prevent constant propagation of p from hiding branch-dependent return types. Base.@noinline function check_norm_inference(x::cuNumeric.NDArray{T}, p::Real) where {T} R = typeof(float(real(zero(T)))) - @test (@inferred norm(x, p)) isa cuNumeric.NDArray{R,0} + @test (@inferred norm(x, p)) isa cuNumeric.CNReal{R} end @testset "NDArray norm inference" begin @@ -51,7 +51,7 @@ end @testset "$T" begin x = cuNumeric.ones(T, 3) R = typeof(float(real(zero(T)))) - @test (@inferred norm(x)) isa cuNumeric.NDArray{R,0} + @test (@inferred norm(x)) isa cuNumeric.CNReal{R} for p in (0, 1, 2, 3, -1, -2, 0.5, Inf, -Inf, NaN, 2.0f0) check_norm_inference(x, p) end diff --git a/test/util.jl b/test/util.jl index 582fcd2eb..eb4657ab7 100644 --- a/test/util.jl +++ b/test/util.jl @@ -56,7 +56,7 @@ function reduction_atol(::Type{T}, n, scale=1) where {T} return max(atol(T) * n, n * eps(FT) * abs(scale)) end -is_same(arr1::NDArray, arr2::NDArray) = unwrap(arr1 == arr2) +is_same(arr1::NDArray, arr2::NDArray) = fetch(arr1 == arr2) is_same(arr1::NDArray, arr2::Array) = @allowscalar (arr1 == arr2) is_same(arr1::Array, arr2::NDArray) = @allowscalar (arr1 == arr2) is_same(arr1::Array, arr2::Array) = (arr1 == arr2) From 1e7a3703c9414d1ef0ce9e06e4d54cb437af36aa Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:17:13 -0400 Subject: [PATCH 38/49] Krylov.jl Extension (#205) * Add Krylov.jl package extension --- Project.toml | 3 +++ docs/src/linalg.md | 30 +++++++++++++++++++++- ext/cuNumericKrylovExt.jl | 39 +++++++++++++++++++++++++++++ test/Project.toml | 1 + test/array/krylov.jl | 5 ++++ test/gpu_only/krylov.jl | 52 +++++++++++++++++++++++++++++++++++++++ 6 files changed, 129 insertions(+), 1 deletion(-) create mode 100644 ext/cuNumericKrylovExt.jl create mode 100644 test/array/krylov.jl create mode 100644 test/gpu_only/krylov.jl diff --git a/Project.toml b/Project.toml index cdbfc2844..e0342cb1a 100644 --- a/Project.toml +++ b/Project.toml @@ -30,9 +30,11 @@ cupynumeric_jll = "2862d674-414d-5b0b-a494-b21f8deca547" libcxxwrap_julia_jll = "3eaa8342-bff7-56a5-9981-c04077f7cee7" [weakdeps] +Krylov = "ba0b0d4f-ebba-5204-a429-3ac8c609bfb7" TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" [extensions] +cuNumericKrylovExt = "Krylov" cuNumericTensorOperationsExt = "TensorOperations" [compat] @@ -44,6 +46,7 @@ CxxWrap = "0.17.5" ExpressionExplorer = "1.1.4" JuliaFormatter = "2.3.0" KernelAbstractions = "0.9.41" +Krylov = "0.10.10" Legate = "0.2.4" LegatePreferences = "0.1.7" MacroTools = "0.5.16" diff --git a/docs/src/linalg.md b/docs/src/linalg.md index ded351949..9bc782bc6 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -358,6 +358,34 @@ E2 = one(A) # same shape / eltype as A Avoid building a dense identity (or densifying `D` with `Matrix(D)`) just to scale or shift; prefer `Diagonal` and `I` instead. There is no public `eye`. +## Krylov.jl CG + +Loading `Krylov` activates the cuNumeric extension for runtime-backed scalar +coefficients and CG workspace allocation. Allow automatic fetching for the +solver's convergence decisions: + +```julia +using cuNumeric, Krylov + +A = NDArray([4.0 1.0; 1.0 3.0]) +b = NDArray([1.0, 2.0]) +x, stats = @allowautofetch Krylov.cg(A, b; rtol=1e-8) +``` + +For complex systems, use `@allowpromotion @allowautofetch Krylov.cg(...)`: +CG computes real scalar coefficients that must promote when scaling complex +vectors. + +The extension forwards `kaxpy!` and `kaxpby!` with `DeviceScalar` coefficients +to cuNumeric's `LinearAlgebra` methods. See Krylov's +[documented custom-vector helpers](https://jso.dev/Krylov.jl/stable/custom_workspaces/#Methods-to-overload-for-compatibility-with-Krylov.jl). +The `HaloVector` in that example illustrates a custom vector type; this +extension defines the corresponding methods for `NDArray`. + +To reuse allocations, construct `workspace = Krylov.CgWorkspace(A, b)` and call +`@allowautofetch Krylov.cg!(workspace, A, b)`. Other Krylov solvers may require +additional integration methods. + ## Not available yet There is no public dense-matrix `lu`, matrix `inv`, or `ldiv!` yet (beyond the @@ -365,4 +393,4 @@ There is no public dense-matrix `lu`, matrix `inv`, or `ldiv!` yet (beyond the operations, not matrix inverse. Also missing: `eigh` / Hermitian eigen (needs `Hermitian` and `Symmetric` -support on `NDArray`), batched SVD and QR, and conjugate gradient. +support on `NDArray`), and batched SVD and QR. diff --git a/ext/cuNumericKrylovExt.jl b/ext/cuNumericKrylovExt.jl new file mode 100644 index 000000000..26428d138 --- /dev/null +++ b/ext/cuNumericKrylovExt.jl @@ -0,0 +1,39 @@ +module cuNumericKrylovExt + +using cuNumeric: NDArray, DeviceScalar +using LinearAlgebra: axpy!, axpby! +import Krylov + +# Allocate through similar, preserving the NDArray storage type without needing +# S(undef, n). Dagger uses the same workspace-constructor integration point: +# https://github.com/JuliaParallel/Dagger.jl/blob/master/ext/KrylovExt.jl +function Krylov.CgWorkspace(A, b::NDArray{T,1}) where {T} + return Krylov.CgWorkspace(Krylov.KrylovConstructor(similar(b))) +end + +function Krylov.BicgstabWorkspace(A, b::NDArray{T,1}) where {T} + return Krylov.BicgstabWorkspace(Krylov.KrylovConstructor(similar(b))) +end + +# Krylov's documented custom-vector hooks: +# https://jso.dev/Krylov.jl/stable/custom_workspaces/#Methods-to-overload-for-compatibility-with-Krylov.jl +# Its AbstractVector fallbacks operate on whole vectors (the n argument is +# unused), but require coefficients to match the vector element type. These +# methods keep runtime-backed coefficients in the existing cuNumeric operations. +function Krylov.kaxpy!(n::Integer, α::DeviceScalar, x::NDArray{T,1}, y::NDArray{T,1}) where {T} + return axpy!(α, x, y) +end + +function Krylov.kaxpby!(n::Integer, α::DeviceScalar, x::NDArray{T,1}, β::Number, y::NDArray{T,1}) where {T} + return axpby!(α, x, β, y) +end + +function Krylov.kaxpby!(n::Integer, α::Number, x::NDArray{T,1}, β::DeviceScalar, y::NDArray{T,1}) where {T} + return axpby!(α, x, β, y) +end + +function Krylov.kaxpby!(n::Integer, α::DeviceScalar, x::NDArray{T,1}, β::DeviceScalar, y::NDArray{T,1}) where {T} + return axpby!(α, x, β, y) +end + +end diff --git a/test/Project.toml b/test/Project.toml index 82d9c81cb..bb08f2cf2 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -1,4 +1,5 @@ [deps] +Krylov = "ba0b0d4f-ebba-5204-a429-3ac8c609bfb7" CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" FFTW = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341" diff --git a/test/array/krylov.jl b/test/array/krylov.jl new file mode 100644 index 000000000..069b1ba70 --- /dev/null +++ b/test/array/krylov.jl @@ -0,0 +1,5 @@ +using Krylov + +@testset "Krylov extension loading" begin + @test Base.get_extension(cuNumeric, :cuNumericKrylovExt) !== nothing +end diff --git a/test/gpu_only/krylov.jl b/test/gpu_only/krylov.jl new file mode 100644 index 000000000..bcb067318 --- /dev/null +++ b/test/gpu_only/krylov.jl @@ -0,0 +1,52 @@ +using Krylov + +@testset "Krylov GPU integration" begin + @testset "$T" for T in (Float32, Float64, ComplexF32, ComplexF64) + R = real(T) + hostx = T[1, 2, 3] + hosty = T[4, 5, 6] + # Complex coefficients exercise the complex scalar wrapper as well. + a = T <: Complex ? T(2 + im) : T(2) + b = T <: Complex ? T(3 - im) : T(3) + α, β = sum(NDArray([a])), sum(NDArray([b])) + x = NDArray(hostx) + # Neither helpers nor coefficient dispatch should implicitly fetch. + allowautofetch(false) do + for s in (α, α.value) + y = NDArray(copy(hosty)) + @test Krylov.kaxpy!(3, s, x, y) === y + @test Array(y) ≈ a .* hostx .+ hosty + end + for s in (a, α, α.value), t in (b, β, β.value) + y = NDArray(copy(hosty)) + @test Krylov.kaxpby!(3, s, x, t, y) === y + @test Array(y) ≈ a .* hostx .+ b .* hosty + end + end + + n = 16 + offdiag = T <: Complex ? T(-1 + 0.25im) : T(-1) + hostA = Matrix(Tridiagonal(fill(conj(offdiag), n - 1), fill(T(4), n), fill(offdiag, n - 1))) + hostb = T.(1:n) + A, rhs = NDArray(hostA), NDArray(hostb) + workspace = Krylov.CgWorkspace(Krylov.KrylovConstructor(rhs)) + tol = 20 * eps(R) + # Complex CG has real coefficients; permit their promotion to complex. + @allowpromotion @allowautofetch Krylov.cg!(workspace, A, rhs; atol=zero(R), rtol=tol, itmax=100, history=true) + @test workspace.stats.solved + @test norm(hostA * Array(workspace.x) - hostb) / norm(hostb) <= 5tol + @test !isempty(workspace.stats.residuals) + solution, stats = @allowpromotion @allowautofetch Krylov.cg(A, rhs; atol=zero(R), rtol=tol, itmax=100) + @test stats.solved + @test solution isa NDArray + @test norm(hostA * Array(solution) - hostb) / norm(hostb) <= 5tol + + # BiCGSTAB exercises the same device-scalar hooks on a nonsymmetric operator. + nonsymmetric = Matrix(Tridiagonal(fill(T(-0.3), n - 1), fill(T(4), n), fill(T(-0.8), n - 1))) + B = NDArray(nonsymmetric) + biworkspace = Krylov.BicgstabWorkspace(B, rhs) + @allowpromotion @allowautofetch Krylov.bicgstab!(biworkspace, B, rhs; atol=zero(R), rtol=tol, itmax=100) + @test biworkspace.stats.solved + @test norm(nonsymmetric * Array(biworkspace.x) - hostb) / norm(hostb) <= 5tol + end +end From ea466360d15d81ff39b5ad108228d5f56f9d2b95 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:25:00 -0400 Subject: [PATCH 39/49] Load CUDACore before Legate to reduce startup recompilation (#209) [skip ci] --- src/cuNumeric.jl | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index c3fe9b29d..6694e1273 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -23,13 +23,14 @@ module cuNumeric using Preferences using CNPreferences using LegatePreferences: LegatePreferences +# Load CUDACore before Legate/CxxWrap to avoid recompilation in its __init__. +using CUDACore: CUDACore +import CUDACore: CuArray using Legate using Libdl using CxxWrap using CUDATools: CUDATools -using CUDACore: CUDACore -import CUDACore: CuArray import KernelAbstractions: @kernel, @index import KernelAbstractions as KA From 8a4c7482abd364ebbc10c45f0b25fccd492c7120 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Fri, 25 Sep 2026 08:34:12 -0500 Subject: [PATCH 40/49] Add NDarray map!, host copyto!, & bfft! (#204) * Add NDarray map!, host copyto!, & bfft! * Support preallocated NAS FT inverse FFT --- benchmark | 2 +- src/cuNumeric.jl | 2 +- src/ndarray/binary.jl | 12 +++++++++--- src/ndarray/fft.jl | 28 +++++++++++++++++++++++++++- src/ndarray/ndarray.jl | 13 +++++++++++++ test/array/binary/float.jl | 10 ++++++++++ test/array/conversion_lifetimes.jl | 19 ++++++++++++++++--- test/gpu_only/fft.jl | 18 +++++++++++++++++- 8 files changed, 94 insertions(+), 10 deletions(-) diff --git a/benchmark b/benchmark index fdc98d39d..4fd2fae81 160000 --- a/benchmark +++ b/benchmark @@ -1 +1 @@ -Subproject commit fdc98d39d8b0a29326bdc3b8041c6fa9ed80b0b3 +Subproject commit 4fd2fae81cc99ce641644e82861b196defab1a91 diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index 6694e1273..60df70eb2 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -44,7 +44,7 @@ import Base: axes, convert, copy, copyto!, inv, isfinite, sqrt, -, +, *, ==, !=, using LinearAlgebra import LinearAlgebra: mul! -import AbstractFFTs: fft, ifft, fft!, ifft! +import AbstractFFTs: fft, ifft, bfft!, fft!, ifft! using Random import Random: rand!, randn!, randexp! diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 8dfb9beee..6a631fea6 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -333,6 +333,12 @@ function Base.map(f::Function, arr1::NDArray{A,N}, arr2::NDArray{B,N}) where {A, return f.(arr1, arr2) # Will try to call one of the functions generated above end -# function Base.map!(f::Function, dest::NDArray, arr1::NDArray, arr2::NDArray) -# return f -# end +for (julia_fn, _) in binary_op_map + @eval function Base.map!( + f::typeof($(julia_fn)), dest::NDArray{O,N}, arr1::NDArray{T,N}, arr2::NDArray{T,N} + ) where {O,T,N} + axes(dest) == axes(arr1) == axes(arr2) || + throw(DimensionMismatch("map! arrays must have matching axes")) + return __broadcast(f, dest, arr1, arr2) + end +end diff --git a/src/ndarray/fft.jl b/src/ndarray/fft.jl index 108006228..68cab8a97 100644 --- a/src/ndarray/fft.jl +++ b/src/ndarray/fft.jl @@ -17,7 +17,7 @@ * Ethan Meitz =# -export fft, ifft, fft!, ifft!, batched_fft, batched_ifft, batched_fft!, batched_ifft! +export fft, ifft, bfft!, fft!, ifft!, batched_fft, batched_ifft, batched_fft!, batched_ifft! const _FFT_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool,SUPPORTED_FLOAT_TYPES} const _FFT_ACCEPTED = Union{SUPPORTED_COMPLEX_TYPES,_FFT_PROMOTABLE} @@ -127,6 +127,32 @@ function ifft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} return fft_task!(A, A, region, Int32(cuNumeric.FFT_INVERSE); scale=true) end +function bfft!(A::NDArray{T,N}) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return bfft!(A, _fft_dims(A)) +end +function bfft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(A, A, _fft_dims(A, dims), Int32(cuNumeric.FFT_INVERSE)) +end + +""" + bfft!(dest::NDArray, src::NDArray) + +Write the unnormalized inverse FFT of `src` into `dest` without changing `src`. +Both arrays must have the same shape and complex eltype. +""" +function bfft!( + dest::NDArray{T,N}, src::NDArray{T,N} +) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(dest, src, _fft_dims(src), Int32(cuNumeric.FFT_INVERSE)) +end + +function bfft!(A::NDArray) + return throw(ArgumentError("bfft! requires a complex NDArray; got $(eltype(A))")) +end +function bfft!(A::NDArray, ::Any) + return throw(ArgumentError("bfft! requires a complex NDArray; got $(eltype(A))")) +end + function ifft!(A::NDArray) return throw(ArgumentError("ifft! requires a complex NDArray; got $(eltype(A))")) end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index a59b7c43d..fd87f69b8 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -149,6 +149,19 @@ a[1,1] return arr end +function Base.copyto!(dest::NDArray{T,N}, src::Array{T,N}) where {T,N} + attached = _nda_from_julia_array(src) + try + GC.@preserve src attached begin + copyto!(dest, attached) + issue_execution_fence(; block=true) + end + finally + destroy!(attached) + end + return dest +end + @doc""" as_type(arr::NDArray, t::Type{T}) where {T} diff --git a/test/array/binary/float.jl b/test/array/binary/float.jl index b3250e5bf..9198fd216 100644 --- a/test/array/binary/float.jl +++ b/test/array/binary/float.jl @@ -26,3 +26,13 @@ run_binary_ops_tests( ), ) run_array_equal_tests() + +@testset "map!" begin + lhs = ComplexF64[1 + 2im, 3 + 4im] + rhs = ComplexF64[5 + 6im, 7 + 8im] + dest = cuNumeric.zeros(ComplexF64, 2) + @test map!(*, dest, cuNumeric.NDArray(lhs), cuNumeric.NDArray(rhs)) === dest + allowscalar() do + @test Array(dest) == lhs .* rhs + end +end diff --git a/test/array/conversion_lifetimes.jl b/test/array/conversion_lifetimes.jl index 9e469d471..72ba9e462 100644 --- a/test/array/conversion_lifetimes.jl +++ b/test/array/conversion_lifetimes.jl @@ -1,5 +1,18 @@ using Test +@testset "copyto! from Array" begin + expected = reshape(ComplexF64.(1:8), 2, 2, 2) + source = copy(expected) + dest = cuNumeric.zeros(ComplexF64, size(source)) + try + @test copyto!(dest, source) === dest + fill!(source, 0) + @test Array(dest) == expected + finally + cuNumeric.destroy!(dest) + end +end + @testset "1D conversion ownership" begin for T in (Float32, ComplexF32) expected = T[1, 2, 3, 4] @@ -32,9 +45,9 @@ end @testset "Singleton vector host conversion" begin # Runtime-created singletons can use scalar futures; attached input # vectors do not exercise the same storage representation. - for (T, S, value) in ((Float32, Float64, -0f0), - (ComplexF32, ComplexF64, ComplexF32(Inf, 0)), - (Bool, Int32, true)) + for (T, S, value) in ((Float32, Float64, -0.0f0), + (ComplexF32, ComplexF64, ComplexF32(Inf, 0)), + (Bool, Int32, true)) a = cuNumeric.fill(value, (1,)) try converted = Array(a) diff --git a/test/gpu_only/fft.jl b/test/gpu_only/fft.jl index 4362aa192..0564de16e 100644 --- a/test/gpu_only/fft.jl +++ b/test/gpu_only/fft.jl @@ -88,7 +88,7 @@ end end end -@testset verbose = true "fft!/ifft!" begin +@testset verbose = true "fft!/ifft!/bfft!" begin @testset verbose = true for T in _FFT_INPLACE_TYPES rtol_t = rtol(T) atol_t = atol(T) @@ -103,6 +103,21 @@ end z = ifft!(x) @test z === x @test _fft_compare(x, x_cpu; rtol=rtol_t, atol=atol_t) + + x = cuNumeric.NDArray(copy(x_cpu)) + z = bfft!(x) + @test z === x + @test _fft_compare(x, length(x_cpu) .* ifft(x_cpu); rtol=rtol_t, atol=atol_t) + + src_cpu = my_rand(T, 5, 8, 12) + src = cuNumeric.NDArray(src_cpu) + dest = cuNumeric.zeros(T, size(src)) + @test bfft!(dest, src) === dest + @test _fft_compare( + dest, prod(size(src_cpu)) .* _reference_ifft(src_cpu); + rtol=rtol_t, atol=atol_t, + ) + @test _fft_compare(src, src_cpu; rtol=rtol_t, atol=atol_t) end @testset "fft! rejects non-complex" begin @@ -110,6 +125,7 @@ end x = cuNumeric.NDArray(my_rand(T, 8)) @test_throws ArgumentError fft!(x) @test_throws ArgumentError ifft!(x) + @test_throws ArgumentError bfft!(x) end end end From 56bef00c29433ea8d4590441644fa6ba1e5670e6 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Fri, 25 Sep 2026 14:53:18 -0400 Subject: [PATCH 41/49] Support low-storage ODE scalar and muladd operations (#206) * Restack ODE scalar and broadcast support on develop * Simplify NDArray fill! conversion --- docs/src/api_preferences.md | 4 +- src/cnscalar.jl | 3 ++ src/ndarray/binary.jl | 6 +++ src/ndarray/broadcast.jl | 48 +++++++++++++------ src/ndarray/broadcast_fusion.jl | 16 ++++++- src/ndarray/ndarray.jl | 4 +- src/ndarray/promotion.jl | 53 +++++++++++++++++++-- src/ndarray/unary.jl | 25 ++++++++++ test/analysis/promotion.jl | 17 +++++++ test/array/broadcast_basic.jl | 10 ++++ test/array/cnscalar.jl | 20 ++++++++ test/gpu_only/broadcast_fusion.jl | 78 ++++++++++++++++++++++++++++++- 12 files changed, 256 insertions(+), 28 deletions(-) diff --git a/docs/src/api_preferences.md b/docs/src/api_preferences.md index 89c1f356c..3d44a580a 100644 --- a/docs/src/api_preferences.md +++ b/docs/src/api_preferences.md @@ -8,7 +8,7 @@ Out of the box (no `LocalPreferences.toml` changes): |---|---| | Binary / build mode | **JLL** prebuilt binaries | | Broadcast fusion | **on** | -| `FUSE_BROADCAST_MIN_OPS` | **2** (single-op broadcasts stay unfused) | +| `FUSE_BROADCAST_MIN_OPS` | **2** (single native operations stay unfused) | | Task scope names | **off** | | `MIN_SOLVE_MATRIX_SIZE` | **2048** rows | | `MIN_SOLVE_TILE_SIZE` | **512** | @@ -60,7 +60,7 @@ CNPreferences.set_broadcast_fusion_min_ops!(1) # also fuse single-ops `set_broadcast_fusion_min_ops!` counts `Broadcasted` nodes (ops) in the tree: -- **`2` (default):** fuse multi-op trees such as `y .= @. a * b + c`. Single-ops like `y .= cos.(x)` stay on the unfused C-API path. +- **`2` (default):** fuse multi-op trees such as `y .= @. a * b + c`. Single native operations like `y .= cos.(x)` use the C-API path. Functions without a native implementation are fused on the GPU when their broadcast arguments are eligible. - **`1`:** fuse every eligible expression, including single-ops. Set the preference in one Julia process, then start a fresh process to use it. diff --git a/src/cnscalar.jl b/src/cnscalar.jl index d1e657c86..8ee80f6ab 100644 --- a/src/cnscalar.jl +++ b/src/cnscalar.jl @@ -35,6 +35,9 @@ const _REAL_SCALAR_WRAPPERS = (CNFloat, CNInt, CNUInt, CNBool) const _SCALAR_WRAPPERS = (_REAL_SCALAR_WRAPPERS..., CNComplex) for (W, H) in ((CNFloat, AbstractFloat), (CNInt, Signed), (CNUInt, Unsigned), (CNBool, Bool), (CNComplex, Complex)) + # Number's identity constructor otherwise conflicts with the generated + # field constructor when a wrapped scalar is passed back to its type. + @eval $W{T,P}(x::$W{T,P}) where {T<:$H,P} = x @eval cnscalar(x::NDArray{T,0}) where {T<:$H} = $W(x) @eval _scalar_type(::Type{T}) where {T<:$H} = $W{T} @eval _scalar_eltype(::Type{<:$W{T}}) where {T} = T diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 6a631fea6..bc6da8774 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -44,6 +44,12 @@ const floaty_binary_op_map = Dict{Function,BinaryOpCode}( Base.atan => cuNumeric.ARCTAN2, ) +for julia_fn in (keys(binary_op_map)..., keys(floaty_binary_op_map)...) + @eval @inline _has_unfused_broadcast(::typeof($julia_fn), ::Val{2}) = true +end +@inline _has_unfused_broadcast(::typeof(+), ::Val{N}) where {N} = N >= 2 +@inline _has_unfused_broadcast(::typeof(*), ::Val{N}) where {N} = N >= 2 + ## SPECIAL CASES ## # Promote into out's eltype, then destroy any new temps (dispatch; no runtime !==). @inline function _nda_binary_op_promoted!( diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 934b20488..daeb02ed9 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -68,20 +68,22 @@ end function __broadcast(f::Function, _, args...) return error( - "Broadcasting $(f) is not supported by cuNumeric's unfused broadcast path.\n" * - "Single-operation broadcasts skip fusion when FUSE_BROADCAST_MIN_OPS > 1 " * - "(current: $(FUSE_BROADCAST_MIN_OPS); fusion enabled: $(FUSE_BROADCAST_EXPRS)).\n" * - "To enable GPU fusion for eligible single-operation broadcasts, set the " * - "FUSE_BROADCAST_MIN_OPS preference to 1:\n" * - " using CNPreferences\n" * - " CNPreferences.enable_broadcast_fusion!()\n" * - " CNPreferences.set_broadcast_fusion_min_ops!(1)\n" * - "Then restart Julia and retry. This is a preference, not an environment variable.\n" * - "Fusion requires an active GPU, compatible array shapes, and a GPU-compilable function. " * + "Broadcasting $(f) is not supported by cuNumeric's unfused broadcast path. " * + "Functions without a native broadcast implementation require GPU fusion, compatible array shapes, " * + "and a GPU-compilable function (fusion enabled: $(FUSE_BROADCAST_EXPRS)). " * "Otherwise, rewrite the expression using supported broadcast operations.", ) end +# Low-storage Runge–Kutta methods use muladd. Express it as two supported +# broadcasts so the unfused path works and GPU fusion can still combine them. +@inline function __broadcast( + ::typeof(muladd), out::NDArray, a::NDArray, b::NDArray, c::NDArray +) + out .= a .* b .+ c + return out +end + # Get depth of Broadcast tree recursively # Need to call instantiate first bcast_depth(bc::Base.Broadcast.Broadcasted) = maximum(bcast_depth, bc.args; init=0) + 1; @@ -124,6 +126,7 @@ __materialize(x::Base.RefValue{typeof(^)}) = x __materialize(x::Base.RefValue{Val{-1}}) = x # enables specialized reciprocal definition __materialize(x::Base.RefValue{Val{2}}) = x # enables specialized square definition __materialize(x::Base.RefValue{Val{V}}) where {V} = NDArray(V) # Use binary_op POWER for other literal powers +__materialize(x::Base.RefValue) = x # Catch unknown things... __materialize(x) = error("Unrecognized leaf in broadcast expression: $(x)") @@ -233,15 +236,28 @@ end _broadcast_tree_length_args(Base.tail(args)) end -# Prefer fusion only when the tree has at least `FUSE_BROADCAST_MIN_OPS` ops. +# A single native operation uses the C API. Unknown functions need the GPU +# broadcast kernel even when the preference normally skips single-op fusion. +@inline _has_unfused_broadcast(f, ::Val) = false +@inline _has_unfused_broadcast(::typeof(muladd), ::Val{3}) = true +@inline _has_unfused_broadcast(::typeof(abs2), ::Val{1}) = true +@inline _has_unfused_broadcast(bc::Broadcasted) = + _has_unfused_broadcast(bc.f, Val(length(bc.args))) + +# Prefer fusion when the tree has enough ops, or a single op has no native path. # When that const is <= 1, every Broadcasted qualifies and the length check # compiles out (`@static`). @inline function _should_attempt_broadcast_fusion(dest::NDArray, bc::Broadcasted) @static if FUSE_BROADCAST_MIN_OPS <= 1 return can_fuse_linear_broadcast(dest, bc) else - return _broadcast_tree_length(bc) >= FUSE_BROADCAST_MIN_OPS && - can_fuse_linear_broadcast(dest, bc) + operation_count = _broadcast_tree_length(bc) + meets_fusion_minimum = operation_count >= FUSE_BROADCAST_MIN_OPS + single_operation_needs_fusion = + operation_count == 1 && !_has_unfused_broadcast(bc) + worth_fusing = meets_fusion_minimum || single_operation_needs_fusion + worth_fusing || return false + return can_fuse_linear_broadcast(dest, bc) end end @@ -258,9 +274,11 @@ end # Require an active GPU target so `--gpus 0` stays on the unfused path. # Fusion requires same-shaped NDArray leaves; otherwise fall back. - # Single-op exprs (length < `FUSE_BROADCAST_MIN_OPS`) stay unfused by default. + # Single native ops below `FUSE_BROADCAST_MIN_OPS` use the unfused C API. @static if FUSE_BROADCAST_EXPRS - if _has_gpu_target() && _should_attempt_broadcast_fusion(dest, bc) + gpu_available = _has_gpu_target() + should_fuse = gpu_available && _should_attempt_broadcast_fusion(dest, bc) + if should_fuse return fuse_broadcast_tree!(dest, bc) else return _copyto_unfused!(dest, unravel_broadcast_tree(bc)) diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 6a8d6eec1..a46dd213f 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -465,7 +465,14 @@ function get_cuda_task( lock(_BCAST_PTX_CACHE_LOCK) do return get!(_BCAST_PTX_CACHE, key) do - ptx, threads, ctx = get_ptx(obj, DEST_T, ARG_TYPES...) + ptx, threads, ctx = try + get_ptx(obj, DEST_T, ARG_TYPES...) + catch err + err isa InterruptException && rethrow() + throw(ErrorException( + "GPU broadcast function failed to fuse: $(sprint(showerror, err))" + )) + end orig_name = extract_kernel_name(ptx) unique_name = orig_name * "_" * string(hash(ptx); base=16) @@ -644,7 +651,12 @@ end else __checked_promote_op(bc.f, eltypes) end - __my_promote_type(eltypes.parameters...) + if bc.f === Base.literal_pow + __my_promote_type(eltypes.parameters...) + else + numeric_types = _numeric_broadcast_types(eltypes) + isempty(numeric_types) || __my_promote_type(numeric_types...) + end return T_OUT end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index fd87f69b8..6b1288f44 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -679,8 +679,8 @@ end return destroy!(s) end -@inline function Base.fill!(arr::NDArray{T}, val::T) where {T} - nda_fill_array(arr, val) +@inline function Base.fill!(arr::NDArray{T}, val::SUPPORTED_ARRAY_TYPES) where {T} + nda_fill_array(arr, convert(T, val)) return arr end diff --git a/src/ndarray/promotion.jl b/src/ndarray/promotion.jl index cb1034098..067f237c9 100644 --- a/src/ndarray/promotion.jl +++ b/src/ndarray/promotion.jl @@ -35,6 +35,24 @@ unchecked_promote_arr(::Base.RefValue{Val{V}}, ::Type{T}) where {T,V} = Val{V} __checked_promote_op(op, ::Type{Tuple{A}}) where {A} = __checked_promote_op(op, A) __checked_promote_op(op, ::Type{Tuple{A,B}}) where {A,B} = __checked_promote_op(op, A, B) +# Ref-wrapped functions and other static broadcast arguments participate in +# result-type inference, but they are not numeric inputs to promote. +@inline function _numeric_broadcast_types(::Type{Args}) where {Args<:Tuple} + return filter(T -> T <: Number, tuple(Args.parameters...)) +end + +@inline function _smallest_numeric_broadcast_type(::Type{Args}) where {Args<:Tuple} + types = _numeric_broadcast_types(Args) + return isempty(types) ? nothing : foldl(smaller_type, types) +end + +@inline function _broadcast_result_type(op, ::Type{T}) where {T} + T <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError( + "Broadcast function $(op) cannot produce an NDArray: unsupported result type $(T)" + )) + return T +end + # Julia flattens dotted `+` and `*` chains into n-ary Broadcasted nodes. Fold # their input types pairwise, matching both the binary C API and fused path. @inline function __checked_promote_op( @@ -43,6 +61,31 @@ __checked_promote_op(op, ::Type{Tuple{A,B}}) where {A,B} = __checked_promote_op( return _checked_promote_associative(op, Args.parameters...) end +# Resolve the 4+-argument overlap with the general custom-function method. +@inline function __checked_promote_op( + op::Union{typeof(+),typeof(*)}, ::Type{Args} +) where {Args<:Tuple{Any,Any,Any,Any,Vararg{Any}}} + return _checked_promote_associative(op, Args.parameters...) +end + +@inline function __checked_promote_op( + op, ::Type{Tuple{A,B,C}} +) where {A,B,C} + T = _broadcast_result_type(op, Base.promote_op(op, A, B, C)) + S = _smallest_numeric_broadcast_type(Tuple{A,B,C}) + S === nothing || (is_wider_type(T, S) && assertpromotion(op, S, T)) + return T +end + +@inline function __checked_promote_op( + op, ::Type{Args} +) where {Args<:Tuple{Any,Any,Any,Any,Vararg{Any}}} + T = _broadcast_result_type(op, Base.promote_op(op, Args.parameters...)) + S = _smallest_numeric_broadcast_type(Args) + S === nothing || (is_wider_type(T, S) && assertpromotion(op, S, T)) + return T +end + # Path for literal powers @inline function __checked_promote_op( f::typeof(Base.literal_pow), a::Type{Tuple{_,ARR_TYPE,Val{POWER}}} @@ -69,21 +112,21 @@ __recip_type(::Type{Int64}) = Float64 __recip_type(::Type{Bool}) = DEFAULT_FLOAT @inline function __checked_promote_op(op, ::Type{A}) where {A} - T = Base.promote_op(op, A) + T = _broadcast_result_type(op, Base.promote_op(op, A)) is_wider_type(T, A) && assertpromotion(op, A, T) return T end @inline function __checked_promote_op(op, ::Type{A}, ::Type{A}) where {A} - T = Base.promote_op(op, A, A) + T = _broadcast_result_type(op, Base.promote_op(op, A, A)) is_wider_type(T, A) && assertpromotion(op, A, T) return T end @inline function __checked_promote_op(op, ::Type{A}, ::Type{B}) where {A,B} - T = Base.promote_op(op, A, B) - S = smaller_type(A, B) - is_wider_type(T, S) && assertpromotion(op, S, T) + T = _broadcast_result_type(op, Base.promote_op(op, A, B)) + S = _smallest_numeric_broadcast_type(Tuple{A,B}) + S === nothing || (is_wider_type(T, S) && assertpromotion(op, S, T)) return T end diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index d56150581..93f692b4a 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -46,6 +46,14 @@ const unary_op_map_no_args = Dict{Function,UnaryOpCode}( Base.round => cuNumeric.RINT, ) +for julia_fn in (keys(floaty_unary_ops_no_args)..., keys(unary_op_map_no_args)..., + identity, real, imag, conj, inv, !) + @eval @inline _has_unfused_broadcast(::typeof($julia_fn), ::Val{1}) = true +end +# Positional rounding modes are rejected by the native path, too. +@inline _has_unfused_broadcast(::typeof(round), ::Val) = true +@inline _has_unfused_broadcast(::typeof(Base.literal_pow), ::Val{3}) = true + ### SPECIAL CASES ### # `dest .= src` lowers to `identity.(src)`. Treat identity like the native @@ -55,6 +63,23 @@ const unary_op_map_no_args = Dict{Function,UnaryOpCode}( return nda_unary_op!(out, cuNumeric.COPY, input) end +# Real abs2 is a single native square. +@inline function __broadcast(::typeof(abs2), out::NDArray{T}, input::NDArray{T}) where {T<:Real} + return nda_unary_op!(out, cuNumeric.SQUARE, input) +end + +# Complex abs2 has no matching native opcode. Take the magnitude into a real +# temporary and square it, keeping the unfused path available without a GPU. +@inline function __broadcast( + ::typeof(abs2), out::NDArray{T}, input::NDArray{Complex{T}} +) where {T<:SUPPORTED_FLOAT_TYPES} + magnitude = similar(out) + nda_unary_op!(magnitude, cuNumeric.ABSOLUTE, input) + nda_unary_op!(out, cuNumeric.SQUARE, magnitude) + destroy!(magnitude) + return out +end + # Needed to support != Base.:(!)(input::NDArray{Bool,0}) = nda_unary_op!(similar(input), cuNumeric.LOGICAL_NOT, input) Base.:(!)(input::NDArray{Bool,1}) = nda_unary_op!(similar(input), cuNumeric.LOGICAL_NOT, input) diff --git a/test/analysis/promotion.jl b/test/analysis/promotion.jl index 422e5766a..3c3d5ddd4 100644 --- a/test/analysis/promotion.jl +++ b/test/analysis/promotion.jl @@ -24,3 +24,20 @@ end @test @inferred(cuNumeric.__checked_promote_op(+, NTuple{5,Float64})) === Float64 @test @inferred(cuNumeric.__checked_promote_op(*, NTuple{4,Int32})) === Int32 end + +@testset "Ternary broadcast promotion" begin + @test @inferred(cuNumeric.__checked_promote_op(muladd, Tuple{Float32,Float32,Float32})) === Float32 +end + +_promotion_test_norm(x, t) = abs(x) +_promotion_test_residual(e, u0, u1, atol, rtol, norm, t) = + e / (atol + max(norm(u0, t), norm(u1, t)) * rtol) + +@testset "Custom broadcast promotion with a function argument" begin + argtypes = Tuple{ + Float32,Float32,Float32,Float32,Float32,typeof(_promotion_test_norm),Float32 + } + @test @inferred(cuNumeric._numeric_broadcast_types(argtypes)) == + (Float32, Float32, Float32, Float32, Float32, Float32) + @test @inferred(cuNumeric.__checked_promote_op(_promotion_test_residual, argtypes)) === Float32 +end diff --git a/test/array/broadcast_basic.jl b/test/array/broadcast_basic.jl index 16b7f710b..c0afa6516 100644 --- a/test/array/broadcast_basic.jl +++ b/test/array/broadcast_basic.jl @@ -159,6 +159,16 @@ end end end +@testset "muladd broadcast" begin + for T in (Float32, Float64) + a = NDArray(T[1, 2, 3]) + b = NDArray(T[4, 5, 6]) + out = cuNumeric.zeros(T, 3) + out .= muladd.(T(2), a, b) + @test Array(out) == muladd.(T(2), T[1, 2, 3], T[4, 5, 6]) + end +end + #TODO LOOP BINARY OPS WITH SCALARS @testset verbose = true "Scalars" begin N = 10 diff --git a/test/array/cnscalar.jl b/test/array/cnscalar.jl index 3ad7d38d0..d0d33ce8a 100644 --- a/test/array/cnscalar.jl +++ b/test/array/cnscalar.jl @@ -15,6 +15,8 @@ end a = NDArray(one(T)) x = cnscalar(a) @test x.value === a + @test @inferred(typeof(x)(x)) === x + @test fetch(oneunit(typeof(x))) == one(T) @test x isa Number @test x isa supertype(T) @test isconcretetype(fieldtype(typeof(x), :value)) @@ -90,6 +92,24 @@ end end end +@testset "fill! host and device scalars" begin + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + a = cuNumeric.zeros(T, 3) + @test fill!(a, false) === a + @test Array(a) == fill(zero(T), 3) + @test fill!(a, one(T)) === a + @test Array(a) == fill(one(T), 3) + device_zero = cnscalar(NDArray(zero(T))) + for value in (device_zero, device_zero.value) + @test fill!(a, value) === a + @test Array(a) == fill(zero(T), 3) + end + end + a = cuNumeric.zeros(Float32, 2) + @test fill!(a, 2) === a + @test Array(a) == Float32[2, 2] +end + @testset "Scoped allowautofetch" begin x = cnscalar(NDArray(2.0)) @test allowautofetch(() -> 42) == 42 diff --git a/test/gpu_only/broadcast_fusion.jl b/test/gpu_only/broadcast_fusion.jl index bacb59f85..d64bc786f 100644 --- a/test/gpu_only/broadcast_fusion.jl +++ b/test/gpu_only/broadcast_fusion.jl @@ -30,6 +30,11 @@ =# _broadcast_fusion_user_add(x, y) = x + y +_broadcast_fusion_absnorm(x, t) = abs(x) +_broadcast_fusion_residual(e, u0, u1, atol, rtol, norm, t) = + e / (atol + max(norm(u0, t), norm(u1, t)) * rtol) +_broadcast_fusion_bad_result(x) = string(x) +_broadcast_fusion_bad_kernel(x) = parse(Float32, string(x)) @testset "Broadcast Fusion" begin T=Float32 @@ -50,6 +55,75 @@ _broadcast_fusion_user_add(x, y) = x + y s2 = T(1.0) s3 = T(0.5) + @testset "single custom operation uses fusion" begin + dest = cuNumeric.zeros(T, N) + custom_bc = Base.broadcasted(_broadcast_fusion_user_add, a, b) + native_bc = Base.broadcasted(sin, a) + ref = Ref(_broadcast_fusion_absnorm) + + @test cuNumeric.__materialize(ref) === ref + @test cuNumeric._should_attempt_broadcast_fusion(dest, custom_bc) + if cuNumeric.FUSE_BROADCAST_MIN_OPS > 1 + @test !cuNumeric._should_attempt_broadcast_fusion(dest, native_bc) + @test !cuNumeric._should_attempt_broadcast_fusion(dest, Base.broadcasted(abs2, a)) + end + + if cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric._has_gpu_target() + result = _broadcast_fusion_user_add.(a, b) + @allowscalar @test safe_compare(julia_a .+ julia_b, result, atol, rtol) + + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + T <: Real || continue + values = T === Bool ? Bool[false, true] : T[0, 1, 2] + @test Array(abs2.(NDArray(values))) == abs2.(values) + @test Array(map(abs2, NDArray(values))) == abs2.(values) + end + + for T in (ComplexF32, ComplexF64) + z_host = T[1 + 2im, 2 - 3im] + z = NDArray(z_host) + @test Array(abs2.(z)) ≈ abs2.(z_host) + @test Array(map(abs2, z)) ≈ abs2.(z_host) + end + + residual = _broadcast_fusion_residual.( + a, b, c, s1, s2, ref, s3 + ) + expected = julia_a ./ (s1 .+ max.(abs.(julia_b), abs.(julia_c)) .* s2) + @allowscalar @test safe_compare(expected, residual, atol, rtol) + + err = try + dest .= _broadcast_fusion_bad_result.(a) + nothing + catch caught + caught + end + @test err isa ArgumentError + @test occursin("unsupported result type String", sprint(showerror, err)) + + nested_err = try + dest .= _broadcast_fusion_bad_result.(a) .+ s1 + nothing + catch caught + caught + end + @test nested_err isa ArgumentError + @test occursin( + "unsupported result type String", + sprint(showerror, nested_err), + ) + + compile_err = try + dest .= _broadcast_fusion_bad_kernel.(a) + nothing + catch caught + caught + end + @test compile_err isa ErrorException + @test occursin("GPU broadcast function failed to fuse", sprint(showerror, compile_err)) + end + end + @testset "Debug formatting" begin input_indices = Dict(objectid(a) => 0, objectid(b) => 1) tree = Base.broadcasted(+, Base.broadcasted(*, a, b), s1) @@ -580,8 +654,8 @@ end * Verifies `_BCAST_PTX_CACHE` grows on first fused launch of a signature and * is reused (no new entry) on a second launch of the same signature. * Gated on `FUSE_BROADCAST_EXPRS` and an active GPU target; skips otherwise. - * With `FUSE_BROADCAST_MIN_OPS > 1`, single-op exprs are unfused — tests - * should set min ops to 1 (LocalPreferences / ENV) to exercise the cache. + * With `FUSE_BROADCAST_MIN_OPS > 1`, single native ops are unfused — tests + * should set min ops to 1 (LocalPreferences) to exercise the native-op cache. =# @testset "Broadcast Fusion PTX Cache" begin T=Float32 From 8db57eae57a05ae2898a7f2c8307b83ac9f38b66 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sat, 26 Sep 2026 14:37:57 -0500 Subject: [PATCH 42/49] AoS and SoA Support (#207) * Restack ODE scalar and broadcast support on develop * Simplify NDArray fill! conversion * Add StructArrays extension * test: add structarrays test * StructArrays: hide behind expirmental guard. More testing * AoS Support: native Legate support (#208) * support AoS Ndarray{struct} * AoS printing * fusion: reject kernel values that carry data * AoS: copy struct arrays with the broadcast kernel * AoS: pack struct stores of every rank * AoS: fill and index struct arrays * AoS: compare and reshape struct arrays * AoS: raise errors for unsupported struct operations * test: expand AoS struct storage coverage * StructArrays: test scalar args, skip without fusion * AoS: mark struct fill methods as TODO * AoS: mark struct and fusion limitations as TODO * provide proper guards for CPU / non-fusion execution --------- Co-authored-by: ejmeitz <54505069+ejmeitz@users.noreply.github.com> --- Project.toml | 3 + TODO.md | 8 + benchmark | 2 +- ext/cuNumericStructArraysExt.jl | 64 +++ .../include/ndarray_c_api.h | 6 + lib/cunumeric_jl_wrapper/src/cuda.cpp | 69 ++- lib/cunumeric_jl_wrapper/src/ndarray.cpp | 56 +++ src/ndarray/broadcast.jl | 41 +- src/ndarray/broadcast_fusion.jl | 43 +- src/ndarray/detail/ndarray.jl | 93 +++- src/ndarray/ndarray.jl | 163 ++++++- src/ndarray/promotion.jl | 18 +- src/ndarray/unary.jl | 11 +- test/Project.toml | 2 + test/array/struct_guards.jl | 78 ++++ test/gpu_only/broadcast_fusion.jl | 16 +- test/gpu_only/struct_storage.jl | 421 ++++++++++++++++++ test/gpu_only/structarrays.jl | 99 ++++ test/runtests.jl | 8 +- 19 files changed, 1167 insertions(+), 34 deletions(-) create mode 100644 ext/cuNumericStructArraysExt.jl create mode 100644 test/array/struct_guards.jl create mode 100644 test/gpu_only/struct_storage.jl create mode 100644 test/gpu_only/structarrays.jl diff --git a/Project.toml b/Project.toml index e0342cb1a..0437a7417 100644 --- a/Project.toml +++ b/Project.toml @@ -30,10 +30,12 @@ cupynumeric_jll = "2862d674-414d-5b0b-a494-b21f8deca547" libcxxwrap_julia_jll = "3eaa8342-bff7-56a5-9981-c04077f7cee7" [weakdeps] +StructArrays = "09ab397b-f2b6-538f-b94a-2f83cf4a842a" Krylov = "ba0b0d4f-ebba-5204-a429-3ac8c609bfb7" TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" [extensions] +cuNumericStructArraysExt = "StructArrays" cuNumericKrylovExt = "Krylov" cuNumericTensorOperationsExt = "TensorOperations" @@ -56,6 +58,7 @@ Preferences = "1" Random = "1" StaticArrays = "1" StatsBase = "0.34" +StructArrays = "0.7" TensorOperations = "5.8" cunumeric_jl_wrapper_jll = "26.6.4" cupynumeric_jll = "26.6.0" diff --git a/TODO.md b/TODO.md index 9bbfa7bdf..9248f7253 100644 --- a/TODO.md +++ b/TODO.md @@ -15,6 +15,7 @@ corresponding `Base` methods are missing, so calls fall through to **P0** - `Base.reshape`, `Base.vec` +- 1-D range `setindex!` (`A[2:3] = B`); only 2-D range assignment exists - `fill!` (convertible eltypes) - `collect` @@ -37,6 +38,13 @@ corresponding `Base` methods are missing, so calls fall through to - 2D `permutedims` - `diff` +## Broadcast fusion + +- Fused runtime scalars are promoted to one common type before launch, so + mixing e.g. `UInt64` and `Float32` scalars loses precision. Keeping each + scalar's type would also let struct `fill!` / `setindex!` use the fused + kernel instead of Legate's `issue_fill`. + ## LinearAlgebra Starter list of easy/medium LA gaps. Prefer wiring `LinearAlgebra` entry diff --git a/benchmark b/benchmark index 4fd2fae81..0d1ae3291 160000 --- a/benchmark +++ b/benchmark @@ -1 +1 @@ -Subproject commit 4fd2fae81cc99ce641644e82861b196defab1a91 +Subproject commit 0d1ae32918c8437b9aa39e9fd5321e8a05276fca diff --git a/ext/cuNumericStructArraysExt.jl b/ext/cuNumericStructArraysExt.jl new file mode 100644 index 000000000..8d11fb548 --- /dev/null +++ b/ext/cuNumericStructArraysExt.jl @@ -0,0 +1,64 @@ +module cuNumericStructArraysExt + +using cuNumeric +using StructArrays + +const Broadcasted = Base.Broadcast.Broadcasted +const Extruded = Base.Broadcast.Extruded + +# Project a struct-valued scalar function directly to one field. This keeps the +# temporary struct inside the GPU kernel instead of allocating an NDArray of it. +struct FieldFunction{field,F} + f::F +end +@inline (p::FieldFunction{field})(args...) where {field} = getfield(p.f(args...), field) + +_uses_component(x, components) = any(c -> x === c, components) +_uses_component(x::Extruded, components) = _uses_component(x.x, components) +_uses_component(x::Broadcasted, components) = + any(arg -> _uses_component(arg, components), x.args) + +function Base.copyto!( + dest::StructArray{T}, bc::Broadcasted{<:cuNumeric.NDArrayStyle} +) where {T} + cuNumeric.assert_experimental() + components = Tuple(StructArrays.components(dest)) + all(c -> c isa cuNumeric.NDArray, components) || + throw(ArgumentError("StructArray broadcast requires NDArray field storage")) + axes(dest) == axes(bc) || Base.Broadcast.throwdm(axes(dest), axes(bc)) + isempty(dest) && return dest + # Each field is computed by the fused kernel; there is no unfused fallback. + cuNumeric._struct_kernel_available() || throw( + ArgumentError( + "StructArray broadcast requires GPU broadcast fusion " * + "(fusion enabled: $(cuNumeric.FUSE_BROADCAST_EXPRS), " * + "GPU available: $(cuNumeric._has_gpu_target()))", + ), + ) + + # A field may read another field of dest. Stage results before replacing any + # component so in-place broadcasts retain their usual simultaneous semantics. + aliases_dest = _uses_component(bc, components) + outputs = aliases_dest ? map(similar, components) : components + try + for (name, out) in zip(fieldnames(T), outputs) + projected = Broadcasted{cuNumeric.NDArrayStyle{ndims(dest)}}( + FieldFunction{name,typeof(bc.f)}(bc.f), bc.args, bc.axes + ) + cuNumeric.can_fuse_linear_broadcast(out, projected) || throw( + ArgumentError("StructArray assignment requires same-shaped NDArray inputs") + ) + cuNumeric.fuse_broadcast_tree!(out, projected) + end + if aliases_dest + for (component, output) in zip(components, outputs) + copyto!(component, output) + end + end + finally + aliases_dest && foreach(cuNumeric.destroy!, outputs) + end + return dest +end + +end diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index ab164b667..2e0bf78b2 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -54,6 +54,11 @@ CN_NDArray* nda_zeros_array(int32_t dim, const uint64_t* shape, CN_Type type); // Internal allocation without a fill; every element must be written before use. CN_NDArray* nda_empty_array(int32_t dim, const uint64_t* shape, CN_Type type); +CN_NDArray* nda_empty_struct_array(int32_t dim, const uint64_t* shape, + int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets); +bool nda_struct_layout_matches(int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets); // full(shape, value) // dim : number of dimensions @@ -67,6 +72,7 @@ CN_NDArray* nda_reshape_array(CN_NDArray* arr, int32_t dim, const uint64_t* shape); CN_NDArray* nda_astype(CN_NDArray* arr, CN_Type type); void nda_fill_array(CN_NDArray* arr, CN_Type type, const void* value); +void nda_fill_struct_array(CN_NDArray* arr, const void* value, uint64_t size); void nda_multiply(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out); void nda_add(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out); diff --git a/lib/cunumeric_jl_wrapper/src/cuda.cpp b/lib/cunumeric_jl_wrapper/src/cuda.cpp index b536598c3..c8e71c9ca 100644 --- a/lib/cunumeric_jl_wrapper/src/cuda.cpp +++ b/lib/cunumeric_jl_wrapper/src/cuda.cpp @@ -21,6 +21,7 @@ #include "cuda.h" #include +#include #include #include #include @@ -28,9 +29,9 @@ #include "legate.h" #include "legate/utilities/proc_local_storage.h" #include "legion.h" +#include "ptx.h" #include "types.h" #include "ufi.h" -#include "ptx.h" // #define CUDA_DEBUG #include "cuda_macros.h" // Shared error/debug and dense argument-packing macros. @@ -66,13 +67,13 @@ using FunctionMap = std::unordered_map cufunction_ptr{}; -CUfunction lookup_ptx(const std::string& name, cudaStream_t stream) { +CUfunction lookup_ptx(const std::string &name, cudaStream_t stream) { CUcontext ctx; if (cuStreamGetCtx(stream, &ctx) != CUDA_SUCCESS) throw std::runtime_error("PTX: could not get the task CUDA context"); if (!cufunction_ptr.has_value()) throw std::runtime_error("PTX: no modules loaded on this processor"); - auto& functions = cufunction_ptr.get(); + auto &functions = cufunction_ptr.get(); auto it = functions.find({ctx, name}); if (it == functions.end()) throw std::runtime_error("PTX: missing kernel " + name); @@ -173,6 +174,52 @@ struct ufiStridedFunctor { } }; +template +void pack_struct(char *&p, const legate::PhysicalArray &array, + AccessMode mode) { + const auto elem_size = array.type().size(); + const auto shape = array.shape(); + const auto extents = shape.hi - shape.lo + legate::Point::ONES(); + // Legate permits byte views with a separate logical element size. The mdspan + // mapping reports strides in logical elements, as CUDA.jl's descriptor needs. + auto store = array.data(); + CuStridedDeviceArray desc{}; + if (mode == AccessMode::WRITE) { + auto span = store.span_write_accessor(elem_size); + desc.ptr = span.data_handle(); + for (int i = 0; i < D; ++i) { + desc.strides[i] = span.mapping().stride(i); + } + } else { + auto span = store.span_read_accessor(elem_size); + desc.ptr = const_cast(span.data_handle()); + for (int i = 0; i < D; ++i) { + desc.strides[i] = span.mapping().stride(i); + } + } + desc.maxsize = shape.volume() * elem_size; + for (int i = 0; i < D; ++i) { + desc.dims[i] = extents[i]; + } + desc.length = shape.volume(); + memcpy(p, &desc, sizeof(desc)); + p += sizeof(desc); +} + +struct PackStructDispatch { + template + void operator()(char *&p, const legate::PhysicalArray &array, + AccessMode mode) const { + pack_struct(p, array, mode); + } +}; + +void pack_struct(char *&p, const legate::PhysicalArray &array, + AccessMode mode) { + // Numeric stores reach every rank through double_dispatch; match that here. + legate::dim_dispatch(array.dim(), PackStructDispatch{}, p, array, mode); +} + struct PTXLaunchParams { cudaStream_t stream; CUstream custream; @@ -440,13 +487,21 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, if (val >= 0 && val < static_cast(num_outputs)) { align8(p); auto ps = context.output(val); - legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, - ufi::AccessMode::WRITE, p, ps); + if (ps.type().code() == legate::Type::Code::STRUCT) { + pack_struct(p, ps, ufi::AccessMode::WRITE); + } else { + legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, + ufi::AccessMode::WRITE, p, ps); + } } else if (val >= static_cast(num_outputs)) { align8(p); auto ps = context.input(val - num_outputs); - legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, - ufi::AccessMode::READ, p, ps); + if (ps.type().code() == legate::Type::Code::STRUCT) { + pack_struct(p, ps, ufi::AccessMode::READ); + } else { + legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, + ufi::AccessMode::READ, p, ps); + } } else { std::size_t scalar_idx = static_cast(-(val + 1)); const auto &scalar = context.scalar(scalar_values_start + scalar_idx); diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index ba281473d..28c2416c1 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -68,6 +68,50 @@ CN_NDArray* nda_empty_array(int32_t dim, const uint64_t* shape, CN_Type type) { return new CN_NDArray{runtime->create_array(shp, type.obj)}; } +bool nda_struct_layout_matches(int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets) { + if (fields <= 0) { + return false; + } + std::vector field_types; + field_types.reserve(fields); + for (int32_t i = 0; i < fields; ++i) { + field_types.push_back( + legate::primitive_type(static_cast(codes[i]))); + } + auto type = legate::struct_type(field_types, true); + auto layout = type.offsets(); + if (type.size() != size || layout.size() != static_cast(fields)) { + return false; + } + for (int32_t i = 0; i < fields; ++i) { + if (layout[i] != offsets[i]) { + return false; + } + } + return true; +} + +CN_NDArray* nda_empty_struct_array(int32_t dim, const uint64_t* shape, + int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets) { + if (!nda_struct_layout_matches(fields, codes, size, offsets)) { + throw std::invalid_argument( + "Legate struct layout does not match Julia layout"); + } + std::vector field_types; + field_types.reserve(fields); + for (int32_t i = 0; i < fields; ++i) { + field_types.push_back( + legate::primitive_type(static_cast(codes[i]))); + } + auto type = legate::struct_type(field_types, true); + std::vector shp(shape, shape + dim); + auto store = + legate::Runtime::get_runtime()->create_store(legate::Shape(shp), type); + return new CN_NDArray{cupynumeric::as_array(store)}; +} + CN_NDArray* nda_zeros_array(int32_t dim, const uint64_t* shape, CN_Type type) { std::vector shp(shape, shape + dim); NDArray result = zeros(shp, type.obj); @@ -243,6 +287,18 @@ void nda_contract(CN_NDArray* out, const char* lhs_modes, int32_t n_lhs, out->obj.contract(lhs, rhs1->obj, r1, rhs2->obj, r2, mode2extent); } +// Julia passes the struct's bytes; the store's own type gives their layout. +void nda_fill_struct_array(CN_NDArray* arr, const void* value, uint64_t size) { + auto store = arr->obj.get_store(); + if (store.type().size() != size) { + throw std::invalid_argument( + "struct fill value size does not match the array type"); + } + if (store.volume() == 0) return; + legate::Runtime::get_runtime()->issue_fill( + store, legate::Scalar(store.type(), value, true)); +} + CN_NDArray* nda_copy(CN_NDArray* arr) { NDArray result = arr->obj.copy(); return new CN_NDArray{NDArray(std::move(result))}; diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index daeb02ed9..b77d7c62e 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -43,7 +43,15 @@ end Base.broadcastable(A::NDArray) = A #* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? -Base.similar(arr::NDArray, ::Type{T}, dims::Dims{N}) where {T,N} = cuNumeric.zeros(T, dims) +function _broadcast_allocate(::Type{T}, dims::Dims) where {T} + if _struct_storage_type(T) || + (isbitstype(T) && !isprimitivetype(T) && !(T <: SUPPORTED_TYPES)) + return nda_empty_array(dims, T) + end + return cuNumeric.zeros(T, dims) +end + +Base.similar(arr::NDArray, ::Type{T}, dims::Dims{N}) where {T,N} = _broadcast_allocate(T, dims) Base.similar(arr::NDArray, ::Type{T}, dims::Base.DimOrInd...) where {T} = similar(arr, T, dims) Base.similar(arr::NDArray{T,N}) where {T,N} = similar(arr, T, size(arr)) Base.similar(arr::NDArray{T}, dims::Tuple) where {T} = similar(arr, T, dims) @@ -54,14 +62,14 @@ Base.similar(arr::NDArray, ::Type{T}) where {T} = similar(arr, T, size(arr)) # Prefer Dims over the axes catch-all: with StaticArrays loaded (GPU CI via CUDA), # `similar(::Type{<:AbstractArray}, ::Tuple{})` is otherwise ambiguous between # Base, StaticArrays, and our catch-all (0-d broadcast uses axes `()`). -Base.similar(::Type{NDArray{T}}, dims::Dims{N}) where {T,N} = cuNumeric.zeros(T, dims) +Base.similar(::Type{NDArray{T}}, dims::Dims{N}) where {T,N} = _broadcast_allocate(T, dims) function Base.similar( ::Type{NDArray{T}}, shape::Tuple{Union{Integer,Base.OneTo},Vararg{Union{Integer,Base.OneTo}}}, ) where {T} - return cuNumeric.zeros(T, map(Int, Base.to_shape.(shape))) + return _broadcast_allocate(T, map(Int, Base.to_shape.(shape))) end -Base.similar(::Type{NDArray{T}}, axes) where {T} = cuNumeric.zeros(T, Base.to_shape.(axes)) +Base.similar(::Type{NDArray{T}}, axes) where {T} = _broadcast_allocate(T, Base.to_shape.(axes)) function Base.similar(bc::Broadcasted{NDArrayStyle{N}}, ::Type{ElType}) where {N,ElType} return similar(NDArray{ElType}, axes(bc)) end @@ -281,13 +289,38 @@ end if should_fuse return fuse_broadcast_tree!(dest, bc) else + _assert_struct_broadcast_fused(dest, bc) return _copyto_unfused!(dest, unravel_broadcast_tree(bc)) end else + _assert_struct_broadcast_fused(dest, bc) return _copyto_unfused!(dest, unravel_broadcast_tree(bc)) end end +# The unfused path runs cuPyNumeric operations, none of which read or produce records. +# TODO fuse struct results with size-1 extrusion and into 0-d destinations. +@inline function _assert_struct_broadcast_fused(dest::NDArray{T}, bc::Broadcasted) where {T} + role, S = _struct_storage_type(T) ? ("producing", T) : ("reading", _struct_leaf_type(bc)) + S === nothing && return nothing + throw( + ArgumentError( + "Broadcasts $(role) struct element type $(S) require GPU broadcast " * + "fusion with same-shaped NDArray inputs of rank at least 1 " * + "(fusion enabled: $(FUSE_BROADCAST_EXPRS), GPU available: $(_has_gpu_target()))", + ), + ) +end + +@inline _struct_leaf_type(_) = nothing +@inline _struct_leaf_type(::NDArray{T}) where {T} = _struct_storage_type(T) ? T : nothing +@inline _struct_leaf_type(bc::Broadcasted) = _struct_leaf_type_args(bc.args) +@inline _struct_leaf_type_args(::Tuple{}) = nothing +@inline function _struct_leaf_type_args(args::Tuple) + S = _struct_leaf_type(first(args)) + return S === nothing ? _struct_leaf_type_args(Base.tail(args)) : S +end + # Support .= @inline Base.copyto!(dest::NDArray, bc::Broadcasted{Nothing}) = _copyto!(dest, bc) @inline Base.copyto!(dest::NDArray, bc::Broadcasted{<:NDArrayStyle}) = _copyto!(dest, bc) diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index a46dd213f..0a1b3f458 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -77,6 +77,16 @@ function _push_static_arg!(static_args, arg_plan, x) ), ) + # Static leaves are captured by the kernel closure, which the launcher does + # not pass to the device; only zero-size values (functions, `Val`) are safe. + # TODO pass isbits values such as structs as runtime scalars; the C++ + # launcher would need to align each scalar in the argument buffer. + sizeof(x) == 0 || throw( + ArgumentError( + "Broadcast fusion cannot pass $(repr(x)) of type $(typeof(x)) to the GPU " * + "kernel; pass numbers or NDArrays instead", + ), + ) push!(static_args, x) push!(arg_plan, StaticBroadcastArg{length(static_args)}()) return nothing @@ -343,7 +353,10 @@ end # checks stay in pre-flatten `_assert_fused_broadcast_promotion`. function _align_fused_runtime_args(runtime_args::Tuple) isempty(runtime_args) && return runtime_args - T_IN = __my_promote_type(map(eltype, runtime_args)...) + all(a -> !(a isa Number), runtime_args) && return runtime_args + numeric_types = filter(T -> T <: Number, map(eltype, runtime_args)) + isempty(numeric_types) && return runtime_args + T_IN = __my_promote_type(numeric_types...) return map(a -> unchecked_promote_scalar(a, T_IN), runtime_args) end @@ -469,9 +482,11 @@ function get_cuda_task( get_ptx(obj, DEST_T, ARG_TYPES...) catch err err isa InterruptException && rethrow() - throw(ErrorException( - "GPU broadcast function failed to fuse: $(sprint(showerror, err))" - )) + throw( + ErrorException( + "GPU broadcast function failed to fuse: $(sprint(showerror, err))" + ), + ) end orig_name = extract_kernel_name(ptx) @@ -651,7 +666,10 @@ end else __checked_promote_op(bc.f, eltypes) end - if bc.f === Base.literal_pow + if bc.f isa StructConstructor + # Constructing a record keeps each field's declared type; its numeric + # inputs are not operands of a common arithmetic operation. + elseif bc.f === Base.literal_pow __my_promote_type(eltypes.parameters...) else numeric_types = _numeric_broadcast_types(eltypes) @@ -689,6 +707,19 @@ end return Tuple{T1,rest.parameters...} end +# Like static leaves, the broadcast function is captured by the kernel closure, +# which the launcher does not pass to the device. +# TODO pass the closure's captured state to the kernel so closures can fuse. +@inline function _assert_kernel_function_has_no_data(f) + sizeof(f) == 0 || throw( + ArgumentError( + "Broadcast fusion cannot pass the captured variables of $(typeof(f)) to " * + "the GPU kernel; pass them as broadcast arguments instead", + ), + ) + return nothing +end + function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcast.Broadcasted} # Promotion checks use the pre-flatten tree (same shape as unfused unravel). _assert_fused_broadcast_promotion(dest, bc) @@ -703,6 +734,7 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas bc = Base.Broadcast.preprocess(dest, bc) bc = Base.Broadcast.instantiate(bc) bc = Base.Broadcast.flatten(bc) + _assert_kernel_function_has_no_data(bc.f) # Things like exponentiation generate arguments like Base.RefValue # which do not work with our pattern for making CUDA kernels as they are @@ -916,6 +948,7 @@ end # `runtime_args` and static leaves into shared `static_args`. function _split_segment!(seg_bc, runtime_args, static_args, ndarray_idx) flat = Base.Broadcast.flatten(seg_bc) + _assert_kernel_function_has_no_data(flat.f) plan = Any[] for leaf in flat.args if leaf isa MatRef diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index fafb0bd87..01374aabb 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -41,7 +41,27 @@ end get_n_dim(ptr::NDArray_t) = Int(ccall((:nda_array_dim, libnda), Int32, (NDArray_t,), ptr)) -abstract type AbstractNDArray{T<:SUPPORTED_TYPES,N} <: AbstractArray{T,N} end +abstract type AbstractNDArray{T,N} <: AbstractArray{T,N} end + +@inline _struct_storage_type(::Type{T}) where {T} = + isbitstype(T) && !(T <: SUPPORTED_TYPES) && !isprimitivetype(T) && + fieldcount(T) > 0 && + all(F -> F <: SUPPORTED_ARRAY_TYPES, fieldtypes(T)) + +# cuPyNumeric has no record-typed operations: struct stores are packed, +# unpacked, copied and compared by the fused GPU broadcast kernel. +# TODO CPU variants for struct pack/unpack so host transfer works without a GPU. +@inline _struct_kernel_available() = FUSE_BROADCAST_EXPRS && _has_gpu_target() + +@inline function _assert_struct_kernel(op, ::Type{T}) where {T} + _struct_kernel_available() || throw( + ArgumentError( + "$(op) of NDArrays with struct element type $(T) requires GPU broadcast " * + "fusion (fusion enabled: $(FUSE_BROADCAST_EXPRS), GPU available: $(_has_gpu_target()))", + ), + ) + return nothing +end # Runtime padding uses an abstract field to break the recursive storage definition. abstract type AbstractPaddedStorage{T,N} end @@ -166,6 +186,32 @@ NDArray(value::T) where {T<:SUPPORTED_TYPES} = nda_full_array((), value) # Internal outputs only: callers must overwrite every element before any read. function nda_empty_array(dims::Dims{N}, ::Type{T}) where {T,N} shape = collect(UInt64, dims) + if _struct_storage_type(T) + fields = fieldtypes(T) + codes = Int32[Int32(Legate.code(Legate.to_legate_type(F))) for F in fields] + offsets = UInt32[UInt32(fieldoffset(T, i)) for i in eachindex(fields)] + layout_matches = ccall((:nda_struct_layout_matches, libnda), Bool, + (Int32, Ptr{Int32}, UInt32, Ptr{UInt32}), + Int32(length(fields)), codes, UInt32(sizeof(T)), offsets) + layout_matches || throw( + ArgumentError( + "Legate cannot store $T: its field offsets or size differ from Julia's layout" + ), + ) + ptr = @task_scope "empty_struct" begin + ccall((:nda_empty_struct_array, libnda), NDArray_t, + (Int32, Ptr{UInt64}, Int32, Ptr{Int32}, UInt32, Ptr{UInt32}), + Int32(N), shape, Int32(length(fields)), codes, UInt32(sizeof(T)), offsets) + end + return NDArray(ptr, T, Val(N)) + end + if isbitstype(T) && !isprimitivetype(T) && !(T <: SUPPORTED_TYPES) + throw( + ArgumentError( + "Unsupported isbits element type $T: struct fields must be supported scalar types" + ), + ) + end legate_type = Legate.to_legate_type(T) ptr = @task_scope "empty" begin ccall((:nda_empty_array, libnda), @@ -338,7 +384,47 @@ function nda_fill_array(arr::NDArray{T}, value::T) where {T} return nothing end +# Struct stores are filled from the value's bytes; Legate has their layout. +# TODO fill!, fill, and struct setindex! call Legate's issue_fill directly. Move +# them onto the fused broadcast kernel, as struct copies are, once runtime +# scalar arguments keep their own types instead of promoting to a common one. +function nda_fill_struct_array(arr::NDArray{T}, value::T) where {T} + val = Ref(value) + GC.@preserve val begin + @task_scope "fill!" begin + ccall((:nda_fill_struct_array, libnda), + Cvoid, (NDArray_t, Ptr{Cvoid}, UInt64), + arr.ptr, Base.unsafe_convert(Ptr{T}, val), UInt64(sizeof(T))) + end + end + return nothing +end + +# cuPyNumeric has no record kernels, so struct copies run the fused broadcast +# kernel, which already packs struct stores, slices, and views. +function _nda_assign_struct(arr::NDArray{T}, other::NDArray{T}) where {T} + size(arr) == size(other) || throw( + DimensionMismatch("cannot copy an array of size $(size(other)) into size $(size(arr))") + ) + isempty(arr) && return nothing + _assert_struct_kernel("Copying", T) + if ndims(arr) == 0 + # The fused kernel needs a rank; a 0-d reshape views the same element. + dest, src = nda_reshape_array(arr, (1,)), nda_reshape_array(other, (1,)) + try + _nda_assign_struct(dest, src) + finally + destroy!(dest) + destroy!(src) + end + else + arr .= StructIdentity().(other) + end + return nothing +end + function nda_assign(arr::NDArray{T}, other::NDArray{T}) where {T} + _struct_storage_type(T) && return _nda_assign_struct(arr, other) @task_scope "copyto!" begin ccall((:nda_assign, libnda), Cvoid, (NDArray_t, NDArray_t), @@ -347,6 +433,11 @@ function nda_assign(arr::NDArray{T}, other::NDArray{T}) where {T} end function nda_copy(arr::NDArray{T,N}) where {T,N} + if _struct_storage_type(T) + out = nda_empty_array(size(arr), T) + _nda_assign_struct(out, arr) + return out + end ptr = @task_scope "copy" begin ccall((:nda_copy, libnda), NDArray_t, (NDArray_t,), diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 6b1288f44..9f6149e7e 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -196,6 +196,16 @@ end # AbstractArray): exact `Array{T}` / `Array{T,N}` / `Array` signatures so we win # over `Array{T,N}(::AbstractArray)` (which would scalar-index). Bulk path uses # `_copy_to_julia_array`; 1-d has specialized same-type and converting paths. +struct StructConstructor{T} end +@inline (::StructConstructor{T})(fields...) where {T} = T(fields...) +@inline (::StructConstructor{T})(fields...) where {T<:NamedTuple} = T(fields) + +struct StructField{I} end +@inline (::StructField{I})(value) where {I} = getfield(value, I) + +struct StructIdentity end +@inline (::StructIdentity)(value) = value + function (::Type{Array{T}})(arr::NDArray{S,0}) where {T,S} out = Array{T,0}(undef) allowscalar() do @@ -216,7 +226,37 @@ end # Copy logically into Julia's column-major storage. # Legate may map an NDArray in C or Fortran order. -function _copy_to_julia_array(arr::NDArray{T,N}) where {T,N} +function _copy_to_julia_array(arr::NDArray{T,0}) where {T} + _struct_storage_type(T) || return _copy_to_julia_array_impl(arr) + # Struct fields are projected by the fused kernel, which needs a rank. + vector = nda_reshape_array(arr, (1,)) + try + return Base.reshape(_copy_to_julia_array(vector), ()) + finally + destroy!(vector) + end +end + +_copy_to_julia_array(arr::NDArray) = _copy_to_julia_array_impl(arr) + +function _copy_to_julia_array_impl(arr::NDArray{T,N}) where {T,N} + if _struct_storage_type(T) + out = Array{T}(undef, size(arr)) + isempty(out) && return out + _assert_struct_kernel("Host transfer", T) + fields = ntuple(fieldcount(T)) do i + projected = StructField{i}().(arr) + try + Array(projected) + finally + destroy!(projected) + end + end + for i in eachindex(out) + out[i] = StructConstructor{T}()(ntuple(j -> fields[j][i], fieldcount(T))...) + end + return out + end out = Array{T}(undef, size(arr)) isempty(out) && return out store = Legate.attach_external_col_major(out) @@ -245,11 +285,32 @@ end # Julia Arrays are column-major; Legate stores are row-major. For N>=2 we # materialize a C-ordered buffer via permutedims, attach it with the original # shape, and keep that buffer as `parent` for lifetime. +function _nda_from_julia_struct_array(arr::Array{T,N}) where {T,N} + isempty(arr) && return nda_empty_array(size(arr), T) + _assert_struct_kernel("Construction from a host Array", T) + # `map` keeps Bool fields as Array{Bool}; broadcasting would build a BitArray. + fields = ntuple(i -> NDArray(map(StructField{i}(), arr)), fieldcount(T)) + try + return StructConstructor{T}().(fields...) + finally + foreach(destroy!, fields) + end +end + +# A 0-d store cannot run the fused kernel, but can be filled with its element. +function _nda_from_julia_struct_array(arr::Array{T,0}) where {T} + out = nda_empty_array((), T) + nda_fill_struct_array(out, arr[]) + return out +end + function _nda_from_julia_array(arr::Array{T,0}) where {T} + _struct_storage_type(T) && return _nda_from_julia_struct_array(arr) return cuNumeric.nda_attach_external(arr) end function _nda_from_julia_array(arr::Array{T,1}) where {T} + _struct_storage_type(T) && return _nda_from_julia_struct_array(arr) # Prototype: the attachment borrows Julia memory only for this copy. # Preserve the source through completion, not just task submission. GC.@preserve arr begin @@ -267,6 +328,7 @@ function _nda_from_julia_array(arr::Array{T,1}) where {T} end function _nda_from_julia_array(arr::Array{T,N}) where {T,N} + _struct_storage_type(T) && return _nda_from_julia_struct_array(arr) tmp = collect(permutedims(arr, reverse(ntuple(identity, Val(N))))) return cuNumeric.nda_attach_external(tmp; shape=size(arr)) end @@ -495,6 +557,39 @@ function _setindex!( return write(acc, arr.ptr, to_cpp_index(Int.(idxs)), value) end +# Struct elements have no typed accessor; move one element through a slice. +@inline _struct_element_slice(arr::NDArray{T,N}, idxs::Vararg{Integer,N}) where {T,N} = + nda_get_slice(arr, slice_array(map(_zero_based_index, idxs)...)) + +function Base.getindex(arr::NDArray{T,0}) where {T} + _struct_storage_type(T) || throw(Base.CanonicalIndexError("getindex", typeof(arr))) + assertscalar("getindex") + return _copy_to_julia_array(arr)[] +end + +@inline function Base.getindex(arr::NDArray{T,N}, idxs::Vararg{Integer,N}) where {T,N} + _struct_storage_type(T) || throw(Base.CanonicalIndexError("getindex", typeof(arr))) + @boundscheck checkbounds(arr, idxs...) + assertscalar("getindex") + element = _struct_element_slice(arr, idxs...) + try + return only(_copy_to_julia_array(element)) + finally + destroy!(element) + end +end + +function _setindex!(::Val{N}, arr::NDArray{T,N}, value::T, idxs::Vararg{Integer,N}) where {T,N} + _struct_storage_type(T) || throw(Base.CanonicalIndexError("setindex!", typeof(arr))) + N == 0 && return nda_fill_struct_array(arr, value) + element = _struct_element_slice(arr, idxs...) + try + return nda_fill_struct_array(element, value) + finally + destroy!(element) + end +end + #### START OF SLICING #### # LHS slices from `nda_get_slice` are invisible to `@accelerate`; destroy # the view handle after submitting the assign so they cannot pile up under Julia @@ -684,6 +779,12 @@ end return arr end +function Base.fill!(arr::NDArray{T}, val) where {T} + _struct_storage_type(T) || return invoke(fill!, Tuple{AbstractArray,Any}, arr, val) + nda_fill_struct_array(arr, convert(T, val)) + return arr +end + #### INITIALIZATION OF NDARRAYS #### @doc""" cuNumeric.fill(val::T, dims::Dims) @@ -701,11 +802,16 @@ function fill(val::T, dims::Dims) where {T<:SUPPORTED_TYPES} return nda_full_array(dims, val) end -function fill(val::T, dims::Int...) where {T<:SUPPORTED_TYPES} +function fill(val::T, dims::Dims) where {T} + _struct_storage_type(T) || throw(MethodError(fill, (val, dims))) + return fill!(nda_empty_array(dims, T), val) +end + +function fill(val::T, dims::Int...) where {T} return fill(val, dims) end -function fill(val::T, dim::Int) where {T<:SUPPORTED_TYPES} +function fill(val::T, dim::Int) where {T} return fill(val, (dim,)) end @@ -844,7 +950,8 @@ reshape(arr, (3, 4); copy=Val(true)) # `copy` is a type parameter via Val{C}, so the default path constant-folds # and stays type-stable (needed by solve's 1D-rhs reshape). -function reshape(arr::NDArray, i::Dims{N}; copy::Val{C}=Val(false)) where {N,C} +function reshape(arr::NDArray{T}, i::Dims{N}; copy::Val{C}=Val(false)) where {T,N,C} + _struct_storage_type(T) && return _reshape_struct(arr, i) reshaped = nda_reshape_array(arr, i) if C copied = Base.copy(reshaped) @@ -854,6 +961,32 @@ function reshape(arr::NDArray, i::Dims{N}; copy::Val{C}=Val(false)) where {N,C} return reshaped end +# cuPyNumeric copies non-view reshapes with a typed kernel that rejects records. +# Reshape each numeric field and rebuild, so struct reshapes always copy. +# TODO return a view when the reshape needs no copy, as numeric reshapes do. +function _reshape_struct(arr::NDArray{T}, dims::Dims) where {T} + prod(dims) == length(arr) || throw( + DimensionMismatch( + "new dimensions $(dims) must be consistent with array length $(length(arr))" + ), + ) + isempty(arr) && return nda_empty_array(dims, T) + _assert_struct_kernel("Reshaping", T) + fields = ntuple(fieldcount(T)) do i + projected = StructField{i}().(arr) + try + reshape(projected, dims; copy=Val(true)) + finally + destroy!(projected) + end + end + try + return StructConstructor{T}().(fields...) + finally + foreach(destroy!, fields) + end +end + function reshape(arr::NDArray, i::Int...; copy::Val{C}=Val(false)) where {C} return reshape(arr, i; copy=Val{C}()) end @@ -890,9 +1023,31 @@ a == c """ function Base.:(==)(a::NDArray, b::NDArray) size(a) == size(b) || return cnscalar(NDArray(false)) + if _struct_storage_type(eltype(a)) || _struct_storage_type(eltype(b)) + return _struct_array_equal(a, b) + end return cnscalar(_array_equal_impl(a, b)) end +struct StructEqual end +@inline (::StructEqual)(x, y) = x == y + +# cuPyNumeric cannot compare records. Evaluate the element type's own `==` +# in the fused kernel so user-defined equality and NaN semantics match Base. +function _struct_array_equal(a::NDArray, b::NDArray) + isempty(a) && return cnscalar(NDArray(true)) + _assert_struct_kernel("Comparison", _struct_storage_type(eltype(a)) ? eltype(a) : eltype(b)) + if ndims(a) == 0 + return cnscalar(NDArray(_copy_to_julia_array(a) == _copy_to_julia_array(b))) + end + equal = StructEqual().(a, b) + try + return all(equal) + finally + destroy!(equal) + end +end + function Base.:(!=)(a::NDArray, b::NDArray) return !(a == b) end diff --git a/src/ndarray/promotion.jl b/src/ndarray/promotion.jl index 067f237c9..1c1ac8ad6 100644 --- a/src/ndarray/promotion.jl +++ b/src/ndarray/promotion.jl @@ -47,9 +47,11 @@ end end @inline function _broadcast_result_type(op, ::Type{T}) where {T} - T <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError( - "Broadcast function $(op) cannot produce an NDArray: unsupported result type $(T)" - )) + (T <: SUPPORTED_ARRAY_TYPES || _struct_storage_type(T)) || throw( + ArgumentError( + "Broadcast function $(op) cannot produce an NDArray: unsupported result type $(T)" + ), + ) return T end @@ -73,7 +75,7 @@ end ) where {A,B,C} T = _broadcast_result_type(op, Base.promote_op(op, A, B, C)) S = _smallest_numeric_broadcast_type(Tuple{A,B,C}) - S === nothing || (is_wider_type(T, S) && assertpromotion(op, S, T)) + (S === nothing || !(T <: Number)) || (is_wider_type(T, S) && assertpromotion(op, S, T)) return T end @@ -82,7 +84,7 @@ end ) where {Args<:Tuple{Any,Any,Any,Any,Vararg{Any}}} T = _broadcast_result_type(op, Base.promote_op(op, Args.parameters...)) S = _smallest_numeric_broadcast_type(Args) - S === nothing || (is_wider_type(T, S) && assertpromotion(op, S, T)) + (S === nothing || !(T <: Number)) || (is_wider_type(T, S) && assertpromotion(op, S, T)) return T end @@ -113,20 +115,20 @@ __recip_type(::Type{Bool}) = DEFAULT_FLOAT @inline function __checked_promote_op(op, ::Type{A}) where {A} T = _broadcast_result_type(op, Base.promote_op(op, A)) - is_wider_type(T, A) && assertpromotion(op, A, T) + T <: SUPPORTED_ARRAY_TYPES && is_wider_type(T, A) && assertpromotion(op, A, T) return T end @inline function __checked_promote_op(op, ::Type{A}, ::Type{A}) where {A} T = _broadcast_result_type(op, Base.promote_op(op, A, A)) - is_wider_type(T, A) && assertpromotion(op, A, T) + T <: Number && is_wider_type(T, A) && assertpromotion(op, A, T) return T end @inline function __checked_promote_op(op, ::Type{A}, ::Type{B}) where {A,B} T = _broadcast_result_type(op, Base.promote_op(op, A, B)) S = _smallest_numeric_broadcast_type(Tuple{A,B}) - S === nothing || (is_wider_type(T, S) && assertpromotion(op, S, T)) + (S === nothing || !(T <: Number)) || (is_wider_type(T, S) && assertpromotion(op, S, T)) return T end diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 93f692b4a..da9c62dd9 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -47,7 +47,7 @@ const unary_op_map_no_args = Dict{Function,UnaryOpCode}( ) for julia_fn in (keys(floaty_unary_ops_no_args)..., keys(unary_op_map_no_args)..., - identity, real, imag, conj, inv, !) + identity, real, imag, conj, inv, !) @eval @inline _has_unfused_broadcast(::typeof($julia_fn), ::Val{1}) = true end # Positional rounding modes are rejected by the native path, too. @@ -353,7 +353,15 @@ function _unary_reduction_axes_apply(op_code, input::NDArray, ::Type{U}, axes) w return result end +@inline function _assert_numeric_reduction(base_func, ::Type{T}) where {T} + _struct_storage_type(T) && throw( + ArgumentError("$(base_func) does not support NDArrays of struct element type $(T)") + ) + return nothing +end + function _unary_reduction_impl(base_func, op_code, input::NDArray{T}, ::Colon) where {T} + _assert_numeric_reduction(base_func, T) T_OUT = Base.promote_op(base_func, Vector{T}) is_wider_type(T_OUT, T) && assertpromotion(base_func, T, T_OUT) out = cuNumeric.zeros(T_OUT) @@ -361,6 +369,7 @@ function _unary_reduction_impl(base_func, op_code, input::NDArray{T}, ::Colon) w end function _unary_reduction_impl(base_func, op_code, input::NDArray{T,N}, dims::Integer) where {T,N} + _assert_numeric_reduction(base_func, T) T_OUT = Base.promote_op(base_func, Vector{T}) is_wider_type(T_OUT, T) && assertpromotion(base_func, T, T_OUT) axes = Int32[dims - 1] diff --git a/test/Project.toml b/test/Project.toml index bb08f2cf2..8bff086c3 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -9,12 +9,14 @@ ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" +StructArrays = "09ab397b-f2b6-538f-b94a-2f83cf4a842a" TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" [compat] CUDA = "6.4" +StructArrays = "0.7" [sources] cuNumeric = {path = ".."} diff --git a/test/array/struct_guards.jl b/test/array/struct_guards.jl new file mode 100644 index 000000000..b9e57bef8 --- /dev/null +++ b/test/array/struct_guards.jl @@ -0,0 +1,78 @@ +using StructArrays + +# Struct element storage needs the fused GPU broadcast kernel. Without a GPU or +# with fusion disabled, every operation that touches element data must raise an +# ArgumentError instead of reaching the unfused path or a missing GPU variant. + +struct GuardPair + a::Float32 + b::Int32 +end + +_guard_pair(x) = GuardPair(Float32(x + 1), Int32(2x)) +_guard_pair_b(p::GuardPair) = p.b + +if cuNumeric._struct_kernel_available() + @testset "Struct guards (skipped: struct kernel available)" begin + @test_skip false + end +else + @testset "Struct storage without the fused kernel" begin + template = cuNumeric.zeros(Int64, 6) + value = GuardPair(3.0f0, Int32(4)) + arr = similar(template, GuardPair, (6,)) + other = similar(template, GuardPair, (6,)) + input = NDArray(collect(0:5)) + empty_arr = similar(template, GuardPair, (0,)) + try + # Allocation, fills, scalar writes and views need no kernel. + @test arr isa NDArray{GuardPair,1} + @test fill!(arr, value) === arr + cuNumeric.allowscalar() do + arr[2] = GuardPair(9.0f0, Int32(9)) + end + view = arr[2:4] + @test size(view) == (3,) + cuNumeric.destroy!(view) + + # Empty arrays have no elements to move. + @test isempty(Array(empty_arr)) + @test isempty(Array(NDArray(GuardPair[]))) + + @test_throws ArgumentError NDArray([value, value]) + @test_throws ArgumentError Array(arr) + @test_throws ArgumentError cuNumeric.allowscalar(() -> arr[1]) + @test_throws ArgumentError copy(arr) + @test_throws ArgumentError copyto!(other, arr) + @test_throws ArgumentError cuNumeric.reshape(arr, 2, 3) + @test_throws ArgumentError (arr == other) + @test_throws ArgumentError (arr .= _guard_pair.(input)) + @test_throws ArgumentError _guard_pair_b.(arr) + + err = try + Array(arr) + catch e + e + end + @test occursin("requires GPU broadcast fusion", sprint(showerror, err)) + finally + foreach(cuNumeric.destroy!, (template, arr, other, input, empty_arr)) + end + end + + @testset "StructArray broadcast without the fused kernel" begin + previous_experimental = get(task_local_storage(), :Experimental, false) + input = NDArray(collect(0:2)) + dest = StructArray{GuardPair}((a=cuNumeric.zeros(Float32, 3), b=cuNumeric.zeros(Int32, 3))) + try + cuNumeric.Experimental(true) + @test_throws ArgumentError (dest .= _guard_pair.(input)) + @test all(iszero, Array(dest.a)) + # Field storage is ordinary NDArrays. + @test Array(dest.a .+ 1.0f0) == ones(Float32, 3) + finally + cuNumeric.Experimental(previous_experimental) + foreach(cuNumeric.destroy!, (input, dest.a, dest.b)) + end + end +end diff --git a/test/gpu_only/broadcast_fusion.jl b/test/gpu_only/broadcast_fusion.jl index d64bc786f..70aa96830 100644 --- a/test/gpu_only/broadcast_fusion.jl +++ b/test/gpu_only/broadcast_fusion.jl @@ -31,8 +31,9 @@ _broadcast_fusion_user_add(x, y) = x + y _broadcast_fusion_absnorm(x, t) = abs(x) -_broadcast_fusion_residual(e, u0, u1, atol, rtol, norm, t) = - e / (atol + max(norm(u0, t), norm(u1, t)) * rtol) +function _broadcast_fusion_residual(e, u0, u1, atol, rtol, norm, t) + return e / (atol + max(norm(u0, t), norm(u1, t)) * rtol) +end _broadcast_fusion_bad_result(x) = string(x) _broadcast_fusion_bad_kernel(x) = parse(Float32, string(x)) @@ -648,6 +649,17 @@ end expected[2:(end - 1), 2:(end - 1)] = ja .* s1 .+ s2 @allowscalar @test safe_compare(expected, out, atol, rtol) end + + @testset "kernel values with data are rejected" begin + # The launcher passes no closure state to the device, so these would + # otherwise read garbage kernel parameters and abort the GPU stream. + a = @allowscalar NDArray(rand(T, 8)) + scale = s1 + @test_throws ArgumentError (x -> x * scale).(a) + @test_throws ArgumentError ((x, t) -> x * t[1]).(a, Ref((s1, s2))) + # Values passed as broadcast arguments stay supported. + @allowscalar @test safe_compare(Array(a) .* s1, ((x, c) -> x * c).(a, s1), atol, rtol) + end end #= Broadcast fusion PTX compilation cache. diff --git a/test/gpu_only/struct_storage.jl b/test/gpu_only/struct_storage.jl new file mode 100644 index 000000000..7f7ec0ef9 --- /dev/null +++ b/test/gpu_only/struct_storage.jl @@ -0,0 +1,421 @@ +struct StructStorageTriple{T} + a::T + b::T + c::T +end + +struct StructStorageParticle{T} + position::T + velocity::T +end + +_struct_storage_step(p, dt) = StructStorageParticle( + p.position + dt * p.velocity, p.velocity +) + +function _struct_storage_float(x) + return StructStorageTriple{Float32}( + Float32(x + 1), Float32(x + 2), Float32(x + 3) + ) +end +_struct_storage_uint(x) = StructStorageTriple{UInt64}(UInt64(x + 1), UInt64(x + 2), UInt64(x + 3)) + +struct StructStoragePacked + a::UInt8 + b::Bool + c::Int16 + d::UInt32 +end +_struct_storage_packed(x) = StructStoragePacked(UInt8(x + 1), isodd(x), Int16(x + 3), UInt32(x + 4)) + +struct StructStoragePadded + a::UInt8 + b::Float64 + c::Int16 + d::Float32 +end +function _struct_storage_padded(x) + return StructStoragePadded( + UInt8(x + 1), Float64(x + 2), Int16(x + 3), Float32(x + 4) + ) +end + +struct StructStorageComplex + a::ComplexF32 + b::Float64 + c::UInt8 +end +function _struct_storage_complex(x) + return StructStorageComplex( + ComplexF32(Float32(x + 1), Float32(x + 2)), Float64(x + 3), UInt8(x + 4) + ) +end +_struct_storage_real(x::StructStorageComplex) = real(x.a) +_struct_storage_imag(x::StructStorageComplex) = imag(x.a) + +_struct_storage_named(x) = (a=Float32(x + 1), b=UInt8(x + 2)) + +struct StructStorageNested + a::StructStorageTriple{Float32} + b::Float64 +end +_struct_storage_nested(x) = StructStorageNested(_struct_storage_float(x), Float64(x + 4)) + +struct StructStorageTupleField + a::NTuple{3,Float32} + b::Int32 +end +function _struct_storage_tuple(x) + return StructStorageTupleField( + (Float32(x + 1), Float32(x + 2), Float32(x + 3)), Int32(x + 4) + ) +end + +@inline _struct_storage_field(x, ::Val{I}) where {I} = getfield(x, I) + +function _check_struct_storage(f::F, ::Type{T}, input) where {F,T} + @test isbitstype(T) + @test cuNumeric._struct_storage_type(T) + output = f.(input) + try + @test output isa NDArray{T,1} + for j in 1:fieldcount(T) + # Complex-valued extraction still takes cuPyNumeric's unary path. + fieldtype(T, j) <: Complex && continue + field = _struct_storage_field.(output, Ref(Val(j))) + try + @test Array(field) == [getfield(f(Int64(i)), j) for i in 0:3] + finally + cuNumeric.destroy!(field) + end + end + if T == StructStorageComplex + for (projection, expected) in ( + (_struct_storage_real, Float32[1, 2, 3, 4]), + (_struct_storage_imag, Float32[2, 3, 4, 5]), + ) + field = projection.(output) + try + @test Array(field) == expected + finally + cuNumeric.destroy!(field) + end + end + end + finally + cuNumeric.destroy!(output) + end +end + +@testset "Struct element storage" begin + input = NDArray(Int64[0, 1, 2, 3]) + try + _check_struct_storage(_struct_storage_float, StructStorageTriple{Float32}, input) + _check_struct_storage(_struct_storage_uint, StructStorageTriple{UInt64}, input) + _check_struct_storage(_struct_storage_packed, StructStoragePacked, input) + _check_struct_storage(_struct_storage_padded, StructStoragePadded, input) + _check_struct_storage(_struct_storage_complex, StructStorageComplex, input) + _check_struct_storage(_struct_storage_named, typeof(_struct_storage_named(0)), input) + + @test fieldoffset(StructStoragePadded, 2) > sizeof(UInt8) + @test isbitstype(StructStorageNested) + @test isbitstype(StructStorageTupleField) + @test_throws ArgumentError _struct_storage_nested.(input) + @test_throws ArgumentError _struct_storage_tuple.(input) + @test_throws ArgumentError similar(input, StructStorageNested, size(input)) + + # Built-in complex arrays must keep Legate's numeric complex type. + @test !cuNumeric._struct_storage_type(ComplexF32) + complex_array = cuNumeric.zeros(ComplexF32, 4) + try + @test Array(complex_array) == zeros(ComplexF32, 4) + finally + cuNumeric.destroy!(complex_array) + end + finally + cuNumeric.destroy!(input) + end +end + +@testset "Struct host transfer and in-place broadcast" begin + initial = StructStorageParticle{Float32}[ + StructStorageParticle(0.0f0, 1.0f0), + StructStorageParticle(2.0f0, -0.5f0), + ] + particles = NDArray(initial) + try + @test particles isa NDArray{StructStorageParticle{Float32},1} + @test Array(particles) == initial + @test occursin("StructStorageParticle", sprint(show, MIME"text/plain"(), particles)) + + particles .= _struct_storage_step.(particles, 0.1f0) + @test Array(particles) == StructStorageParticle{Float32}[ + StructStorageParticle(0.1f0, 1.0f0), + StructStorageParticle(1.95f0, -0.5f0), + ] + finally + cuNumeric.destroy!(particles) + end + + mixed = reshape([_struct_storage_padded(i) for i in 0:5], 2, 3) + device_mixed = NDArray(mixed) + try + @test Array(device_mixed) == mixed + finally + cuNumeric.destroy!(device_mixed) + end +end + +@testset "Struct storage round trips across layouts" begin + for make_value in ( + _struct_storage_float, + _struct_storage_packed, + _struct_storage_padded, + _struct_storage_complex, + _struct_storage_named, + ) + T = typeof(make_value(0)) + host = reshape(T[make_value(i) for i in 0:7], 2, 2, 2) + device = NDArray(host) + copied = similar(device) + try + @test size(device) == size(host) + @test Array(device) == host + @test Array{T,3}(device) == host + @test copyto!(copied, device) === copied + @test Array(copied) == host + finally + cuNumeric.destroy!(copied) + cuNumeric.destroy!(device) + end + end +end + +@testset "Preallocated struct storage needs no experimental opt-in" begin + previous_experimental = get(task_local_storage(), :Experimental, false) + input = NDArray(Int64[0, 1, 2, 3]) + output = similar(input, StructStoragePacked, size(input)) + empty_output = similar(input, StructStoragePacked, (0, 2)) + try + cuNumeric.Experimental(false) + @test output isa NDArray{StructStoragePacked,1} + @test (output .= _struct_storage_packed.(input)) === output + @test Array(output) == [_struct_storage_packed(i) for i in 0:3] + @test size(empty_output) == (0, 2) + @test isempty(Array(empty_output)) + finally + cuNumeric.Experimental(previous_experimental) + foreach(cuNumeric.destroy!, (input, output, empty_output)) + end +end + +struct StructStorageRetagged + x::Float32 + y::UInt8 + z::Int16 + w::UInt32 +end +_struct_storage_retag(p::StructStoragePacked) = StructStorageRetagged(p.a, p.b, p.c, p.d) +_struct_storage_bump(p::StructStoragePadded, dx) = StructStoragePadded(p.a, p.b + dx, p.c, p.d) +function _struct_storage_add(p::StructStoragePadded, q::StructStoragePadded) + return StructStoragePadded(p.a + q.a, p.b + q.b, p.c + q.c, p.d + q.d) +end +_struct_storage_padded_b(p::StructStoragePadded) = p.b +# Wrap indices so large arrays stay inside each field's range. +function _struct_storage_host(dims...) + return reshape( + [_struct_storage_padded(i % 200) for i in 0:(prod(dims) - 1)], dims... + ) +end + +@testset "Struct storage across ranks" begin + # Rank 0 cannot use the fused kernel, so transfers use fills and reshapes. + scalar_host = fill(_struct_storage_padded(3)) + scalar = NDArray(scalar_host) + scalar_copy = copy(scalar) + try + @test scalar isa NDArray{StructStoragePadded,0} + @test Array(scalar) == scalar_host + @test Array(scalar_copy) == scalar_host + @test @allowscalar(scalar[]) == scalar_host[] + @allowscalar scalar[] = _struct_storage_padded(9) + @test Array(scalar)[] == _struct_storage_padded(9) + @test Array(scalar_copy) == scalar_host + @test !fetch(scalar == scalar_copy) + finally + foreach(cuNumeric.destroy!, (scalar, scalar_copy)) + end + + # Ranks above three pack struct stores through the generic dimension dispatch. + for dims in ((5,), (3, 4), (2, 3, 4), (2, 1, 3, 2), (1, 2, 1, 2, 3)) + host = _struct_storage_host(dims...) + device = NDArray(host) + bumped = _struct_storage_bump.(device, 0.5) + try + @test Array(device) == host + @test Array(bumped) == _struct_storage_bump.(host, 0.5) + finally + foreach(cuNumeric.destroy!, (device, bumped)) + end + end + + # A multi-block launch covers the grid-stride loop for every thread. + large_host = _struct_storage_host(64, 32, 17) + large = NDArray(large_host) + large_bumped = _struct_storage_bump.(large, 1.0) + try + @test Array(large_bumped) == _struct_storage_bump.(large_host, 1.0) + finally + foreach(cuNumeric.destroy!, (large, large_bumped)) + end +end + +@testset "Struct copies, slices, and views" begin + host = _struct_storage_host(4, 3) + device = NDArray(host) + try + whole = copy(device) + rows = device[2:3, :] + rows_copy = copy(rows) + try + @test Array(whole) == host + @test Array(rows) == host[2:3, :] + @test Array(rows_copy) == host[2:3, :] + @test Array(_struct_storage_bump.(rows, 1.0)) == + _struct_storage_bump.(host[2:3, :], 1.0) + @test Array(permutedims(device)) == permutedims(host) + finally + foreach(cuNumeric.destroy!, (whole, rows, rows_copy)) + end + + # Slice destinations are transformed stores, so they copy with a kernel. + replacement_host = reshape([_struct_storage_padded(100 + i) for i in 1:6], 2, 3) + replacement = NDArray(replacement_host) + try + device[2:3, :] = replacement + host[2:3, :] = replacement_host + @test Array(device) == host + finally + cuNumeric.destroy!(replacement) + end + + column = @view device[:, 2] + column .= _struct_storage_bump.(column, 2.0) + host[:, 2] .= _struct_storage_bump.(host[:, 2], 2.0) + @test Array(device) == host + + mismatched = NDArray(_struct_storage_host(3, 4)) + try + @test_throws DimensionMismatch copyto!(similar(device), mismatched) + finally + cuNumeric.destroy!(mismatched) + end + + # Struct reshapes copy each field, following numeric reshape ordering. + flat = cuNumeric.reshape(device, 3, 4) + flat_b = cuNumeric.reshape(_struct_storage_padded_b.(device), 3, 4) + try + @test size(flat) == (3, 4) + @test Array(_struct_storage_padded_b.(flat)) == Array(flat_b) + @test_throws DimensionMismatch cuNumeric.reshape(device, 5, 2) + finally + foreach(cuNumeric.destroy!, (flat, flat_b)) + end + finally + cuNumeric.destroy!(device) + end +end + +@testset "Struct scalar indexing and fills" begin + host = _struct_storage_host(3, 4) + device = NDArray(host) + filled = cuNumeric.fill(_struct_storage_padded(7), (2, 3)) + empty_filled = cuNumeric.fill(_struct_storage_padded(7), (0, 3)) + try + @test_throws ErrorException device[2, 3] + @test @allowscalar(device[2, 3]) == host[2, 3] + @allowscalar device[3, 4] = _struct_storage_padded(42) + host[3, 4] = _struct_storage_padded(42) + @test Array(device) == host + + @test filled isa NDArray{StructStoragePadded,2} + @test Array(filled) == fill(_struct_storage_padded(7), 2, 3) + @test size(empty_filled) == (0, 3) + @test fill!(device, _struct_storage_padded(1)) === device + @test Array(device) == fill(_struct_storage_padded(1), 3, 4) + finally + foreach(cuNumeric.destroy!, (device, filled, empty_filled)) + end +end + +@testset "Struct equality" begin + host = _struct_storage_host(2, 3) + changed_host = copy(host) + changed_host[2, 2] = _struct_storage_padded(50) + nan_host = [StructStorageTriple{Float32}(NaN32, 1, 2)] + device, same, changed = NDArray(host), NDArray(host), NDArray(changed_host) + nan_device = NDArray(nan_host) + packed = NDArray([_struct_storage_packed(i) for i in 0:3]) + retagged = _struct_storage_retag.(packed) + try + @test fetch(device == same) + @test !fetch(device == changed) + @test fetch(device != changed) + @test !fetch(device == NDArray(_struct_storage_host(3, 2))) + # Records without a custom `==` compare with `===`, as in Base. + @test fetch(nan_device == nan_device) == (nan_host == nan_host) + # Identical layouts share a Legate type but are still different records. + @test retagged isa NDArray{StructStorageRetagged,1} + @test Array(retagged) == _struct_storage_retag.(Array(packed)) + @test fetch(packed == retagged) == (Array(packed) == Array(retagged)) + finally + foreach(cuNumeric.destroy!, (device, same, changed, nan_device, packed, retagged)) + end +end + +@testset "Struct broadcasts" begin + host = _struct_storage_host(3, 2) + device = NDArray(host) + other = NDArray(reverse(host)) + empty_input = NDArray(Float32[]) + try + sums = _struct_storage_add.(device, other) + projected = _struct_storage_padded_b.(device) + try + @test Array(sums) == _struct_storage_add.(host, reverse(host)) + @test projected isa NDArray{Float64,2} + @test Array(projected) == _struct_storage_padded_b.(host) + finally + foreach(cuNumeric.destroy!, (sums, projected)) + end + empty_output = _struct_storage_float.(empty_input) + @test empty_output isa NDArray{StructStorageTriple{Float32},1} + @test isempty(Array(empty_output)) + finally + foreach(cuNumeric.destroy!, (device, other, empty_input)) + end +end + +@testset "Unsupported struct operations raise errors" begin + host = _struct_storage_host(2, 3) + device = NDArray(host) + column = NDArray(reshape(Float64[1, 2], 2, 1)) + scalar = NDArray(fill(_struct_storage_padded(1))) + try + # Struct values cannot reach the kernel as broadcast scalars. + @test_throws ArgumentError _struct_storage_add.(device, Ref(_struct_storage_padded(1))) + # Struct results need the fused path: no size-1 extrusion and no rank 0. + @test_throws ArgumentError _struct_storage_bump.(device, column) + @test_throws ArgumentError _struct_storage_bump.(scalar, 1.0) + shift = 1.0 + @test_throws ArgumentError (p -> _struct_storage_bump(p, shift)).(device) + for reduction in (sum, prod, maximum, minimum) + @test_throws ArgumentError reduction(device) + end + @test_throws ArgumentError sum(device; dims=1) + # The host data survives each rejected operation. + @test Array(device) == host + finally + foreach(cuNumeric.destroy!, (device, column, scalar)) + end +end diff --git a/test/gpu_only/structarrays.jl b/test/gpu_only/structarrays.jl new file mode 100644 index 000000000..7bdc6f9a1 --- /dev/null +++ b/test/gpu_only/structarrays.jl @@ -0,0 +1,99 @@ +using StructArrays + +struct BroadcastPair + a::Float64 + b::Float64 +end + +struct MixedFields + a::Float32 + b::Int32 + c::Bool +end + +broadcast_pair(x) = BroadcastPair(x + 1.0, x + 2.0) +scaled_pair(x, s) = BroadcastPair(x * s, x + s) +swap_pair(a, b) = BroadcastPair(b + 10.0, a + 20.0) +mixed_fields(x) = MixedFields(x + 1.0f0, Int32(2x), x >= 2.0f0) + +@testset "StructArray broadcast (experimental)" begin + previous_experimental = get(task_local_storage(), :Experimental, false) + input = NDArray(reshape(Int64[0, 1, 2], 3, 1)) + dest = StructArray{BroadcastPair}(( + a=cuNumeric.zeros(Float64, 3, 1), + b=cuNumeric.zeros(Float64, 3, 1), + )) + mixed_input = NDArray(reshape(Float32[0, 1, 2, 3], 2, 2)) + mixed_dest = StructArray{MixedFields}(( + a=cuNumeric.zeros(Float32, 2, 2), + b=cuNumeric.zeros(Int32, 2, 2), + c=cuNumeric.zeros(Bool, 2, 2), + )) + wrong_shape = NDArray(reshape(Int64[0, 1, 2], 1, 3)) + row = NDArray(reshape(Float64[1, 2], 1, 2)) + wide_dest = StructArray{BroadcastPair}(( + a=cuNumeric.zeros(Float64, 3, 2), + b=cuNumeric.zeros(Float64, 3, 2), + )) + empty_input = cuNumeric.zeros(Int64, 0) + empty_dest = StructArray{BroadcastPair}(( + a=cuNumeric.zeros(Float64, 0), + b=cuNumeric.zeros(Float64, 0), + )) + host_dest = StructArray{BroadcastPair}((a=zeros(3, 1), b=zeros(3, 1))) + + try + @test Base.get_extension(cuNumeric, :cuNumericStructArraysExt) !== nothing + cuNumeric.Experimental(false) + disabled_error = try + dest .= broadcast_pair.(input) + nothing + catch err + err + end + @test disabled_error isa ArgumentError + @test occursin("Experimental features are disabled", sprint(showerror, disabled_error)) + @test_throws ArgumentError (empty_dest .= broadcast_pair.(empty_input)) + @test all(iszero, Array(dest.a)) + @test all(iszero, Array(dest.b)) + + cuNumeric.Experimental(true) + @test (dest .= broadcast_pair.(input)) === dest + @test vec(Array(dest.a)) == [1.0, 2.0, 3.0] + @test vec(Array(dest.b)) == [2.0, 3.0, 4.0] + + # Both outputs must read the old values of both destination fields. + dest .= swap_pair.(dest.a, dest.b) + @test vec(Array(dest.a)) == [12.0, 13.0, 14.0] + @test vec(Array(dest.b)) == [21.0, 22.0, 23.0] + + # Different field types and a nested broadcast share one struct result. + mixed_dest .= mixed_fields.(mixed_input .+ 1.0f0) + @test Array(mixed_dest.a) == reshape(Float32[2, 3, 4, 5], 2, 2) + @test Array(mixed_dest.b) == reshape(Int32[2, 4, 6, 8], 2, 2) + @test Array(mixed_dest.c) == reshape(Bool[false, true, true, true], 2, 2) + + # Numeric scalars stay runtime kernel arguments for every field. + dest .= scaled_pair.(input, 3.0) + @test vec(Array(dest.a)) == [0.0, 3.0, 6.0] + @test vec(Array(dest.b)) == [3.0, 4.0, 5.0] + dest .= scaled_pair.(input, 0.5) + @test vec(Array(dest.a)) == [0.0, 0.5, 1.0] + + # Field projections use the linear kernel, which cannot extrude size-1 axes. + @test_throws ArgumentError (wide_dest .= scaled_pair.(input, row)) + @test all(iszero, Array(wide_dest.a)) + + @test_throws DimensionMismatch (dest .= broadcast_pair.(wrong_shape)) + @test_throws ArgumentError (host_dest .= broadcast_pair.(input)) + @test (empty_dest .= broadcast_pair.(empty_input)) === empty_dest + cuNumeric.Experimental(false) + @test_throws ArgumentError (dest .= broadcast_pair.(input)) + finally + cuNumeric.Experimental(previous_experimental) + for arr in (dest, mixed_dest, empty_dest, wide_dest) + foreach(cuNumeric.destroy!, Tuple(StructArrays.components(arr))) + end + foreach(cuNumeric.destroy!, (input, mixed_input, wrong_shape, empty_input, row)) + end +end diff --git a/test/runtests.jl b/test/runtests.jl index 6d88ba144..997451dc4 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -51,7 +51,13 @@ if filter_tests!(testsuite, test_args) if !run_gpu_tests || !cuNumeric.FUSE_BROADCAST_EXPRS @warn "Broadcast fusion is disabled, skipping fusion tests" - filter!(test -> !startswith(first(test), "gpu_only/broadcast_fusion"), testsuite) + filter!( + test -> + !startswith(first(test), "gpu_only/broadcast_fusion") && + !startswith(first(test), "gpu_only/struct_storage") && + !startswith(first(test), "gpu_only/structarrays"), + testsuite, + ) end end From 6bec8e6a69ad187e67496b26fb4319d709c813f5 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sat, 26 Sep 2026 16:23:01 -0500 Subject: [PATCH 43/49] Inplace slice assignment (#210) * inplace slice assignment * fix: in-place broadcast fusion launch * update comments * out can be dest itself when the eltype matches and no input partially overlaps it (#214) --- .../include/ndarray_c_api.h | 1 + lib/cunumeric_jl_wrapper/src/ndarray.cpp | 4 ++ src/ndarray/broadcast.jl | 40 +++++++++++++++++-- src/ndarray/broadcast_fusion.jl | 37 ++++++++++------- src/ndarray/detail/ndarray.jl | 5 +++ src/ndarray/unary.jl | 8 ++-- test/gpu_only/broadcast_fusion.jl | 32 +++++++++++++++ 7 files changed, 104 insertions(+), 23 deletions(-) diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index 2e0bf78b2..290822970 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -117,6 +117,7 @@ CN_NDArray* nda_unary_reduction_axes(CuPyNumericUnaryRedCode op_code, int32_t num_axes, bool keepdims); CN_NDArray* nda_get_slice(CN_NDArray* arr, const CN_Slice* slices, int32_t ndim); +bool nda_overlaps(CN_NDArray* lhs, CN_NDArray* rhs); CN_NDArray* nda_attach_external(const void* ptr, size_t size, int dim, const uint64_t* shape, CN_Type type); diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index 28c2416c1..f5d02bd60 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -308,6 +308,10 @@ void nda_assign(CN_NDArray* arr, CN_NDArray* other) { arr->obj.assign(other->obj); } +bool nda_overlaps(CN_NDArray* lhs, CN_NDArray* rhs) { + return lhs->obj.get_store().overlaps(rhs->obj.get_store()); +} + void nda_move(CN_NDArray* dst, CN_NDArray* src) { dst->obj.operator=(std::move(src->obj)); } diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index b77d7c62e..99989318a 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -176,8 +176,9 @@ end return nothing end -# Un-fused implementation of broadcast tree -function unravel_broadcast_tree(bc::Broadcasted) +# Un-fused implementation of broadcast tree. `dest`, when given, receives the +# top-level result directly if its eltype matches and no input partially overlaps it. +function unravel_broadcast_tree(bc::Broadcasted, dest=nothing) if length(bc.args) > 2 && _is_flattened_associative(bc.f) return _unravel_flattened_associative(bc.f, bc.args) end @@ -194,8 +195,7 @@ function unravel_broadcast_tree(bc::Broadcasted) T_IN = __my_promote_type(eltypes.parameters...) # type input arrays are promoted to in_args = unchecked_promote_arr.(materialized_args, T_IN) - # Allocate output array of proper size/type - out = similar(NDArray{T_OUT}, axes(bc)) + out = _unfused_output(dest, T_OUT, bc, in_args) # If the operation, "bc.f", is supported by cuNumeric, this # dispatches to a function calling the C-API. @@ -209,6 +209,23 @@ function unravel_broadcast_tree(bc::Broadcasted) return result end +@inline _unfused_output(::Nothing, ::Type{T}, bc, in_args) where {T} = + similar(NDArray{T}, axes(bc)) + +@inline function _unfused_output(dest::NDArray{S}, ::Type{T}, bc, in_args) where {S,T} + # Elementwise ops may read and write the same array; only partial overlap + # (e.g. shifted views of one store) needs a temporary. + S === T && !any(x -> x isa NDArray && x !== dest && nda_overlaps(dest, x), in_args) && + return dest + return similar(NDArray{T}, axes(bc)) +end + +# Skips the temporary and store-back when the top-level op wrote into `dest`. +@inline function _unfused_into!(dest::NDArray, bc::Broadcasted) + result = unravel_broadcast_tree(bc, dest) + return result === dest ? dest : _copyto_unfused!(dest, result) +end + # Slice destinations must assign into their parent store. @inline function _store_broadcast_result!( dest::NDArray{T}, temp_result::NDArray{T} @@ -269,6 +286,12 @@ end end end +@inline function _identity_broadcast_source(bc::Broadcasted) + bc.f === identity && length(bc.args) == 1 || return nothing + source = only(bc.args) + return source isa NDArray ? source : nothing +end + @inline function _copyto!(dest::NDArray, bc::Broadcasted) axes(dest) == axes(bc) || Broadcast.throwdm(axes(dest), axes(bc)) isempty(dest) && return dest @@ -280,6 +303,15 @@ end ) end + # A same-type identity broadcast is an array assignment. Use the native + # path only for disjoint stores; overlapping slices need the broadcast + # temporary to preserve the original values. + source = _identity_broadcast_source(bc) + if source isa NDArray && eltype(dest) === eltype(source) && + axes(dest) == axes(source) && !nda_overlaps(dest, source) + return copyto!(dest, source) + end + # Require an active GPU target so `--gpus 0` stays on the unfused path. # Fusion requires same-shaped NDArray leaves; otherwise fall back. # Single native ops below `FUSE_BROADCAST_MIN_OPS` use the unfused C API. diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 0a1b3f458..164012f92 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -787,6 +787,9 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas input_ndarrays = tuple(unique_ndarrays...) + # Legion forbids overlapping input and output regions in one task. + output = any(nda -> nda_overlaps(dest, nda), unique_ndarrays) ? similar(dest) : dest + if BCAST_FUSION_DEBUG[] tree_str = _bcast_runtime_tree_str( bc_scope, ndarray_to_input_idx, actual_scalars @@ -803,23 +806,27 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas ) end - @task_scope _bcast_scope_name(bc_scope, ndarray_to_input_idx, actual_scalars) begin - # `blocks=1` is a placeholder; RunPTXBroadcastTask overwrites grid dims - # from the local PhysicalArray. `threads` is only the occupancy budget (tx). - # Scalars after ctx: num_kernel_args, arg_map... - launch( - fkm.cuda_task, - input_ndarrays, - (dest,), - (Int32(length(arg_map)), arg_map..., actual_scalars...); - blocks=1, - threads=fkm.threads, - taskid=cuNumeric.RUN_PTX_BROADCAST, - ctx=fkm.ctx, - ) + try + @task_scope _bcast_scope_name(bc_scope, ndarray_to_input_idx, actual_scalars) begin + # `blocks=1` is a placeholder; RunPTXBroadcastTask overwrites grid dims + # from the local PhysicalArray. `threads` is only the occupancy budget (tx). + # Scalars after ctx: num_kernel_args, arg_map... + launch( + fkm.cuda_task, + input_ndarrays, + (output,), + (Int32(length(arg_map)), arg_map..., actual_scalars...); + blocks=1, + threads=fkm.threads, + taskid=cuNumeric.RUN_PTX_BROADCAST, + ctx=fkm.ctx, + ) + end + output === dest || nda_assign(dest, output) + finally + output === dest || destroy!(output) end - # Fused kernel already wrote `dest` in place; promotion was checked pre-launch. return dest end diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 01374aabb..46e6be468 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -332,6 +332,11 @@ function nda_get_slice(arr::NDArray{T,N}, slices::Vector{Slice}) where {T,N} return NDArray(ptr, T, Val(N), arr) end +@inline nda_overlaps(lhs::NDArray, rhs::NDArray) = + ccall( + (:nda_overlaps, libnda), Cuchar, (NDArray_t, NDArray_t), lhs.ptr, rhs.ptr + ) != 0 + # queries nda_array_dim(arr::NDArray) = ccall((:nda_array_dim, libnda), Int32, (NDArray_t,), arr.ptr) diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index da9c62dd9..4469b383a 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -166,7 +166,7 @@ end @inline function __broadcast( ::typeof(Base.literal_pow), out::NDArray{O}, _, input::NDArray{O}, ::Type{Val{-1}} ) where {O} - nda_move(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED return out end @@ -174,19 +174,19 @@ end ::typeof(Base.literal_pow), out::NDArray{O}, _, input::NDArray, ::Type{Val{-1}} ) where {O} promoted = checked_promote_arr(input, O) # always a new array when eltype ≠ O - nda_move(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED destroy!(promoted) return out end @inline function __broadcast(::typeof(Base.inv), out::NDArray{O}, input::NDArray{O}) where {O} - nda_move(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED return out end @inline function __broadcast(::typeof(Base.inv), out::NDArray{O}, input::NDArray) where {O} promoted = checked_promote_arr(input, O) # always a new array when eltype ≠ O - nda_move(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED destroy!(promoted) return out end diff --git a/test/gpu_only/broadcast_fusion.jl b/test/gpu_only/broadcast_fusion.jl index 70aa96830..1808be97b 100644 --- a/test/gpu_only/broadcast_fusion.jl +++ b/test/gpu_only/broadcast_fusion.jl @@ -52,6 +52,26 @@ _broadcast_fusion_bad_kernel(x) = parse(Float32, string(x)) b = @allowscalar NDArray(julia_b) c = @allowscalar NDArray(julia_c) + @testset "identity broadcast between slices" begin + values = NDArray(Float64.(1:8)) + dst = values[1:2] + src = values[7:8] + @test !cuNumeric.nda_overlaps(dst, src) + dst .= src + @test Array(values) == Float64[7, 8, 3, 4, 5, 6, 7, 8] + cuNumeric.destroy!(dst) + cuNumeric.destroy!(src) + + values = NDArray(Float64.(1:8)) + dst = values[2:4] + src = values[1:3] + @test cuNumeric.nda_overlaps(dst, src) + dst .= src + @test Array(values) == Float64[1, 1, 2, 3, 5, 6, 7, 8] + cuNumeric.destroy!(dst) + cuNumeric.destroy!(src) + end + s1 = T(2.5) s2 = T(1.0) s3 = T(0.5) @@ -634,6 +654,18 @@ end b2d = @allowscalar NDArray(j2b) a2d .= a2d .* s1 .+ b2d @allowscalar @test safe_compare(j2a .* s1 .+ j2b, a2d, atol, rtol) + + original = T.(1:10) + parent = @allowscalar NDArray(copy(original)) + dst = parent[2:9] + src = parent[1:8] + @test cuNumeric.nda_overlaps(dst, src) + dst .= src .* s1 .+ s2 + expected = copy(original) + expected[2:9] .= original[1:8] .* s1 .+ s2 + @allowscalar @test safe_compare(expected, parent, atol, rtol) + cuNumeric.destroy!(dst) + cuNumeric.destroy!(src) end @testset "cross-statement fusion into a slice" begin From 999d634b430cc8a29349edfe3431be0612ec6498 Mon Sep 17 00:00:00 2001 From: Ethan Meitz <54505069+ejmeitz@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:22:43 -0400 Subject: [PATCH 44/49] Undef Initialization (#213) * Add undef NDArray initialization and avoid redundant zero fills * Use uninitialized buffers for contraction outputs * Cover undef allocation edge cases in tests * Test undef construction across supported types and ranks * Restrict undef allocation to supported array types --- docs/src/api_initialization.md | 12 ++++++++ src/ndarray/binary.jl | 7 +++-- src/ndarray/broadcast.jl | 4 +-- src/ndarray/contract.jl | 2 +- src/ndarray/detail/linalg.jl | 18 +++++------ src/ndarray/diagonal.jl | 2 +- src/ndarray/ndarray.jl | 11 +++++++ src/ndarray/random/generator.jl | 16 +++++----- src/ndarray/unary.jl | 8 ++--- src/ndarray/vector_linalg.jl | 2 +- test/array/contract.jl | 13 ++++++-- test/array/initialization.jl | 54 +++++++++++++++++++++++++++++++++ 12 files changed, 116 insertions(+), 33 deletions(-) create mode 100644 test/array/initialization.jl diff --git a/docs/src/api_initialization.md b/docs/src/api_initialization.md index fdca35848..96d62c179 100644 --- a/docs/src/api_initialization.md +++ b/docs/src/api_initialization.md @@ -4,6 +4,18 @@ Constructors for new `NDArray`s. Default floating-point type is `Float32`. ## Basic Initialization +### Uninitialized arrays + +Use `NDArray{T}(undef, dims...)` or `NDArray{T}(undef, dims::Tuple)` when the +next operation writes every element. `similar(A)` and `similar(A, T, dims)` +also return uninitialized arrays. Assign all elements before reading them; +use `cuNumeric.zeros` when the initial zero values are needed. + +```julia +A = NDArray{Float32}(undef, 2, 3) +fill!(A, 1f0) +``` + ### zeros ```@docs diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index bc6da8774..2ad29d92d 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -134,7 +134,7 @@ end function Base.:(-)(rhs1::NDArray{A,N}, rhs2::NDArray{B,N}) where {A,B,N} promote_shape(size(rhs1), size(rhs2)) T_OUT = __checked_promote_op(-, A, B) - out = cuNumeric.zeros(T_OUT, size(rhs1)) + out = NDArray{T_OUT}(undef, size(rhs1)) return _nda_binary_op_promoted!(out, cuNumeric.SUBTRACT, rhs1, rhs2) end @@ -142,7 +142,7 @@ end function Base.:(+)(rhs1::NDArray{A,N}, rhs2::NDArray{B,N}) where {A,B,N} promote_shape(size(rhs1), size(rhs2)) T_OUT = __checked_promote_op(+, A, B) - out = cuNumeric.zeros(T_OUT, size(rhs1)) + out = NDArray{T_OUT}(undef, size(rhs1)) return _nda_binary_op_promoted!(out, cuNumeric.ADD, rhs1, rhs2) end @@ -183,7 +183,8 @@ function Base.:(*)(rhs1::NDArray{A,2}, rhs2::NDArray{B,2}) where {A,B} size(rhs1, 2) == size(rhs2, 1) || throw(DimensionMismatch("Matrix dimensions incompatible: $(size(rhs1)) × $(size(rhs2))")) T = __my_promote_type(A, B) - out = cuNumeric.zeros(T, (size(rhs1, 1), size(rhs2, 2))) + dims = (size(rhs1, 1), size(rhs2, 2)) + out = size(rhs1, 2) == 0 ? cuNumeric.zeros(T, dims) : NDArray{T}(undef, dims) return _nda_three_dot_promoted!(rhs1, rhs2, out) end diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 99989318a..8bf5029ca 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -42,13 +42,12 @@ end Base.broadcastable(A::NDArray) = A -#* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? function _broadcast_allocate(::Type{T}, dims::Dims) where {T} if _struct_storage_type(T) || (isbitstype(T) && !isprimitivetype(T) && !(T <: SUPPORTED_TYPES)) return nda_empty_array(dims, T) end - return cuNumeric.zeros(T, dims) + return NDArray{T}(undef, dims) end Base.similar(arr::NDArray, ::Type{T}, dims::Dims{N}) where {T,N} = _broadcast_allocate(T, dims) @@ -58,7 +57,6 @@ Base.similar(arr::NDArray{T}, dims::Tuple) where {T} = similar(arr, T, dims) Base.similar(arr::NDArray{T}, dims::Base.DimOrInd...) where {T} = similar(arr, T, dims) Base.similar(arr::NDArray, ::Type{T}) where {T} = similar(arr, T, size(arr)) -#* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? # Prefer Dims over the axes catch-all: with StaticArrays loaded (GPU CI via CUDA), # `similar(::Type{<:AbstractArray}, ::Tuple{})` is otherwise ambiguous between # Base, StaticArrays, and our catch-all (0-d broadcast uses axes `()`). diff --git a/src/ndarray/contract.jl b/src/ndarray/contract.jl index 374277b1d..8897cd335 100644 --- a/src/ndarray/contract.jl +++ b/src/ndarray/contract.jl @@ -392,7 +392,7 @@ function contract( Am = _require_modes(A, Amodes) Bm = _require_modes(B, Bmodes) Cm, cshape = _free_output_modes(A, Am, B, Bm) - C = cuNumeric.zeros(T, cshape) + C = NDArray{T}(undef, cshape) return contract!(C, Cm, A, Am, B, Bm; α=α, β=zero(T)) end diff --git a/src/ndarray/detail/linalg.jl b/src/ndarray/detail/linalg.jl index b74e09d80..aa65b9d34 100644 --- a/src/ndarray/detail/linalg.jl +++ b/src/ndarray/detail/linalg.jl @@ -114,7 +114,7 @@ function _solve(a::NDArray{T,N}, b::NDArray{S,N}) where {T,S,N} ) size(a)[1:(end - 2)] == size(b)[1:(end - 2)] || throw(ArgumentError("Batched matrices must have matching batch dimensions")) - x = cuNumeric.zeros(T, size(b)...) + x = NDArray{T}(undef, size(b)) isempty(x) && return x _solve!(_linalg_backend(Val(:solve), a), x, a, b) return x @@ -246,9 +246,9 @@ function _svd(a::NDArray{T,2}, full_matrices::Bool) where {T} k = min(m, n) S = real(T) - u_buf = full_matrices ? cuNumeric.zeros(T, m, m) : cuNumeric.zeros(T, m, k) - s = cuNumeric.zeros(S, k) - vh_buf = full_matrices ? cuNumeric.zeros(T, n, n) : cuNumeric.zeros(T, k, n) + u_buf = NDArray{T}(undef, m, full_matrices ? m : k) + s = NDArray{S}(undef, k) + vh_buf = NDArray{T}(undef, full_matrices ? n : k, n) svd_single(a, u_buf, s, vh_buf) return u_buf, s, vh_buf end @@ -283,8 +283,8 @@ function _qr(a::NDArray{T,2}) where {T} k = min(m, n) # CQR writes dense column-major economy factors with leading dimensions # m for Q and k for R. Square buffers give R the wrong stride when m < n. - q = cuNumeric.zeros(T, m, k) - r = cuNumeric.zeros(T, k, n) + q = NDArray{T}(undef, m, k) + r = NDArray{T}(undef, k, n) k == 0 && return q, r _qr!(_linalg_backend(Val(:qr), a), q, r, a) return q, r @@ -336,7 +336,7 @@ assumed Hermitian without being checked, matching cupynumeric. """ function _cholesky(a::NDArray{T,N}) where {T,N} _check_square_matrices(:cholesky, a) - out = cuNumeric.zeros(T, size(a)...) + out = NDArray{T}(undef, size(a)) _cholesky!(_linalg_backend(Val(:cholesky), a), out, a) return out end @@ -349,7 +349,7 @@ are always complex. """ function _eig(a::NDArray{T,N}) where {T,N} ew = _alloc_eigenvalues(a) - ev = cuNumeric.zeros(_eig_complex_eltype(T), size(a)...) + ev = NDArray{_eig_complex_eltype(T)}(undef, size(a)) geev!(a, ew, ev) return ew, ev end @@ -369,7 +369,7 @@ end function _alloc_eigenvalues(a::NDArray{T,N}) where {T,N} _check_square_matrices(:eigen, a) _assert_geev_available() - return cuNumeric.zeros(_eig_complex_eltype(T), size(a)[1:(end - 1)]...) + return NDArray{_eig_complex_eltype(T)}(undef, size(a)[1:(end - 1)]) end function _assert_geev_available() diff --git a/src/ndarray/diagonal.jl b/src/ndarray/diagonal.jl index 27a06b47a..fb06ef06d 100644 --- a/src/ndarray/diagonal.jl +++ b/src/ndarray/diagonal.jl @@ -342,7 +342,7 @@ _uniformscale_mul(::Type{T}, λ::NDArray{<:Any,0}, a::NDArray) where {T} = a .* _coefficient_as(T, λ) function NDArray{T}(J::LinearAlgebra.UniformScaling, dims::Dims{2}) where {T} - A = zeros(T, dims) + A = NDArray{T}(undef, dims) copyto!(A, J) return A end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 9f6149e7e..3b87618fa 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -786,6 +786,17 @@ function Base.fill!(arr::NDArray{T}, val) where {T} end #### INITIALIZATION OF NDARRAYS #### +@doc""" + NDArray{T}(undef, dims::Int...) + NDArray{T}(undef, dims::Dims) + +Allocate an uninitialized `NDArray{T}`. Every element must be assigned before it is read. +""" +NDArray{T}(::UndefInitializer, dims::Dims{N}) where {T<:SUPPORTED_ARRAY_TYPES,N} = + nda_empty_array(dims, T) +NDArray{T}(::UndefInitializer, dims::Int...) where {T<:SUPPORTED_ARRAY_TYPES} = + NDArray{T}(undef, dims) + @doc""" cuNumeric.fill(val::T, dims::Dims) cuNumeric.fill(val::T, dims::Int...) diff --git a/src/ndarray/random/generator.jl b/src/ndarray/random/generator.jl index af489568a..3b91081b5 100644 --- a/src/ndarray/random/generator.jl +++ b/src/ndarray/random/generator.jl @@ -134,13 +134,13 @@ function random!(g::Generator, arr::NDArray{T}) where {T} end function random(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} - arr = zeros(T, dims) + arr = NDArray{T}(undef, dims) random!(g, arr) return arr end function random(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} - arr = zeros(T, dims) + arr = NDArray{T}(undef, dims) random!(g, arr) return arr end @@ -177,8 +177,8 @@ end # No loc/scale kwargs: shift or scale in user code (`μ .+ σ .* Z`). function randn!(g::Generator, arr::NDArray{Complex{T}}) where {T<:SUPPORTED_FLOAT_TYPES} s = 1 / sqrt(T(2)) - re = zeros(T, size(arr)) - imag_part = zeros(T, size(arr)) + re = NDArray{T}(undef, size(arr)) + imag_part = NDArray{T}(undef, size(arr)) _randn!(g, re; scale=s) _randn!(g, imag_part; scale=s) _pack_complex!(arr, re, imag_part) @@ -194,13 +194,13 @@ function randn!(g::Generator, arr::NDArray{T}) where {T} end function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} - arr = zeros(T, dims) + arr = NDArray{T}(undef, dims) randn!(g, arr) return arr end function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} - arr = zeros(T, dims) + arr = NDArray{T}(undef, dims) randn!(g, arr) return arr end @@ -232,7 +232,7 @@ end function randexp( g::Generator, ::Type{T}, dims::Dims; scale::Union{Real,DeviceScalar{<:Real}}=1 ) where {T<:SUPPORTED_FLOAT_TYPES} - arr = zeros(T, dims) + arr = NDArray{T}(undef, dims) randexp!(g, arr; scale=scale) return arr end @@ -264,7 +264,7 @@ end function integers( g::Generator, ::Type{T}, dims::Dims; low::Integer, high::Integer ) where {T<:_RNG_INT_TYPES} - arr = zeros(T, dims) + arr = NDArray{T}(undef, dims) integers!(g, arr; low=low, high=high) return arr end diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 4469b383a..0091e9c96 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -86,26 +86,26 @@ Base.:(!)(input::NDArray{Bool,1}) = nda_unary_op!(similar(input), cuNumeric.LOGI # Non-broadcasted version of negation function Base.:(-)(input::NDArray{T}) where {T} - out = cuNumeric.zeros(T, size(input)) + out = NDArray{T}(undef, size(input)) return nda_unary_op!(out, cuNumeric.NEGATIVE, input) end function Base.real(input::NDArray{T}) where {T<:Complex} T_OUT = Base.promote_op(real, T) - out = cuNumeric.zeros(T_OUT, size(input)) + out = NDArray{T_OUT}(undef, size(input)) return nda_unary_op!(out, cuNumeric.REAL, input) end Base.real(input::NDArray{<:Real}) = input function Base.imag(input::NDArray{T}) where {T<:Complex} T_OUT = Base.promote_op(imag, T) - out = cuNumeric.zeros(T_OUT, size(input)) + out = NDArray{T_OUT}(undef, size(input)) return nda_unary_op!(out, cuNumeric.IMAG, input) end Base.imag(input::NDArray{T}) where {T<:Real} = cuNumeric.zeros(T, size(input)) function Base.conj(input::NDArray{T}) where {T<:Complex} - out = cuNumeric.zeros(T, size(input)) + out = NDArray{T}(undef, size(input)) return nda_unary_op!(out, cuNumeric.CONJ, input) end Base.conj(input::NDArray{<:Real}) = input diff --git a/src/ndarray/vector_linalg.jl b/src/ndarray/vector_linalg.jl index 7fbbb5449..5981b9dec 100644 --- a/src/ndarray/vector_linalg.jl +++ b/src/ndarray/vector_linalg.jl @@ -75,7 +75,7 @@ LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED function Base.:*(A::NDArray{TA,2}, x::NDArray{TX,1}) where {TA<:SUPPORTED_ARRAY_TYPES,TX<:SUPPORTED_ARRAY_TYPES} T = _matmul_eltype(promote_type(TA, TX)) size(A, 2) == length(x) || throw(DimensionMismatch("matrix-vector dimensions do not match")) - y = cuNumeric.zeros(T, size(A, 1)) + y = NDArray{T}(undef, size(A, 1)) return mul!(y, A, x) end diff --git a/test/array/contract.jl b/test/array/contract.jl index 4d829a585..066693b71 100644 --- a/test/array/contract.jl +++ b/test/array/contract.jl @@ -72,7 +72,7 @@ end @test size(C) == (5, 6) _host_contract_compare(ref_ij, C, T; n=nk, scale) - out = cuNumeric.zeros(T, 5, 6) + out = NDArray{T}(undef, 5, 6) contract!(out, "ij", nda, "ik", ndb, "kj") _host_contract_compare(ref_ij, out, T; n=nk, scale) @@ -196,7 +196,7 @@ end C = contract(nda, "ik", ndb, "kj"; α=α2) _host_contract_compare(α2 * prod, C, T; n=nk, scale=α2 * scale) - out = cuNumeric.zeros(T, 4, 5) + out = NDArray{T}(undef, 4, 5) contract!(out, "ij", nda, "ik", ndb, "kj"; α=α3, β=0) _host_contract_compare(α3 * prod, out, T; n=nk, scale=α3 * scale) @@ -218,7 +218,7 @@ end ) α0 = NDArray(α2) - out = cuNumeric.zeros(T, 4, 5) + out = NDArray{T}(undef, 4, 5) contract!(out, "ij", nda, "ik", ndb, "kj"; α=α0, β=0) _host_contract_compare(α2 * prod, out, T; n=nk, scale=α2 * scale) @@ -246,6 +246,13 @@ end end end +@testset "contract with empty reduction axis" begin + A = cuNumeric.zeros(Float32, 2, 0) + B = cuNumeric.zeros(Float32, 0, 3) + C = contract(A, "ik", B, "kj") + @test Array(C) == zeros(Float32, 2, 3) +end + @testset "contract no cancellation" begin @testset for T in (Float32, Float64) A = my_rand(T, 5, 4; L=one(T), R=T(1000)) diff --git a/test/array/initialization.jl b/test/array/initialization.jl new file mode 100644 index 000000000..df8b0f11b --- /dev/null +++ b/test/array/initialization.jl @@ -0,0 +1,54 @@ +using Test, cuNumeric + +@testset "undef constructor types and ranks" begin + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES), N in 0:6 + dims = ntuple(_ -> 1, N) + for a in (NDArray{T}(undef, dims), NDArray{T}(undef, dims...)) + @test eltype(a) === T + @test size(a) == dims + cuNumeric.destroy!(a) + end + end +end + +@testset "uninitialized NDArray construction" begin + a = NDArray{Float32}(undef, 2, 3) + @test size(a) == (2, 3) + fill!(a, 2f0) + @test Array(a) == fill(2f0, 2, 3) + + b = NDArray{Float64}(undef, (3, 2)) + @test size(b) == (3, 2) + fill!(b, 4.0) + @test Array(b) == fill(4.0, 3, 2) + + scalar = NDArray{Int32}(undef) + @test size(scalar) == () + fill!(scalar, Int32(7)) + @test Array(scalar)[] == 7 + + scalar_like = similar(NDArray{Int32}, ()) + fill!(scalar_like, Int32(9)) + @test Array(scalar_like)[] == 9 + + same = similar(a) + @test size(same) == size(a) + fill!(same, 3f0) + @test Array(same) == fill(3f0, size(a)) + + typed = similar(a, Float64, (3, 2)) + @test size(typed) == (3, 2) + fill!(typed, 5.0) + @test Array(typed) == fill(5.0, 3, 2) + + by_type = similar(NDArray{Float32}, (Base.OneTo(2), Base.OneTo(3))) + @test size(by_type) == (2, 3) + fill!(by_type, 6f0) + @test Array(by_type) == fill(6f0, 2, 3) + + empty = similar(a, Float32, (0, 3)) + @test size(empty) == (0, 3) + @test size(Array(empty)) == (0, 3) + + @test all(iszero, Array(cuNumeric.zeros(Float32, 2, 3))) +end From 985906c801cc4236d56b18d984b3c817ba6494dd Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sat, 26 Sep 2026 17:54:34 -0500 Subject: [PATCH 45/49] 1d slice set-index (#215) * 1d slice setindex * update comment [skip ci] --- src/ndarray/ndarray.jl | 23 ++++++++++++++++++++--- test/array/slicing.jl | 26 ++++++++++++++++++++++++++ 2 files changed, 46 insertions(+), 3 deletions(-) diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 3b87618fa..0057c629c 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -594,10 +594,20 @@ end # LHS slices from `nda_get_slice` are invisible to `@accelerate`; destroy # the view handle after submitting the assign so they cannot pile up under Julia # GC (which sees each NDArray as ~pointer-sized). -function _setindex_slice!(lhs::NDArray, rhs::NDArray, slices) +function _setindex_slice!(lhs::NDArray{T}, rhs::NDArray, slices) where {T} s = nda_get_slice(lhs, slices) - copyto!(s, rhs) - destroy!(s) + try + # cuPyNumeric's assign broadcasts rhs to the slice shape; check first so a + # mismatch raises DimensionMismatch as in Base instead of writing silently. + Base.setindex_shape_check(rhs, size(s)...) + src = checked_promote_arr(rhs, T) + shaped = size(src) == size(s) ? src : reshape(src, size(s)) + copyto!(s, shaped) + shaped === src || destroy!(shaped) + src === rhs || destroy!(src) + finally + destroy!(s) + end return nothing end @@ -674,6 +684,13 @@ end ) end +@inline function Base.setindex!( + lhs::NDArray{T,1}, rhs::NDArray, i::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(lhs, i) + return _setindex_slice!(lhs, rhs, slice_array(_zero_based_range(i))) +end + @inline function Base.getindex(arr::NDArray{T,2}, ::Colon, j::Integer) where {T} @boundscheck checkbounds(arr, :, j) return nda_get_slice( diff --git a/test/array/slicing.jl b/test/array/slicing.jl index 2e2a539a4..f189c418d 100644 --- a/test/array/slicing.jl +++ b/test/array/slicing.jl @@ -143,3 +143,29 @@ end end end end + +# Issue #211 +@testset "Range assignment" begin + A = NDArray(Float32[1, 2, 3, 4]) + A[2:3] = NDArray(Float32[10, 20]) + @test Array(A) == Float32[1, 10, 20, 4] + A[1:4] = NDArray(Float32[5, 6, 7, 8]) + @test Array(A) == Float32[5, 6, 7, 8] + A[3:2] = NDArray(Float32[]) + @test Array(A) == Float32[5, 6, 7, 8] + A[2:3] = NDArray(Int64[1, 2]) + @test Array(A) == Float32[5, 1, 2, 8] + @test_throws DimensionMismatch (A[2:3] = NDArray(Float32[1, 2, 3])) + @test_throws BoundsError (A[0:1] = NDArray(Float32[1, 2])) + + M = NDArray(Float32[1 2; 3 4; 5 6]) + M[2:3, :] = NDArray(Float32[30 40; 50 60]) + @test Array(M) == Float32[1 2; 30 40; 50 60] + M[1, :] = NDArray(Float32[7, 8]) + M[:, 2] = NDArray(Float32[0, 0, 0]) + M[2:3, 1] = NDArray(Float32[9, 9]) + @test Array(M) == Float32[7 0; 9 0; 9 0] + @test_throws DimensionMismatch (M[2:3, :] = NDArray(Float32[1 2 3; 4 5 6])) + @test_throws DimensionMismatch (M[:, 1] = NDArray(Float32[1, 2])) + @test Array(M) == Float32[7 0; 9 0; 9 0] +end From 047acfb4335f0632afbc517042edf9c1c4d4d86d Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sat, 26 Sep 2026 23:33:14 -0500 Subject: [PATCH 46/49] Fix buildkite JLL GPU CI triggers (#216) * fix/ci: solve a problem where a PR has no wrapper changes, but the develop branch has changes. Limits when JLL GPU will run * fix github ci workflow on PRs into main --- .buildkite/upload_gpu_ci.sh | 34 +++++------- .github/scripts/check_versions.py | 69 ++++++++++++++++++++++++ .github/workflows/ci.yml | 10 ++-- lib/cunumeric_jl_wrapper/RELEASED_COMMIT | 1 + scripts/wrapper_changed.sh | 15 ++++++ 5 files changed, 100 insertions(+), 29 deletions(-) create mode 100644 lib/cunumeric_jl_wrapper/RELEASED_COMMIT create mode 100755 scripts/wrapper_changed.sh diff --git a/.buildkite/upload_gpu_ci.sh b/.buildkite/upload_gpu_ci.sh index dc1fd3ec7..2558339dc 100755 --- a/.buildkite/upload_gpu_ci.sh +++ b/.buildkite/upload_gpu_ci.sh @@ -4,12 +4,9 @@ set -euo pipefail readonly JLL_PIPELINE=".buildkite/jll.pipeline.yml" readonly DEVELOPER_PIPELINE=".buildkite/developer.pipeline.yml" -readonly WRAPPER_PATH="lib/cunumeric_jl_wrapper" -readonly WRAPPER_BASE_BRANCH="main" branch="${BUILDKITE_BRANCH:-}" base_branch="${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-}" -pull_request="${BUILDKITE_PULL_REQUEST:-false}" message="${BUILDKITE_MESSAGE:-}" run_jll=true @@ -29,27 +26,20 @@ if [[ "$message" =~ \[skip[[:space:]]dev\] ]]; then run_developer=false fi -# Keep both suites for main and PRs into main. For non-main PRs and post-merge -# develop builds, select the suite whose wrapper matches the code under test. +# Keep both suites for main and PRs into main. Otherwise use the JLL suite only +# when the wrapper matches the released JLL source. if [[ "$branch" != "main" && "$base_branch" != "main" ]]; then - if [[ ("$pull_request" != "false" && -n "$base_branch") || "$branch" == "develop" ]]; then - base_ref="refs/remotes/origin/$WRAPPER_BASE_BRANCH" - # The published wrapper JLL tracks main, so compare against main even - # when the pull request targets develop. - git fetch --no-tags origin "+refs/heads/${WRAPPER_BASE_BRANCH}:${base_ref}" - - if git diff --quiet "${base_ref}...HEAD" -- "$WRAPPER_PATH"; then - echo "No wrapper changes detected against origin/$WRAPPER_BASE_BRANCH; using JLL GPU CI." - run_developer=false + if scripts/wrapper_changed.sh; then + echo "Wrapper matches the released JLL source; using JLL GPU CI." + run_developer=false + else + diff_status=$? + if ((diff_status == 1)); then + echo "Wrapper differs from the released JLL source; using developer GPU CI." + run_jll=false else - diff_status=$? - if ((diff_status == 1)); then - echo "Wrapper changes detected against origin/$WRAPPER_BASE_BRANCH; using developer GPU CI." - run_jll=false - else - echo "Could not determine whether the wrapper changed against origin/$WRAPPER_BASE_BRANCH." >&2 - exit "$diff_status" - fi + echo "Could not determine whether the wrapper matches the released JLL source." >&2 + exit "$diff_status" fi fi fi diff --git a/.github/scripts/check_versions.py b/.github/scripts/check_versions.py index 0d8926aae..3f7e1dc99 100644 --- a/.github/scripts/check_versions.py +++ b/.github/scripts/check_versions.py @@ -2,9 +2,12 @@ """Version consistency check for pull requests targeting main.""" import argparse +import json import re import subprocess import sys +import urllib.parse +import urllib.request from pathlib import Path REPO_ROOT = Path(subprocess.run( @@ -18,6 +21,10 @@ WRAPPER_VERSION_FILE = "lib/cunumeric_jl_wrapper/VERSION" WRAPPER_JLL_COMPAT_KEY = "cunumeric_jl_wrapper_jll" +WRAPPER_RELEASED_COMMIT_FILE = "lib/cunumeric_jl_wrapper/RELEASED_COMMIT" +WRAPPER_JLL_REPO = "JuliaBinaryWrappers/cunumeric_jl_wrapper_jll.jl" +WRAPPER_JLL_TAG_PREFIX = "cunumeric_jl_wrapper-v" +WRAPPER_SOURCE_REPO = "https://github.com/JuliaLegate/cuNumeric.jl.git" WRAPPER_SRC_PREFIXES = ( "lib/cunumeric_jl_wrapper/src/", "lib/cunumeric_jl_wrapper/include/", @@ -149,6 +156,67 @@ def check_wrapper_compat_sync(pr_toml: str, errors: list): print(f"\t\tOK") +def http_get(url: str) -> str: + request = urllib.request.Request(url, headers={"User-Agent": "cuNumeric-version-check"}) + with urllib.request.urlopen(request, timeout=30) as response: + return response.read().decode() + + +def released_wrapper_commit(version: str) -> tuple: + """Return (tag, source revision) of the newest JLL build of `version`.""" + refs = json.loads(http_get( + f"https://api.github.com/repos/{WRAPPER_JLL_REPO}/git/matching-refs/tags/" + f"{WRAPPER_JLL_TAG_PREFIX}{version}+" + )) + builds = {} + for ref in refs: + tag = ref["ref"].removeprefix("refs/tags/") + m = re.fullmatch(re.escape(WRAPPER_JLL_TAG_PREFIX + version) + r"\+(\d+)", tag) + if m: + builds[int(m.group(1))] = tag + if not builds: + return None, None + tag = builds[max(builds)] + readme = http_get( + f"https://raw.githubusercontent.com/{WRAPPER_JLL_REPO}/{urllib.parse.quote(tag)}/README.md" + ) + m = re.search(re.escape(WRAPPER_SOURCE_REPO) + r" \(revision: `([0-9a-f]{40})`\)", readme) + return tag, m.group(1) if m else None + + +def check_wrapper_released_commit(pr_toml: str, errors: list): + compat_ver = parse_compat_section(pr_toml).get(WRAPPER_JLL_COMPAT_KEY) + recorded = (REPO_ROOT / WRAPPER_RELEASED_COMMIT_FILE).read_text().strip() + + print(f"\t[wrapper released commit]") + print(f"\t\t{WRAPPER_RELEASED_COMMIT_FILE} = {recorded}") + if compat_ver is None: + return # reported by check_wrapper_compat_sync + try: + tag, released = released_wrapper_commit(compat_ver) + except Exception as e: + errors.append( + f"Could not look up the source revision of {WRAPPER_JLL_COMPAT_KEY} {compat_ver}: {e}" + ) + return + + if tag is None: + errors.append( + f"{WRAPPER_JLL_COMPAT_KEY} {compat_ver} has no release in {WRAPPER_JLL_REPO}.\n" + f"\tRelease the wrapper JLL before merging into main." + ) + elif released is None: + errors.append(f"Could not find the source revision in the README of {WRAPPER_JLL_REPO} {tag}.") + elif released != recorded: + errors.append( + f"{WRAPPER_RELEASED_COMMIT_FILE} is stale.\n" + f"\t{tag} was built from {released}, but the file records {recorded}.\n" + f"\tSet {WRAPPER_RELEASED_COMMIT_FILE} to {released}." + ) + else: + print(f"\t\tOK (matches {tag})") + + def check_subpkg_version(base_ref: str, pr_toml: str, changed: list, errors: list): src_changed = [f for f in changed if any(f.startswith(p) for p in SUBPKG_SRC_PREFIXES)] if not src_changed: @@ -210,6 +278,7 @@ def main(): check_package_version(base_ref, pr_toml, errors) check_wrapper_version(base_ref, changed, errors) check_wrapper_compat_sync(pr_toml, errors) + check_wrapper_released_commit(pr_toml, errors) check_subpkg_version(base_ref, pr_toml, changed, errors) print("─" * 60) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 182fcd96a..b5ac72ef5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -52,15 +52,11 @@ jobs: wrapper_changed: ${{ steps.wrapper-changes.outputs.changed }} steps: - uses: actions/checkout@v4 - with: - fetch-depth: 0 - - name: Compare wrapper against main + - name: Compare wrapper against the released JLL source id: wrapper-changes shell: bash run: | - git fetch --no-tags origin +refs/heads/main:refs/remotes/origin/main - - if git diff --quiet origin/main...HEAD -- lib/cunumeric_jl_wrapper; then + if scripts/wrapper_changed.sh; then echo "changed=false" >> "$GITHUB_OUTPUT" else status=$? @@ -86,7 +82,7 @@ jobs: jll_tests: name: JLL wrapper tests - Julia ${{ matrix.julia }} - ${{ matrix.os }} needs: [resolve, wrapper_changes] - if: ${{ !contains(toJSON(github.event), '[skip ci]') && !contains(toJSON(github.event), '[skip jll]') && (github.base_ref == 'main' || needs.wrapper_changes.outputs.wrapper_changed != 'true') }} + if: ${{ !contains(toJSON(github.event), '[skip ci]') && !contains(toJSON(github.event), '[skip jll]') && (github.event_name != 'pull_request' || github.base_ref == 'main' || needs.wrapper_changes.outputs.wrapper_changed != 'true') }} runs-on: ${{ matrix.os }} strategy: fail-fast: false diff --git a/lib/cunumeric_jl_wrapper/RELEASED_COMMIT b/lib/cunumeric_jl_wrapper/RELEASED_COMMIT new file mode 100644 index 000000000..113f5b469 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/RELEASED_COMMIT @@ -0,0 +1 @@ +50e3f8ed4b92bb2e4b44a90d156f2852f9b086cd diff --git a/scripts/wrapper_changed.sh b/scripts/wrapper_changed.sh new file mode 100755 index 000000000..c9f39ba0d --- /dev/null +++ b/scripts/wrapper_changed.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +# Exit 0 if the wrapper at HEAD matches RELEASED_COMMIT (the source of the +# released wrapper JLL), 1 if it differs. + +set -euo pipefail + +readonly WRAPPER_PATH="lib/cunumeric_jl_wrapper" +readonly RELEASED_COMMIT_FILE="$WRAPPER_PATH/RELEASED_COMMIT" + +released="$(tr -d '[:space:]' < "$RELEASED_COMMIT_FILE")" +if ! git cat-file -e "${released}^{commit}" 2>/dev/null; then + git fetch --no-tags --depth=1 origin "$released" +fi + +git diff --quiet "$released" HEAD -- "$WRAPPER_PATH" ":(exclude)$RELEASED_COMMIT_FILE" From 62615c027f424a6941429dc56a21893176b6c571 Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sun, 27 Sep 2026 01:04:44 -0500 Subject: [PATCH 47/49] update cupynumeric-compat requirement (#217) --- Project.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Project.toml b/Project.toml index 0437a7417..eff1a1ad0 100644 --- a/Project.toml +++ b/Project.toml @@ -61,7 +61,7 @@ StatsBase = "0.34" StructArrays = "0.7" TensorOperations = "5.8" cunumeric_jl_wrapper_jll = "26.6.4" -cupynumeric_jll = "26.6.0" +cupynumeric_jll = "26.6.1" julia = "1.10" [extras] From 462ae6472207cd74e7988751b8eb8d5edc33ef78 Mon Sep 17 00:00:00 2001 From: krasow Date: Sun, 27 Sep 2026 12:12:29 -0500 Subject: [PATCH 48/49] update compate with benchmark and make [container] --- benchmark | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmark b/benchmark index 0d1ae3291..389002a47 160000 --- a/benchmark +++ b/benchmark @@ -1 +1 @@ -Subproject commit 0d1ae32918c8437b9aa39e9fd5321e8a05276fca +Subproject commit 389002a47ecf5ae2626299b8ee0aaa356ed8297c From cca25961fe14affa21a6844d0a1231aa5d6f228a Mon Sep 17 00:00:00 2001 From: David Krasowska Date: Sun, 27 Sep 2026 22:02:00 -0500 Subject: [PATCH 49/49] In-place FFT patch (#220) --- src/ndarray/detail/fft.jl | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/src/ndarray/detail/fft.jl b/src/ndarray/detail/fft.jl index 6e38f55bf..25e0c4683 100644 --- a/src/ndarray/detail/fft.jl +++ b/src/ndarray/detail/fft.jl @@ -195,6 +195,20 @@ function fft_task!( size(out) == size(inp) || throw(DimensionMismatch("FFT output size $(size(out)) != input size $(size(inp))")) + # cuPyNumeric maps an aliased input's store non-exactly, so once the task + # partitions across GPUs the output can land in a non-dense instance, which + # the GPU kernel rejects. Transform out of place and copy back instead. + if inp === out && _LINALG_RUNTIME[].gpus > 1 + tmp = similar(out) + try + fft_task!(tmp, inp, dims, direction; scale) + copyto!(out, tmp) + finally + destroy!(tmp) + end + return out + end + axes0 = ntuple(i -> Int64(dims[i] - 1), Val(R)) unique_axes = _unique_axes(axes0) operate_over = _operate_over_axes(axes0, N)

iIu&D&OD=ud)1dfhD7yRx0arpGFQ8w3tS^GEc%rb%9XrWc zR(VT3`2=c>s;FPkG{770pmzqPNnB5O~raqWwO9A)%JgbIvQO6!cJ`!gK3Eh$Ict3oW&$%c_~ zGFEB~A@?{G<9z(#SL}(^`<-!bi7n<(cNVK;hO6XV3-%abkBKg39MFK_05$qUDdEgQ zdF+|ey_dYjd%u!1lH^2sB)ww(ZzuDJrZvK775(C+IV+V{A$EHyzAVlsA0Y!LisP|^ zo_3KZgXl-$gU~m<;n0V;Ha!8`e;MP26??tM^q3p0gP^+_*tbdzFH)4yaX=2!n2THI z)y;#lVBL;e-k@~x5 z7j<>t3F}j-QyI_sJCb{bsbXUG_2|4^jRVEy2$@@mx1L#uib#a4IdS2@V0=sTL93<< z?=EY6KVggolem)$)yBk9O?=kTKr_8nnl`J#x}Or#_se0(Lm9k&D2>@Q-F5YU@ID~M z5gI^E!C-pkzNWw0cWO05>hN}672?)2W4UiWa~#ugx;75-{|iU1mp`@%`k?g|dE9np z@GaHHvFloJ572-c`JwSCLYR6BP-&?O?+7InrO0ErL@z`a%U~2ae!uw+wexKLXCHU% zX5?Q9I5QB}iH~l;g2D82?OTcKhl??HLk|1@WXKmqv)7?Fy>(xde{zPkoE4@$H6+fj zjYbU(Yztu4@o51ZjtQU{DZuZi^i}Cl#=tT~{9YoDm{D@bINu9@_OQKpHuu^_{^W1Y zzbBi}sL+Vu`MiVr*T8LGImYZP#Me)mxRgq4);kg=y#o;E#(C7vp4l>{SenWC%$s>r z{e}3?Pk`Ga1=#5#gkruBPCEqT+pA&XVinjHa7H?*gxq0D_%oMD^Nf76ocy}0&9G^1 zLgF~yeb?(yZcYC@TjDE|^6^$D9i4|`vEp+mBw9sq6!(VhWh<<&D|&FVNZT8LnBh9wRN;6iThSJqw8$sNN?;@|6-Z(+;1Oi~QD zUUeAuwGw+;N?;e33um(wsA@-ZKO2kWb;#CLEB5A#UA-A+{IRS%Oo+xY8Yq0u zo%b^x*c(zGHJ|(EOT;BS=waANT>7Q~UbF5co{_yOID3cl&aG-h!1a2}xSF7U+eXqtH*TG&_ zeQbYdfG-aXVLH=@y^%4*_0(@}ruLn6vtAIdOXhE^&3T^~^ya?xoYAWB$+!&TZFvva zq~VNA44SM5;LT!h%$Ib;;dRU}^`ZAVwaCPF5Mw-f;jvrYGgD$4KXj2 zTD0xd9-lFV`6n|>inhRN`mEgk^ZP!bUUC3=6c0I%aYo3TUx$#=Dzu#`foEbas*WWi zPIWM}I|4CS%>#=M(WAxL1RmT;mjBSi0pdDY3z?PLR|AuOYvN?P4i>U!kGpP!yB^Gn zqh>cZ(*m03EwMb$nm94}+N`^lGno?ap%3d=H(ZR}KWmZN%3PY3V%(dWjewd&sP~Da zPSqdfXI#-hKS{B@5%V8)Fzc!&W-TW^SV!E#S(Cn_TD*UCvFW7&4CI+7&RyN)#TKv{ z$b5qL)^ONrOWrnjeXRTM9r6hcxM!~+Z}B-fIcc@%oKt}h9z_^`Et6PWJS*2%$a&is1XQx-4Uw{?Vh{u{+*}+_s z`@ye{*vPv6SE$GN(2Ot2O?czb04dKJdT*D(_i8>GSEr%YAqK7+`(tsW7t|!};m+Mm zYJnaUhtc1f`;C2@wXpjHb^A8-1mHgB4)>(i_vnYLV~WJD=GdETg^#5+xSc}{5_PFw zK2CpiURf9DF>;4Iq`Qq6#CPcS7S2LuWym>}hxz`gBv?fwX?_s=%{{P{&%~oR?$FzH zv4hX8JYeS)48*3|dV6;FF^X41UvyOQ<8_Y4E{KdS!)IM>KqT=8HlO9f( z$k-nwN9wVY@cdut)|ZFN-kU^QgD-;0b}Zm-*Is-$S z%Y4aM)F81}^WZ)-oV$(uSn98+eb#fNuXv;hdOMm!b}@Hl2drVRkGqsL^nxs8HWcfA z;~m(#rx|yqG$K2^p1F_YquwpSrJlL4ew_r_<|y11^@H+y@&x6rkQhXdjdMB}$z9-` z@7z@-a8Jd3;Wj=~pA2+x=z%Wsmm6SHn=vFNnZct6eFBWF@kHAeexL2Ie31hJSa1Q!rm6NnF;-Vh3F0l6jIe)fGm3YK zG3{k7jvuaoM`s~c$z-ByRUE=^hhTGoH-0;DkGYlpcpaJ$hY6AWRDiriLfm;RBp*SD z|7{W=LRUx(ig{_o7yi?tj*-~6+&p7c-ZaITx8^u=+Y&oQT62Fzd|?gm!G83-WKY$( zq833Z6=+B;gtTu4CM=0X{Qu~8DeZ*?7s`;>;rUa54?LIanPXj-!Ayr{ z0e-d#Ftkn`&Y@bEQ^@^RG`(}L8Np7)6py*nQG08FygDn)VcjDu{@^(Lo-3&vsH%k- zxsoyO3b1r(I>ug$fyt8oSoYilZp`0P-J*{t)f!lMgjsiA1k8^YqA%}*4g9H){DCpJ{Vo`hx*l+RY=zOE$Q5~^j^7mm%snT-s6_-V_oT`gGxkz<3e{nvB0%qT0s7V z5W{$0+oq5c#CA>Y?vZuHU=}{FPEdw4s zME`Q`{Risf=m|bs-;E*sM4yoz=2)(4NsT;rIPB?mI8k?0B1U3iE%ds|(N0gVp99jF zK^ucXtAg=(WFH*;!Tr-=U8wTCY0*-{qp@m`y)3|E_R8n>2yp%p^;UWU_+J)6VWB22 z?$qIILthXZBYL)&kiTq(9DfUP#mMJi-PxL`_ zd~)iKF;y})deQog&__;~{bP%~w(BgLI8y+mVe z3}C$92!6yaU)V5HM#B;otLRyY;=DE9v2OVma?Yq%(77juhfN(`3M)04L|L$ zmAo7Eqe4{dR;JIa5_7hd@aVlV^`t7yPgcdn7y;I>&wIa;IK}{a8t`5iUd`MI84EPl zSYgj;8^k`d`@7#)KSJG88}-gJnYBbLa`BrgeLc&KB+oKq`& zeNS&nBYXVA&->*RJ-#lHM{%mEP#Zh1>p|#3KlNUwI5yE7PF z;Jc`6i|wo{&wlcgDmBl!O$fW#fKJC+6j)V27)oBB9(jGlsoMATN7QJZ(POQ!)LIwf z8>&#B&VAS*ImnQ|eSNJQ8n~nTq@awcQ;BC(X+VlR*+$M3R}UKECAGjc4raKSX94x$ zR)4iA@9z@da3FVteS0YRlP@}Ju|2F3v!4}VL{t`B4dd~oUl=wN`(R3x9WfhbuB{dD zJ!VeUHtyd>%HsPaS%?$mFz~kmde^I9hOZFTZJMYZr$hd#9Z?2LEH#vozAQ!&VeCL8kp~+ z1kPgIhEW;z9?Zi|*;E*89n2iCK<=sBu*cMb8X@|K8L1#v zD2L$PJz+dS8kNS}$r;HaT1Fm~{H+(BP{FiKYB>H$K;4@Q zwAGVmm&@Fx)s1*$)_^{~3Y0?PQfVZ7mqEW_a`-w*fjyBT9HNyFeN+iSEy|D>r2@xEs`$jZhlow; z6Ej;gznS_=X2YIpfYMxgxLZ`BpG+~fU&w--&pqmvDmr9 z2=8$3yrhKKex4Yq z%y5=M#2^WH-0Oj;$r6}YAqkhgl9+2R2|Wb~i7{}MApZ$0 zczW_}SHY8vUND#?g}5g@n1lUGl)th|WNGwGH0VaBsB7nI(T)DkMCFZ-M58?)h$^Pu z6aB%>?tUlUeS&(NKt5w}&8U6H?9Mk0h|;cunq?K#50*f7elF6QQV?7c4fiX-=u_&2 zCHmyM?lZzIBYI39R>aGjGN=+r!Eb&KJi6B_K0pZz6>W??ih#pNJlex+ChBb4~Or z|B`6e){=s7lZ}a(TBu9rB|CFsws5EWldtHx&vo-iSt^#MT6k&07HtxkGA%6q4 znXiH%+UNzz)eiKBqwdF0hdc!}+!855&bSxm4wZ&TvIiEg`Xu^q>kE;j!hO++SJy>n zM3+VVe|Fb>#=5DGc>hmrLCEK3B#|%kh`WWA#GmhQ$G_R23XwAOJ;}?%0tfErYGOI# zhT#LXzYp8oaQB4`5=yB{=+MT$H`w3y<9jMAkAQ%lcs<~^XkSvNC}YPn(Ws7lqJgD1 zMV~KT{rh>v>~*AhW}RmLzCoY+(4b}*DKx=`7)vVg!bf%0Fnd*ofUW}MPRvAPdji@I zM080QLn@XnaN`OmB%#T{Uk8+A^3#8`f*qFPfCCQ&j-GU$P5<)pwU;@!EU(A214Ty4jU6fBgpccOk#x-x@q0~i`kny}5c_B(C#{`*V9RTHES=xI*O~<3)=3yNU(W`&_+;SX?FQV68HZ@);_;8heYrB}ydO8?v z8OQ;VsY#>ctOqfsa`~U%Jsq6ovB-6JhggT8|i|IwYm$i%Z@Wz0A-b39!+dEh{iI}^^6@KQ?d% zk7FOCmb&6dj5EG1qF0`mHD*mSWB#fUyy&a)NpH8rH34T}_aOZAHG! zA~B*y)??+G8eD%$Ur3V@ypGGm!fol;J1hYQ^`epdI1DZUL8zS!hHt_3J(}FQ$RM&$%XOcQ=IZlWv{$TGlCepRX3f^H(nRgrp-zCIJys2Xp@shTrQ0vAYL%lg-}HQSm^XpBqf5!L91y zh#>M}BkL`YP(w|Bq#+b#y7jQrSa&9$s}4R_6Zl-6ebI~=#G>WjGXE(`j698c1PGbo zW>f)b<6=NFm->h_{M)P<#u-r#2P#pW{jr4Zg*@Tn~@{&i;NAFWDc8%ohB3A~xY-+3* zWg`4-B0XiIVHOYuE$0C2d*ws@nmb;2yP)Nm1NYdr_#(;qG1(k*ej350R1X7-=s%i7 zOrAK%pSn+YP9EerIcgiv4-vKW!^vl?rvCSb7=>@@(c7dJuhOWYZz;jnj6Aq&a;I}? zC@dtRkS6R8l}RGJ``QN`yPaVyLwuQ7^PP4JjHCYS(|tpXNzlcf?bO5azA7W<@7J*I zJ?x)*p_uoA@(S_^jCfXa27knvrGJwcvpniC$B?}xv4_cvieTB9jXgb?#iUPN_lhvo zS@gpxTQ8UdxuEWmEipxE<@cH5(Q6}mO6X(fYi(49s^j|qRM4LE%KFc?i$w%7Fv-xQQ4yV+ilD=b?V#~0;r~s|nnK)gQK%c892sZ{JeqLX! zaHAiD&<+Bl57Xq|$>i7UOz_V;y&sGnVwh?N|1& zXfuao9`lE!b#aM%9LqW6N_ME>8t3ER)p8Ih_Jp8Xf?3`_MN8$r|Gkg;b#a%%Gvjd% zaXa?V3)ssx9VAv+Q4h7i8hrmu94({>WeVBwaZMslKL|T61;e?>2f`|6_6U|}8fORr z^>ohMt85^q^G=|Oo>P?IsUwg6Z+b%MmIQ9N{uDh~@>O)jznj;m!aeL6&cHFD^pwja zpHPW81omR6P9=9!q8e?kWvF2v>5ukRlDP1lr4Udm#p%Z`IhuY zGIixFFpx7LJ#y@>%j5iAS&Zu~4d=N%@XweZqL7teMB0Zx{_Q6xv5y_IpZrZ5o>iuQ zvS}Ofw6%f0+qDqSt3-`|348)_ak(Z11@EHqU3UOHHu@lFj1xU}%~9sAM^1`5(m8|d zBxhjbWd$ryR-i|V99lD&X%Q=ps%QyZ`1w<`_QiKm$*Qk^`}S=-BMbNrHAwSpzTN~E z%SOCAT90odYLFwWz$DWmydZD$-iZW+7Dp22>jydcKJ@dq!2uT|ESyhFuu>K4+7xk* znZ}mei`Vhfc0*#zf~PoK+#` z%8(js?JCqPFTt{sT>Nlm=F>BJPD=$NugnuQxpwGHA4=`Vn#4fq4aEL+W0gEgN6SNg zzXGDYl<>ev8KVWviV0RiXtW}B?o&Vs|1D8x6#mYk6W;P38phAuK<~Wsjl`fE;77hy z{Q=_N)H*9a&Vts11azK@z^_p5^=n+=M<3iGWqNvXu6;B|8DZ0ib*z;~c!L7=Cn#Yp zakT-mYLMsr>o!jfQwFJFbrgNJ|5L*l;?#fU?%OZfXYf8a7(mWuVIy9jsYio%4W1?v zTe(<(WkDHm8y1Ii+e4AHu`j-zqAwb;p&wVZ>4&QZ?JY{sTC9Nnb>vFkQG|M+3KkF} z&F)1XszDm`EGCZTL=PeI!?nk2VKcSRH&}P=Q+l}#;khB#i~+yJXg8^cqenGz=@r;~ zJ`YKAQW2v!1S3}yhwksmOd(tB_+*Ix#dFWFj=Xa6P~N&JV(cPC46IVdO)oX(w5!AW zh9(MDQ)hRGx}aIa{E0EHI>OvdU1qAV?ym>L>wJh6p5k0IiP`8Sb*Q4(*9v-2W*p0b z(u*YKr$o`)lo^v2ZWy-H5~aVWSsck+sh=uHXj8&j10`liD$(CX6&}5ag~#xB579v% z&Q3Z)db>p%;{0+WtaCD9emJ#3tb3oc(J>Rw3geq_RFBW{4&FZo^eA{;goM3WQ0)O8|@Gone?~hLiJjFh59$k*8EpMGm))i!z-1Qlno&jFLWILDJl%dFWxxA!>Dr z<1`1GAmNcIF3^wd-$qNUV%>sk>@{TAKZh|tWMuh`z!&R_4R~1Rq_)cviKZaa>gW39UNi#wfcU@A2 zW=QF0f!}?sFyEdYdaS$s0_TL6&5+?t)acrPf>7>ZWy)djJ0Iz%)95=J1LvgvcoyzS zy^9UBUD+cgQh#GWjYyU%E=a2)GfoxSscO*76Jl(MCSHEifpU=nIue-|*I^3D6Xv+g z{rZyI+;=~-g_w0;@@y8K`@mQdgMO$z+Z&eO%^|f; z7ef}2XH1TWcshO54yd754*_0~yJ2Olj>Tu_H`ZGhxhsg#?J(wkl)92B7BK9v!gzjP z58vA%gmusGPB7s8w=kv=x}Eei8(M{A*AlGL&Y^cnBE0rSFoOixKi(O=l$mwU-ST(t z5H<%3kl@PQLNhV`sRDrBc`w!J6-bSc>UkYp@ibue1;1x`NkKi^Bgg>g5Ipz+6Xu;L_h9B z8xC=o!+q#Xp3xhh3t)eZo^D$7`S8}k^eOr{KHmu6eAs7=HAfS%r(^W|e6htACal{= zod0NLGqOI5Q504O=OLAFB*%K3W){49#pCJdFboXni=eOe^wKiMyBS*8K1qlS?nBRr znDNfNP7fa;e7O&`77Ouaga(X9YcoSe53f!e!iRH0`DS`SU$lVVRV(B#vcY)vbc#!e zB^ftEd!U$}PPLHQQI41Fr7Z5JA#*FW&JTi6>F>^b6?1Hf(8918%Jiv~$D(U;P|TEv z^g(&%N6KUC40%`_R=|{aWo$mFhP+G-3?<$#ZlqU*j3GT-jUl<;6dSLb!;*EwX7N7y zLC-q!frjm_!A!|=EGQ+HZbB+rB%{fL4uG1bE7~rYV!~zO%9oX(Jcf4wJxgS}$f*)3 zU|PF8(uT^T#*6pO93}ksOcl0y+~>^I;=fH7E%)?s*~17sis%`fNIepJx(5?DE8J|t z%>xY>5>tZ{xn(eznTLn@DOf)7|A@M(s4BNEN_U3{(x@ml2nM>Qg@J*nh+?4v5{h(( z-Q8W--Q9`Z9iWnepcqL1Q}4t7z_{Z++_COC=lk|vYt8b;@P^%R$Gr_k+1laqY7^{m z*2T3rEm$tqLSVQy_Ak-Kl5<+mXCn02-mjaG6J~T{q^nAEx%ZjCqCX zwwRwue1&!GW>E)pg&e}gNw}*Jhg*e{0rrkgNu#idTIWe8x}f}UD-;~Jf&DyV?Aofs z>;NtFy{v`fr?laBQ5&mvQJ1od{Ba#E%&4z}_bv5N%I8jZnJId@)Axk;g8Fq!v_4BN z(Lh^xvTmO##P|=SVs3B}vaZCU#r(wNYW>KO7FH9__Y>aJc<>z>0iY%gb*D*9uy z?qSw--JA;B{zm(SvsX5|?TH@{y11v7lB(AQ3)*Ur5 zf$#CU_gd8MXhP+=2K1dY@vEUWYMhvrLeI-@iNs`W%+Nf^9Pyi(CqCVZdTDD!v93FJ z3qv{YZsV-mvM82kIe*{e3Alf1B+5;^;LYE+p<)XxkF(^w$o!K~O_a@1N26d3guc{7 zk(UN`J;gq7u1kN?BLsH4Pdn< z7IxLd$pQ`PQi$Ui>EMKqKAJuo|Q9fvlWol*!t2W29edN74(=Tv7^^0>I(23_{z&h^t^@+1{ zH&>w@kCvk%aqkr8uoYvVK64OMXE8%$5xt_7ZD6Zvh$d4t$ca+LBzsk!xoQYqppGnU z4Rqh4fvp|1aN9%|a@GJ=)2NZHHp5v1K2Pth;6zUJAq@uvwrhkLtot*AeoW+zjGmed zuLbeQu#CdVtD)HV%O4A52$W{^K=+N#czn^8{`E$fxmAN6Z7K-NRYsF?6})Pr27gm^ zY;LT9dCj$;>8p#rZ4IfvHbu}y>duQAV8lafWGu5omK*u?i|Jd0Cx3kQUs>+=24^~*a-4^-uwz-nu2;r<7cZYi%?lqRbq6hPv zL_FOdi+0cH`;k8Zue{mU9r1!oGobbF=I~wA09%icAAd>(1F4g%^Qw`f*)`I%j{^Cu zN=SIF3MJ|zv-|4dQ>YQD%1qI!h~5cn8o)}dF*29F$ip1)clZ2}{d22d)cv%ee@-ON z%B6AW;~a_oPeX92XbePWFjU%hN27sl&}N)&}WTS@3mI~CRRZ5~tr8p~B%jY)4 zZW5JY$ulK~vu4#xhRUT0%-R|S?|uWJk=zBRiJN3NvUhcvI4>`i@QvT3 z^zcu49$hK=O%>2+iV{YtsKEb^3Uk<1QD0jfZ<91|<`K1{tlO}dJ!K?!I>ZLk7N=ln zMIwR+#X+${B>uRCBI$`gz8)EZ_$&{2Sh->rJtW=JO;Dhz0gFzx@~Ey%1_ykV4x@`j z{Zxr0zAuw?#^s_?Qz1uOs>C9oNW-CDosbo20hMDqcxkN!r9;1@pW|1Vy1!7; z@(N{-aj|^pUn2dRl*-xxCGuufv6Oc&5?!4l>3z6Jl0EtF{JhV8@Oj=y{?~2hq!Jj3 z8^zx^E&-j!#~|oJIDEP1j?o^4AyETya79-H4{w9$CU)qXZiF3i)Ipk8$)|);3GG%W z9gXtEP%&SYsN~CoEx9s(#V47Z{Xy>KzLg$B-^hhmuSG5TwN$b0F7Db(4$_~nf;h^x zR7@;S#-JeX8Zu+?=Q4ZR!Vr23_+xUI4<;Y(h27rm$J93Y6OA%=we~R+c zB57ohFReG^h`rY*X}S4>3?KDcwjX;cvuq#AiGlZI!jn4^zV7xv+_~Tnf8UFIrtR79 z@Y%B9xwW%%6254~Ba^uEoiuXmg9GRh=!a9Y`XkVVIKsD92zh0TPy_Nb7pi06$r`y) z{9Oz;e33bsxw6RZqr4jMRvwo=mAk?BWaQAB(l79;j5oL}QJpXT+Y=rp?tK04nbniL z(UIJpRZ{Oo8WvNXHx3+Z7Nl8 z@o|-GYgHyHEsJE|{2WQmcqb9QPbI+Qj$B!DRSY*?5U*ut<=(*4|Mszl+{ga)q_`7E zzrgj}VFl6~czZI;sf(#N7>h;u5#*1BVp;of=vL{=TomT-9ujCgV-9Oq2lUUYhj&YK z;I~r+Hcx8giP|qQJy9&JoIZ<2=U1{R^uDxReO*qhxhUSbXQgT1|LWSX=TP~;Jytrs z@;Ub(yqpR@H*%s>nNjyM4inErAx3Kw2JQ-ER{BU}uk?o7L=V_6Xpa;7n!#(m4aPP$ zh2cFN=p?A2JWdgx6Dq{$X^EJh`XuW&ypWIc@5z+;H)L5_8W@T#9)gK0f8r6u-StgV{>{= zo~Gv``DZJ0HBnNm0`IG}l4V>bpB-{#wLYI^izm`Y{l1(zc>7=7sl59)ej&GS7ytNd z*?Z9o%RCkB#wJ7IH~oV)F{jQV5|i76q4@fE=;r$2#>;_Ntlbk@>h#74V2ev)VA3C@VyqsZT$gxFO<=rwCJb(}s(NMq(zKvzs&)efIBTEarr5#b+g zvFtN_&U(`evIF^}%Xn3P%j1zw+n{9By;UZ)^uffh?$R&tBfXzj zrr_e0zj;^O_ur2}rDp`Q215BAj0ZlBzz}jdPTcaO4!axr?dXUF%Jjk9)e@Wh>B+aI zF`^gRW8wwQyxp0Z_msGEXFa%o)WY`#|KZMySXY;4R>FOrTha9Y$fEZbISp$v*?({s zeRV%`%OhhE(kc?}cf+9CHxTw8M?w3u4}H}8(>uWf+Npx5PVLc>n9jh?t>EyOyQ20@ zaJ`V9y}%mj_slVWq$zz5x%=7kAMX5@V|SZ%bsuqVATBwpFpd7a#F0Ly&>u7znMDcE zz8{P6Mp-k*6~dL+AHK$0J?o(cz99*LIk<(*;M5w5D#b1#D`XB6w{h z{3y3&E}tdFuQEr6?$m1%A6&q?W~{rH-}8bP^4&g?bFRU2>sbo6H6j=LSpv3FOA)v- z64S;`LCImHW~zV||4y{_2N(;at@x}mTGJy^!H#+#@XXv6(n#)d{% zZ9?pqz15%R=J1{OU)?a)&ERvjhtJic5I$QE$fYbwMcX|oXj72H_ayI|MAcf(`GneVy!jQ zep;ZGTsz&R|Ka=XSl6HLQ6=A_3Bx&;9_8-tBJ*MC!%;9h38#O@)1xg0w`-_}?H>xS zr15BQZUin+)3>G0litGJ=+EPhD{EZoY1A5Sku9)dg%jGXa>R~Y2NWdRFlWL7&hNMj z&ooAT@^}A!&qX{p+7a6bKgs;1?$ia%C+}=(D%NS!|5-T+4Lil-F7#nuos5wq5|PlKc^7W%Yx5>jR}qAaA*10O z=z~ok`ohzr8=f@nfMHf`;JnNU&6x|QW5hX1&jwZ38eqq6GwRrku$Fkv1Vip8?A76- z{GTrMEANHFycaI6CT~=US|lraxAU3$s=;@6K>~!nR52M5*n2A!<;mkw{>%^ati93Z zaZh|s?*!a)M(R)}ET%W#?#pgVD63KjvK+iuk7e=(XP! zmn&WI>1;E!AZPxRjV1iW46AQbL-EuA7eaIqKz`+K8x@>=Q7etv+b?ffCZQur{+*Le zc<*0*!M$B5?}r`aKyhw(@t)bGYZ4Kg5Qmu^qsUjBh}Gi)5uL`oE3-iuzP2Yqwspk1 z(w2Z{LtG7RKu>KGT(vS__M{H65G@Q+R)_jSCG<3?7WKwI#Ha`FgE{$PtdR4s&w0x} z=jOe1Jnu)XN*MDsEK{+f2Y*}gRTs35gw?thE^y!L z2(M3;xYpW)`$b*Ez0kxwW_Wb@rh*&;au`E7p{)5K#W6*)r(doNCC~2bkN5xjx;DgN zAM+hL>%_jJNg6Kfqo4SyBwRZdk4+b&kr5n@7R&{DoHiQIPY%W!|6XY6&=E=>nxn&X zY7xJ4-%CDdJ68={{HTI9Kb4r@tB4<}wPM2WHgRXE>@+_#va#KQFZzTl9Dd8>a zY8BFh!k_nFt2BJ>n}Us7Iir-vV#6u(LgYFZM&XnkQ1nhFT~&PvhJhL z#8f<}vo21>oyW=av1Rtr6LLkonR#C_0rfmaqm}g_c%AA_zudMsd$b{K?9E^jrb~_` zvHMIV^zW^RvlkUna8?Hhz2RqPkKSrU`o3=c13sJq*x4kQvjg`}#fk zgiW~T*v`xp@}5>3B*0*6H2vNtVdbhoY-%zB`&tbErPS!dxx0@M@%1Oh+^=av<(w*B zDJx^$XeD~RD`U-TRrImZM9d5w<|gXn@JQ|$=;dN~*ciKWOfX|~Jsiq0!+O?j{*v4r z4emIGr^4_y?;rPgoC%D=u&glTw`I(-q4|xyf zr6EHr6&pAQ9`cWad80_|vJ8d8Q-2KqF$A;c^~Cj1H@MROwt~9I(Bpb2IH&=qQ>r+7 zQw7i4s-o;MXOtD1)b#4&q_H6e5qk+Hw(xDCIUa{Jz%Ot18{_FY8gGXotXoZ7QtMh8 zY|rv8I8094j98fDOvb&D!6-=@jp`rXF#XvL+OwSzdD@=;FZ9MU*F}eZ+{^Y;L(iY8 zNZP1|_8m3Qu~-Ws)Ql`hFhX0$dg#)Y`pF#)Fh0`?!-{R7+tMDop$*~0x|c5f-Sg8U zD1}FnYSOa=0wnV>0HdsiE&YU`hRekuJ)>Fp^zQ5Ps ztJ7@kG|Vz5 z9!gx&`9(B2c$0{01Y)VK-#?$NHd8xc%q#j_zoS0xrXgl{YNN*-4Qwx0NB3SD>pN0HHh^PkhxtF?ZcGFA+Vu#*&4rpvc-8JjJ-^|^LMH)Ku&dqd+CwG^g z48HVzy5UdkZwN}yctEwDD+Z8@@Ue{ zwAwp{y!yfPqwayZU0u+>!XEd&nL_QeF77YTLJ5e~Rche{@lemRS}-AI=;Er4)8u=d zp zFao#e;X3(^J5)Q8Yr5JD4OF#|Mqjw8=C#sPp;l%dtCJos3glka$??KEQ4dqZ8b1{@ zveLl)AN0Bk*Qch$2)Fx~LX%j->J=7vUdWsV_EsJfh#&bd&+1enj&F;>h7Xf4<97fC z1P#aDm!1fp)d6+S$@N%h3gcrM=)YW%nU1xR(7sMa&QrkR3VPVaduU?Kd>z~h(?=h5BWxjdX2QI}e&pe~v+jfe>^0O=>0wN->*LWVm@*N+^~a;r z$D#Omuosj+P>)jI0b9oMyxpdTcb62P^rlv>kLKMslAm2k5f`XAd5}>pkMGp5lT}3B zN);S5(ZG~$+W4VGE!qMD?0iRk^j8y1Bo_UYy<|D>z;|C$u;M#=n*eeOv?fA=ncH3G zg9nRxV%NO3C_QP1pUn*M^R_Dbdn@4T-dgcb;(mRKA`U4M--xJ{Jv*voQ&hE_h*7}W zJB68{O*N1MSa=)Is;!iU$$5;FyOh$7!11$tPphfA-NW9n7n(Wg$vK2XA& zhHA)Zu8Bhr$!}|{ho@Z(;LYE8A9w!CSa%_)hj~k{+y^|H9q9?UEeuVrjzyuO5Bk}A zpkTW*+@fvJ_Zj&SN0f2avql!Qs+J|q>!klH1vHFP0EX2_*RV<{)2Wh0!|TNBwGz%a zt3e%_*qEL(2&pb6HlI3+_Uc_uiI&5G#u%FaXn!i@*ntM4)rhX zy0~Bv_fb_5dboF53HuT&C8ywz#N}1VtNX9UPp9-oK-<`4W9MoW`TauAx}?p(nmH zZ;Q*rY?%|Phb`QZ3KCX^2oW6Ih1TL#zjVSb1KRjNz`VXDY64PJ~B93<{a|kY*G>&4({mw(89+ zQdfF?*rDkN?%g7k;B~QF@`#&Q7?w+Q3ZEtWYPtTbN~-z!Yq!_Rk)}!*_C^(6ysLEB z^Bfw#~DEitH%$`2NJ)yoPkcrWNB^SjPS#nur?2;9n4(0Moy+X!BRY{(P0(LD_CjUnrOV|rF z$kRvI1Y`W%Lw!3vEcQRP#JX4;9A@1`b2)FnNX3yGN!U3s4r^OQVDkG3ust^t(Om|h zn_nliwP*qvV1kGZD%6+$7X5K0a_ZApIZUi-``sUMZQF0@;9M!*_v>UtrZP%JgZ?f$ zh&)d%&ItNbpE1LzS_@onNM8=-n>6A5tG*^3pS7v8CeA)eGaeH?BB2%(f{44L5&n4~ z#-sq>f|-Hc+zcwAYG`R!E@{7uCI0poi8@p)f{c;J%a?*B>)@^lgBGnbt6vY=@<+tGkofLKp6wwgOUWI3gfVHzqE}F#S@dpQ5RLEqCz1!ocj>`U z{JPTxT<4;Nvu2T4n6vwrD1kK3WgDrw0;x|HzoSmCY92DZp9rMvzr@hl3!mghDv$j_T`<--1}PWw!M*3;(!_Ja;4$Hui|?4 zmso79md;fQh_vI}?W~I5vFeC?tBK(%s`@uj~z-f^{`!`PTF5Clm)8q<>uEHQu5@v?3(*Zw5#6BlFS_0 zJ+VlBU;HNX<9|war*iqKS1BGmL!*d?PxRxyhjnu}cW3raN7|oMWD-M}nU#S38)I;M zPdJ7p24PFu2;>jyk3Kq`=xg2tssW}@-L8TF)o&6yBumzPc`l7QK9RizPb6pU3wd$$ ztwhYp7X9MS^5#gcOh1q(rz-PhfWsFFdsr;zj7sG*>*jOM5Y3))dG|Exx>M-+mx$fm zGfdtYf&KkMury!{Iq`$AJGmRC>9oed2bLJ8%IsD93Nag2iZ$)^hsq*u>J zlG6XFTvK`>qn%&Mr-v_O>HFuRyy&^?81X{f9=sHryjN1fx^IXb*>g9!a8?@fzNFyB z;v_h_lHZ;Z1%q$Y((8>!pFm#}U+ImpQ`%whUFuL~7$JXy67H=mmJS=vLvUcZHb~ zeyIrVlZ?V42{3LNgN7r*VHzETxH@KsUl@S1&xBcNEiftD0yf<1cIfm+(x!iwh1t(# zNRtQhYT6w!K9ecukKB|;Ew4%3_={q=_lz|2IVr=A9~U!&<6<%Jm{hW^-fwcD_H&2z zD-EA6q~btHGS&=DMETxW+@2o+h4~?H>g11hLxy04O+*Rrn;EGvD4UN z4d&U=u@~>9(cJOZGUG}y8J$`&1CY4$i`(HyNS}Zu`lC@WW)ODl=z;Lbu9&&85hA?n z;lw2k9Of>&dS;O{ambeP;V1CUkcN|8&;(=uD`kBFdmb}Jq^z0+A(EGg98+h;-S%PZ~t-mQU_G|JAkN z9o+K+wNF#od+@wezm!ZsQx_iuWm8x z`Ve3Eu!_%?0&yH-bRK6?n0=Cj>GcwD%{Kl?VW0QYh2G+=n0FNzkF6V{v1?~IPK^)75AwiM{D)z7cQ1Uc^}q-1&X}Cr z7EV8jLyo1lRgnWu7m@S3vcbQc6ORl799E~Ndl-3-^l^xP@t@v#JnNqOLH_l1p3&R~ zyPr>Eencw1O-^C&oP=AvpS(uJK#opEs|O)ie1S8}1wRC|9E3~hec)-|6{EQmIHcnO z&+Ha3dC(LUXpG;{#GOA|aj#q-&;5)M@m3GFW@yZ*Xc8js2=P@|0?P;p4+2|qBYVq$U{D4f*}EhFwp)FcaCG-1FUP7$yu0ndy^m3GCLKf z#M#r?+aB;pz^$AZ<{pwi@;(%+ng-B!WF(%Z4nduFKXgp(fvxOEo`%!k?wm8GHExB3 zCe5MF+<<-U8{y|ZW=s^3Uw6Wcd4whia{Nyp>d(4&`8~J3!8>Uz`B;~Tvp-5jmvrjN z*?)AmO2ExSF<8_l0xRc)VgNIimFkbglCOi2bC?+6WUw?`(xjEtbn<4nBBbt*J>T}GF zqF^haojK;sFu{^U=2ArIqGR{}aM&rV+l)Bu*geEW9XT70B;U=A_u<-PSdUD^BI<(l zT127cV`BRaf>35Tn%YJB70m1l#hkA2zuq3&Ct71nBR)&6jgYzA4xeVw2dWjlm$l3g z?q-BSVw7&d+;t?Yq45}%f4K7(o|6SUC;xC>sZ(XX7d4Y--;sl5oQ!+i3r}qqi#>-U z@bO_N(i7=pe#H-JKL%1~))VdPcf#gw&N#HEDXeaAr@haL*_-t0qsQF>Cu5YI)k8hb zqbE2IC*~-?yi>Kj==kT~xnae+jkD;>FoXWn@6zy^Se;vV3a*DI;lk31i07?YlL()Mam;xjH|NMi{0R?4 zr^u10_sSc0)OzC6nhvn{Zh@WjUFtNqK2$gph%;xgB^tOw58&##iiq7-Avr@zCD|lT z&a}>wCoA5_<7O}a-TN85_g#p?+Vh`bFGZ(SU-Hbp-C;Hjxj>9$G zVVE+kAG!qsJB!=kE}x$xrqo1C)hDK;f#<|k3S$*ezq(R1Fa8jpj4$%!Wwwm#!yeN5 ziL5tyAO*E|{`EQc+2_RZ9SUno%#1kfj{_+fzmA#_>grZribj7v$2)pWplAGO+_4;t zypBDw*|9xNUT=ze_LhhkK_dq+ zzaw=SH^u(bHL0lM-i&>n<6-LfIIFCDnTn-NQXuye@n}vQj_X9>dR-V2v;r`r7 zZ}|3FVE0sJT_sRgbkZ2@JgFbpS|?4G{gi$#C1O4CizIPIe0cDqbXfC7Hoku%p?&Yk zm^-)ReaUrcHSenYV%=1pRh1jryFcXpmz;{lOOs)kmw+>EV==FB1TOo9AS>1%2gC>K zGJ9e9y7s6&(S-aBbDY)DLE;QWydU*ba<&yoaA3ZSPS2I%_3SlNvc)apjii`AleNbm zh=-%-A`HJ;xgQGMS_zjU1IZIq{fxE*eF;;ke8G`TgRN2zD8W zqNQEoI;job8radlk7v(JRZO4%NAkB6i$`3Zyh+XxkJUL6v?xc`U;ZR7K7Wvz7H_1m z_PMkQc`DD8p2)&8k0hIQ`|!*tolFj;A9*TU=nqU!)%Th7pRkUicWW5tc+rc0>@b}B z;ECyu?ika@3HLsjqsLrr^dGLk{J`&W3k71lBS+Lf=1BWLdE)rAKyq>mn z>%rWIwm^sj-j6kgCHIdn8dpfyf?}C{IZt-a$d%^T^X1N=BGK_Jl}}y2OStiOIkKlz zyaG!k)UHH!pDB?Y^eQ^dy8c;wrd&9uu#Y^wG8rz@=$+mx27BmF;C?I!*C&lc#dt3q zKLdRCr6;kH6<%iQV!f^sR(1R--|81h_|$yS9atb!=^3NSnY>b=T&y-#N=NQb-6vJc z?KRv@EU1y7PIaOf&HXg%z9b%MXh05DbSnAZNjSD59OqsI7msQ94iz)k8%m z?vtpk>@|*lW5k0MSohgk?x2^lu1^Z8wGyE*CkDT%IeyVT2)+|W;NZso)FX7p9qJqp zs+%J?LkkN_6yVvqTv|RXlN+PTWc9qClB-=IZSU2|2R#)u_1A!OW)DkTq30t*bWt_I zL$7+6aic!+T$u05yLZqb&I;q$tCuEYf?om-B}8LI&O{Q(0_eFk4BfOnvHU@Mln!W& zqfMxf`>75?;x@y^a&MRSQ!Xd|61~E5+1;gD&W`5}c$z8(=V{{6H+mNiFvNW$6I3s$ zhwkRo=3Z&Q>?mslux`5y-oNxyiJeVt_nvrYhEc~pHVpe!$07Q)59aRZ3HL8;;Td3$ z8Tp3LB>&S#y;d^${A}_lm%7M5QvXt=^j*Vw)SEmBV$mO{d3$t|T5oFCpLI9I-k)Ze zvC{&3Ct6{+HFGCe_sJ^er;smMY0JEY>c81iky!U71cTK5VbN_cRws7Hs4;EGIkd*T z_WHOQq=I+hHR83RLLMKgkQdYrO#faj=TsH&=Y$e$64hXOUW?h@)b2+bA#5h~F+-?5 zu%ceH)C!}B&FtfyKXo2;Pvl&wUQ9wU^^)o#5hx{}O~-LGPQ4n4KTEs7&b1}9GWw?y35%qdilJuuawhXS3aY?nZY={DWZBc@^of@`?7IgaQA(wi3YZX)U$u`5{ zLl*RIr|+$uE&8+W(KKTGyQrC9@94^T*ED@HI(24t7|)9)I|m@_cW3y8H^coI=9tl6 z8#;^lUGAz8#|_oeTCr9Zo7KtofI5lptpNRIO3)2ZMPDCH>e+R%{DlF6%Z!nBuO6_R z9L9c@E!?7iz*q zg}NW(T6wBWF3zbsF(4i~;Z~j4eyx)?5sH{Oo_MGu^;b32G*VxY%-zqefu?w^#7sou zN^`1vawBac-9mg`+9>5uE_*<;;^Ne{b=0f7EL;7+2%EA@h7oTrFV69rbgAlzz#g z$NA#hD@RI>=Sq{0`O<=YMJ@7W``CPWbM=epxO|tPP8Cvbg#vcCklPWZ4$CRjf2Gm4 zVVnVME9o&?K|Yc#ce0%QvdiOP@+cC|_J`o1!WbBJ@J3_ZE=bp=?r1Rm{9{$2oLeU5 ziTP5mU9NoZo-bDh6^iNC0$Ff0PnuuJ70dAjGHh^(+$;DcZ9V&vtV1omEAvCBX)|oXcj!zo#vU7m>$AMjwz4x?U1T5SV2aoW%6MY? zUHUc67mw0haf>OCpy6L6vvr|-$;p*!r5tI+9c{p@BKdZ+OcL!YXyqncUGsNP>tD6+zsk(j(|_u1l-s%65VVDaR1+l+$2Y=8DNaB!Q9_3 zE0y=x^5nf=p44_Kl%dsMq=a=}D(A}6(VxWf-DlC%DikH|Oy^pcbEi-v?^Y|~PAe5; zk*D7U|FKb3hci2swb zX_TXaNB7jw&r$<#e{t{3J5c>WG6vj>L*nnr2rLi6p3WoS+_fKkayp=zvt;Xy1_*TH zeSNA}o(JVhooSwIomnWG+7?N~r2=`k^0N%l&z42sv&HXNuGkJPmSW={vdp$ZHiy*8 z@D++^dWx7Z`AaQVtKr1oy18l0Bu&P-^Ko!^I~k*H2jRyLW}`jsi;q1!V1suP?VUR1D8;aynpE7yKGki}< zjU<>TVpB6^O#GmN`gyAVKCewaJqhn7WB<)KnC4GL@RJ~{F7v~zbA54MkDTUW2b_1* z$NAAUQc_qbvFe{?qEn7MRx6NA4Zg@+g#y`N@>w2l&ytEM+48zouADksByB>=vkvLq#6o;3L^ZJK1u@h%@lSwBxEA1Ri=tteuBEGbv-pgE#I%0K{u&(-_G<2&?#=eH}xE>XO33DgF?cNAzZSO~q^ADy$R#IS$9+&Y~l1?#hA=P}|b%km}ScAo6={VXjsKg!L8pCo-_z68gV$ky0j z((6T)OuVjuGXctI_=3OfFLfYM^PhiQM>Xb^H|0(tHXb=85in>QjCDIk;@9^6=v>wj zA!8b$h}{2srBzbZEl*V6y_XApKZweRPy7yZMdN<1xF+-ST7Qy;U2|mH=|VY{^GzCX zC$3dlD`Vy=!G1h76J{F7nW=@`KpmXoxiQu(9W6$uK7Kx12jKhE!~&rO0(LxQg`C599*6yFZzC#MIoFG0zS)*R-fgIU7kE@QY=oy zD$7?^Not@Xln<-m1$Tz7_sL&bu8Wzk=x4-p^0X7V!0S?QWk3SjcaK7(X(;c=G32KX z#Hv_9Np=%xO`x|%RIR)Y&6S?Hp8QX^q}8H45}_y9NxSV{l+h(OYtiiMf~VnRhFDN_fd9s$dL^l3&rTl zH(9x^LY8+?gz-34Y~p?{{Q-Ntk<_YQGp6rJJsjcA`}PoOVjNOoMlO3^Sv1BHuX>w0 z4pGKFxO~1l4pM6qyxojGoJu&ih`P0km*SW3RKhwwmyM3E#o_il(Yu=^19p6t8P^Iq zUzf?Rja8yGTM7E)2Q~hrh351%Q)SOHn^=2dPjf8g{j1LV*MvLvzD0=`l@)`Rg_H0q zBmn2O`6BaYPwX4!3>U8k_`=z5`^Yb{ zQApbyioh2B%%mKIUG=(PE_DJcY}uc&U;UJpCCh$3kghJ7(kk|rv^BjWZl51W@u}zX zZ1xBFeuU?z#Sd{iTP<$XBKPgdjHhRsI2EajaBV{@_cXy@jn}4Q)E*^p$H3Wl@P-7W zXGBB&=tOjHFdmK#heETx2VR7>M$b{qVBf6FoakIxs{2?XU*41%`>sj-r`KgtZl>G_ zd?+h_yp&g`K1#=FMKZ+erx@L+k{6Ar0rysc(q?t~IcdT9r!Km1K016Q9mz-15WFb` znhEp;u8Bdw@o?;_3`B@IGoB`SqWc3kDE6mM_j|q*OG+f#8O_c%P=6alqM!8rem{PJ~P|1$5UUvb1zYeU4ARY;-kJL&6pTUxKWBm=F^OPJ{e z@$LhKs`uizrO+rl#Fs9Bx1HfpwprIKSTohZ@*m@ImsbmVK9Q zRnKKu<7=|{<0<*sne=jtJWbys zKUr6sxU>IC?!M15V{t$#tX-4oFPebzl4$&?m_(j+AbM2MbK1N=7Pjq#mDRkTesX?y zQNq#Axw5eH0~u;{NxpU@<~sO@oUA+`BU*69$#t-Hn=81t(-jf&?vY&PQ;KutHhK}S8$;}fTssfGPp?`gV3KPLuAL1>%%}=VnuTjg=+I{6X4 zN+x)&kk)6G{oA+y;T;_FoPEax-v3+D@NE*k5s&bFYL|%B=VOt|o@;SzDB7xxLzfg^ zB>(cn=Qo{j;dKjWePA}3BK_mLsnSE|j~qz;BI=W}1Q(vk=I@yzW6#UF(!`n(jz{ZTQS^zWhGK0XY`c!c zHydxvTiOG9j>MhUH^uXNmQYt8?L;vcQv2HkbqdBXnndI*Kr(qh_pG<+%tt2$oOF&JZ7|f=JQ2zX2M7#K- zY1g6f(d>)sy1->}N$rO^qOFP*@)GG6afuo2Jv4BGzUF7vRm%B!UuEN%Y^h~$;yduZ z`1Q+_b2Zoh)t${dY3CPmp~(MQ!e>jpZyKC{{c`mBu(*V2yJJ$Zy_mj0eo4XmUdnuIH3dv9D4>oJz$!B2QYZ z%odaF@BY;-V%?F%ozsqzbKai(vsnJd*Ou6!xw;k1i|S+K1bQ-QQVY{S6F10xTUl1ay??nZ zXh)8g!W202>^T38{E@SKk9!kmUqa9N>GZo+PC@em`j6Dc!API{)x1eK7B&HK$Hrim z2J^$;FehMJ5A-Jwz28rAh6lF57-RbKRyx4bpSz+2OI*KXhIBt8*mKstvrq$So~qba z`(M6uch=1(F8S;NXQhG6e%hP{yY=Lca)!8UpM=drxDPB1vLZO!CboyJRO?D)}Q{oO&VbXInz_ptzfXIJ`U3J^kqx->m}MawedgP zxf|<#W!)dFo5Q-(_%7LRO2zV4DVWicIUk|%FghHKaSy^#aUd9}Dw*@1N* z@_UZp_k6t%-{Tc&NMDeO+4WNBfsurJ)TOS7j>eNI;aJx%7;lUGv2*t@xR!b0N7r7+ z(gEIGZU>9gZID&31u}Ou#g%)FvGt|{dMVlBbSiVM?$<-hqek#vsgJco|ErtB=V~gS zD_8beRUOm$45UFPi2E#RCi|UBgbw!r8(%YP>lk(PT_)h|sxcU+du36*DK?-q{%M?mNKyvMnZ;Fo$SNeLOyD41Laa!?)>RZM4?E&zsJ=1$>Y6 zcGGXHAvOFyoJ*Zj;XacdFYbxhV-|;>%${-&n25G>0x@{}C}i{)f{WWdaYCmnZmYLv z57-)x2b;p~J$?Cx+TlC(w|kr#K+nMp7QxJ;+Q%OF2)(Lqsv+;d|8dy&JU0%m;oP7~ zzZ2Uuba|BmTW#tH=Oy5{RV=#CjzFuWp{TGOkJ$tM`p6DMT@8IHR=eZgt2S6aqABJq zVGhnbE6m9@NAFAgUa5_lSg40rrJDF|r;06{msZx1*UA0(-{04Q=T*}zatnyV<`IXz z@-7u`h{GBahppL3{OWBqe8@=(cMGO3f>pGo4)mHiMjh-K?%C!Z&HY!miub~t7vvNWhyAsedZ0O} z*j}6rr~QerSQH110a1v~n22((Ks<6A35_^!6hwHy`9^y@`5#eN6&6&&ZAC;>ln@)m z789_=SX)KxE<~jzq*1!;4zL3$0qO3R?vfOw1O>rH6v4QA{2%VaeDlrI9^jm_W0k^u zG7gzS?>>&9#nxgpV;=9-?29GTI*{h2{f6b|Zfr1n2P64rc$mG!xItBzlVAR?58C(f zU9^z7Bdmu#&NyrW?}meRGgmg#hUT!IyvE#&UP>6yf;vsk|52vm$ZhoTFz5?w)&(P{ zk;;)Vq)Wr8g!`e?#=lT|&$`H&U$AgE-?vsbVUI)=a?TaO=Y9@m)MP+R@cF;K@5Ft7 zc`AFF%kb@P7W`V<5m+`DymW6@KQc%4h9;K2oxbP?vz5jiy zc+SZS;U3)2{E2c$y2krT=5t%hJ84B)9pe3hi**PWlmR$kKTM{arnY z)TW4%^JxK!n)eyY{aRtrUyqI#)mRWyjs-kx?s=90qdkeR&WeKN!!Q_&hd`w{5MR0O z9iHpHKVm*mE^~rS9cjpE&Z#S~p&dglnYY6EDINNBFh!k=NA0KZcS>|OXenJjID>|2 zNs^MzAd-{%f{hoO@$O(X6jRG^W)J(C1m;6!eHQ$dr{LD4SUmI$$BA_zC_@mo91QrU zOI77>#zF4+*RFBT|LZ{3`F7+JZB1v)PSWcKoO6|^L%)oU&^p7N*LKmE<=tC%4rtkB|&po8>h=Vp@G#JQy&GssSQ zBrVSoAfw;!(4k(7iNA``xicTbj^$zO-#i%n%7c$!E-oy{#EgbiB;8L!LQ?{o#wEaG zLOe3L?z7iCOKxQT6Yt(LzS{Ghi8X{X*!LpJlTimq8lt=HtpS$-m= ze;Z2q2|bX_Xu{za<(TuV0I$6Bpw^#{9*Giy$wFp0p^AVJvi$cjrs>zw=ENj;!gkBfnwR^nm@fg;>kCS4xW(3LT^q5u6dDwuJnZ*-MTw zq_h!zxIeEIclq7a8&ZPx8bzqcD8Zyx6$oYT7<2z7G#a!bT!wkzxo_~$J&5ai<}inXpJhSzCq;R~Sb=RTn7@$L0xDq)TJ9s-T)*I4Ip3lCdeBk9+;_!a zIDYXjy15sP=f1m_J*aMKIM6QEdP#?{57d1V%Gs;Sv*Z!F@_Rd39b!-7)iU(=J8OCG z3sCZ!PQLV$lW$HQST(xCK{1NoHjLNDAP!z&6HiTl;$s)O7ql5 zP)21R1o$4hepMYN#8ko2y$S;#)nf7dMu>O6;i2av+CTrm8s>D=@m*mk>wK}tZ`kHC+x*6?L=nB>?x3U{W)V#P*|k_c}QxK;;Q|m@llbY&dJl`2q~(4 zJcLw?zCozzHRfNfgXywbRH?nhx9V4@<2&W9{hhddyBofEq| z4(Uy%k?gxKF2Z_%whp{m`5F`IxMmjL1s&gDOX*vvXuiV+zCFL(`yH2N{Ke8nLDGvB zX200~G|vkNtlL}f5WWQb|h?U!>FV;_}$iy8(TVXb4@$)n2Yp-^)cAUeUW~mgLAIb zSM%@MPMBV>f4=915ma3jn&3xmm<5V|UoqRO*(DF5F zscY?g(!V&CemxPQ42f>O#WD`Nlrb@`HEB1`PG34OF8(d7)84>cxdR*5eMClO5B@sw z9=38I$tnv`G7TXT6Ctx0<}q?6W)IIABkG-KGIOq$ZnmO_V@zoV@7K>}s8EOa7K)s{ zl(Lnj>3obZnGNp3w}?s{HY!9c-+)aliqV=;3L}?NTx%#s*^6T6>zCt%Xf49Un~}r% zq*u%NIeXQEbGv_G*D(PSxIT!oHH7J0H}^okAN^ToPiuTFsd<74>)DRc1}Kxn(~YD^ zi%5Dqb1o!=sOL^Q<{zlQhV(+b5-)<|!cq(!RnE7vQXKIwf~{pC{%$G(^eSQaqaMR; zTA{hM6P;H+!_4_RW}f)bfbgjr`y%lftG@#6#Z=*kAV zn8iA)*l|=`Kadm>+EBOT1s(_#Vdd;%d_G@>#GmE3R7X{c+y)Sc|U5rTU7tgC}_fu5S zdb%IIfOA>ck6loJ)OWtd+j->}B~gUzmSWsDEQk0x?g=|f(cYDhMIUnzZ>oiHd?OsXTCs$4#8yasfLtQ;hkk#>F}_XS(08KWtQ}07U_oushBPffgMQiX zrMrAX5O_J4B z`Pf`oim{`s(dAu_8txg3d)hFa`F8rBy70*21CDcDc|9ju!JaDSk`}bw!;q@aYtUNB zy?mP>%4bf&0fCXUko`P06zXyFU=ga8=3|v<5w7yAJm5n)dU>Ww$jimbrc69;$bzR& zKBoREMcBG(NSD{+PC_#tEqa4Ji{9}qiS>q^AFz=Bz8PjtlzGyQ-mc@Up=Lu`SgAox zdV5K9*jl>xb`BlwlAxzny$Bsxhw-lpaMC9ilRg$e!@d+kQ_J!CZ!uPH%ERHzER=u9 zMnx|Fp5@Ci|3(cA)-*tH&ua`B_}_P}3x>}>ViM=3>HPORpJp#J4?7Zdv7ivv-poF( zNn^(Eql>v~>6i;=*Y6og!X~{a;5lX=V+s%RvXRR3v0FwlD*l#YZc#BBMe<>>BnSHv za}k$c1b3|pnJ{R9+mU{9FCvpimJtf&Rir#W#T3CTw8AyLe8cH`MSk$ewU9E{A`6y zEWopCrTA@MjjI+7I8oP%zpQ_|X!r?Bo4?_}bIw9M#u{|qjedkWk?(ZAk)5_AgA`-( zRo13^R}avRr;3~xuz)@kjiL`D`&kECi@jkvFm+1Dx-XeMyc1-SSl zAE~uP7*zcN*EZE*U49EhvN(s2by0Gh2PpZjAD`=(E44_FE`8(qxsZKxtXTImfib#Z z6S8`CjKbC_lX0>VsW!;boj+s9d%_<`aekm-Wj5w+NrTi+#w$v*q5g$EJM{`7AykOz zDTT0nQi?bGs}X11h+U7~qCoK@-ks{fL|67wtzitBeORuF4Iz#1JZrFD-OF<3nlYc@ zX}l?2pQ%gd)~itVy3M4OvxsK5j;Gg(0u&_h3Z>53ShXev@--=rg{C)Hik`}LV_iz`ClG*nvs2@Jb0wn7zM4AReX;R#9n#Mb~)==KL7dp@d zM{CNcXYFsH9(_D>gtp0SBkfho=>F+R6un^(d*wBuVQLP(-$=s#@I<&fq+-|7Ow={x zVEMCrsI#x8q<005%zOofoHqz&eq-+OUQD(bKn8w%3#c2yIqAbGz(t&rx&J!MK1J*A zIFQvU8)_VJg1l}VC-<+a^jUQ$X~?alx0O>Va@r8OW88*2Rk`T1O~mV{IJok^9Yxa+ zYLbaB7jp46ya@f`m1sZNfW(}3O#J=@8*KYAj`5EZE`w?KSN7Fc5To1dg*u(*s}sBn zDt&XH!~Hg-wdo|y*EFPz3mT+Yv4@&&g48QHm$`%am&{ek`dfgz+Y_NXDF)W&u^4eS z5nZd&;JQ5v{cZ&?`|$!_#Tzhy=a6pZjSQak3j(eKDSa@{?L0%@sTQNPS(3Du=SdeW z#>)1x4+e8ZXPn`Aa+NW)&C{j}CpepZts)ia&!@+C*iU);H`wua=(SoBE-j0KhI=GF zCdKeAKN08urlD+nF7zjtVX0pop7TD@we&M?l>bDEiU2(g7owKY!^qE2l>Kf-(iWcE z_c-t!IEa1n&6txLZAsn|rc_d;L&CGzdusM(+InIUO^_Tk z4a4Q1;V=}B!J_m;%wCv*d-jEp6|2UfmS)@-*oA3hzaia(wf%e9JNe3B+L1Vn@28?P zoc9!+JIpix!+u-RtXF(xMLPU!8CdF*Ywb}|m0^F${1ueCVG4Wr4W`L2oAHMAVavWm zV#AAI%)1th=38M{@gf>?$0cL7TsA6IxF4T+iRO&g=xyl4;7woA9{CfGIXG|Tvq5Aw zeF$p>dEfUtMYFfD$JldwviN3Ay)Gx{_#e)5w9=r=$$M%12BO3tGE~nwjSWg4uzObl zavfu^(Ju%?s{JucDG+8Zq1ZMz2D@ES&|aN`kF}+^pIQS2evdWJwc%lLC%QepK*yyQ zn}Ys8g7@FC3GCJ9>_l=0+3V{kW1*Y{6t8PcGM}}n^7}#NO)1g&FLE^e9p?)1T&x&X ziRm{JQL;A}lFHAp-pdzOmj2k^9)gmYF&NEQu-mCDbU!J;I|ts2-&epXzZNSFH=<{3 z8{B_)K$GiMG{q7m2ci@hDJF#<(SEP%ci#SIKO&F3QKG z^Tp8ZEyE+OJGzpeD*>*5(2;tc^PO(6HO)vcr+w=U=uwwC>(%zq?mVKI*EsXLPn_my zcjNkL)?sGHz|_`{{~tVYZi)x~^m}4qDQhuD1z_WeU<}zD3KO9)j4KMm+OrWzkc;Md zG7jHG6Y!ktj%(zMWjVgvrSl#3tUaAx%US8JCrMt!m<+pEOLkG2Y*ue#&defOWX*Yg zA%c{-ls%oh({Pi2KAF#6@Yi;OaN84Ty1L_cE@Q3zUdRma!XYzH+`r|4zwI87-|vN# z?x*;c;{ES_HR~;N3l8#4bg~m2l69bnL+nej&5~~2G^I~B^yuKzqvW<=C%tFfL`rHF zO-~c0?!+(1XAVacV_i9itz(LS}#~$_BaOq10o@XFot{?p+*mx{U+Tw-7w@ zHtyBlhD-5nyyKtu@_W{!TJtl_bH~9f2YRrJvwsU1cP=!iYStVbB27Aaem}J@-#}Vc z^3<<9fyCw6r-A1+mkC*jTN93f>OL4V*&Q;&U7^+T5T&2)p(XwnWGCOiygOGQUwsMF zE?>f~CzsGL_7b|e?m3>_J8z#N=@|AhWgm{1S?oJG-iF3!@$Wjqm?~6scppAY_o}v$ zpX^Fn{%R&ApAe_LKD{tz&26ku8m3PO!)|9^#J==^w4E!2Z{0`e4aR#luOjL81%!5- zh3dC6P;K50lVjzp$GAXHy@W9kSutle-Al4dvHyYv#8 zt@TeT^tDOL{B8odVorD z#;K`S;e7TIWS3n;lH`Sd_jY66flz1y zx#OVy11LYfgRD!pkTk&MAHE>OJ?VKB`x5IjKi8A*NhYkfim+!*9plakmQ>9?+d{_; z=-L;aGu4&Jt#T{v%Un%DtOd0jJDJjKB}l4Qh`I&)@a8UaQg|m)QLljy>*yz0v)*`2 zG-^%-VfdtHh#csRkmmpCtmko0^5l28 zfQ1fqk3LF+`*zdS%NuB~Dtn7BnM-*slc;09Bt3OuUHvyfD%|iJO;fv(QPB?XrS-6> zFM(e{7Si9Pz{x)WPC2pve%_y4cRk}r=J!sKGvE5(4|Sq-e0T7Ow4=&ToVOuuK}kPM zD0GcJWmKt?bjtx+mBD#{!bDg0ET$P-W|QMBX*&C44D}d`(`QE!n%Xj$mL~|%#dGY< zAl-#}rDjNMti`?o75I6q^k3coI9T8L_Wh7~Si0lbM_?86M}IodX<2)^8Njy(Sxb6n zXGWInc~%*u#X33_@^;)sCe0hjW6w$oXIy3GZ5fK*Ifb)`CeX~K?0GzQB+UpOK^l98 zP=@tD+CQAJXQwZmtIZxA2JQdq)^Ock{&o}xv!D9B|9QB}$T`xNGxp>rV@t{HmbChU zIq9?a^E#hnv{#;UbBgxT%%7V{3~a|Zei;oPC`-@x%1|kL#YxI=u7vjlI-Wg-Mu@TB zbL4PRvJxicnS#{Icf5A?tNITw`^|ODc~8=cV4QFX&j7|wq_>UppF-`a#K@M^c3IJ) zi6`j2rxE4k>5xdYDix>fr?O+)Xx2Q=g1V$YJ#QD1yZbyUDPSCQ)^xHSHiZ=YC(;AMDthLXUaT z%;y}@U_|q#a6aDxHPTf(Kmm)kllnYG5>8!7n)?@%;razs;3G1KHFtncNp)Zt3vqRKc79XH91pkr6b9)E;V_tEfp-Z zq8iR=J#^NHeue0etfwm3NAsQ5a2s`OSx@5&6ewS15vA;$PlK~$$kSp5WhP9euqcS)!D>inFgc+of{j|N9bO=JzO<`5S@*SVu9&i86l&S(9OzG?rdo;5=IP zRN9-!d0(UW{-q>H(^*gda{ljs&(&mpeW=BJASd(Yu zNeaAXLT4H|$NIK9`>Y?}o9b4^>WI`B>(CIHOJ5bHQs@Epc~Tus;e`_9#Tf;m3px9v zfU|R|m}9Qm2TbgSne<0QUhVwf^W4wc3C3ZM8#0!|IP9!Tj+EVDPX<1=RC0>B8%s~n zReK|Hmeir-*GFjfh&}8H&RIPgD`-RSJQDYqLb+qc(t;8(+RnNolW&75#EkB5_7aR+DPuk9SH{-CMra98f9rh%d%06q07$a0Q zBexI7sm)!JTK6ea&Ejn|C2kEhxXDq%tEqI#aWs{xiBN2dAcb50#?2AmF^F$A6;kc6 zJ=DPUD&X*;0BsAivGZv<+w^cl4_;1Tj?OH`MQ3mh5`QCeZrd@J(3(zro}^(j zP3U{QE{(ENrK09NbP^k=e#%l3vtTc6<1y?l&7QE|e?#KXCq((aL1T0y)Wm9`%sWlX zi5%!lq~Kk23`zz@AZ1l3ehv%9e>m)Q?)!55tZGvL4hyi#y&-Mp9h9abN}ZL{Ep|G5Bx0KrSrXz`4c&2_7omzL%&)qNW;^V zI?m{m@OX6!6x&B8yEo7+g~g0 zm`CX5kCYQ$&^q)4T@H`%v+V&QH{Zt_{&`pMH$&EudobU1zIrbFwr4%3pN?eu-yYC3E`hg`Ecr)#Ak-M-O|+2v(`Xa+I@5^(KM4DQ~GMAh6d z?CcGI+a+(Tzu^WI(?_V^^Z-5+?_+G#UA_Y`$E=hmyb6Srrvm)=6 zoCkQ!fVRoAXW--moKMe~v)f|I-#MB5J`H1S{ulJrRzWK>19!~g@U}ey+h2!apI{gk zuMNS;y8)=a;)7AoJ>Yfy3A#mGQM%_bj>tYjJl8F0Vm+M#^9dUr$u`HHQrV|aKf!_| zMNKJVyB>+WJ<491yXcD!$iRLs?VdcE`kQ#S({0B46L~mkn1GcQ+vEM`AVFm z#m|iB{u^y-UUrByq_$9v{W6*8bT}WUnlx3>!yc7YEY8kQT@{=Ap1L0ScCpNWB)0s#B5dI~|Kt zAxZdfEe+Yn(s9Nh9V%7NalY$0OvYy*vpf^NL$Wc1>q@3FkK&mV`P^jQla(z!WKCU* zk|{~c>5n0dK0vsR^%<9TYju(4I4m%dmG{2mIu#+DflRu0KuIJ$Sg=kUPA_c zbr#@LT?NkUdWEkFtuSPr+pbgZ@z(V-64ZL2@7IU7tU1)@UX*!>Zw*zBG_Kr^rg9&s zdt}DGjQVseRh3?7?4tM~Ybbw`49(3KqtPWjIBijb{-A7_K1#tk{ziD{ry$iQ9W;ZV zZMQP0&#r?MV<}}e%w2i;1%5I=u<_t8I0y;Q`bmSxbFwhq;h9~)ocD$Ejx=eG9p{i) zlGSGu5?-iFI;Mvyte*H+!mt219+!WVJ~0qf!hwTuZpZAN!i(S_N04oLymQ`@~ou`;ydi-q0~5J1kDW$^aifcc zcr8$Z4}Ilux>Sq9%UaMG(TV+O-7u->M~sUAZJES*h1o->;=*uRH%**cS-1F8nD_3P z|M`4aQ_)v5nzC1)R=z$;Z{P2rLBgv@CTRwVvbOW2U^l+Zsm1iK1&}$F3*y^;*ncA%+`PhytOQK?cdo;@w+cnQ-b{1eFQEo0X}Wt#n7k7^vG7_Y>NE>+ID)xV4#lX+ zE=B$IQuMqm#_XD6$XS)6Z)z>B%xuPL3C=#)@(KGs^gu+DeP4|PID2&vYvhG#%{RtF zpYx4yy*+isFz;8M@BBNDQLVNz^Jg}am-Qmby(C35c;?NhYlotB1@!X@VKb}<{&Pwp zAz6;@{8E^Q@qeEzMB}OweBED(;MRJ0>9itWp%cn>pE2?DcPudbg4_>x9IZ#{ffl3# zA^tae63w<{T_W?mebmZr=-MXESKgvWf3_T^KWDbkTH7U5+9E}-QwLILMLo8fr9-z8_?vLUl-GtOPtrPN*Q-y*x2GjzAzBuHwEF!r!X8Wh(VSH<2{?7qb4*9V{+NEkTKMm3rYdi zGKg{A2*wtC*bm~*Ni&LKjg{@eL$o}d{Wfalso>sN+TH#Oui7i2)}4%>K~az&9gV(( zc&v<1MwL=B>nCHOw>A`OivwVh7Kp@+VHhe913~LV_^wHV{FzL~jC0Y#y}81;7##;n zaF6TG2zR8Vx}52B!HkTD>++I!h$6$e54p+H3FZ;Ux%WfYig^rolHhC|iO`X(0r?Pz zrUl6uCY}rlgIIL?g~D@FAiU&)@X3eo;k#nk>oXB9c4_EH&cu|+Tr6K!2!s4$h-Q^w zK=UcGP37LpS|_zk)^=;?lIHV6>>a;}Ms&*aY(AF#qxz z$gE4kqe$i_@+>#bFATO1gD~PjFm8xP;4;5&r;!Z%3F+`TpN+zZd}N&|MxtdI*1TrD zBR|j8JZt3T+R(9i<}~e_F0FA;p+#MrXh7ydvT7SkQK|iC;9kFVUIGf8!Z7JW7`>ue=?OIgIH9-DS zGbVQM{5-*lidY|^$y&DUMfzmWeT43P-%1VaZFOXp6y?lgy>es)ZhVY`|E6FJdk}=K z1K|j7i^0j-c=+bA*6MCFd}O2G;TeP8(Mf12NQam``(LJUe=Mp*dRHA@k7+?_{aXw@ z)dlk|z8&!$Md7n8d+?s*+)@K_^-`tSDLY6ZZ8?=yOrp_5-kghTV{c<#%onj%^9t-6Q@hIXMI-O^}cUDCRDyhWPl2=IeZo?sO<}oyUgIU)v zWD5?YkG%W#xbrRalN}X{SrQug?i!{^zpw12hmN4eqciF97S_S;XhN4u5_5e5uhquR(TB0JW0fm;B<^lEr4@+C3Yk?LVrUC&Yk@Nz1lwBR|c}K zbugVEw0=Ko4pNs`>M&j{(w!kcG7 zcw*>R2b8{>nXIPlvC&lLl4XHYn_7z=Cul7=073LvFWh4=YQa1DBgv6H%?%U&#< z3k4}{IOph^4X3$0w?9oFZA_-koY+gCbG#;^Uc9*?Q#siUyn!qZOotH zWA)Mjw7YWaX zfn>gi%{lgF-unl!69nlN?`4sEA5_ZWU7xwnHp8vx^9551KcGudS5zopdNX^BFC@iz zW2oBe8{TbX9*k-rA{$)MTzwyl7C*rG>yMGx>VdMK&v5iy7@noX<4{mKHn{OO`9T@N z*VW*tawD9@-=d`b1MXLRgYqBNqwpST&3ou#eS124i0_fk=9Hy*oHa_SbZX^J@-$T- z?a0X#6Cp?no>kECjKr`qcO3q94~02*m|JiUl>G>m-5#i5eDsG=82Y$ZjQ^U9vPl_e znUsr1NkveZSiznGFOjpP32(TrMFjig^0#9S>x}|0F{gp&N(s&eP!-dn%Od+JapO7~ zcVZ45mS(M*MLWLQq++kH4}2Mu+M0C>4^?i#W9(fl|Md`ovpjJ4tS`z0f}k)f4CSnm z?Ya`ee2Qu+%S1va5ogpiOxvFta)DzD4!H2fAdZs09E#+LeCLLN}r$XD3 zH%V0kq$^BE!G~S&m*X zjD8B?c|NH0dWO$^0nk+sL1TItu5;anGWH1k=0t-O9NGWajz)!8(Lxz>ny7r7E?cRQ zSI929!}qeNV>77f%TU_2?KPf%;{L~2Zcp1?OjW;$#qrm%X2eZ=EOSBN^*gYVW^54m zFd*zMHv8U%cJV!IZFzv+uaA(!IBf#go%4#ZP#Mk;4RfRmn)akT#D>Oz^9D~EQRWXV zG9Gl0OxTxq^-eho+&YFbU;M-_C)S_4XyyH*Lnd9Y$%Bv6IYvBT!&g+PLdJT&HSCQ*` z8G{Wk@NRep-dj(>pxqIs=}y>{atcRAo%*K}5qi(sGE3gS4V}m+$$@0o+R?=@E4t>! zZj5IP=uff+Z4cW=4lV0xgSsr4D2}D$BL|RTSq&u26L5=s@oMQKYzx1QzDdqlz4kh$ z-@1&jvUB(<<%AGrYh2Q{K-G8)SZ%aG!<3W%aOeLxqq7-z_K$L+TZ)dVjh}*D# z8)vDw7?Z$c9p=@mklp-k?BSdW9cEgm`o{;YU$XMWN{DCph6 z5z(8>qq_tb=Q9v(w}rzvOQhUBiFd=e<`eUOxU(AXqPG}#{=m5Nh*(F?o3keq#+@(N zSkj9HW_0ScK5ZVWPU_lwDb;H|U7IUUV#6nrD*Kj5m9a13_c9z=p8_G3a0FNR;p8MQ z9GvwS3pd`z1BolH=-6@!Ff0#l8VaU z81{7!f!QD*j6e4ne{S7I?OvXr!ww5)QI2X zV0*eK$GPQUmQ?W4j1Dsvn$g5_$RTA?WxewUg*9~doha=V$nf(tggq z%iT1H%7q#CH2aRg&0TP;Y+~+RHJ-GWKuf6rqX*>vtJ}wQFY_E7ck93WVb(se4r%`l z2YPs!^PM^C@oD}^8g#*g+ymKTbhbJjmpwqpGKtspN)5KmG(mXzuu3Jx} zg6HF??eu8QpX5wDHTD@{Zo58bK>a!M2i1wanDyk_zq+1WcRu5iwomzPH;Qq|B~B#& z(SepsVNLrZ8}gB6zg#mjsy$*rJIb}FX24-`KC+wa*c;gUoA4p}_G1S%@!vNtK!L<>$&)RAN5-T~qxDCn zX&>|d_veoz-Ja2uw28B{TSm|+o1vsuC`5YGId5k5|9;Q^@i%H4*`tB^8!61+_*m#j zuXj6;(P}#~=DFvuHfJxgzF@qP9@&(uQP*dl_oTK`&3EQj%5sLaHfxQiP9wYLQdGkG za+TyL3NVo%Pi;{eZ9R;3b_h}M8379Z*$*eNADjd7KRvrRe>Vd8yHQ}kn3*4YaGY_Z zP}VwI-?AkI*3cgqe1dY88Bs>eF)}qhLLv8d(|W%R^j>)dEjFD`b0elwd*(O_U_aTQ zT_fnQ@Gz2CAWZS)f^?ttG0G7=FctlP+UswTyy7+7q?`VICr|TtGL^9$H^yO2nadD7 z#gS4sa~?so4SkxzT)VYq)SZ5u95ppbnDz6&JGPR{4fYX#E=RvCr%|cb7}h+GAjO|T z)Vx4|7JBwW;bIT2uxD?i>>GRyuE$m93dFZDPy0{-`qOj&^#v8~3vMZl0W%Kk#5k;` zKWjCe?P!XZH3i&0NuI-5j})y-eQQ){r}G}#)TBs1^On#Y#hJ|WA46g>LrGEdFFvnk z+^(w~W0hOrJGLHWmK7MTQh=JR>DW9q5pFkPApJTL8kOPya9DHhgE2wOE8soJnE$rR zc&@AtwWR_3t>{&_IkoU^I%<$MC9FL}BN%T;vsgv?#;lc}IiCGqhSG7~XUm7Q;ZsR9 zY)ngVnD^}$=Q2<&o`hP1NaSn}MEX%54E^AV1~CuxS-9a7&pEfb?`z#VMU={YpZA7R zJ<7;_KAgnB3+9<@?Ue#pfM`=bTR4|1!g!XvS*? ziem3)74{1F_>*&i>W#?c)-ihgM1?jz-bOp@S8~40EV@67-w$KXvQlcm(TZ$D`LVZ3 zO9-@w1|Z~)FFKBS!STK;XH4A1z12Hb#?p5;IkriM9GsI^D-38gs3pI4avs27?u!)U}1ll z41Ef*Ego2`=LWMgj}hB)4+fephzh@k&bOB_x$F}5C|`ofp^FIRy7wEH%OKA)C1(K` zd$W(-WA-`Iw4jaWO^B>@Nz$D?<&N#3@RO@(!5sF2pC?8SYM5)wenh=DV^J{E4|%`b zaUjwa!Gf;XR{R*UM<1bTM?G$Jc6XGD{G^kp#9Yoz9+iEXTn2lw%|Q#;a%KYeHVkC-o?2sywh;qQ8~=#%j3-M z1_yE+Xh;1*{2ZS$<8OyP4Pfo!yNkPN@lDWfx7pOC#Q8_{Uy!F%gr`p!i-yl|k*8*Age;kfE4=4N|hqKPL0*Lxw#_$lgMKE(^}2@0`{)m>(e z*)9jV&py2ynDf1NwFwnG)FFX76%yaTe$&U6QF4|vU2G7dSx!x;>q^BbKgP2AJ#hK8 zD^@Ig0s~VI4BP4hU8MkA=nqCzY$*1>4aKn+p_t?vhSwGms1J)m5n`~3>n>)VS@9b7 zMhvqj`Ce;6^$9AErqSJ0l(2>~gSFu7EZCjPdxDVSHaDl*(U1T&WTFxhJHZ;C_GI zk>0JbBkRkSw12EA*>2+dH{bhRws78j@=}_%Xc9?Z6eR72I*8~dW3odKGI#i3nus^P zul0pbZ2*othU4w?cvuIfV;Kgv)k0pc2}8!e!Pd@B3^DqG&33(rJn|O}d`DWb zgta0G4z%=+4SD9CAkmKotYg=pL4WqpV{MwGF;F1%z6>10DBYAORTTgo4; zV*~NBBm`wWeeH&FOTOKJz(LDXn4$t$V9L_hP0}M3OKi-EP5GzKdyBg(IXe5VOAo z;u_!U)?bW3+MgI45=+L#g;|{ARt#$|=8J~2=Gwgj{S}{ZMf?Z0aW;JH6+!xaUzi4e zXWw?##e6EV=l7U-xDBQx{aBaW34x?a< z{1JlYH4(Vu9F6?3@%TL_1rKj#VMc5*npJD?%A7R^&K*z{`;4cY+xNlaFYXNVtb=Hk|uNBzv!? z4<=>qf#)7O(L>f-t?09$TmfTBG}WSX^ZnGddOb~aok!mVIqPRsH=YhF!K~7Fv22G%zK0>7vOwK9Kvoz zAkK|%So31xo)?e3zv3}bG9E3$3GBU@0(Xlns9Y|Bon9q|2GqmO$XLN4u zh3=}qc*woiV;AQE{I#VkM^Dl_YXgeSS0necyU6+bDr((1gK-oqTL(9ltj6y4W|9Q%&% zN}u4we3x6iC&bG+QM{KewT7Ia{hT}Kw_cV0TJ4~pdJ6Q+a4PlnvTpW#D-;~Eftpx^ zI7Q+2#8~9KO~B~2NqCkMk5i1J&FG6n+T$2p(n-RaV;K-y$r)m-^)t|}Lc#WW+}+rM zZ~c5LeA$JiO^kOxVLX&?M;$4qG%Mp69o9NXslkerGI2ibt`es(<^$?)%Yc1#0MunX zaq)u(&W-Sa-4TDJ)dfJ_#}9^seGppv6!YACQ8YOSN11yRQ51(9_Y@4)$wV3Rb>v0M zQ2ewS!a=VPTf=jPvLpL|TJ!hcguagBTf5|b3i4k|-yLQ%pLYc9O?ZQJ$5e=#`s4U% zPsSF!u<^AoNQatmUe4aYLB=E(ph>}__EHfOjD3&IB$?KsWOlz9X~s#Ieb5iZ z<(`l{{S-Lrhi|ij@GU72!>xU>Cf6NZmt7Gw-VH91UZ`2@hc=U7h%aS)tv&|H&k`Zy zkcNo1OjIl7!iMMdsCN#u;fNLIDH&0ay$1D}?&fdFDsrisMi(~;a|XgIJa`w6E!y1o z{(2(qus1GP`9qg~rlEDL8_@P)Zr>B!y#Elxavni-j63|cc;o7Pe~i8rjK=J6h!14m zom2uQ-cH8*f>fyFvY!iUG@JyjC}o=w6+BSqeD_^cyJjVMmQA6Fi-lkOb)7-Fedi};`TgmNc?()WwY<0tnMC~@*lzKo;!vN^2W-y&v5B< zAZGQ3z(yhh!Rw-M=VKJcaa|W*2dZGbfP0=Hf0NZ{=i42mtG|M#8cWm8>jTM5u@YzW zqHxgK8&igQLO;YCXHNw1w>AVfCIsQ^YHwt{c?799cadj$56LU}X4vWm580n z%OoRlW&cwQn(TqIeNVZc`QuM@Fvc*xeK^+}?tYIU5pa+F0Pn-O@Da>a-SPVVQ)F&@ zhH>EmDEb}@hlVhI7bEapF%ozA?^~448Oy&dNo0``sVl0}5AB^~p16X>t4lMkBtTwx zfox&!JN%AJ?sCK1CNKON>kqY~!MJE1giVP)*tYly_M1F_C*PQ7Uvh==7*9ai2TJ4o zG2bHywxh#vW_=_i9z|oQRSc~7dEVaUK%4hikx#M_EfUlq+ub}+{hA)2yE$S(9~5dCzd~Y7iEm;moMcI4|N0!+nAHaUci-*DzK%%nJ(4M=iYI4p*nA*lFd*9xlNM zKek&a~x;+ed4`G@)3rpJ;9Ndrx+LDk9|e_tg8E?yV@JER$k~3_lEH~KgjI> z)Gx;&Ga?xt^E05^oQva|i*ek$9L@FYt#Z+k4!>h9tgR_65j;lIy!MmG>a`@NJd5m- zIm@W+C05Ug#6z?HW9hoXa%|r}$)-Y*nN<>5S)tEqCbA-;3)VINrni$K`pR`@XL8`piT;+7*q#J+Y|WngY?|%*S5N#b4=M zWLan7u^}%j^mEgW)~{BA)q4> z+zXp)PK9?3$W~@Inb~rd%QOWV>pF}=?3*y|&N*1{_b9v^iIOAHkUE%%qx>1AdA7^o zJ!FGV4iuDfIq#)_HO|br%a-CD>qUP%-o%o#_t19uG44e*A(MTw(JA!4}jL{#UvA-LswY@mZ=l%!w zX#VJRpzqAH-??Q-TOR1pttsj>v`~qnZ%NUen_qCfUnS&9QdmJs4 zDocZ8X%=SJL=y#HO?seX(j4_;#bhWF@L)s4Gqf1p{){Q_V3 z-1lS7>ZBvpac_z~?;UQF3gkw7aR7dpK_(!zBY%V z81vn`e+p^GANJl1wWGLhGuqqlFs+%ghvtr0N8Mj%QQhm|v}81A;66?}K80~C&voNCTYYv9mMgGd z=)Y z_TbuyKe)>Gvcm0*0rSjJwT^SX3am)I;uxLxJ3x9Lw~ZvB zV{90PEcSy`m@i&M2O!%p6eT9n2r5iM-x1kR>1C}>)CG8G@+{5I#X8lyuz&mzEu7ml z{bU=qtmBLZzW-jgVV_8ZfL_3nj%o5atjYX|(IJX;+eJHU)=;;n0`-0xMuSY6pyzr9 z4T;e(8|se-JG>!b;e*pM0hnYIic$SyFor#-Pj+Wxxbq2o@jnNt`VtINsD!@iHMA+; z#`N6}P~rL%lK;E!hr3Y89{~~jtam=NrT#_cH0|#ZiYVPn%W5`~KP@CzU1{2}ob#gX zFGDUc1#dKhu%p-uDH@(|G4O^;m_NE_hoU+<8W*|}F;+hvogvwn%yWTh^a;$Ibe3oC z3lRLQK$pd3RPnl2j1gvObH_$2=Zepol5GJAlg0 zKf;Ssg&5cxfh!I^nD^WrAD9QM5O`r^tsg>LgW-KD0+)8jaF=)-OjHsucRBkrR`K3F zDHHR)=HN+bJ{B{cGqI9!{#?#84{@ef`yD9$3cts5Oev&+ak_@xG<|_8St-xuJkC+1 z^!yW4(<%{hIvJ&2fzW&DflU+L5Y+63L;0Q)0Pl#%I@I3F1UpXE) zx4{cv%Do{o#Runk-I;tQ6;I{vmQ2p}+{)kedpp`S#EP^g7_;8~AgPSqNgs{Y(9^++ z)Yg9#jkNB7-^6Q3<-K`eP#7|o`yk@1JIn*!FrvW?Cx5zO)~n-)o#z5+HAgrL?Vylq z2UTAOJd<@o?HxxveJtP~ySL?0VEFK$NlC5LIGD(kU+sgt?> zQX2GPGHI|E@b0B1%sX4kUWima1HPDVup`%@f?e6k|a%tc+-F(gE_Z<~$w8 z)!Z$qNCw(cB=+Vv#u`1u>#@aHHIU~zv1r!Mg`lb~0P=TyapSuuhEF?=b5-`ZG~N=M z*PG$h5Hs}5HN)p0CjWTnV7_~==XXivJl})Yvu;J+iL}`7oL<4*tcfOcT+x8;JlCR! z%IdUWo-$p~=Ux*JG3qm`8%FtcP|__$r*j@&Jxf8$hd2yd5rv=qLowo^AF6XbVBE(Q z3ZhOpvE2cAIrgX>XZMeH&i=}{8NW+YyM&}#CZMLbPSn8X@hir%s@HOUo`@lx;5qYe z#&()FYz-Z6oktJVCX#6XA=EgAId8qkoC|*y-lCj$$eO>;`MI2XmX4nV@v!a=M_ooB z2AuQ7?pkl`ZSumlTb}sCzi$R(I%yR=XH92ss5EE1csbJvawK^X?(ej=q=s3X^-|9L zyLb1Hs=7M;@m@|-_9#$WpRuI$o4Y!ue8Z1Y&XBXO$LMRE>9hL^E`BRRuiaU!(awX> z#x#^9CSc~iScv;ZLy2`R|KrQ5nNMlqdDe6zf8Y8-GGN|Wj=8ca%scN|Y)$h@OzE`N z5i-BMk2dt#P7SNplJ)9El-euHzQr+A_HhU;f6CpJFBy}zVh@S;OB{Ot1c4Loo&c93$(u_j_dX>!C6cZe1Yora`EoQy$ z93y&FtVcnryGee!I<~Z!8+x1Fth5A^W@) z+b=)Gu(`Dm8&iXAQ<%e9{9o?$dOjOEUND#A%=kCY0FUwnbXCHcJ>L$rB$jzAB})?U zU2!z$%sf!uOP!myQMo<)2kn+ppT3GTzleR_qsEatcLQd6iO~$sGYKx|&VzxXWYzNr zzM@}Y)$sv&BCqk9{l^<0|Cc-6%na`(WH9PK1=Pb%{ z6ZVxKCPmI&`TAl9-<#CP^S26Z^i-nCu?nw5fyjvhL2*>dfykdkAN9e!ztQ z=C1A@p>b}y^s{{@-D7U$%A!?tX!jD@>M)0je#udQ40k`aO`yupW9fy{7@B`ZlIq1q z(A%P6B+wp2V@pIyM_Po`Sa!<`k<#+j%sVZm`=94hem3V^_fDnM&PjAxaRR;V8cW+v#*l-! zB)#b1?B$A~)OX$>S{&P-#47v!TeqIS=kxqM$ME-jp+!jkP3-qFa;ENmj&#Y*j;_R5 z(VBgxG%N841#Z_R*{Yqi#73R2k6T5TIOlqD&>XtHYzA$PoI+!gCeo;`aim%=P34EB zsQ=_q)IC<5jQ$RxGwhkF6YEc&YX7affY%jiuWF4@BluFb{u1H5sIv}*fOil3xNrx#Bn;|b%*f7EDN^GAZ3Hi%JQ?iJm- zmwOMd^(WoPzEnBq7oK%@V{*k8R8)Ta$HNxzzOnf><5-)RkL9zWoM*iIaH8fk_L_;> z&=e`|V)8VkFOsZfDBeZ4&Tpj8XH@8(*F16@%l_u&W9hIrceO4ZLPLfRAj|drXm?{D z>UaAWE=Y2&pyvm4@_T$my%BpV8j!=9{@=V$`td$#c9C#Ec-VM82O}bxk7XYA4D+xO%)^$XuvfFn znoN(F(Uz%4>7{T#HB8}*x38-y{2=qNZj-2a#4vi&^bJQV8`(E_2PYW|HSQ?GQHx@z zbB2-i+5$ws%D{@HNl+_{MfARCRI(RVb`EEe@aNsY=l(Z)z6aZ}mZ8y^PK7&C`93>3 z-O3zckukmBeu$vV-FzbI6rZ9@(@x8hO^pP#g#CnZa09IMu3)bDIW$)1!?QOFQa+=0|+s=If1`|E5Ez>xV;-lInDpl_E}lYY+}3agYRH_lwh6mNte?GpFW6(ZwsDxPS?qggkG^IM`2 z`zj1~hXr9$qz~+8dVnf?=sjtE`s+Ai`Eu2NyF2Oltg9~7CIf4Dn5wM;a3h%$c5RnST z-_`&u8tjYkKo7oK9Eb6FA#|05h~Fv1_kIGH^SV*@89VC{QpGyPp-UXeN!E`1XIiq> z?HK9u+4rkilP-){OBedhCEa8x`egl!&-D8^B6Au`wR}Km%hOsdJ(&P3&Zz z!Q==fo!d(iPt>V##1i@z%)OuQxX;q24%JIeLdbYvu0$y8WP{LmX)u;Q41>;co)of@ z@a|6rCQZ%50YM>tI zlAVv$v(F)ENI4?hE+ZrJ27Yb0jc}t{oVI_AgVP!@D)bfF7+290XU>fGi`)V`isGD& z{n1BBp-hK<6>cKg&?R)sZ6YWog>BxjGrg#(m-nlOz`Ex+y0H~grhLHTnI9@G;ov?4LfK< ztx+cQg*l{E9h%fKZZ(;WQebVK7`66o!%MqzjGCE&0qo$H!xsd@|CM)8QDJ z&G}5MO=TX!uP^)1R^GsfJNJ-o&fe(kM#w0&;iJbV?6v;}+4+6QoO#J4Yu1>VaTo3! zzUQAjMon7|P>{F=S^ii-GIfkEaR1ldna}X~IqM&CQZe#q95O57(5{sP38OSDyO@bx z?KxQPQwWhm=OB2?{xA0L1+w=%V(26MvV4K}i(cc$`Hx7y(}NMriR!&)O~Eog4|qmz z{b@-`HHMrcw~y|rZ>9$6rL=YABvRt{XY# z8^$AYF@9?S1dN-B^^`!vfIa5+x3H@K~FT ztFk%xH6;hS&AIrPc>?h-fukR_AL6xc9}5&|VC+3^F)d#W(FFc&i>rXaXG8860W;J$ti zyr1S`czib6|D>a-Aq|N+nYhJXzGo?{KYm`!TvG+)thqldsv1h1nYj4uBP=apKJ+PP zw;vRcqz(5Rm6@_XL!Ww~NiHu}($Y)vw17KS%(#~ow=cpzJO}FDDOk6ebK@Rn;_JIy zjJ3_f_kt|!IhqO~^P%6zq++5(7EFKS!!_qL`oBMqw`0mN)Tau~3vRG?@-_?)*20ao zNCTocA3?{4E`K$mY1{WvkFPq-OIb+M%cSYDOb@nHT*1xd=`a$H;OyNH#y=zQ!#@uF zRwrZYk`yF zoGsxQR>_%0I$D#Qz>qe`@1<>#Y9!S%k6!pnQXIc?>tioM@kI)@-44gs;b9nL5QR7P z3HW^~g>_9SkmLDg$|KfQIfdefdl=f4qtKNXhw8>8Onk)NioV%sI+2eP7f)eRNf8PL zm15Ni)&)K1tehe%O6Go)cWPQRXAq}z7bsFppAjUshC5iDO7UqxGAK73mA@H_<&2by zVM%Deor(e{JWhq?jbqE>(XT8C&5M}Nnv#iE zAG2^O(S_&$cReIpQnmhJ>W$b%c3U~)NbE3l0TAGBuR@WFzp`H5zlH;t=f0+~TQZ z#Kfn-lxGcda-t(^`fPQ7pyxx3CF*XctjX6$ola<| zF_%&ha5o*F{8Mm(F*7B`LZ_bd!dPd<>{j@qJ3RoZ=Yny`EDR!iW-5G%#vGeC2xXaT z<7Y!5hOU-vfFXcOR7Og}2MRFz}%dmW>U-wgLRzCx&9JVgyd|oG@K77Jb>z zuz}x&M*h6F`0i)E&63{N8BkByPFk0~k}lZDQC`v@>L}#A1+8Kk`i-A~F;*HSx-q6hTg`7k8#Y`uA5H2ejz*zh?H z)BnanBZ<%b{Z6b$v7{bux!FEp7r8TT(ehc2UOXH`^@dN-KY+7CpGPD6Pzbb~!eF)_ z7Gn;k;NIR0+*+B2tkzhJZ48FzA76Yi@P*9{fAq8jVF719b}Ws6$+aj%TQMKb9=zY@ z6X7GDgz=f&%^u=Jt&IEHeLqZglDjERXEo(blBWdjAJbp+6v{d$@T@PtJ4=H3nGD5h zkr=25li?wmfrMXa=*v8gbX^!WX$L~GJ^)I~f>Cui4C{79;-P2^8gIv8RDB|xAEw|j z((&6m11Aeve^u{9p;=ZG*K>pvZTHXvskJm~5NBx#hjIq>GZcS3fwhd2Ykg%eQ)VzW zv2LMyRua-I(s4OB9YsTupracFWqy9I9}2~V-f)aJjD|Yz9TQ6v@%Bdw%#t#2m3hQ{ zFY@t7b!K5XM&)QfhJWNFYyj0}tW@7X{*4Ml##NkKG z2Pd$HSXhQ)zN6R(oarlj1=gk*k;zuh-H_Wrx?=NaVDJc9>G=veIE%(?{)~zKuxRzi zz42j4XP=M8l@$C8O2@}%saWzQ5eh}DkNuK>QKwmZ%DLD|L~yT8H{Syu`El-& zFD43up<~TBcNBluS!s}HPlJ>^XO^aK$~uQH9=FP9X!%!l?Oam>iga#uM4-(_Vn>eON2;xr}vxSFobH1}Q5!k3)^S zs5$SfW^o4=B>#j6&sV+O0&)~}AZcw28k%yL4s%z_pY3Z%YNI^OA2f(oXV&6fMh0g5 z3V`N1PjvtEg4Du51SUsd;(~a*Volbl1sOQdm4jC| zx+5WYGQ^Yh10L9E>5JF8A^52f4XZwh@Y|7wSNn5tx$`8(JiLIw<(HA~QiEk~^_ZXW z0uv6sfm2Hd&L;L^rfNS*WGv@*IOk1Xcckmt)@0Rjj0%f%=}znxYFM^}=4o=K`1>zt zA6bI&1EMfj$_qvbZg6_&iKsSzdQ|q3Ko~ zu4g_&%;eY5Iq(@D8h>EtwZ3$Q?Tt_`}$gL|yf1qKqbKH-dY_*%xt8 zguZCrgoa}hW`6d8Maglz&vL^@Q*T5C2IAYRa5S37!L=$C59IQ&fVC+`n&lY%>?*8V zZ{zNtde)V`z+=w4?AY3YNZp^f!&qIq1ncTo^L>yr0WX!C(=zU7pJ$*&TX*m*TRNK- z>JKO9_0JGznTNXwLf%d{ELrc0iR;~=Ywd&Yra`d!6oJu;6EIFY1H+pOa7&!^eYTtn zqILy(%D1qWGZHRFG+?7M`>p&wK$`ErHLEyZl>GyD?%GolbD;wd7*UYsewt*wnN}QF zOvwYr(y5ajF!L_OnmbW=Ams@&BO&ziT<{~o9RurqkizG{s*8~zzDL+U$Y35YA49WF zV^#Zkd|=(O>h;Tz9CZuAgj(EK^#m1+#ojk!uFiz>COHNFjkv}+`Ou~+rU`c#m6 zCfdj5!n|)G7Q8wQ(G$h!!`W5w0hiIjID=h)3)Q}6eDEE66>V+FQ_`GX^cN2xa5wnsg4NP;sEiPLNqUSNB=!O$bT4!53stB+ODt%m-D?l?JQ}0k`a42bm`%j zt<>lJGIAR+nV)mcWwoe5A@7UQ%KmUkbU|sQ9b~lZ@h)G0k*&D@m)4M=E1x_B{uWkf?^ffqKZI74i&9pXFe@v2h*$6v>hrsn~Zqn=na z)e{3eJ$RP$K(T}ueC2&0*~br_7yRJH>rP|dIrgoP(qsg5hB*bNee7RIHm4odtiSHV zUB?=msWE&h&3rhCx~(`1_Q`#O4#>msi=i-x@j(9r0$8kd#D+IcII_?ME>~TV6X%MM zV?yMsIKln913Y#(Vc{$xj5&owVu$NL-dTxx=Qqqd_cHHnvCWBWSQBM7z=|$Q7;`4& zK^oAiNty%MOQt`YT1Ji}!^xj{9=?qF*mS(g3PGxe7d{SiL+cY)Tw?6grP~#A3|wFv z>WH3T8w6gl#OfAHygFwMo0+xQLHOI`4eT=*{O0-qs5bo1{!ow1c!QH;x$bJpf|0nJusjSy>Q_Ay_j z^w@-y+ze=hwl?+s&Y4Mjmyy+KIcnWMf(A+SqUsoPeY)3BxcodePvNtpGMi_ZbX-(V zMloaT%BRDTP#=VS-~CZ~*AM0!7$0RV!Y^KTRu8Yq^PL{^3yD`agZV4Z@p~OeGTVkG zaF)*xHTKh=(xojLJIH&+S`w*PK!Z82{|R>sB$xii<_)jW^!@<`rQO7x5YA9CFGr$6 zF`x0LP;JiINzQzhjbZ$7VFKQ@#lbc>_8;#&><6EPl`f>nd`jm8A+7g!rjzR(X_vSi z$@^K-_ie_s`&lObDFq;y7OrWK4Vx-PmdB-DNnDx62BbpnLF#0K$ zbIy-x%xx?TyNbQ{%W;A8H~)&UUc34fN}B)k{|@Ipy7V^d3O2H?K~G2r|2R{B_NvU3 zvnSn|))Y}-N>w|K(7WgRXvId>8V*=T!&(>9h!%OW?j27{ze&(0&Zg@eEJDi)exdJ` zugKoY{dK3`p((tXv5BYH+F6SopKe2?h;#0g|MUMIST7-v7}KmYGM-lKo;J8Wmo{PP+1cl)xZ?wJ$S zJ2_CuOk0YNwV0j@wnKC1g2OWzOfBb0%dBnM%sfCXj*QSoWZC z=UC2YDm_1n+}<(2#od9oE)S-FYXc}PUzElv{kQH~ey;lQbCtu-)tV^ICEm|Eic}{m znCCzzSqFY_js;yDbc`zG^=WK`7E#_78aQbUUAwe|LN_Rq{mSWdOk^VUl#Sv4B}pnv zxEnE1f`XF7=<&s&nPmpN#=ny9lK*jB2x`1GUfp3EZRkj_HLy4H&tl7{(P$KBS&*pCs1#aBnjq;Q(@Im zdR91?O#1Pi{aJtNxWl}f<{wmBe8W2BPR{iEgtLP7e>`j>&vxa!Z)|2=VHfkTR$0#U zfw}#kA$By@)`~tfnb2q3!?dMWhb~xX5G$kU;G)H}TUVY;E{~`AvpK6#axne$=|?2n zhXUsGA|c~D#=P!Ar|w5c3~$4QxF#r!=gz4kPw>>`(LWyc56`m1IM%OULK;&mpg+tb z6`ge;?_IXE=%fV|?L9`PejFs7ReMNszB-97TS3vAXOmPBcM@FW{pwIZa-QCUu66AH zXnlpuZ!ZzN_60QMo?s+R=}03@LFAk6$dMk_@m-qF0lxo3r^AOEoUFn+QS0c zZR$i%bJ_EC+nT_BSowuVX>II&3e(t5KRZIt1BSnDc$;Bzvms0djmCt=y1hlK$-C>3O5zHb&<_*{tPbI`_-Id(_Bukbuu z5bH?K4>3phfpsbs##E}!ygv65<&RdUsCs46TqH-%O(W>z;y?JE*M#S3)p+e(hE=9# z@p0ivw6jmcQ7jj8q%$ygW)h~pkHMzoNXV~`Kwfh=`cn8m9(FIE`+awDUcfXV=Yw+w z2J^59jqI<^GpEJQhE!LnOM_arQ>Vjf>ci)+nmubIj74ZbMI-u*yNnX)lbHM_9e-t$ z;A9n#3qi4%bT$f&tHUsSN+6m#d=OOR1#{*oM=taN$~}?EpSOnJ-MjKUbChrgua^_$ za)zFlgf&U@_?#a>ArYnclNzZmE6}L{I!!d)Yr1LfJ_m^=hC>IM(#o_*gFdQ=v zhKm~Wb&Rh_&-Q0**c+DP+#y!LnJTr;SjvC=tOWR)i-^yV<2&Z_>=a!26394_2OMpVBka;~Bo6Yx z&t5MyZSiHl3hUb5`k;HXH#UCq#Jd?@s9-(o>ntC{@b7(+_vR<82XbM3>|NfY&P&;m zhp`2HWL`fYbw3%N*-9<9afAH z&eQkDt*#)ny$iwi(V_6$6@u)6A#m&rf#7)Z+fy5tM zQO#jvx;LKliQ0G2g(~)Y{g^>RUk;&5{?8GiehMGdBXNnJqtYAh5UO}!;bw2NefGz( zqEJj969ZetB>e7^4)+;ZxFVN>Y}q`F7+DDKL8tLlzX)QpOK^erX8p;`O**npRN9_m zIxINXoO365mhybHnJRpiQfbsgTE6TrZmq4xd%h$1{|>@;e=q!9#CITnZ(Jzy$Nb@; zP_&Ikd1ex#?`5L=S3cHeoPze`bJ%WO49$o#L|?gtX!q-Q`nCp-cyAumE~I-8ooF~~ zpVhfHJ9htJYTTkl-`v(y{=j*3ea|R5!oJD&CL*b`fb#R2S@5cR{E9X_a)_J>AtFrMZ| zVCWL|AKgvC@WdR%_dkt>p(S|sw-VK#uHkuf4SqM(qHN()T+(QUyU{yDaDME5VcAMzH|fXv z7$r~6%CWJe<4s4&{o7u;6tIyj1C%Is`WV`Iz5~6L7a_GQ2}Mmoa3AQ8VMc*CzcLJ; zmPTWcSv;&Plc5`tfg-1T=p~+o%)2rqM_%FHl3S=fP>U^f+)u35j4fgB(etB=XLr6I z)d@&B$DXFYG^ekJ4wLu3J=7byj(H75I{QGJrfztJvk^sD)({JSxggBU4#2rRA+TUg z#zfUPJhe+i;e-?%{Kfq9*IdqLJdL}%OW5mh3404}U|b+)_l|i4<%~v1_P)Vzs}3YI za8{5|KxNnMXsenT_o(R8{yI%+zPXad%FH0&4TH&`Ik@#fs7MS#4WDI^ zr(*E&L;|W9y9^K{^M1)bu!~vfG%my{_KU^`l)-HHWyI#*L|X4X+){gjlr7CLJ^CKK zbC4~9jz$n#w*p~1ZquV(% zm+|L-N6xfg%$B;B8k5^fUD~yL3n`6RN@-^%P*1>b)PKB*)#ozNq91|$cfru-+_oE5 zG3fM6#OttRj2)hYBZ~1@#{7HW@I)*XNyC%;Y#7NDqCxWv(yEK0)K!5KEmxtv;x<-u z=4bvD7h1*|&3Wr=XwhRM+O4pUWOk`j0QYq)Fq5WRfnQOraRt@9|K$G+$CBS6_?R1k z_P99w-H?ojoKI-l%=uE|Vz}=o5{uKLalJX7vw2e>!F_70XXW8j7xx2juA$q$QX~Xi z!iR0w@v)qHX&ED26lP77Xh^LZjK$Wdk-qGF8uUn#){Xp(FTXEh{>W7HdBNP^pfC*H z6$K3=&VtBKfs{=OrW(aVmGk802t)DHISh~HuvZ{C4vzInXsb!Xsh%vXOUp;(Y4&Fq z7BNR%0@*~~pIKLa@}UKF)f}SAM~o-DP@$ChQz`1A2=&vf!Jd#bT(k>B_(*T)B>Eyw z7y?(x7({C&Vt_;v1n*+e`!XDFS^HFC9D;I@a6Dceg^gOV7=1VacAOum%N&<S0O141Up4(9 zlgT=-nQ>TGm4vapt`l>AzlMasZV~U9rJJ_jiqZUU$Mcw617$d(EZApcz!=`HIEWnt?WSG58CKO^b$# z9{bW76S(`BKW_=oHWlpKzphKWqc)M$iiLFdx)d$m_8FEaLxOf3_Br{(ev&V|`UT-; zC+CN3O+?Pk6v*6UpRsKO0?zv4CHpqIo^eN+syn=Qc;X13$$y9X;Z<3h$J4y=hW<^03_bYDW9n&vJbcP&XeaN<2Cd@07mj2NsH^+&d&FT|UJ zFn&ZdGEF#FBs>-3j}uYg7|uBu-dKFy70_aQ*w7UQYHk?(-5oPiJTWoR8-gBRJlGh3 zwgEwCV9aub2G8qopnc~}C}`AvIwii5mfo00A&Dbt{VL9U)GLBAbD8AihkopHI-(bh zx_i;+UX_e|*6EAPPr`eZa43EE!T?1VObT{lkEj5Kk*=64m|4Uz{P{!)L2ry)V}4_#&4% ztD&=E5K)s1RnB7^-IB!EZ#c^Kc%#?G1&!8DDBCQ+_vNl|O+SvFKJM6&?SThjUQp2X z!K0(T(AdaXiQ5^EB?k&SYeN3t_tCnd4W#gUE}sP>=wN#*7$m}q_6W!f^Ti+LlYcJ^ zhRmcGw2E?$t5-UnzeA>-N0cy?FG^JkAQdw%X7$1WQ;ta-@4t%N7~uk&L6jW>Ru@x{S# ze@HYi*ETp1o(8PXDR!U*=S(S8pi2YN)Jbc^0`eL*iaCzg*gW?P^MK*#;u%${#S1}4 z1KHojSh`L!%w;oh&pHj0B;!zXfHjkxGr72%?+MXfNd3d_lA14uhx);2c>whH2Vr$O zdk$^Fa7aBIYx(_^AM3~-0`|h*KR^}CW$aqAkhU)wO`pDUzJA>|z#Ik-e zI0o8k@t8j|5x+gyI}qqdW8SG0KxZfKLvb-k+8Peoj!?uc2uCl^x;u5F5pg^gb@B=5jo>{d zG#$U#1HGsr4>3`!w-)2P$@3PZxBn1rwb@Rm(w5P+Z4>CzRPKw&DMgZhBo_Vfz-`vJ zK6vYm2KLRlHpJnRY#P?Jus75`jXSLqF*!6A=`&cPGCLkx;z>~dm4doI>5z%ahN)Bm z79Ti`+po{#YfBl#bGR2LnlYnqmh{u{FzpW7NvCgsgq-_a!TEYGzFmSCXIZxO@x=P) zu8`sHy3vx~`!zARus#JPf($sUBCvxWJg?6lcm=YfbGJj#BgcJydpd4XtDUh2BKwj8d!dk-g*-m=AQF z>;lpL?r_ubhsWed{5hM5X61BrjL1X<=gGD*|GR{98SaMW`A{IaXQliO<30)B@I?aqrH93s5nr#9&eGB%XQ) zshp3jsMCn;qHnQXsRIY=neXQL%ICQ)ZPqX$HTKbe3*1b4yBAXF97%e#treHf=R?fU zABjfp-J69RWWK6 zLUH%B3u@&Zpgs0D=E?fP|6?d7SH>dcTQX+IGAEakhlc&9nGa>omb+XoR92&?supL5 zaes_LD<-OZ#DvS=&|2SzzQ1uHRrYr$GlycBYe6HsI5+;~Zc4trnv^Z&s9l4zeROVM z%f(nYpLT<)u^m45IHB;82RwQM@PsoNOsnIusc#zhKIfo$@+ou$a<-a66+T*5W0_?w zO0u3JIlBe(61e~JcMt1L`%oS0Hh(fU=)c;LR;F5!!Av7cysASFTQ|@!(Rs93Uz~g| zG$8N^XA->P`#N)SqeShX?Bfd4C)|(e%zE}z=A%s#p~3fs?S2J#GQ0?99V#HDa}9qV z+(xRP9ZB6l4u4`IGSfCtO1p`Px5H`Q)vzrkL?oBa2stg9H3jCtKzoSAUZ1nbFh`1G>)I=e>rjspF9x zpT+&@UfV4Uh)RIXd3R*pu*SLZ*7&&384XiB&?ngsyEQ}c!6*i2-Xuf7{$H6fCvdWf zyO^X(;2(YwH1HbE2ySEYrg~i3@eI*?-;Z9zm>K6m$vfE7oYj_edcPqh7Us*N;N zWdRB2jH35{T5+eU5OduE_^$5=r*sQ^6tRI9bCKQ^x~63g5Y% z^j@!^i#H~d4STaza}NEHzHwNoz#QCVE67)wL#N0F^G$`=BJ{v`H9xc}hG4o+B#eug zYfVeQ)xSxQd60^bqD=f{t>vwCC$VJgSyVHg^WDdVUOnM_?>mlkevK{7;~bqk2M<%0 zv=$ADRHe~Z6)AI-IQ{W!LigB0T)Y~HEGuU$US@%mZgbq)X^XTNA(W?i;HenTksAY1 zvyktLwV{yh3}fyx0%R49ADsKV#vuuwPgDN+2z0U-=T{YyM2QoPdud0WthKY;XGF6H z>(bA8TWQKkWm-Ld5^ZwitO3?jKjHh(gAv~F_p`@oHw!rWSmMG9I}F+8g55#x5GnJ* zJx?F(67@yqWM9;M;aQ*cL??WMcs2^buH|8v%Xr=T3f`lccLwJIuV&uaKERr`-Zr5D za}Uv>qFuCDeJ!~yo=wu+S0vtJ zd@$V|IZE#AHE_f8P&bIaV1Muvp6?iknLWV=p1kf`=AEB@U~ENBKtI{dMM%1o@9&EF(3%nsx9MI;+~$hQVFD~$;esXGjw6PB z&r?I(@Pq$bknh5oV?wC>aKUlg<2b~7&ymM&DE;5(ea_iEGuXR$N9dpe?I zMIW~tQ%L?niZR?tTG!T4sf8lF4;)1g26eIb_XcdMa$w9FmZ{8jY6f_t%-Rd0&%Lmj zXCirhFPzo!;F*i(=qs*R`^**E4af1Q%ni%a+|a=5syfjF(;hrO-doSnyB|6p%8j`V@}Rd2NOy19($%n~r3IrC=h7&pN zv8UQe*0gWFDM@mt%6!GWG;`%<=J;38AlsQFI#Y^VgG4FMo;%S37(2LKjpouSOk;g{ zbqQz2TqwpJy|cW(o`CgM&LNS>!qUVHjQE(2)8Et3%j-VrVV~qFzMuUOQYQP_CiHWr zN^=Ktt!9jrv3`$KLn_7rdakTV`UUG~o9ZI+(94oD(&hvtu)~DB#T|(h3Ik$(u_%&8ASx8%a_u8$#_f`>}?%7u5rL zuosaiTZM;>hceadpt0sx*>)ne&7K8eZP}t&)rC82Z0_PTtGVAGpKZ~4E5}gCbAku z4*MnOoxd0zksVHRG>6jC{6Q3J(4P(~aqf`bU$p)Hg_*2%`=1ZDhSv>z>_Q5BXZaH< zq$%43C;3iq&~85QOvbI6G9s~@B+|2^CvzL_F-uckQxobTPtIRa&})V58AqD#lo ztmk9s>l!JtRUA#{E{~%1ZX;-hj~K1CVh`SjL3Ci=K+5kk;Gh3@5wF|)fX`w6o-ey{ zcg$kWS<`i*JJt5osANNXw(&DvZ%8U(deoJ&o8*L>=*6#ggG`SoAl@o0Xo-1)g0r^Y># zI?PS};%>O4xfe;0$QE(>swGCXzlM=s#Sj`;JCK&V7Nz7w5t=W@xy5t(&>wz(*YJDv=MsB0 z`8}#;uI@lGzrPNS^vB+g#5Jtwdmj6~796IVXLM-c8V&AzUQ0TYmXN|>1=5j~p~0U< z)4>7a)NC?@2Gk9tPi_O~ZBT#i5f-KE#eGTH=Qq}Oe}i#L7ql6R6?^^}KX~0x-Z#$j zz9HU+yXqbaC~~qh{op&Xr?xG5oaH_h&LuV*qEDJ;dq^fnoo;VXq0a~BQ1BwojDIqU z%6>6#_ds0MWMx@C;@ z-{O6;`KFK_n+Zt6$%(Yf>?!$`HGPdTBYT6RoPW8WH0rlgsOnk@f1^Zgf2NY^bbe-w z`cq+Z4@~Oc!o}t#JlI?HVPykU);z(->-7-iawe1FEzXR;hLs*yQ0j3RXKq*FHLrVw z&xHj-)||NsY0_uboaH-GnZS;OVpjC)6K8+0SLXN$En>qm|Bj61&6lS`oay?kjQ1#y z56HB5i~*Uquq^os%FHie(EE#M{#lOk#wD2Iehx>=PU5j^0h*`f<62`L>bzMG$?MAU zIT+74#{~r;$!>F|(0uN&{$@iL?yx4Pz>t269w2|U9du^G8VWu;mwu~`qutB-bDFe4 z+O7seVoFdkgL`3zov1`C<{2;5KN>qmN8{Q4XwT`|R)f}~6-$j6~d_NOW?BOU~;MG)e{jA4z8!R@L@(aTT#qKt-_=QNd2s zF%`Q-6buwmknZkoknRrY4hiX0N=2m<3%k4Ho&WoO@VQ^^ADQ{F!IoC0TG2>Ba~j3-tFw2tsC}{uO?u1>W=~l< z*d#^$Y2#_j<AnXrDx9{ctSc+9!t06EUxc6s8(Ixpm}^+a!^8!l@*qfOEg zxBOVkTjvM~5hu)_>I6alypdHr+t1{Vt61K<2U`-!nbNIO`lQT$)r91eGo82^aGeKFPzog5&$+8QT+Tf=;g1GE~Qa5d8f9j{!F$}>_k?$>yt z>wvZ@2mBFof}*@L(qmaO;C=f7U(Ra2u_gBr*0i$MjLe5KL-wTxos2(4b+Zr9Z++fB z@199HU4!Y<$aYj*Peyo|4+={iAg^PAU0=-+w9N)Wu8yp)xnjgucT`Mf4V86c{R8f} zU*iTL&U7!@;fa7hUN}_gjYa%!He`S7IwNL%jIbgtNmFW6)1!=Es&xOEBAw?flAZAq zQqK~h!0pd)$h;VF-#Pcu;*67~mXLaHhLChCWIl4hT?JPR{^W_NjlTH)-4D-9nU6Jx z`(yU`V^3ZHyvGD%oJ}ZlM)E!JlXXI3|J4H?y+j(X5Ka(^goN>;Rdatf|F&!E}*73WP$q6F3}Sq$2U+ zQzV}Sqp^K0_Xp`Dpz2;S)9(wpleAd>67^$D%0dX z9xE|sNW4W`St+y=La?=hSxoU(_{o3TtR1j!iYro`y-@7ykLS-qp?otMqb(Ak<&umm z-aG>?NW;uKnNXaZhf_^Oi2TCLSKdnwX}6;qGixfCVn!7ydbEkzS4U+PXx$xIGFiWn z`p%4^A*;IKJ1Px_Onjhn$^m12tnqxV0}kZ7u#V(~e{cM-opWbRCwcG68B*Ji%!%BS zgUE<{I6p0d`Lr@abfXg&v+_xqyPpF5wHMPR9E9FFCu!1`VJ zoAp5VO|=+sY+~KJ4bELRkw5eSNyKgoA5o z^~@<`#d`$XnH4DHXD)n*JJSC-!TqosG&tY1E-nZX_rg&n7>zyiW3gs&0{7yj;DkyR z4*e*=rTJy(c~ynIoJ}2etrf8!ZbEEAKmK?;MKtekj#%@)M#7ZV`fIWMONmBZ-%UTT zoHqRuB`0P6e=Zkbk!=umPjf?>rV9ebc|v@yKjK^2r?D^!BdeqF{7*FcPsHNJP}bj% zrNVDUHa0{Q;KlyhCfjI`>B-}?N@qK1 z1urI@=rMG9{v8;yzN%>K52?X!NSf}3nYG?%lng?-MFbv|MdP(C`xw%BKDsj!i`mm) z$N4JBu8Zif%D{$T(T`pn>voRGu=V0Q@V%+NItW|g&(!<$*HPM<{OAYDXMO9M1C{O$N zo|1e#n{Mh3p(&j$C}q}xvZ613k7NH0uiNf{{t%Z4$3vAEj2s;c8=m_wyBUO98Gaud z`lEeR5JdNe;^UbJ9Mp?O^q)AiYbIlOR65LA%e!#82t_`eFJlJe^;!c`9ej@7ua~2% z$}(g*WCr(#Gmm>kBQ~6igM5|`M)N#$+IDZ0`Uk>rawKdx^JyZ^9tY0M^-c4~pAFvV z%Jzb4p$|T(`XTsr0L=85Wycw}x3i*fh5w#=`Ftj4kKHQXSM^9U+iR6BNj*@a?)IJZ z;(-)tC5|B-s~-FiPKEgaZ$vV?Blou>Hb{FS;!_|N9OSvKAJ3Hil5p%J&-ZgVBf38j z)}H>bpBn&Uy+Fuj2BAzO1Pbi6u;q;A&E!}#?qv`5y;KCVMp&CEF% zl98mi$l>I>;W};~NQ7vrC*(~Wv52*yLEKkyo97&bg|V0>nGC_0Bq(=9qrf@{8U;SM zGSLf*MtY&=k{4$iy>WDx5AL#0%ZPs)g;zmX%KDjQawNhgavx5yId{HmQPDW|Z3t{3 zaoKq!rXxTr*eCcZB^IT29ys{k5u(#wvDw=f2OGm!8;i$dvx`{343{alBe5)(IULRI zDDrcGkBAFQj=JDqybEhsuGqvm1Mh4PBrAAh(_3H2e)Yr0pX?!=YtCE|O=?}QKr{6~ zr-sa?sX~M3if#j3ZbU&o-W_L8IU({WdyqIYiP#8y(@VtKuoOt}jPNH#;7hs>d|o)? z&^F%deC1gz|A`l~XP&-2YZ(qWeccH^Ke(X3!woH**PZ*!j?zoaXzhFrs_QvIZ%jAR zea9Je^2tw>sMjDjjQ75e-ME*^iMhD$h-OCJn2Ax)=}tmk9M8+^lb9nDjvf5H9#(Kf zQ=Juiy(~CiX#ryy3-lCN;DLt~mR_`lGw&&V2OJTr%=5!L+|^>r%%^B|()%V)-Vz&V zY|}LAll+DxjY??U3`5>BH%!ZRLO6eqy=MY2XhAf_e&>#d*fjK?O2%rraJUTP>#lLY z>p>Qr$2LXXKNEOeFu}yDCiud->6Foykoatk6YTFB`p*{6yY1+!GJC6;FQOGAPxn1! zX!z}^Bz*rf>`yZLhq)8)OqjiA?1ZIn+;Q+4->cW6@uA})&#BUpt8x(r8zUfB>;>U? z2UMw8U}Byr_h*`*PS6C?{upDHtSM|Snxk>46hM5xoX4tXk0 zl3~B}R1y$jW)w5wo6m>9FUtk)?;PJVz-0aKn~u0hrRvIWnshq;2DDyg?db>tmoh%O9eg zOZlS397aETsL!&)+0Ql@rf&mmv&HSR_89fa0ew6lZn@=%Pk-#_uLLvOe(;R$+EIEt zZWApLpFzpTzF~~aW&VD+_hgbYUWz;5!2?%_Z1qR^&}i5$O@aRv<{_=f;PpQNiQIql zOw=2TjodKdnhW+%b-^=tXAIuvj4ocU&ke91ujz)n+3xtV+5=C;cn({{Il38IbSprS z9w`wW$(%*PjlYmyScwU%L-5ng2|KH8aq_$i4%c76*wRRh;NSJ-vrHWA%*4WwWPUy( zaAq~nNk91DxhwnpMtL(o!3*yryzpqNH;DHt5Be_P&mDgpWxwu%mCP4a;xo3hHi^`8 zc27Z;vZUtG$pwQc^?oge6okTr_cpJ@t???OGBX}K=O*$w<|5RYYd+;^Hl$`1V1=GNz3;Q6ZQP%^ zV~#S-`@Wklc1Y94d86q;$#n>I#GrS-GcKJlgOQI7#<{yALpKOHGI2=u=DD0fHbmEC z;}U0(oc5(4O8X*aC8uC*Ng7(8W?;vjY^>7HgL*>|I0b`|$7>+Jxe48#ydK3{QDKMy zJ(_uzKCDsNq3>Rj~V#-?2{g6itaHM_^##x)7O5Gyv4JTZqDVc$imQd z+_6O2h**<_e4e*lZOO&bZG{l(x`dz$muzF`9ema-niB1JHRci5Cy$N!|+92N4iMqNzEVUcJiNUY2jpyA`%u$;? z*@Q~hYtW%finRO2W_mPd4(U|=f>C1`GuZubo!HR^ZT*dbl3Ciq#s|kp1m84E_3XO5rJ1+;|84{SKKHd+MvS zq1Ts9sbZlvoj9sQAwfGya`9p+yE~j(%dSF)nE^SLwz%?M57SSZ;`C8x%$d%f!iI2W zStr1m&s|I2W@9n;@b=vB3DuW_~fD;&fIQGO5W zbgYAhY&Iv8ce=E`;S8y*I6!HqR?*CRqGV;&kL|0H@t5@j_X2(V*ld6=!|d?4%>%*{ zg3xdx8htzqls0AFNqHXLWS2l>XEm;U;`tV_e*m}9>TnlnsgEJ`|qbY1NIxfz|8j_arf&l zIKH+gq3xU-C=0T2=&KopS8-t47uL1}r$;iaXt%h`Z7U zoym`|k#vvIb|fCvHAxs9m4YFU(jmMy2Y;;# zq5p>Y1rC+aVLiuAg}wFrY-w34=bhQNvcXMOb?%Smc%oyi$IOmS{F#kpDK_miK$G|-%9wB|n z*f}X3Mn|*I#`@JbepgQGwIws2J7>$9F)vo1{0^U|+z-d7iF?b&ERm*Jiesr|e?Q`W zGcQfd2PVHP5n-f{UGe%ju+9|g&Dm3D=8U|R9*AKd)bE$fbvfz}qrCy>x)TUB=Mac- zK4!R7G}IMikLeXs`_J_ntw$EB-2`ECz=l3CXIrk)gko8XO^a3~C3!`PUc~28ZD~69TbQ|| z&!NLzM9uXP*qG{pYkaLmKWy+S)((enIzr^73)-FC;n?GW4(_DeO4F$aA)pP6n(*F+5|X}Tskd_G0zILD@HE<;fUv#FYU zAf^uMh4;S#^oT|wjG2q^i#@PC+YQO3?$Bq?USBt7R#KP`DCvWdYrG-(#S6#Zdc*Xx zFFe%z;LrL*Enj!>ylE3dGuhKrEca z3?)7z*HyD;o3A^ZwWMHWp3|-5Zb{bLReUTdImndS-{{je0Sy|EJVmxu2k3|e&tRWR zaQ<4Da|K^fweU86>sBF(Ia^<4(s55M1&bH5K3ANAHT=vye4mU#&WX^pkH?y;+^fAU z9*f5&;N0|t|2UuHSv$Mu%fJCaQe9(CV>FFu8FSV)nV+Sd>zJdOv4b8< ztfE;%XVGA{G1UI>8}roe;m`CA=uB_mj^}D5hgG6Rp_020Do_(yj)?OmkU3BU1^oh4 ze$2;ou>yR)l8^8FdBxbHb3Tsm>uWskW?eQ|%bMmpG6zK2gcgVEkzDq9%2}vHgRUJQ zzx$v)r=%(V%T)5>KKtoE|6%phH^AISm=$~nQe{1;DeuPb{4Ny9-NHrh>sV^jhVlAW z@w&eme>XMbtZUPM`N0ePuFPV6Ws!hA#dg|~!e|@XVrNO(%=)}{%YYnSY0}TCGvv%X zb@5d@Y4nq|{QXGMU-KzsQ!|#HwlME_!9VoGe#6#*PndQ5BT{F4fOp|rUUy#t!=AEE z_XzvUm`yY39{j)kKZm*Rhdte` zz_9f@pV`0Q_XW;O%>MY_99D+sgV%W9v5z=c$7`Z$s|`KN;ddAB=WkvxCbP%7bcB5~ z%Kb|8!c~sMMs24A6VBAPNs&M^Ytv;@$fSA#)wc-Kx+v~uO%kHvB0{u4Yz#H+8bu|% zZ|O1;q?||rT4OYXgi`;%AGVLLJCSE9#e2A$;G`YJf3P7bUKf4EEC}=1Tfq9EcA`4{ zh&oM6nK`X9dKVeY*+|8cmeXy&1@x$ACXErALLQQ0^!w*{Ixu29J@OKzhlL_kK1P@x zUJ{}edq$Jj>57pX4TM4ks$knIOaI-oS0#->jtiLOZ$UNVt3a1J0@e*&o$v42WZ zl-i$hUe#5Ip2m!(E9#@@Q1ZzCt~-IRYt7&D5dNM|vp-0=(}q^oS&`;FbGp>e-Ojpt z^a<+JwfHnetH@Kk%TDqN*g(P#%jnB@30ky&CRv2>zPEEC`L>OxY-aeoF*jq~C{gy^ z3e$q4W9fy!7@DwxXR@4MHCG=&|M)qI;`>UO?<-fnukKCZIjoNjIdU#&#x`^69n734 z8D=y_t5Nd5le8xA5Pcl9jWiY4Qg`BF@{5~IUFqT^dv`n)O7M5Obu8(}3(-h5A&ROP zLpEzilg_Y_B;P)ae$N%8P0NSUixdI+#n)ZO&(U;#j+XIrl;6$lpqVzL&Ux6G%q5Ob zHY8=vB*k(b_Mz!9GW)uhDprH$D=ni(F>?vMlj($|Fu4be;{RtD_3jm<(%VDHBx@+; z{S;uf!w~XZJBY0I{>BQ6ADAG@y@!h5@P)6N^@;sgu{>wLZbxZWwq);uuIyaQN{rK%Wk_mNoCi&eL}(Rs5Q z-G$u{7}%jD1+znXO0?UFHEN^SUP5!*W&+KW;8Q$NmV^+a=8+y`aNwJb< z^mc+F{hOdgO5B4GxAX{Uxv-Y^VL2UdnNBN@jiQPhpYT)Z4*XWOqWEPk>uHs+ep`Vu z_UMT4KK4uWWvtLIczZ6(hUOXx(v{S|}X#pMvo6R1oz2 zgYj~Jwdvm>NH`yiU!MXvi_2PRqaTDC{2?$a5Lz997{i~py!!vMe6ndabSlY`R;ink zfQ>%AmrN`zFV5>TvwM6Y2A^WSm;40`JLLg1yWv!&J3h02 zw93vKFB(}d*7SnZ84rkXHZ^z``e1b#6KA`WUwb&;;1-9~67~FdN7lBmQ~glAJI1UHZZBc>qSs1)+`Kk+asax4zMaHfC7R?Or2# z-lD~xA7yHKagZ*=%TU|2ndCY9AFk9iz-)6A#3fyD)|;6$WPqga#@Og+&8!?J#Jlo* ziStp4jsBRd#vZ?u!7y;;y|QEoeEx>Qsxtx?TcbH=7>jiFs(7#;SdFXg2lybjPCJ4AM2YiGr)0wBZO?X!h?BEIFrnKSUq1T ztqMZ$cJ4UlPT=_d7$oxypq=%qA3cd!shNV#y_{VL&c?R?okN*HQ6p_ld%W5I%=wr{ z56;rTTk;e!g=l5PTpD*&fX=2dL-|1zJdZoW@2oMxE*RodfCV10pWp?5&h{{0{2kz2 z+3rZZk&Z`hdoun!OvQ$O-1D_P3;H44PdBUxTmO~9EwU0@`P_Z^G<)inSy5w}F^MH< z(Zcno$ySi_B7UoBjM*d-(0&aOlS2F*%f>n*<_+C9g!BS44AOUi`E55$-R_H+7-oJg z<~iYCX6CV<@zz9U+Hl9?P$lj<^ecf+*=5L1s)f+sEx zP)Ulzi_jEYypWBRb9lSm^;gNS6oQ!Yco$4JZ#0+qk4bMY=+t4gsGs>yf zp_^&S^!w3%>g--k@5fCd5uKO##e8n1E6hs$ZVg%HDxJ%_#Gc{ef~yc-a1Ax~UFiLO50<^p_@Bi2Uv(Sme`iAP z#I;DlmwQUy?xe5v(iFgZ24C48{C$;-5s%$5(%llN0hUuuLb43ES z2_~U`%|+}Ul#WqDn9s@UUh#-hOn0on@*}m_hGum6cc5TgH|j(mAU2fOB5`Y4T5U+p zeyU_MTb{(mZlnb6wDea0fs0=%kTE6%5u=W1_g*6V$e`c?B*i4>F zwyV(6VFzi6;Tl>SK8aahuQ6G#5Q|u=i5=;PiB-1jbzv65dSBEq=c00RECxgqv9}@t zCuheY$0P=MJ2=<(EFRjUldv^^Q`-Tn+VS*AK=@(EO@Q+wYH2b|B(IdCU&j!hAH@!_1(X$k`NO zT`I~`qD%feX;Luv;O-kkmmhWGN=^!{&+^8uAv^;a;e-?0Jh59S2-3{?ROVg)C+=sT z@h}!!V#46HjWf6&%wdoSK!af*)N_I`N{lnX6o!TNkvN+di&n)X9Dk7tqfPd7$DTVa zDtNDIcY?V)?4R(Iq-F1i)4SH|a7{~uUk7I_Egg{&?2Kw{ZwNATxFs$Y0<)9xIfefo z<}gnS48j`LFLqDnjwjC9kK;L(9CzMt-R^_auY3{h<{S@2HqaWV4@sDzow$ zCDf_2>@az!t|J5O$+Y(EOO!k0!RxgT;`^)-u+NgaS)I|B!@2SuQTSb&!aTklXjpI- z<3}dR2+HSq=m}Gj z{KBj{`GaIVcon^TD@s$CKOQ5V0r%Zr@VRQuy~#Gv6>!5sWr_FL24Y4!wac;K6a4fD$E(X!S9FOCLd$La)#&&k3N&TuL& zWG2GqG-ee&cP!inm+m>Be5@D#@_V|(g?o3mmy3}@bIT?$%J~|JJc|U%0 zbOD;#XWq#1UINV0&&3=;@UHH6e21IQa2!r9Xps{438GGL12>2_2Y zX+&}U%9QkZ2b~R+q@*i?bZK!5?94yg+o3=LhOK9WoUR`KU)5E<)U5w?7 z#-Vs4D)d&SotJiySkD3qF&#>3!Pbjtm%JFhf-1mB95?kLBj=2t@cP)w1nADV?;|CV56@d(r)Qt%^Q7u2sME8Py>8k z!`vzEn2?e^Lx(eWQdTeP2FnF$vUxLf6(aDK^DU0%max=u#G^Gn(Eb^Lj?rll%`JrD zi!!WuRf1sdi!tzvK|gyXEG~NBtB?zlDji_+$QnNy&GE6@6k)GSVff38=hi%rHZ_OP zF=l?m7?Zk=3e_FiOu%|TAsWo8a{_9=J^d-Iweju(egvDds1 zH+{-+<9-<^ArIGL60ziTC^GoH(jm?MjhAk?&A-vpzm5p`;(&XD95Ez|yJ{4j*rV&j zGYOuhXqwQTzh@~TbuV4bk*3_MBgtrYJ2K`+;aIIBZnLi4z;mm&XZd|89|pU3JhMqG zfP7IoeEF=IY{uLPoiy$~;C<`SRL(yX`Fe}=d5$k2Tz3j_Chy*XP+iB zGbGQBLQk5~&!?(%B;Wu^nJ=e>SB2>2X6_ok&+HOj3r_trLhLn5B(8DC(eWXecrOX- z#PgxIrX2BiScjZaiivhPNMCXht;{wM%#XsO6A@T=n7uPiA^5=g?}|Sm&@T;xJ9~)I zIOq89A>YsE&8Uy}>Z0G}I7h#RUVIj%oT5AMmg6po>CWh$V~DnDQ&>j2V3Pa%R+may1>tvT)5;Qp{Y?Qzry}{f! zawG@wqU;5>XO5YMCOy4;l>Q#wK(U*d7yI!kl)t9IUfC5+Qw&fUWQZrpj%dGm0e53! zAeEj8p|VSG?&Z%bS&90EWf){rgtKMbll_vpIp6YNxhNmY67%8C&$vWtG1jKDKij4X z*-G`0*=$ckBP__PQ=4{|9jC-;Tj;UpEP5sR1)hopFx=^hHfICO%hAQsS2moV^TIqU zKIassVxfB>B$r*rR`P6kci!Kxp0v!#W}%B3^A#~+9y>IJye6CgXhkfsDd^~oS zJrD!x@XMK#f+Nm8W?f>q+&-Bq}osP$7y3m zqcyW>z3_)UWJbcwqK?eLw6mAc(0>_+{@1PS+6dQ8%m`?xA^bUH zz24)4iBDbdltx z4mnq2{4wI}?-ze~?ukK?SQ>g1p2|1h#RQK-sl!c@4kV@j@_tK?uUsrcf~lp zgmdvnh-dzS--;o$sFr%Bdi5B< z9gwbO=Jiy*pM@S{n8zzj=kva8m>o&CuqGL8LeI0c>Bzm)^uuW%^Eg(JptA^FHMxTc zmlDyunRAX|TCm@wft{<3@VD6kS0p^~d=vW>nejblMLfnBvPX4q4)Uo8yJwbR&$&v} zec%XS}u}!!~O=5NSrvqx9*@5>*lnJ3`BRI6gZti!S8< z!sLUOv1y4vb1qD=AW;LtjoSED$ywCHPKaTS$S6n7W}FQ}*_>F2olb@Y^Yq@Zjxla% zK8&sx;WBro?{TQ+tVJXKPH)8to;xe^-1#BTo&W5$pvP|vDZfdBq~cD}n0Kt_V<{bC z4Jo;$7sGv%v68)KYCrVx=a3c_73pJ%fECn-I%6Bpos%8?urQrj4(piDEy5bFhY=L-CNnuG13GT3v40S?Tovco{0U--dR2uw8w?v{hUY) zn;VTciLp4hg1tfPKUDk6dX5^eDXiC(^E=D&sV$Yiup%F0Gx8BMppRG1lW+WSnnXKk zljjo39xF_nhd+doMGjiIv!px19zwMy(5^GV5n>M^GfqlJy5TM7BA%V&nP+$)^jIS; z85xR|qr=h3=QOP^ygw~Y#I`dR(aic}1)shBc#d-9i8Z~KVx676!6uWmDP!sxT2yg} z9`|mbVdArCd*cwQOKfLGKX-PGxPX|!&WJ0ugIKgZHrY9$P|FSW%p9+1ynw>604!O+ ztemC6u;khNjg#RRRTYKezp+>{I|2Jyj~!QUPqWLIS6gU9O%awfc)BUwVK(m)XEn;n zIYuthJL!`=bKTxgAcMpYa2sBaSC6>w!IHaU&ay9+eVDe)QWwr;?ZMa&r-g&CDk=m? z-MptZ3d1vpaIB2uekku46sW{QWES@Z@paE|R&pNCG!~p-Zft`UeICNBz<5In`L0R- zc-DR)`VhTV+(d5z=F`-3qiNS#=Iw5+#m$T9kmbyJa&;(oqA@@Hc_>c4jerMpOghct zFkUSl$6UB4;X^z!wGyFvmHW(>rQ#LeXD$ExybtYZ(ErY(_igE0u{AaMT2Re!6Pg#L zO9rv36smBH3TN-4>kn4aZDxx0cnWe?!vF>~H=}O-CHC88VeIuZ=%uCMQAq}jOtbOK zIv0Tx^YAYz7ngSABK2u5hIr>gTe}cW)|!e^M9Iu4XYheO-V)-*^F!jknRv8e-OlI*7~GplU%a*0G1b z{&fSkOEzSL$`GW`>My7}{0z_E-@~eXcOd<%4{k^A z!XdgJL)mksC3qM5?|Sjmz6aVHdmwbF2ctjs;3t1x=`TEgO}3|FJ-ny2vn5eGYkI-G zmc_v)tZV5}-$r$k9e;}6?ma{@`?gT{n&o7b&pz}m6KGe(NRq%1YX9>SYiEANK;367 z*8GA^@?T*w;VZ8SpRwt}2k3Gu;CSJ;aO`}8qSQC|UtV&~cY9JOwx_3?xKgU*HhoO#iX4+lNMA=q|2wqau@bUYUAwl zwqQY$OcbQwUj*r4hai>p4yD3l0`%bVVA}P05bN55$esN@|I6V%W9@rI4SO19GRJ)= z_q3m~q0%9IU+0^VzL*jHx~@%6e9lqd!sB%8+dg`In`qpSmGnJ)0bRumGMGJy+Lnk? zw9Ghq^i7ED?+B6gjImVkTZrzB6QZ%sqv`(Bk(5z2f`%^`LCH7&e-2y3@8}ba_B2(R zGyI&h9_M94hxb^~>^bJNqSlCRw(HQk@2cD%dXh9l4pQ&@tt9np4NadRMStecB?U~Q z5ho|n?US7M78p+hU801^=@zgwDlm;=Q*!|;JdYLCgnF2zzK4Z*(pErrGyNIvb zDaGppUw0bMLt_sye`J|CoxW>C3V(D+WxE>P2|Y;zGKc8>zpeCe_&OS+AWe;S^GG~< zI&(WFlS2PQQsPdmW#=YPLoRcQ(4c*?qm~s!# zAxWdDbUjXtu9uCc5(l1hn~IWtk0@n75GDVqqIC75Fijsbj&}A7k<~6CdNo#v{+FY_ z%4@KB5&Qaw@Qj1~3`0d(7ml;zhd!-j&)t0+WxkK)X_V>?8q&9x+Ab}m z#y!*N+llcMJ!uU6I6jh@RkI`~sV*5BR{ojbD!JlNkRQPvt(t?dnIEZGVp!bKhe3t~Xd4 z{suZ5-lCVU>-d5_1scp7+ha$koo$$RZb_ds%(yGmkl8j`^hZyHI<$_Gi}nuM`+605 z9%t`Fw=g|pHhJmccUXMpA*{lCu_UGoMnAgv4Ap~CWqr{2&<9zQUhZk@#_0REp>^vv z4&Lm-zJYF(@^ugLUe`W#gb0Fdh8@ znV2#*8`F|)w!JN`ig*6xgUvH7QE;Bzr&gu9YEClQ|^q zI)X9|J;ZXJ9h{k;f(teg=)D?*2$w)?925-sC!rX0BLcFnQSjkh-xse)%w}%li;ze> zw&Om6E3vTSPCEzQ=iKFe&W|(9G5=*vr_FhvLl9Hml8?qZjhrq11> z6ch4>^M5t)F&_~IEpU3uj}^Z?uGZZv|GrUr0}Hd=5XqopaLbjF?fR2Yq>CY+r88W-uq%9dgI=jXoG9&urWNK&bHVzT+!%SDD-M z)SvgWd@rc^M?>*ZELQRycJUWGI``3TiL&pA#lE;}emSDH%M`=Xl8 zXQU^o@MmA*zbz)PH_^d?NBVeYZUJFiN7RjSM^&K@GsFU6`81TdLXkL>8->gI(I^j& zf$2aT_b#xu!y==?iHtlo0E01>#`{Zt~wxI*d6bgO?1aLh?%k6A@?E{&mSa$3#zcPECt1e zX&5P+iRnvnp>UxPX$~b=!h1TKZOr?6Y)J+ajcIe5CRxTSQHIqXdYHDHK93NkF*6>( zzc3Ao+ubp7iXm2Q)5eO0M$o-xhcEta*c;-Du5H0s+Y^bWC-^)npNjl(+^@xc;8+iy z#l9?riChVUo-s3cZY{1#HNltPRnZ-`^nkr08rcT4s7Z}(RVdK#r(4K%U_OO%e!Y;I zkhnDrLGx^JXNe9L?$N`EIaYYd-q7>kxVKuD8OeKj*6}X^^O)tceKC9NeTuNfqZE6+ zE+Z(U3L`l)NPW#1_L(w#=|Ir(-{KX`QJG>2Eni=cCur*-UY2Pc(EddXa3Onv?(K%O$$ksCCCRd}yr~zkm zuAoh`4SN-Dz%i*CcgEjG-iN0U&*8nsfHg%Y88e4KorE$KX-nT0`h9gC6}}uygTrfK z|1bzynpW^G(Z$cLCb**S#IsNz7(eE0x?dbrmZ!jUeJNxKlzzWLL6Cp=lHb*G8gi%qMVz~piG4$J;KMnVHMLprP0dE|i(G8y>%Z+N zLh`RmI8#>!!P}KMs9z7|JuT>NyMaqf>}fUoj;Bj#(~P~xX@j6F1x=Yp8RCO!*x*{$ ze}Zu%)DGe!OgN8VgNCgh_^^d_(XVlEkm9Z`@f-w7=3w)BM zc{x*gNXLewX=b3wju6Q4#`O{55Krd0`ur@MWD7ES`WA4Ib{j#7d*PWSzV@{@90$fn;9gNAW^k6lOOSJI4{W)c#E_)-Dv>qky5BG34Bz#yP%XHG@@*IJ zaJnVVdznB(iDwIseIYrUxr=34Fxyd%;cMzKae5<4hgQQaq!2xvZ_nV#!z*!$#GPtHsn+lJUfcYlwnVCvftKKlDJr6~k z8}}e;*@w&3}8@ z7jB!G;Zo^}OP8FHu+ABJ%yT$@%LSed?CaUiI$fq7rEq4kYo82FJ3fhYhAlO!c@=4hksuRoHmS8CVA|rHRQlpRrm3f* zk#nd|Wh{_+)EY;n-BGfWwIbt-Xpk#L;Ep3JpkRCh1je;OP5dg7=TvfbJQI~Ikucij2VHe{ zs4%m}c9I3=$s1rPYqzgPYQgidCIr4|!b3$9OJCSf>s(#>n{24$Q*09`=q1F<<-x ze>Z`Su(Y2G)I6$IWjxq^77#O`nd?rhlvHwt@(a{M&;? z&l2#&)CHA*FYR= z@Pf#87iSH4YFv3xiGTwY_>p-T-Z%4HEB=q9^A5zi{l7RNq#-*=8Yvj zEecVgMU+UC9U*&0B&&#wB&9_{zxVg~{rCLSbGh&PGp_4>&Uqcj(M5PXy9BL;WiUEj ziAU#ZaaGcV?i{nB%pdwx*&3TgVdsR}}_fzWzs!`ikEX7?Il!&F!HlX1_E z^*soyK4qq8Iz}rMVqH@yu3vo(AKljo@n->shWke}g5}mB>6;g)J7GGtPR)eT?`90fl!WCjRx~NLfgO#=&B#YUT zFe(uD&PSkmbuy0TWuuwB5ARo(;kIiz7L~t2V@VYr4zGjYsCu0K^8xZ}nxPr?2?_>Z z;k4j4HqGrvw++wvhuG6GVKb6%KSI~`uctH9)F^k`Aew$58FiB{A~N3q@*|G}A5X$p zz#UbauHo3c`-uGd6uPo$Skamb!<1sU4V1x3wGu_qH85ue=8E%8@V0Bk;HRIls_7e^ z9`1%s$pEIP3(=ZeJpY~SKxyFHQXZZbQAv7c{H= zfc2PONc9TP@+M)DO?IKaF^=>_+KLQ&PS8O19*Wop${$dohCiP%XhJw97Vz_>PX{yB z>OfxNI$KbW@)Q3U6X#F($?H`SWr%8Y1(#9@l+&>)k8>XarSLT+!<@%Nq=!XKKT zb*VNExEf(vhzmT}2W&cw`5&D56zhw^2ImA!7f*wNMh+5o7Gpw1IcA@!#_OB!vFUXa z3_pLu&*@!An)?T(X9P&A-i2HOocIiCLlfqhQB|NeUG~~W#cc~H{G|-7?0AQ)W%tm% z!5Qy_bul4I7n?M#P!P`?`Ndb!vm_8of}z-Wj{VN-d0+lA1zKM+(RRE5_McwyyQBik z>Z{RLU5`b7TQI%2ljk6u@#gcs&}3)wb+seEG2BUSbArB3+ed{Ot7yTviL~#_0G4k^ z#g9?t2>XuDZe`F3FNRp(z z#??5??Bv8x%=!AEk1(?naFMpaiT=~*@x6e+Qg3{>xd}IB)#lYbfR0BL?}=g|@*@$0 zgj4bJSO(swcoAJ*e& zU0xqn%ceku^QN+1Q$8WI%-oA#`0YBzAw*3|2a|NSO?O0ZA~xkoTU9cb4Zst zNV2AywEx*`nzmMkhTdpL_5B2xuD=eMerECXv5)K#cb2SWriifzirgrHlFi*@qCPJ`}w&p9gi=(Zc~g4Sz9t|Q^Aq$df3ueElc*R8q+Ly zEmCjTL7Ib!=3Q5%+%sZSxvB<2F){da^ClK`c>-T9AZz(Wj2Z5U$!6ZT{MaAMKXYbc z{auI{hv3F*?i4%q5SIo;qT^6BBI084^KSx9@cem29y1I0edztckt{f)ayR%CojJ|D zc!heD5PX1&6gSY3zYqu_2JZ+te6*dK!zv_KECIP2t&i8XohyK2@BjlXW3KS0^eE=e(7b4aMwY zeNNJd{(Lz`dnW9pX9_DRtxugY;-#oSr3XS2-{7R_3p8`zNZhPQ+*}yJTJ~d_fa0*E2fym@~?3?fkvwd=K=}0DK_4E8>d@d|? z3NdNJE37OpL+#(!Fnd^zkMAq+oHdW?Q`I~Rs6{)k8_?lGd#$*~lV|StZ#j~*xIG;| zU`djLKx2m=qmBHTtw*gUhtyd#^!FJ0Ja-tS&lVtS|4wF?H6ilZd*~*LG z52rUlET9D{nQc6K_=FD|KjE?aC)9}z=FOgZD7Xvs%= zS{`FV`)62^MVm3*(AFW3`TOa}wvDuG$5Kj-SEo?Z(G)f?oW^#EP??wj^L_td`=O=z4`jd<`P~CpvF;y&WCZ@x=|=Jldej5AdvOH?GCj*(PX#|)TRVl?9X zV47w-m^_R`DgBNZy*MUL2M>x9LqUe&LMqz z;Y1r+nY*cJM*%^nNX7jm3EnoKiN43kW#?`Z-@TqZW0p`$<#gKmWjxKbQzTU@8L|lC z{@L*16rnksruGh}Kp{zb9wkXpHzY}K%*ojmK`&hP&t zLwV+3=S1&EInsO9B^8vc>B>5D(s^!3yQXSWiu+ztT(ptW*MnY5%pp;~$uvlNEP0xb zqWH_QG{#n%RK7~lbSG)fiOZ0^tSmYHm7!rvWN70iX6rtYqL-JYX!1kWgn8YiyzXFL zw?=|BJ6`wNXh-^e*N(K+t?9P5ISq6e(q>0(vK81zwjVc=_Y@+tQFBQ%dkQH%R3WjK zid6nyj&A12aQ;q)PGreYa&fbQI>AskfCe8rRiUpG$sC!{*Q+&%HOl* zGu8+H^7YK$b1mOX`wrQW5cABNYEM%80z;a8S&Q}-?IC^L4Yc40dq{@PrW@xc(bDb8 zRL=QSRWBJ*^OT|uN>cPKQ;H^(axT|hhO%vBNb`j>U1mM_{Cd`(7E6(o1$PniedN!6 zhNsN!%dK{%A*{n{Ty`MYLR-@L#a%8@X5@EGpT6%pN~=HYqXO6BA$%D3p%0@>)#22| z>)vH;!}zKTiHEV)J(0iX-}YqNW<$lC`&#$fglc*4gJN{poTh|u{aA-X+Mm@-$1Q1zq1WauSI<~K!YoR}zio3UQIVK93j22-yU`Tu~ z9`5Ty7OzW9oOR}NLs7pI&F*%fwVYR(Is6o_V@7XSn>={xFg=yt#=Qtczssi4r?^qH zAXki@dv)WSL<_9bYoR9k7W14cq4&NDUnAFr8IzaJaT>CIFU|62hS1)5WSg!`q8G*K z$-}P@ajb;>pByy3dx2S^DHzk90uScD#GlSW=ZrjzTT;LrwF2(LEWj;(W_#@|M$d>+ zWct6xVqSN6F8fuO<*L!*NcIxkMIdHPdZs7IE<~R>FNaC{`c}FzbU8`*Orq&ihtmwL zukcYUhOTk~o(zq^F3$XJ{Kefo3j?zwEsi+sFIs_9VE= zr=p`i4WazrZ;a;Iw-x6$COT44vMqW3w4`E96Y>;fp3SPgq&;gL`EQs->Be%DF3|&f z!6LLwiel~~_u_5xL&IkuSbX=z$cfis%=7t@zB@43>uoP(wIoix^N!jcRPZQ>S3fw?f{tbi<3czsY zZaTH|%>PLUrj&CB1iu5temm2&Uk(%^VM8+GPtso2VQ0wir-tL}=*-mVbb$RWPB}Fg zNe{4PJM(caTOn1)1iLny!!%2b*52dgDg zFq5?^?`?J%^VSeWmd3c+b_&JJ?%sIA1H)@QG3>7|wAn+jRrL-u2ZeBtODIM>55sm- z_KPl$K%M#%1hWoH7ZX^kbs_tO&NMa6j;FF=l1!W8-pjY?5<<$TaR?WbIy2n>$Rq*u$_c81mf@F!^94)aO6J8pRlF z=X>hK>UfOe`N3r4G;Ea5g5rDbF^P4eFWYUYMuof6-sn*L%-!_efM~DUc-rnVfES}u zk!^hyah!23+J6Gd%Ek!(!JhZ~?$~$23y!D!q4xY1ruf{)&e;)|u_gv`Q{oYDH3`x^ z&mm^=0w?P-AjVvk=j`dwswwMNV zj0wY$OJ{LC!Vp&PPGA+!9c+Tm;5M^)?yT~MZN_bMa3*!((8pMu@(gx^k`b~p4eiRA zNd3cp_u@QEbxcVR7kpQuKey&{xg zkb~Lr*B}yTjg)_SP+Vn8&PzQ<;u_Xa@8)12 zm04dr^N{T;g>r8NayHi>A-MqxS6Z>)E;Avl97x2=g2ueir6)&r(Y3qFs8eYSiH+;R z{={g=?BGsi4HIlgI)V6QmZ)ZZq9@}DioFA|@MkEFZjFLUdn~#VlVN-=9pg9UAjN{8 z$LC(5_W5gMe|iI>K{c>FU5^*+HMKhZ1#!+U^vT?wMvps5xo5TLe8pz6UNo16%Z{WK zvej5nb_XBCPP6yNfVrDSFn!~IhKI~0_Tuicf?z%uJx1g6Sa?%17T3PuE=l(D`*Rk2 zYB7$!FGW&iIXvn)Pta6@+@lRhoY9Iq(Vg($&CjYWb~Mbvga-8Da^9s#Qxa zl06G?#jOOhJ4+#Qr5v#uZ&7!tjv4b!xE9`y?1{WzA7)DpVa%Ugu#dXRh^i!1Xz<+M zJbU4@ls+%ad0!Kv)+wspZHi_0+n5rCzmnK9fIi%lgd5c-|Q8NW2V?M=l+hg4XO zeSyqjX_y_EhR;*du}dQZ+BumhZ_mYXi(;NXmc#rB^J90MqOe3ATJ5rh0xM@z;f>++ z_*MmED{sNV`!wgtOtHt@62s3i-$BJ2^A+xL<~s`VXOpl;G#x(%W#DX8D$eI5p-43W zBEs=-*N(%$omfc!=3eVbvFNUhgWaSglsr#`jYuY%Or5F!izQtx)S?>|nsiuiD$QFc zN`b5L@cKRTEpIx)-@^=Buh`KLVsT>jvRi8Qh^Lu3qQw{y|LJQ z?+Kn1MnWy}5d?xBLa`wXt#iZJH}e2)7sBB+;xVd6$G~p}`!ugxP>twO+U2~Ke)y}> zyDtJ1=Kli5UEb(iW{-|H=8%kb!u!yR7&vtkKfbazBr_hdJR97?dgv`mhoN->R&I`h z>eB~kIv9d*?tc6qiy>uh030~m&?tKorkqC{x9k=QhTcWccPHLQnv?VSL$vqqNc52_EGf&;WKMgr>7)i>f}tEAD@jWW76;=_8B&A ze+acvcVT-W0Mg3-aOZqmc#jv<*?%>Z^StU`Sr_%cf;oG=@nDt{{Y*5WAOCiduE~6g z&K^MIu6g(P6-ofy{iBjDT?hNAT$_~*_UT6^wDydKOP@cWn({m5>7Rb^jIhUwuclC3YmZ6k7x2tH06&L4hBVIrHP60+NLUSaUww~b|Ee)R zs{~y|84#_D$Mn@vn8JRi3vzeSvo-)n`Ln-i^~IiTzA$;leN*<=a97rm^!*IzlG|q5 zm8MP(fkHHIUOHk!`L}&(i}kb2VJz(e>bs2UrMF=}B?jRknXuVij-k=-uwiZkEY4L! zxu_6tZls{%=M!9$2!q7UTM!oY$IS%JGOWFXf|BzX^3DSiDIOSF?Sb_B4y5+y1nmyp zK&FY4sK{ync$9?uPTZ{)V1xItL^x}VnlV=($$8-2{;`-pIu{Y1l^ChnfQCmOcz(rQ zl`i=(bmn=D+ap{Ey2Cmyzo)i(Au+)N4HM47#p^UQi=EN<&IwEBIpNGU2U5t}R?GzFKH+?CBsGL!#Fhm7`j`*(dDS@K(a1TNCM3I8 z597^{(3vP~U0|kwi<-ZVRZ&n`5rB1N|Gqy`fVy z$SsLGLmhb@e=-JpIG63k*_-%)Q)pG@j)i%?m~xxhuAE9bn zgWEcJs9GBbt7oBzxOEe+&R>D`JN7qzae?ySB-#60rb`2n z>&WY!3SFJa-RVc4U^@45*gWL@+r?AJadN|mW?#7YgmH&vGHx6w#;1}xM9Q}y*sukc zXVgKvHy>+);;}I^485iS2vfa^9h*JC#9;XI_qOr-NywF$po2B&pB4OWo8mw&qxDEY zR)dyJ8ApP{y13ge3R;{Yf4;~PKD>8KpYH~jGGBB(3xiWhGVc5+W`^oJJUQ5k&e_~u z%d`A9e~Tdho%O-ltaa4g!iIC474hNji@ls-Ipc`1UDg;`ZGqXBPGZw>Q*<8Z>x#X% z^+(s!#^CW}weSZ@{>RCz@IZ{WIh-F@u)fXw=iffq_BRwSG?VcpsTfb1-l0UT4O81b zVeN$v*fqBdI`U~yspGv|SqScP4tYr|^B4DDz=5G|$arxY;!m9*zQ_T61GY#_;m*NH z%mO^Sfh@PGl0suQHt;>4Ir=;nwwmEbj~ON=aF_6IZ~WHb?#8-AW+D_pcH=wP9BqT^ z{?9n}>?0D^RO00m?vS)dz+=fstk@R}m4`Rsv(pDJTQ6ge^#y!z@qpl_bMV>d4(Vx* zj9 za7X%;XhZ^kw^7pQ8MGr$n6uU?xcvSS*7EyLd8z@H2->5r?Gl#W=RV+!(tRW>Wd2H3FkIeq zKhae7JXf1gpU7@nJ201w^dv|tHV5lgFf&TX3{qLTxTI!liyb*wJtQGw^nF*Pq zH+cKD2_wBfLpJv_t~xZsm2;;x4dpNsEk^yQT%57XKU5$mMO)1>mn@A$w%znJa{rQWm`i5 zY}JY(INq5ythOMBH-~B1y_IBtNr|-On~|r+-5~Sra5Y^EZA0{-jZ zg!1-66ppUtF8XH7NM{|6{aUttEf{vU3GP!rKyYb2QrEu2pd+<-Jem2++iI{=stzXd zJUf2a$QfW~GCFKU_m^waj#uj`)NT@4RQ-X`>?hE4Kf`^|+9-ajjrFe9n3}`Bj8nI8 zX=@A)=wu-}w*vnv8#rsqeLWeUuxIus#Avpmwyzbdx#!odr4=2ITT%9-6`PGep+K*L z`zXFad&@7xedHPJ7Hhg>p+|@7w$fzQZAxmz$h5su}$wKcQZ?11loG!ukSp?8JXzTuL{_9{3B( z@dETfMTnYK4<>ESEsVRyv&aS$icCF3K3CRI`KSpb*ZdPjr?}g7i7SGYw75g~7(zuY z;aKhg>4Y0twD1uW?j*zKVje_hmBT-_1}|qfV9TG62n_#(i%Ol?KIl6F%zk0gpB~ih z@5eHGL8{LdCXXp%B&6ut#7TS!=fnKH_%}9SE&T|_ zL>pkzyVKas{>efs-b>7ghUIVO;)vxllcEfR##O;y?meEr;%;M=PY}rXf@m#zr;Vp;Pl0&fFhAKJi_Ame`s^usvQpW1l>;x<{#4E>d#u}N)Ssy7O_?ogdCLqmnDy)Th6Varz6WP*R3sg_z;a z{Xzq|4j6OwI6CEY5ogHzT19vK`E(VMvxA_<9RWV&Q4sZuhx+$atoojXvc-j1b+r^h zoc9fyQUlK)_3&$I!9urAe5?8eWuC>#&2*-xoOe#rv>>;%6C`$hAC*+EBDFUYDR8m? zxh;Q=t;`IvkTJt59UYAMWd!YWj!5Ts^gG@oi>?TS=aCQ$|M&=r)1M-ELK0s`Y3RP0 zh4$6?e7}{TS+oN7LbZ79+lb=mPiSEc$0@{_lsh^5(`ZGlV~x3&@hE3-Hq%!2HOwrL zB7>XnFqfZGH@}{NjH3~@jWxj3^_J*kCWmXy1&sgXja^zdarwesxbr*Uk3|$_oQuVl z`+PoK_yQU7ne3;{!$9jRXmK{jPWC-|Sa%NWcc%A3j+7$K{myx2^zxuCwI16;^=d0= zwZ%kchzip7+b{9v9M9m^IN|qnQ~cU=5>uYoW6N+iNOATy{-rl|xN|2fdtxt2^O;C2 zlKH_gFuWLtWk-`yIwu`UqPfVKUIIt23e5WJLbj{eE7Qd6cxGMdlv>c5d4_bP>M%2s zHqqd*^GI~IEa}^~V%p|t^xJa=W55Y>znwx`mmRh~afPQA=OxTMVffk?XFa$VX3{+f z$ua|pJ^e1O(a2sC2ig8)j1JDgN&7sgvwvzP&qYnGc^>QUL}n}O$$SfIp>s{iQdo!F zxto5C<4ThGK8g4BqSSilHD;d+gGjw6vd^Ez7cW;>SDwX@QRiWF)DsK!eQ{JI0Krf0 zqLsPkEr}7h$lmkES+Pi3pA45#Mm z{Z%_md#`VzRr?mu8~xRs94e3 znI_b}|2X+Q+D#v4uBIWLQz?48BqfdMfLKK)%*RCVdE+*s&j!G)BmhD!x6slN%zNl? z^vOnJ_~CdsIXs8S?KBi#&&2ELc~EdJ#@MFUxU}gl&honY{Om~N-k1ahC-T~4Pa!L< zNzUmcIquS@Ykv=s#XX+!q$7Ylv(c5=k|mv8Z%j%>+T`<#ec+mFSp%3&p??+0L}xImCU&4dW@7Xc|bxNOn4bS=` zSK84z?%5LO?!))Hn2~6Gh)j2FAw`MhBwVCMg9YR%a-%3^kMBVbwPS{46N)k$kRsg( zkKSf1T=yAAu72Y)=TH3MJ``nzKZy9$i(^Oo;iV!#Z;lAkS6=s1rwgs&`I!}K48m!S zlm%vtau)V>j5#68kW}@wX!rcxB)Vc91scyMtLqa<=(8+M;7+xLoUt7#F@OaNdhu4i z7yn-NVdy0R5`Qy@%I*l$IL;>5OdCwO!Gr0$jwmfR5u<2raq8i9@BVP13D>!YkeN1v zr#n#`&!Q(5+0e%$mc&gkA$FuU39Vtaw zUq{g1&f&Cg?+8+5Pfwqz6otH!rh^VL)XW*#KZY{YbWw&g&9bC(TlPQQ`58YeyC1X8 z-^P19KPSq$=Ro5cZ0XtHQzV^eMvC70q_OrWEvei|dNu1P?8;*5w3to-*C$ZKE+vw; zlBcIWGL&;znli6QlY52?N$--QAFSo|&QqXY>IyXUvpgMX;p<|R0*x(H_)mxR;dL+Y zy19HW6RBOUUxF#eu( zScjFWaHd#uCsH?dpjCdh%(%9q(F@E-r1J!Ab2&mDUv^M==~^0Yu!yE{U;F4S<4Ig> zG|5ksWBpy4zHruhoTU`iCP>q=1F}>WDMw@0%Twoac{2YeN81OuSrCr!mtGW3bp z{r!cX84j$gm^ss#R7bk8*`8dQt;u_;1s!;5L?vR!$*Y!`t_wDhMeP!5Ixv;~`i!9^ zMN;IkMS_mR3?^4CVcOm%#2skDq>wR~y356AjL{I%o+m+4TFmygmLQ832~y1-N-l-N zsFc^WYjL4>dt69b(U~+=9cjfuJIb>@MFo6rTKia^<`f^Hr!qSz_{J)V(3njc!&E3# zbtIkSth~u)3Zrk-A@RotEE?U4adquD)7b%i*-mV9`iktDE?gM= z3waBBu%FkxS>!?yP0nN}>%{Cndm8R$O-t0xX?3ar&AfJ$3WRsi+dC^rFJ&q*WsM}g z1?Zi_2TUDbjK7jEalkqiDhqfHJR}{f{$@eCqJX_JrQESljyKCHQD62J)7z>sb|}vX z+8gjU_#-O${hN^BLT+BpH0hutg~@Y9)Sda-lT68Kvo7iLXMHx5b8go2C}X)2eKZxK zK|kv-XLLH&+~ICo)~JRzgkt&T2Y5ay0w#gc_{ICmtn%lWb%t4!1sPbDkcrbrb8vG) z0iIqg!JV+z_`~~+6@lCVJkFWsS~$=#4;$(qVL>rwJokTdlp1$3!$*R*>P4!wz-kCh zvHyUTt;w(m2*EP-8|a&Q4NjKVV6}!lSg!)1$6n!(O%IU69_f+%cj>)hw(75EFcL_> zrD@M$+mZ&mrCGSbI&3FDtSe+aRo4sSq|W1mo(B#@UO@j9PrUW?!PRA~fmYmtnQk!D>_ZXH4BNh!ychTv ziK?S97((&ricE%PuM73A2Q)={|WzNwx8^cRv>Hoc$$!!D zB^7<)zL-1A5x&PwaO00Dn$2u+Pv;CazBrHCMlYy+y@ux<0npUHi|^6*ao6nuZgZww zGAaTAy-{d6`3x&<5)nE(6_Z$p9lY3?JTvX+a~bQf(+ubhdsE~FX;R&~nN%GzoP;h` zp*a6<9Q$)3oydKeEqVPiqX+Iflo-5=V(mcDF)Fk?u@~xR zk}#^tlk;Te5bf86W1kUfbM3Ll@*Hb`SI}F;TsrkT=+rE*fHR5%)&sUdH>CfjBTJ9NVVFLHGU(W?^Qd!n_cF<4dt8^9{^YYdE*? z4knfLm^ZEw;_I8Sq?~7155BS&`!{+PvWM}iEtRDj(W=`0G`oy>eKq4KF}fS>wNFug z)C0Mj4RGRs4tmyGVA&COoSt<-z!H;28!n&tkH35e#3vhLQ3c#09;0fztrKX;I)z`KxR-X$4R}8ahm}?`R_o_s%!t=WwW-7plS-%=vIjQmHAHjD zASzvk(W0f8T2z8uwGwRZEP)5-h5neo!Be9e=yW)d!wGZRYmj$MbUm&JV?>Ly!4vgIz1Jw780yftBdrT!vAs>%W*& z0Mkdg_}Q0*OIpn62uX*BLpo=LUqV^8;|>=IM;U8gu)VbQJwUBN6-xkt1-CM7=f|x!tib-bA8X=Lx#gWqy+{ea?^b*i+AMi6eGI5bvwljn?=fQ;?!r5 zk3~EE@#?!hMtwGgv$Q>;ZeBpd>Hq}1W54*0G#DIxg$(Z+^syc~Rkj9J%_Y$4$iTy& z@tCONk~X-0^E@u!55Q;NCpd5}8>4J$pt+y}b*??|QR{~^d(IDa8pR6b%MSwE4W zbfG+a{pt&MJzIQmH^aec&Up366SE8MVxe_B*5nq!G~xpS+rDGz3Nm*tZ8sC}194 z)DWYwcQUZ}=M}tAJ%!)5%S2T@R0 zAI=(dpmbvurucDg)H)86r#*loXEP6aUqP0U2jZKVjT-HU2h}zx6t%*1ZCiS{;t=^M z%_pUcVzhBp22M@9!p{~oG2vb?OS|PeL%XBnjiAOJP6IjJ0mv@O2WTV`4(I zQR*Ky&H9RJ|60V($%hPU;`Z_pxV+&uUn9Pdmc4}PgPi@p?TSJfXJ!T2VO$3DfsP&K z-k}Awe&!JB?s|zK3$EbKPz&s2Z`nvQCuHyO!m(q)nADVr4?VA-eWe*nHr;siT#$q# zgeWFSfCe{w#}KOqxPB{v6myw!zQtguK4-x>7b`Nx2ZD1hv(CnI@~iCkG-WpFRy%T= za+FSoEvCiH09(QSUVXi*i2ijFms}07q1g^mTFf3gaF>0T@i3cFf*lGUamx8Oegq0q zmy9qyZDY>Bm~Qmgw&12#B}^9NVc)UmXtjL8x#|0Gbqc~0M`kBD`ap_(YBizEo?%|x z6=y9P>$RM=)sLjW7sbqk0tF zMX3u@LdPJgwET-0xi7H%^d7Q1UNZwD2RZG}nHTvKvA@C*H0D0+R^CNy;w_YNM(AOo zJw>h1;lANjbm5CUH9mWbjzaEJpJV~$!^d%;+6;=z&f%8lO+;5efrCaa-n@Co{>3hA z>H3Gr5+V9@Lx={n|KZ<*ADEigiYB#pSarM{m%8)OJv#$#?~`!&axD7UbK*1U33M7B zqb1ORhHIQ4n;H!|v~4U&Ep3MP^gFPTvWEA-F?=6nh#ZYGIKgL-{Zl>t0$u!F4H;()b$6n?Pq>ng;&#QDW?~4QUBCo)(=RSOdQZT3S zHB!#CV*TH4tlKI``OAc;a>yW36CXg=?rx0O(FGCKq6L?23X9onx(6@T#i4!YZly@M8?@%8y9^xXf1 zrQh1|uD1gRa(FJtp7KcUcD}l62^~2(l4|Z0VUyhzguT;4U;aUmfg!4XpT*jR>}BqW z#2)QTbf2lhY^!#hi{jqvBmW@#U4XK((Yi+|&VMyckmNp(!OfV+{^8mXbGZIJfO)0b+>z;sg48RJ zH@c7Gvy(A+#w(bd`+!wRUm@AenSqP_cr5-8Dwh8cRsRp(Qv|4ld&}-B3()_br*d}$ zD7r|HYOV`WVx$N)y>zDWXHL<1A02wVe-rJhP^az~0cJow#n{2z^Rwazj!rxbr@WIm z$okkc=RmyK!dbwIEF4<(7Mqk?V0ir-d>eja=H_1Pz1D}hF#~wGTYxZHkR&1n>FUNo zRIx{h`!GewPeYXavczeGk_*u_8w#FhK)G7GXyWLF^ixujl-U~+v;Hc6vfgcQ{Qw+C zo?xELX$U&`pkh!M4#X#6c3mNM4XQ=`?iNh^{sl9;z9S>=H@58P#aiEfY@aAVk4TWp zj}4*~%|i4;crYE06k|q#1W6p@SU-vK_W4Wbll5fZEvBky?bYh&FxD3&|n7M`GWhj&v~@lsk{qd+>>KVZe2 zyLdm&7UJWMV!%!t4i2Y~z2!W1ow*JtZ`J`hM_7K2{e(ez7$RK?ab|j-lX!=dS&fi; z+RB;X4&2cDhH=+_;X~(N8kD*f52&Xol#szKW!>X_a1BpJ8Xy+p*$e9V(9#k$3n zi1y~(%I*e?T+|GSo_0)m--S$%Kge?Bj5jlDZ#{7$C+2J$-Z7)(jmIg^dpnhDEFyt} za#RrBz|Xy4^m{Oec$Y5TJY(H?uN98I=8kxSs~BS*h*;H7{NnxTvfr_or=EgOQ!}yq zUOsF!U*WkvpY@ZgF{is8A-h}g~f`iTp^qHIBv7i(G`}5d&!y8KL0^lkbj14Qp(eDrq`3DL3 zvhxM&`&n=c&qu(1W+(Q&fp)?>9O-X{f9e-3{mB{JBxhPW$AQuePm!#I2^9|3qWrwA ztb;71#6WrK>1{=hB|oRa&!eZt0v~=c%gxK0HS;q_bG!&W?wl)M$?W`9cah@s09i#* zSi)K3=@rjeGtPkU*<8@c5**j5#ITp|k!{j~ZqAkkkK_)(HIDTCvJFw>Ns=hlqrS)c zXx-B_)a0o~Z?=n(oD}afHr>IsNOvfC*&^YTE!ufjIEuU3%mO{JhGzg91p^VVAsC|z z!5TJO@$3v{G@nH~b7p-1x6e)Xg}5>MNh|N6 zHYgmABx7*!Wdi3=Q<1Eih5geCu|=dDc4D;k2Hl>D6gu(9)Vgc%V6r8cN@fvIGS7T?%@`9h&VLtN*_c+wpgx&0s$`$^Ki{F0YrBE+(JD55BhCi>63zhD4BGp)X z8fdqs`4`P8rQ3kA9FNi{o$WMaBhmO$HS*Pxrc}RwSP|3&oB8FK&wLNV=lK{MR0t8Z zQe2F#gxtpWSXSJO(aSoZ%zKaT{=ZON{|64Y2cVHHNDVDQRKn|uHZn(jIrAZ^oM-`S zbjwX_sl36GE;^df%rG7Dd%2HfEHr7@(1ld9U;@chaR1?jL3GsOJB;_VVbIQxP`3Ms zbHT0HEB*y7DL*jCy_Y#;0+h(i)$u~YwDPV9rCW%Sy@5ChPLZHiUiSyH5@gvARk+ld zR_<~n)hl*1B!t-~oJG2KQJ)NV9w7(S?G(0fB_VSr6$+0fyR{?f+g(v=ex1gn0lZx(NTPVM(Ug?3|~Ej4qO{bS0;=g>wP0BwO)!+UQ3gdzYICJ%hHk~F%x1d2fasp|Y&!pTA|3G=O^%0U zX>`I!a^}2me#HoqXp$uD*Q^Oj%TkJjJa;lH&=k2*w8?)IO&~>jHf=OrR#W;BsYdj+!Dsz^z$qbXQHi6YFENd3DK<)@CJp<&AZ<*e`X zy2*DqrJaa$eLPieGq*%;*04;k`7O} z(9rkeXv0MXvfC#`N-u_!lj1Obt_>wi&i0O+F3HayDO$kVesrcRk*OSswaC$BKY8-F zr9h(!Mo}HF8^)T-o8zpTabNj@EsivwJzzp*%)#_ON!nHh6!GgQtts6}eRI~5it2pw zzNAXy3>28VKAe{DHEelWg#KsfnK}VQl**|>8qR4gFaoFtbCAcXfrM0&l-JG zoi139q9ShzitQ1kbAdgGp8W%VySwnFmd}H#J#cL5$B!mK`n8Go44*|vlKY6e*=JF= zmi-l7Lr5iRC^_)D1+Uph!5R8)zLuq*F{^%<4e9rq)0SU`6m?vS6ry)g-nlj8x@$Jg z|ENS2%Z8AIa1TaTHbZM#9o|}2K_j^ePHnZAz4ikRSMphOLnk&l|G+T$ZkRs(!(Np> zyf0&hO`jlhHic;@ubY$RLV0}e`g3+;$T{w}WRC2k>1On#N{<|!4w8e}CMrC-m|Eo~ z(A!l*>9l_rZ|_YhNuol?9=CCh zz4y$ll%^D#T1Hlx85xPPOW9jQWm86?6hcBNB9ZF%`hEYnANN1^L!X`$ z!UtcM@`2>5fBkf3)uJhq|04gix8*i34F+WAW3hWG#%Kux%UifmTI--vx*qQ5H)7Q~ z$sYHSE{l`|ycvENZ>>_0n|cx-d!5C@OY+^cx{Adq8NwaAjg&_Zaqzd~>GUIsVPTwf zEQsFLiy3L^!6p8)+3L$E?jP2T?uVN4nB)a+bUKf;YU#2yib90wNZGQl>yi_N!Oz!Y zeNXASRoNjv!@ZcYH4(QIj*8bRMYK_w{brp;U-6mWKXw%_B(J%$@DAL>Gcru}v0eIw z^27cB-puylnBw_V-#3LCD}=+awKcCCR^){t+0YfH%I#R`D^2vrVaouSF@-~2JkkE; z(U>9JxV^bMaY=0-E;l)d1AUL7qv;8_m`R>t$Qe8i5S_dDs`v_TVu@&mvt;dU7#YEc z?;-5hPS*O1p4`2A4(l+QWA(eT|8E2O_!psfkYvQv)<7f853vUqLSM2_4fAClUb0Sl z4>uz^Q0CKT;-$|y2^}hg*Y#a^Nk>k@blEw~^t*_+3$G$Z{Dg)FZ{zQdhZtQfd*@5C z2F(d(MQ1-Yc`F??w$qs3mHcqXiYmqb(9P{CehuD?@SvqQ7v_$)qkS;6Ar!%lYhd$B z<}I4yhqm4i&C`cbtS2m4E$K^DyNGV0<8N_JhjZHuJo=Q0rDb=q_10rloPLgjqQizO zgj08MAXN`9WP*(&SIi$v3)gN;kPZRY8eza^??=2sB#y3^{-YE(^zQ4Ae+E$){dE%- z_T7zn1Cy|(>Nq;zJ_V1y7ZEI84)ZQDD|(!bO+I&#rt$#6s~;n2)H4ifd z`?}^~Y}eYKZ38`+;pf0+eFn4luQuYtX@JML%V_EzgJ-qg=rnCUKECzFfTT$5{uP7U zi=_jo<{E(*G(*iKUNV42b zci@FEhz5q8#FmfZzwLet&9(0#-!2E%8PCz$;}u5UeGMIpH_#mN4(i=MV5LtP7Co$h z(z$BH%ba14baqE|Uc{82bEW%e1dlv#&;H|7=+sg8It?2zKidndtevpB!VCWu$g?wZ zGk#Pgpr|$#54v1M%NN=B_x%y>jeCxjFGL?NF2=2<;xRNYMMtYLWdD%)!@6>e^R0sa zwqGz@P>0=Cvi_+B^Ff#wUv8hl5t^W*RSQPvf5vK4=`%E1j(0QWW9gOo=osve==!zN zgCcC&X2)^+?+m4EZ8F@N#S(^bM{Fm3Wr<9p%G+ zN$&qI(z0c)nJeB!>p<=t>&6}76WM-tcQ)GU@x)B=(Y)G@orS(oC~!oFn;w!IibTxG z&1mV6h|xXHV)UL&jF7yDReb?AJBv?9y6$71ltJh9S7a{`U$=S{ZWmPHQ`~o92v%cR zuNovB|BE0M1-|r-pncs^UOnQ>;de%IVykvcIjkVef{QQ~|Gf8O@e!y-?z-`6m0{UZ!GQ~>j!_n14SOumyZ*fXabeX1&Ox_c#_T&l$K?Ufi6Q3-qJ zD%^Yc138y#U?E*%!zHJAX6s^pi=V^0@dMd2y9KpQlwpM9VFb>MfR=&_UZ}aDSC(`f zi5~i3&3>qNIg2;ew_zG9S-#IDsG9TzLr{Ujs0u7=TLF)na?z*D(IM+AB3-}W&ydfk z9Pt@#e|*7#&ficey$~g?;r#bw5!J@dU~IoWTz%J&9Ri=Bv@8ytHZO%;;e5QGybyZ3 zL=Qc%4ar@E^BHs%qhlZA^v+_mF8K^g=_NbUvkD*8zQNH^=3dsHW&T=zsk;hbWoi`MW;PDR=D~*eV}Z+0JVur z(c;5ejBL6GM;E2Rt3@V;+rGfIO&?*iw-Soet8vipyJ&mmIOy^TF%>09y6-_d zPY=a&-Hn)GcK{j|=V7aI9|K=VzF_57Ea?6dzOQ~Eb;x(5t@(_ua>o5$TY&o1=NLHm zG0eu?Mej-SjYM6+Lvp2o~hEVv{-FQ1)LkY8I|S!yYfB z1^K`)cm?j3ZpHNLhw)O@QR`N@usc|S`)jIjHvc!KF8GaU&8uJ=^$~sA7h;5Vy^mLNxUHMWnGzU;Hf*yxslVV-%|yW2NzMY@>22BXPm&S=aMIBC;iWBWVS3_dd+g|z z3o$y?P4eBr$e6qVABCG2r*cVnkhxfE_6d3I>tNMeReIYs={#SHgOk)b>W&f%-v5R9 zr7B@3m*R-{VonWshMcVXm@_R4-z`MrKc9}{;n$FPJ&1?4&y*ZpXSNJhWcZD9@cOd` z{cd=nSG^aWby$wV`CBpXhIGRIzsD`@1?ELppk}?CsjsRtVvRO8P1WWGH+8BHRpeC9 zU#O`r!_M!o;i>-wmbbH!YkCcr#M3|j;YnG4PhjFcVVwF0vhdb4jy=xEXMd=YwbmGqtNz_H>K^!%MEI@fFH{{4Yp%E~N!s>$Ixx}2h=!)JQxJQvf5 zu6?R7@8UbGeDVwmBX8r|*~_@z>Lf1P3qSCDBH}a?5EZx=>2m{kJ8m*P(ygi2_bZYb zq<{1FazxsC!&Fx^!lX5r9*UY;((&x( z*OIF*yhd%*9>n+tq0Y(^ek%j9ZM^i9o=Lzp_Y2r}C9 z$l04)iSOzS9{KK-?#Vz*`00kEcT4g7%R2E-?L%YJ^B5cO7^8$0QoghSiAOXzyPKSA zm5hX!WI%^7E#4F+)yc^}@Wb;RtcN_o@ryTcC`gzt8;>F-ECI>C<6wqu(Ag_{{G$PE zt2l+mJ*?>_T2QE+u;{l1VxNmEUP&+Ul*4O~TCfLex1GUDjfXfq`yyBXBA`y zvd`q{+%%>=ZSyLT)$<^{Rs_p^?2LvU-Ux}4EW3LgW|pU6r{i5Detd@!lFJFXs7CEt zT~=lrGG~J@!Gm=SOAkpk%%_j22;sx&sd&`ly=fX+U*xYi&WTh zvwX(C2D~0`NVP+HED=uLz6fRUvrFIUt}2*`9&eLdfG&7~s{QwHyhWDuT4vzq%5-_& zLRh(W9zB$Mv$DPLFDfozjn*o>JLU+l2q$z|7lc9el4%M*gjLJZ5n-E;JL@aqFh_~< zA4>m*K0~YwIcuUm$4%B|W{SK&Y?au5RUO{#mRxB1XZT%s2Nk*3T=QOFR*iU_S_-Rr z-7ABIE{K;9?k=ZCKT7;t6>rh#<4;?%%2sb7N?V`)jIHdto z9;);E8eQ)FtvWTT#f+-!bfxyhRwxN z{7eaD@JM$avmL?o>n*5IREX<4cSuKq2hwgj;INk)j(M)cy&UOk^-aU~=sQsT?=6-t zuaP;d3fG;|Vq&8%YZvNqeW?zQAJyV|cMUr9P-E;nWnN#b$W=}axZL;`NnQTH`gtu* zI{txCvhdqxdDGoy0-N2Cy(E6&_07Y0KhGa|eh$J=kzD8-$-NKVjH6kHF|_k_1SG$} z&jA&9VW7ZbC3P;c)aEkzt;gBuNKQ(NeQ(HTj*=d_HtN*eD>J4}DpU|noQm>~zcOD> zP~pc2Ip=HpvhetH&OO_Msk)jRa{4MvoujbVa}LgX&P1;mAIwWwi-Skw@lN*~^i&_B zZ9)l@HEQs7i4rY`sM;zCr`e;qwxu(G{4eG2Pug<1g>M}o5XO_Q) z^w()J(Jq3+$1LHh<#TDQGmulKo3dF{zHo)Mp-k5OQB7xH+esH>cU_M83%238&v7h^ z620=?3!D}o=G?|VD0Efg(Fj$BRjU1ulb33!LEpt1)Y_xLuHu3D8==9T@_w_stI6+E zwCU4H?g_~pjg$P#%1tip=02L+g4#0Kt`<$Klkww02(Ij&iLZ0!h)+Pgs?(!!{7?c; zhMa?`I}L6I*RVw!di== zb~@Da6ULvh^e%Y?@y0_BM#WCymc2b0ou z+)<_4S;={&Yjcate>G+98(J@(hJo|BsP{M~&*;nm2MsPdFWs2?W3b`R0(d*jfrj{V zf6Q8eqrj@`vp^tKcb7* zW7suCt~OES<}eNJy&`N;%P?9r@#mn=?lfIJooRapNGGH@8()i`aPq_}eD{5W@f$v1ZrjhWcw2!- zKGkq({TEr!6_~I|g&EHxS!W`780mLb*t?MBW9P{}Pq;yMI`WI`P4^$klFa-Td`npb zJHz=H`_Bt~Iz%GXY6G^Z@5YScLkReohNc$64D5CtS--OIH1z=%H+_m%#rdf0RU}!? z5{$I_jENVkVEwNanvn{!enj&5i*U9|3T9ED^h~9@vg?;AeD$>-Pc1d)qN1Kl*Gt%89`HKpkLtcr@_9F-Bw{zxBa@)qF&UY5r}4D&CFIS$0qv=`@iATUr`b>O zQs)&0_J0e-Rb_Id{f=%A{-A@*+Jii$M{KCPr{kCMgQ^$fM$Mt#p^=<9qZ92PYH?<- z9Q1RQ>`Axf_&Z&AI!=pm`ga6|k6ef6wp)>AzZ<&Z&x#mw6g{HS&_iZ(nepjJjhB4? z5aD3iKgG?uS1^89f@J-2%t-wynUzTPmcG)5TS9riop2bZF5*@z7hY>Qg;9z9=_mWW zA~}cbI&%fHK8bh3c{zHY4#LQL;kYqvHTGQGfHiNn;lj~9u!v1WC*f1&wm*$KT`r@K zoIRR+7cRw&9L$Q&$CRYEP*wkeZF_#eM&_brc9DFsQ~KI^26OO}#XND+g9DDv=HgMK zxMEBf#+w+@*XymSPcv5)$TNVE#|b@2*coQ(5u}IyMG68r}!8Kd!NRX z`pbyaza?F@_waYp6FfTb3g6O7kooZ|G&}ypamnLUy9?vfC7hF9hVa~ef5zJRa7N60 zmdbl1KwjnHX^(uGkDM@$Gp(-G$Lw`?0md5!`S~L%ioj z>=y1w?B~14K9>uH8830t^DR`Ge8$#$-|>0HAGDFV_ww!1Kd2PJdDfvkS`-@9cupP!!K$ zQL{|wc|4S}$8+2j?`r)|=}Gtf3a#1I*gd=sUo8|VGa4qI6FoF6f|l*WXmlu;rhAr3 z#%UpKqzhue=BX@eIhe=wI#9{kkVy-wG5mY3F#gkpdno#E^;tCTxd4ZClC|%33oRGj z$CLffa5VolJjAzgC`|Y(2G!6%_y^g;ghzZunTs>k7%2Ke!~@9>zKY{qY zoLY>#<0O}?SS4K^zcF{d?D5rM&DgSIdfQ7R@+;!c84~j3hSY%U5Tr%vX5`^3FrFA-t~F~ z%;cU)&Z|T8OGQ4OtxA{P8r-~Di|sRX_@Aymt9=Zau(iqmbSrPk4;D;_BpapsP&(8c z^MkoPTIQtd7V-R0nWy!cO@H%o{5GvWEyL`1x>a+wjL~I4h$?#yR^;@kM(FA&aBRFH zS9Vt6dOVrm%(MP=24@L-O`ET-Vx+s_zWA(OQRXRkb)RRs87* z^6$Mm?Bk$MJ<;!W>lyQ0+h!bd%9O_SW_;o#8fHZcZd%y#e?6`DBy)9Ot>nu^cb@e) zjLoJm<9@lXH;V3j&C8orvGX}<;0!jTj^e`dzU(dR+Qu$c)c$SC@wJji^*5yb2?Khy zlWv_e!X8#P=Brc9c*U|gC)S!vcG-d_hP33yd6vviZpB`It^TJwAFmb`##YfpcL{g% zPZ)mb!l_Faa(@$7zMVaj36W#z9SG`Gb>WADHf$PZ!E?^0v`ZG|a$-|@ zy==;r7AExRY04%k&1qICdhdajG`wNSS-V@YY`qnI&$Q>ulM=qa_cGFk{{d6Lxen z=9G2DTzT1qqb27%v#mKV&lQbX(UKYZt+=9~6)mq@vDKT_v~FfC95-2i<$G=+`hE|2 zhTGf?W9pP;H2oCFA(Dq}e$kt^+PLsd4+oxVJ%&|Z`ZIY*XHNRmhUV$!^gL?9Kfk0C z_lFTfJ2zoNSX16eFrmA-8I|Ste4=f^V^%GBx~3)XeYWKLmR4-Ks5MLFx^u+~rM^h| za7C*N@s$1a^$_+Hzl?F29|wLDW>cCICsj=4lW)VhAh$Qi2DE47B?~?aH)fw012!|( zsChgz>rK6{oavue!PuCh6I=(b?p|7+frE%{Kcn=ET^!6>w57E&qn0kKY1E3ea_< z^eFBslx)Ep*t9Ceos|`^tB_2^usZyZv&=iueCvvp8LOwp%P|@}t*I@{dFgjPCEDcT z2zE9N%gD_mHNX3(p_3wuOdp+;V-?6PZXrTt53ztQ4)Gg__Jp~2HU|I<4 z@ugl0rnsI&IL>2E#5GJM#$moXmkG%%9M8Fi{?(~ zo{mHB<-$-5IDq8vBe)_tjq3}~BRB6F_E}_N*zmh}V)6)S5l>Mo(>T2Ib7GMAzMjpKXE-c+DD4~_eS8l^0(OFe*gk}pYDUJv8z z>(Mt-xGJ+_(fdpsKAsk~KZ_8rW+-F&1hV+nB5HWM^8UYR%r+QIhr?~TO3sq@k|R4m;VFsp7XU+0KyjUUoId3JldU6Z1&u_=0#AT># zy%0H$ey~xFz=%z2@mymI9Jj>5%`5@8yB~s+dorA2Pa!~G<}(Yg;Nf?f$*z^&VB_1c zEq{RWvrn-(S(u7B(gn~WoNcZKQoo}QU(`GDcgi>}8s3ApU79eoZ82U8f6ur1N+cfe z#Q7y2Fp<1j*Z-oRd~*|`Qg*?rhp^CQ9K-z7G~_=QcIv$=xKMusi&e8wZhHrp#BVod z#$za*d4>o*$)a6&i(}`-%U~Zyjp@?q^~8f^bq@4a7{c*NHcXhOK+DNjF$VVFlBa1meOM25%qVq#P2a;Ptaj@f?LdM<-* z>-lI{?uuJQ0jP~xi|QLYFd_G#cx9z4x8w>mO*8T4lI+pr#Ebj)IegZ>!lO5@@u>G3 ztag2e-?KizYi}7G6~95Nx*EB2B;UOugqgp5*mKofe$yMlWB%tZ3Z) z?GD>sj<9wX4%DF)(35UM-;4uTIpGw3s9i@_^LyC7;t5(*=VQ!*BJs+U;QN$PjFY`y zNb|3lcB>qP)|E)h{Q>oJzkwmbR+Bk{iFmCJ7I@O!eL7pJ^yeRCGwDJv!2s1n1iuT$ ze-@7TF?~K@ABbjG)}zzIJ#djc<-sxO2+n$dr55@4KJ^X04*ZA&g)cbY_M3QUzhgsD zH7?3bQFCMsD)eg+-KGwwmN%lqR3(O9j9|c?0CsJ%fWL;0r{3w#?C?+ao=-FJB4{H9 zT6ka}<_ZVe6Z`X5VBN)7bdWsXSK$#q?Q0vjYx&{No5oGF>G{)%jxuf#<^WNkSZPDAO=&I_B#;a&T2W14VZR*2VVQyl(S z`Qqa6xwx6_hJ|L*`x7rr;s6rP9qTFCSVlWBUi3o{3))Azu2jCi#k zpN=koqmd&P1^Pn!`WjeO?nc_Olelm^8yjWa{gzaY{q1FD(N~dGl}enduFP92WfuEc zkw3dBa=5*G##{AB7*dDJcVxa(@&|)=)?q?G1K#L_QLlriFnh+(?W+wJpZ^6DVMvYJ zw;aZ?&hSig!R62poQ>az#EwZAdg>CQ7UZDV{R3P+{K9bg|L1I0;e&yy%>7S=@duPR zT|<$Z@5^39tq$8?)?l{iT+WlLaA1LCKy4%^)Zzzv7>9E5#08w;I+RHPviCH4D;;3_ zP?!*a`ku}x)bzrB{gtru+m6*{$rwKMCdSDe-PE!I7t8BWl%T?)E!A0bN{w2+Dl9rA zzSWX?EL~fR?V_u{dQpxFzfVwE{vK_XzJpuqcQWIYIg6I~+MJwOV?Th~(hYe`_+t;v zw;@Tv7i~wlz-hu_R4rJGb@h92Y}IKT5Uq39__yen`4g%#mptLE&OJ(+4De8=Mu9Tx zcQp#bPP}acs!;Dv zI~rrKInP`8?rv~?9*ppn8xUtCxza(GU_U+=A(KAg>Wn(P8m>yClbUpO)Z*nh>Atxv z&&a*MxR6kVrMF7a=jm&x**^PU*TwS@S-5U2zM#|T*cW;YFRFw1y;_(T{e?;RRgt+) z=h1igTEu?vfDnei4S5NDt?MJqSL12IIxQ^n7h8HeYCf#br%uA2Z~dSQFmRH>DMX zMQ&}x=IwM@sHZ^#$(k73eXQ@-Dw zixKLv$iC)-kXTpb91K8@Nik?yegH4eU&hQ|&oJ^`1=_w;qVjtkUUzQFx0g+M)ZdI( zQcUO(YRLHc+RV0B<=-6Xi72c>`RDiepd+1&BOl}4kK1TsMbTwNaqMe$e7^*Dl>>o+jBZ-LCceqwfpDx0s+=NUN%?F(+sq53V-U>K21i-96GU24GJcHLe*i(x_dpqN69NU54(iz8_r;B8{tx4@#P=u!7S^i z&E$Y9ShajDhW_@H8Ll@LcNV7i<8842nhdSfEGSKS12tXgd}*e^)V+q(SZK=9cjh$k zHm7mZW^8L_Kz$nx_HL@cijqqFlU#g+%*=8PC2zg{Jg&bwfl2;HV00i6mVt}tQv|9$ zQ)AZkv$z=&g?FM!R?YLl(s`?(dUhwKO%_Ip+g+sFy~oPDf5DJUTPhYT zv~R(WGn-NOzCPFWQRfANdN{~_eOalj!KHWc>+5A?m!@Ku=+77J6QJ09H%_1Rp=m*X z>A6v%m&Zw1JFb-Zj|cp=E`q~}C`7H=g_u>RVE*d9winSTv!X|KS}QQwff?ltz?KLB&3>%vL0rJoK1ccs5&gk*Of zFQQRL@=%>BZwll7n&WZ=!v#TAo|tlaB|d)GiGmR)@k8rAd|W;vab+VuI%#wNo2Kl2 z$efx@Ea|Xl6zjaWVVetFXl95;9_0Q6wh;#WLBH&R!ARZ5!$EZKu^v%e&4R3hr&7Z z-j;^wpNqL@?g*Y*ZotT(bd*L!VLY9XH{BW2GeYpU?G}tsk`6y%mR9yCM(2uJSsygG zf2|R{%FQ^{PWn|IwPb#YIfn(Bu(gH(uPo7I>RsV%z5ES@=3kI8r%3t-r7y07aHFqe zqV}}(+LbM(TiIBSSZT@u?e8P>l5{p^IH4+j4i1|yk(q7`vW_KT;g4(ZTJsX~*H+_* zk(%g`1`O&ea~tvLADGaRb|b_uRwQ|aU?ZlK>#%r&I-^Cu(eL;hm$l3Bb3+OGH7~^1 zuspP~e}au)gf;SKG6NJ^vx|QrMs$hAIa62EZ*V}nFP_LzS_89r@mOMh0Xec)#rbk1 ze^H`Qk}mI;7}K_gxy&sr_)_u$+V@O3O;~bA>qHA)s?Fb`iT9E@<-z%XVeR74j;sv)j*s7kTlvNtT5YBy^o27z^<0jU89UHra2hNI-oxnd z4=C1Yz{}=Zyzfx7v6^o!7@nrMSJ$E)$p zaTVT8SLP?t@0?}sTV3qR^NNFce@IgXiRW|B$#tl^GaDJP)3C)w`o%k}L5^QM{+vCJ z%6_tDS$u)hMFp;yq(u+W$ggZM=0Nji%uO}n(+S2*ywrqk4UOd4GT^4Jdi-s#!}VKb zW-4ojiEw$|td4+l zI2f8m( z3r}J{0tQ~f!PZZpx~mKa3&f*5MU8DwYcn=Uk1KEL^Z9puE-TRIg(Lc$AlH1eSC7HJ zME^Xb%aLDo`ShV4CukY4w}oUw6qm5wlR1G8bdc#V#z8dy z5jy-4twTc>U2b`-$7S+8Se}kxxaiLM(JtIMZZwsx+wx9c4dS*PgN<_#4p=#0@!8pE zdVdK99g!ZbA^Xuv<~2_zJb+`HHwau(i2~m`T-~9_y?K&phAKxps&UOMb+$E?wRo2X zR~Tz@-Dgc^P15Fy;kry&rq6dRBKb@5&f7Z+=ZjPM-@#sd|5J~z=HG*#^F}!B5{-Sm z1A0Zd;rWaecpte9FVc>p(&QQj|9y;^?}{<_o3P9+tKsQhi~ha-;e+_j9*vgs$Yv#O zmOS@;$*pwM5PyTR26dvecuq1}Q$1zQz9WQPUiWBXnE+O6E21pmapu31iD)d!Pz?PX9!~ zkXqbysY9E%2CN;a$dE#1F8dL!(0rt^+3_rNR;&0jGw#rYGF0#kHT8JYxZ>4{? z6ixSj!JCKQaMijRPbSr3c(QN>zAG_os~SsgM)H)4FywCgQ;5r~eBi*lLkID~=~m)5 zsK&D$shGWX6*6Znz^sujm?^*Wnd7T5^ujjy?@7ejD(QIoei4f$>;1kg8-7n7pnK61 z^m>^uU1CKTc~baZ?aN@d`Wq5dYhWkAn6G&q zt)FIL+=&OcFPX`O;V*Gf`wd*fOL0uK5*GHgun$(?>as|B>O|1EUkHtp7IQ#`WN^Dn zf2HvtCVptmiR0_>Y0VW({IdxQ`UWEWhY$Lj2I0~`Vf1wuKYwBzN-yk(^@by&b)UfK zCugy7+*M5aCpk`^``C3v)}FSn@O-(jbzXeI!h^!HeOeE7$!nbs{HoEp^_gahXXDQqz-U4jF^-VNtODv>J<-#9-RkSnO)H8{>}b zN9P?!khm%hUED9?&Y=uA9lnRiF;8JzSAdwFAMmExH~cWI#jEWKoF{YOAtNMz^fQc2 zYl7LW)lwEc^=ONLW8ry3Lw?WEI?_Kc{EG{%8xbvQuDkfz+W*qx)yaAsB)Uds zhG;k`vOa7KW8CBrF23N;qmq;CVYz@SQXIIn^GGTsb*EWe3wk|N;sIJ@#B-$UG;}M&-KJ z&m);%C)v;e;hZnNx0Ive5$?K}zqWZwud*W}#!q77N3zSC4m?oUjE2+IY5cGj*JCSS zQ1cnDzI?%;uy5!p^SO}xzxc3Hk(0Ak+4X`ZV-0j!@kyVd>x@|6MYO!)W>gs6{C~OQ z>F*->v}Yt&-Uz4DqEP-dlT48JQkH#MDEg@jlWiR6Vn2!@mwVIXk`1>jn6UF@E&l7N z!Y{sx9Brh?fYwTE)LW zpFTec=eagpi-kS4sV9EqzmuuzZuh*n?x{1kE}707bt9R#r!V{T=)~JG zt+~O)oM&Bx^>fUane&W!!AKZ%N1JhjXxn`*w4mEKOD=b@V$aIf%;;v#CJ{E0t+wSl z_jat2@42>0^s?o`u9d9H!GofiHCe_-a^0xd#hh8>C1*Bg#$B1tnckyV-@PAae(S{G z>1{Yz&f8C0oATB)6K4N1<{p``#0X>AZd!Au)VJV-)s}1)W5si-ZKQ8rp2I*JR==}l zSBG~0)AudEiDyOh{h?zc*xo0M&qarQrxHl75n!P{CA#{E)(qOwhRd7U zutSn93+3}V$(mF>Ba&VfAVZq4w`6|D%00Yot~s+BS}?~+{6()UxqY1# z6Na_nMY*oZn@A>fiezJ2I6Kb@2)Kjv@vE?tqGlXNiN!p--wDJ(Oj+6+0$2xw~VFVxT`(~SR1m(kS4Wo5YVuYFB~BF4fSxE&E$x&RsB8 zV$K{@`o?SU*gx@Y28njHFP!$Xm$BdR0NS?o<$>02Ty=RSW0gj+`J(P@D_QW>BO76V z_&GMo%y31ibjf%h7yffHte>Xg$fv%894CPU7(&{n>g{8_x1jV_{i7 zy6ry$f9ceH?-hr%usC#3+>1W<_QS*UFj|Nw>{rMc>`T0Yq-(dNH}xK7ch1Ff)fb4L zE?LC5_h>8gv?9r0Yt}?^N_7NhSA??YZy;^dM4NPzZXorU;xQb~3%%@F9UyZlqYqf! z@*I@Y;xH{#`sTdXqiM_rXdd5+pB8bbSQn2F&m?Rgb{uYRPoZGmC3IE40UzNvtqHmh ziwC(#>-iFG7QaQ)OJ#77cXQf?2)4brjGcrNAJKLpt-YPN#$+OcFa5`M87-KhBAN9+ z=~#I!4ja3#7N)mk3Y}NNSaQezt=@#aJGYC@vR+2psFBWwQq@!>ov`*szZMO?GpnUS^=IZvSvkB%_q_i>-m zuHpo$EY>0Lfghaw7h-a&Ahh$1!lA*NVB--7D}((AA9)z(3zDV(?-U$kFW_zaYgqRx z12%VsS9t3#W}kQjtIyJ-*<1V;B_)WtDt?jqVQeh)=g2Hi&R9K*6CH=qcWFD8$E&bi z^Gsyw$KrGRQt0$^!>Oe{__se2A4bHWMm-KmNwUA2D(3~8)96)v0R?W?5cfmwX_>P% zHG6<_PLC0PSoZF@GDQ(jzlpHjlXX5zXO#3!2rDS1bTL=@yK=7TWDbw%!vWV#IAf2T z@unX{_Pj_OS?db3Q46qfSAekP)?#j+=sJN(__8StPlYd&fBOc!k7i@guls1w$wj{{ z&(X)=C8nGz!1X>w__Xsa)(ayzx1HpZ;=kjDv+SKu$~kn@BK9_%M}8c^Zu{CxZ-=t< zRHS3Yr5OAi1%ok723z|+>(P1aQHsu`2 z0j?Q$00u=tI1@il^2hU$zalYpmszF%8Z+MFT^1!4<{3n{|s*@4Cy*Pkl&Pva5^f*oy)>oRL zIwuPMpnY0226kV7<5TBg@lY>xKerM|7TaMMbQE3mucAF4;pg7h*f_cj9wVx7IQSRL zB%@F_=pTyG8erE%fvUo^dmf>{*ct`;8cNrP%p$gHtMP$#1W!qCX~vd$EX5FZnIs-K z^{+7Od9R<~Q+S%quiIRf!E_6uHDngIRZUSa(#9 z%_VQf8eOLUmdr@5HccJmEVN0374lhE$#G0rb;TD}Y z8;J2uNw56rN(Hvm$v3A(r>o{y9zt-M~JWk#pl&!!W<4<-ktVx z&Il3S;z`F(Xi@+d%$>v` zZLD}r=`Aux?}1j!C77M+3`fzB@`tX$M9H>fbViv`?yagvP<<78X?;Ms@Wpz}dyJb(ci?*SCcNv@@x#H7 zm7fODR`P&dWd55f`jOuPM^rYrVC$?1B#Y)=;FSWG$+uzAyaejv0hxSBiMPpklw!Q_ z>e6Ljp>_Trl(U5O>KuspOh=@y@shb{6x5IJhI82&Om%pSM}5CwfR7UI&eG$Mai(1H z$WrihW!h) zR`BB7kZyGS`c;_Qd!X?~y5Q!>y`1MKme2LrwO~JT=3IuFT^@exe8;XdVH!FZa&2vM z&NOe$Z&Pd;KA^4mM6GF6Wy$qV%~&P-^P6*ZI7c{_xv&1B4CQDmd7+!8d4TRC!c;Dn}Ue zO-f5vPPF07+ICE`YtQf(Hj?|WWcnIY9!WEh?j{Xd%YOT~Lj~OD2*>s4BUn7ViDuEl zpc{S~6UMo6viz;E%*BH67>q1;L+1et;JP6UyEbfv*6rg6{g(w_Ct=U^kj|SR$%je5 z(3KmOY(3JJFLPv%P}rXRrrI*TSu0+NG^O%v(b8_IGih=?E@pgzS?^b<)EAcgOX)`( zBOTxOPQXFtCg1(7nI~MIs~PKY-CzMO>~lrX^DxZOh{eM-lC$}d4b`v`EOBkX7$Y6F z9NLU?C%0lro-G&7vf~t2I~u&UrFMKP&J-=l=c(ui-84ARssRO|UopDi6{haIk9N3@ zCY#P-hmTPx9QtoX)wAQ+F)tg+Iwh!W-T*TN z9sVBNjAxx&@&0dHYV5FM`&)MW=+TxN9j&PQ&y0U`joAB?CXJsda8`IF^!pW~Ww#t$ z_;pKsNzzT2a2j@sZX7bN9h>?*L*I;O{2xi@9oKXJfB&Mr_f8s`8Y(L}$KE>`Ng=xu ziFVmDnORxMRz{Z)viC}MnUPf?M1}Hue7?WG-1y^q)$4k{pW|`Pxu3mGowL5{tMrZnhe-w!sL%O+_cw`TVq|h@RKWzDxFx|&yMK*Koo?u|sE%a)25n(cR0CoQoHAP)G{6St0aHx8H~1n zW?;&n^IKA#wR#Nu7k>sFhooo zA$-(u>0(}tOOv+4W!Wh_AO8fJ&NcX(E;Gg%h74knGBJ*Wdxmq&I5ifA$>0`{R6+YdF4xE4H4|dMj2BS5T z(X&?|UVRLctl~VFNp^oxi(@c)a8EeDpK*GF5_N9su|t&w(>B}DWUdnr4Rhf=8)w>Q z$a?w8hQ8M<_->{#b*}5OzF3`Ss+%&fYXiQB2fBrPuZA4?00qMtJXhM4kzVRN-7^<0 zmPRAEG5`tlgVC@|ys-Dy;OFKXY=3hd-(I}K)KUc+jnihgPo{KNZq9)dTd>(ZC#DZ{ z;<9xPymYcTHRoAzwv}Wm!VP&kS~zQS)!9K+nTY|4{P6rQ{tU0f4V4%kU)_(f9Zfj! ziFkJ3&&8m#^82a{Ly@O&5+r|K?63>!h3AmI>^bst>rg5`V~m|JqK;WfUdE1dGFvd_ zeha=XmwB9YalcBnX2lJ6ght_6~FsnW#;ELrEbn_F8e){S3IQy zE#f({(pI9$u3!Z29E?l-#zW=He9=QPk-Q}zMPG`M@Aw4|TPd;Ea$P=MZ^pF>vNmq9 zx6GY)=<=d41}bC@efVkuYekcc zupu`l>B?;XES9Oy!-M64*lIHrimk)ZEMN(2M7J<`T7Yq$k8q*ztFYOca>xnEIa`ZA z`mwpp-z+)Y(~2twSTb8?qBqY953!9YOJ^8!NtSfhSxHZxt3LOZ=y8T*(Q`W_QFY=J zUcJ+YBmOk!7Ns&APs)IQ^fg7wa}u4L?_uFqUM$p#hbF~vrYE{VEa?k|S|T zVOPmJJIov{j4_E^NuOZL+=p1}yc&IujT3g*P!v5LhlVk8(BDbUm6G}IlXwx1lY|ME z{~5af8sK(GiA!Z42$H_TdFdMT`YSn+L7JQ#qsb9r;vo{j>DT8b~;TcKT_hvEg- z5uqxb*-ff2JF^y=qI(|`f9m2DP3WB}-H#zkoS>^rjaAA#AoIFt1yu&mRpSE5n)#J# z@up`g2kORg*WjsKA2^I1jQ!ZE*pPoT9$-+$TDVwG!K(C;D9RrX^@=%gKk^rp`_yBEXTR$Ihkz=v_ZHVOmr(;>`ML4Ci z0Rfx$VU$S$BDAid&H7?=&wq+Ju`e(#W!67vW6uqtNWFs=q zG@-lnTz)x_%2^qSJb59S8tW!->+>Pp-m4wYEHRSYR0*6fXJU)$Y)pw7ixm-*a6=&p zD}ODAx=SXW6zs>gcH-N=drnvvSJ7svc(9%C;il~)lnyAxr2(&mHz$lF|4-=E^(#W- zeqm**ocYBMJXkV|-_9rTWZ^7cF%c%=k5P;s*Mmz(IrGi?CN#QOh!Z(0;FJ&rjq?*R z=KBmp9+{6}k*naNxmCJ$_o3D9BRJG>9RIaEgZGISP&i%sv(Deahu;rGhb%>P`fIfQ z@&VaxhyCXwR3M%xIVV0i9bOz@sUP z5T-l>FZ)i%T8k94^;` zTb@bRbvc&GUE}rUTI9`Dpr&L8`}R$tpK$Xgsm|mD$0__EoXZgQWOlt9e>Ig{WnB@H zleXdRn*}%|o~#uuQ?O>>BJA~938ezzRejzD^_pG6`Z|D=RYwszwE!nyN?%6sO;kya zU~SklxSkT8?#nOGl+T-TtRfpEpK|V63Z2I$^7Q>#{QWqR_6x_+D`g1B`1^B~nhnE7 z{eio1Gg|s?#@MS%(8XjS%AYR5+#4&UKV$>0wc3Us<98wQ)&a~LkcSJ}!nuyQj4Lnh zAW-)S8pS{CuTg=-Q8j2MOc*n3C7!Jp&bFN}&6AVZOf`<3R?T48 zR@^+&kmo!d1aoqhQ zfP-vg)_%^Bp7rXy@AC??ww;1f(|vdtC0)N=cA-`Desn*17#_nViXo%-h2Ux3fJK>pa?rYKSA%C zW$<#YLNkY7i0bkW*JST9dM#S)4^389%Nm(t$ba(OkH=C;VaKN0C2{J;IA*S%$**%G zxZvIQ-gl@~tneZ=_TDK+N{Un8}Dw24`R*`_Y2?&RUB$(VSYM`yR4!;1G95)>!_3U9z!KyfX5ces3?`6V^$b;Tz9p zjo`)4A@>*Jawat7%EJE zYb|qjwY8#^tqrpdNe=b2J++ftaAP|szEW}Fn#->IAMbjn=%Llqg$LPLx;%#@aq04S zK0YLVis0$=w4cImTSM6@JAhsHcjs;Ww)B%dda#_GRm;t3^-X$ijZOG)oC!xvHDyO< zb82s~?{BD%2@m&*cFEnMd z3DWH=@8Z|GW^5B^%Xz!)c`Bv_bALNZuG^W{?Oi$Vy4(Nh`%f!UnI9@XEYbH@=_c{E zV>}1j%%Z}EDDJYF%*4JSJfS#*MTK3t{;Myy|L4TyTh>f2Fs1(^BkC#`@l{_VUL0V| z`@+^ zZZ|28o=svH|3}uX(h00+Gm1lh_Gk4W&`8yjO$XUBf4Fd^lk^$(UWci{I&4`X+OL}) zr#Km~mt=02{W4{(WPZ3*1WObhGj2o*;Sr<|5_?-Z=^8oPLgy` z$8%KHEWUdl#b3WB@yCeK^jtoWhgN}C=ezM-z2p_d9~ZPh_RzvwWViYWyWn4#7*H=< znFiQyQshKyWsd%&#;q}0td7*>t#a}9xEXQIY7?$-G-pJCCAY~s>3>+dcqdD?`(Gk= zO%pyt!c6wv6+zoxVLWtV7%QWDaO_@h&Q!LR?qhW-$bB*|;4Q{Xdyf4j!Xu1$i64?j z8@=!YZfkwPr*}WGrBx#otDCS{D-~|9Q=@X0COgm8;kG&Y?6|;)@eQf$oF>fc56LX5 zP7uTSY&ss9&Kq%)`LOM1n%NJa>lr_KthZ<3EXgfbeMS%QzMT!ag1qGy;Hz~B%4Jt! zvF{do1Q(-m)>908^cvq*e?m?6H@K{;!?k0L!oCqrSZ?ks`(oQ^G?I0_ZmD|nT+3FF)+_fMADXp za9)s(4#(F+zzWhuvl9!o_o8x04(bB)MC&<;HBSn$Cr|pYvad;(*c}w_dI%NCjP8tk z2k*!NM{~Tw* z>cK*66))kTnbPMm>i|@L=b@`^0T$#JqR#aqY+ql6LD(%A3!}KU%YB?5{|GfvC3so- z61|e&A@5`rwoOar>HlTwY@;|~RVXd{_Ti#d?zHNvCLROn2kyBVoiw6w`_5QgwGT&+ z<#W*Od^-N9Z9~ld9PIWlfc5)}NXoi_P2=w3eZqaLS@syMdzYeb#}_DwekFaTWys!9 zjsnF`$lUVHiIVw$-50za9_wNIp%zLy3oH@n!o7TsLI^J0I|5x_Gp0XJ15|Z<^%aC*s4JU^Kp; zg4KcZ;IMTq29@kVz>rgDH|jR_&XMN{!!$qS3l7$Phl|B8{3(#Rm}ofrJpZEip}&Z! z|BFvQ8!+8Qfjz${Qrkm`_Z-DXxI3CD`$BkZNLLQ-VMEWf3KTEQ#`n&#m~?g&wv~tC zM#tI6$(OnKg6$Y0nYw?*mEw>fAaVF5xbG^#y@b!mN|gKgEG4cH?&VF%Fs7$!$Qe(AwkNK6QTaQ|F0rb>8l$!C`rtj2fQAZif2AS_!t z8fQMmz}F`YO|EBQd3pgRmlmUS=|?;*`3s$XD%3usNo`?5cC*&wXUWS9AFan@`g*)S zRhN#E*%;hbn_UC7xLIE~g7dVvFjbpPloL76Xad*w>B_&MW-NO92*w@P!usk&>~H?e4=Huc8DGq^g0$IrCo`o(ITyrKXmTjs;%T`)TI z83mme(_!nm41K5WkUM?>8Z#as-Rv`7il^zwK~2Wp6L)%FQ{I5^bo!W!$4|18>Za0% zZ7gRK@!Vy|d0KcnJ>`6o=P7d}$ymgSHoRtHEPeGxuup^=b(MY~vqv^&Hkk><6(e!# z-B|QJlZ=XoYv6bC0G_;+jL4o>Fm6?k8J~p3Xr)i(OQsyV+me&tTQOedgfn9;XcKP6 zgPtZ_yxWkbH}t6aN@iApn$({s{VwsMvzv)7EayhQorAczvKfmPORh2KXweOjuc_f$qHCDZNgZ|n|v4VVD3lhWOB9PlpOJAO1`aJx?FOE z4|IB)F_ZlbXuDj8H=YQmZ@usnt(4gXoghx;)O~B&OUV!D#8T#=rxh54Ch=+J~6Nay7L~uhr4wg#>`d<`}jOxraCTi4qcocfVO`q2% z6f4}uLoqJ}Zh>pzntKoqo37%?m^T>Sy#Zq%X)xQ?m@m|>qe|poc@M=IkvuTak2CKzKyl)3q?FEv z=B-eK515LgRts<{ViWv#<>8v%T@0%Egzf#Ba4CeB)5eVF3vIZ0ssp{tTkyPx18*8N zr}|y-DmFLfTo+x|-BjaJd+`Ye)#A(dN>~=Wk*=a=nCbTz=bujI`%&)VxqXhb!NQl= zFczl@gK%wg6j~)L!?L)Yk^w42_J5_wjH<;0Uv*Zw8*_%I4J$r6Qp;TKI!inl;O`;Y zoEv-SI&=S1d&c&+VdhW^uB?``<8)mb4i+wzoV6X2n(*1mf5?9p#(lyKifDZeKJymf z%BIoiGd>i@!s2n>WHlD_K7gFytEgD`7LBS39Mwsiy~V$i7;8tr>nr3;BCN26)F3eYf!&=nAbon=IMJVTdHskJrC$OX_ z5v^JTqel4uKRxE6Q`LGnX&uF%-FKi_RUuvN(v`Sc)}d#XG-%#}56xOqH^qw{%HI5* z<;6)AEqSKQiP0VHgpXy#4zrD^-&coIwZ$_cbAVd&pYX}9MAJWS(KBF_=*0SLbMYV= zHD_Y@`%#Flo{W!Q7vTCaVfHRQjyEX}pnJC(ei4#mXlcl$zRjrM>%^lU+*#4oTXGgY z(p%Gt#%oW* z!khMG++uI?w)nBLoS0eIoOvD1`Bqh*UF5(0=S?H*Grz!Y^J}Q+Jc6U>lUc^YxKjL9 z3mGfH9RU5A-+4|LyS!jNgU-01Aes~f$z ze5Nlagtn%%RdCr}4-U+BrbUGqen*D$JIThDK(Ea`EOn5K z<}LB$D`ugGmt@GQ0@3yPB;1v`dem59Cigo98azeq%b)10rOu{T%%G+ZnO8| zbVpyd>MlJt6Ma}>CCrd07fv^^=k+#n9?dtR_dzXgNK<0B|K$1$w*f!sI0VOLvKc6TXIz8pBzhyP+dnJm3mTc`2!`?i}Ig&2x=DjQHx!nhl=ZTO)q(^D;+qZccsC%_`Cic}uv#?RTlLhqEGstm^Uo zf87P~6IgWFLA1;(7`!JLC>f05I>QiFEL?uQl}L}+58KSESaP!*156b;W}6O2L|70` z4!pe9jT4f*XmZe-pDevOXt*bb4{@Xabw^IDvgOmgR&)@?*>(@%Qs-;(T$KvHe`&(a zorMqHX(~rXd+}=83%tx-3eEk2Fg`E{`+7~rR=XwmZ)6tAo)zN4>t}FYTaTVQHE90W zgvWn3XVzV3ezx)8-iux|KJCQ|=^h;B=Eh$AoY=?Jp8Jk8V_K;>Q_79lxs4vZ9kqDA zNR2L{Ck;M5jZw+|qSO6GO^?mUY8#4<3H@=*G6Z-$2R-I)M0Ue}Sg=C8rlY^2^*zaC zMj9|fW)5ck#P^}=#(8eyRnGC?TN8H%407c|BPW(+*|YgO8$N#|nH0%Vj4P1YZ?GN@ z;5eeVn9k)s?2jZcG+`!Qu>O8h&fRmDct=s?nT%a;^B>+nlXxjA`7- zkfmPwoV#DR>rI6lyJZN^CfM*ulVWVlU5qg8K-^Cq2-hnU(5`VIMtW^SpZx{cl>Znn zVt!z$sw&GC8SwgfOV-V_W72#_;Z{0J2bvSlJZQm>C+yjBrY#Rjw!gl$6+4eI=athY z{FZIRHnP6E_!{uS(Kr^}7)57OFIsv3g7@fc&{v#**f#PzX9kJCHvv7hg}Htw2kUNL z$JN&FuwPe!TPuY1ks+G$X)7A|+j4y?dp>J!&;Flmxw=<#Hd`P)!Yi!kEt!R}+GhM9 z*&)MZBWi3gq^IO960aq&d~q0GE0KlPx;(S95No81hHnFK(jfqcZ%)P0;3Zg9u^qO< zPhsPyN4S?)jVn`>7&c9^y!FN$^+LMfT3a(*vPEsqSaD>UoG)QP+e>DQlI+p*K9YyN zZOFJ%@!u^ppmCM~6@rrZ_WcxAKkdtct#;I&UXI4Z4fwM+3^%$B#)=6cDDEg3WXZvv ze|8Y6ayNZr`x;;U>QF72V`XO@_75^-mZAwgg#EEWxFeerjH$ZAh{jtDd3&M(=Znt$ z%3u1M-sp0agY>rD(POK*$t>51;=C2ZXzSujw@g)jPC9|}`E$@~$#584ABIuaB4DYz z1UBVc@!skc{_q23Hs&i! zx2-bcr0e+hKNV)IR%3pohVUFym@#-ZT_ig&ciupbobO3xQ}NKs|I5u27NX`~2vko8 z;a))mE}dS8tn(YC7dsboYo(Vf;UWHozlDFvXPmwE1G{JbLe9ln^c5eE+mzo($Pw>k z;a?2XXoQ9IOV4Nc{7eH^dk6V^+?{IOKRS>F2RU-r~snH3hKa*bsYpIC@rrnzv{e24QzHwu%v8Grf; zOUW!3i)$8Ox}S8K=uJVxi#Uv2zXY4T*5kWdHar`1arJgShTS=ZiZN%=%l#sryuXU% zF1OM7Trs?RJP{Y?3+$G&RYqUw2(tc(g4_n|c_e+u?#UckHk;?Vi8m=Hl+h9Wn6t7K z-^napE`!2x*($7&IcPsHO4!RW*mrN9bf2xjGsjK1o3|6YFYm{gore&VBVN|=Cs5nJ z5V{SQ(5~zz`aLbitqvuKo?nKxcPemWa}AEZZ@}00siN;DGg%{^@xx}YWb#B#{WXkj z(3!u!*;9(wsoQuSgYRWPIbWFKdlQg-c`kkqSO%YxwMYx!f(B!mi<<6(%Y_^ajmm@8 z+yb1BzJQV0H&FPp7@fjP(f;ULw7Bva$@X>F|3-lclHqHap2D@q6S#bN4BgI$bLa3- z+Pv;h-&d`<-OiNH)=B2zxiJ4`uY-E@VhqS$gqCh;*lWBFDW5Yjf9_6%SM0^PfLxf= z{fC3rXVAjy3ThJWqVcHs+f>Tn{z&FZ!dR=>`41);O1xB?%G3ABf7ET)sEu1 zx(O_-7|uWO-DrB&ReHCyc(C7dY%Do~>OY&HDr~ir;+rwpwHfz|vM_7*KEw$3*=x-) z;W3@TCetfurB(ztvl8f(y+P)jD(DRQg@0`nnAT2N7@+Eml)1N7Z^?N7kv>b6c)spA zlk2_0<;)byM2kUu_t%db<}~NB9l}qJdW3ml((mznk2wD&XRNdjo%$WZ{K{j9>2Vsf znq0(uotrqb@d3PkzreU*A2H1HJ4UbjgYDCsu)i=+eNJn#-6LHtkU9T@IjK}RmCV59 z2~2-6i+{RB@sR37p6@V{tA6!l$gNgP5`D4wRs+&*KEdbt7ZLOK6dKD<FO_e4 zr7$8tkt3hS(kU{Ul@F$}$DlD(G8@8@-K1863)9+2miES9;lRJafA=55e$7K<%b^I@{flqYjv`?uI%2A6v1wqv z9VbxA(u7E`ER0}Wdp4_;g6iTSJ<)2xCK{Fa^eGd?#VLA zS+`2%x*XvabWdd8(pXxW&E(Rh5zOBo#vt_(e4p8y!`rs!>q=)%oM_2TlMPr@CG6>w zTD%gdO`k#?eq1A5lV9>aWiPghlbrE8D+YeH;g=*kHi|x%^WBl>>|MCQ!;K&1xfdiK zI%u%0(?61>&sZ{q*>OD8FNO+%)A+T^M9w@kis|zP@Yhr_&QUUxau3-he!)X}#=IaN zi|wg~d?;s_Rn_wM@1|_rVZoq_)*Sm*{A;`HdH72UeyS25YpDwt6uB|vpgX_HbBjfH z&IwJWlTiu{8xonO7|&U+VtCJ6dN-C#Vy~3Z4EQsMFE(`Ht50&rc5&iB?Pe^IIYEbN z6P6Z9Mk-DGZRceEu)uaCdj!A8zCc66OC9>=_8o2rw`|T^kZf_H(Ka4V|No{Hk1CL z*n2vnBkSnnOp48zODhuVrVsjzSY9tN`VD>N;F)g%5%##_;Z>tdu9sneWxLVFPkt& z{75ykg#o=enIVG{IB)$dy6Z);%w!_lKO4y>8NGS!i!Y77SaFG_WG3v(rHB0v#!KdR zd*nrU?7AX;{hPS|`7ZJ-9%9|jQk>f_Ow^H;q8a_b{+i$TeMN!VD&j>-l+GG^O)fj9 zLk-E{_VpL-`B4&AT#Dn$#%ONV4yT6A818I8m|pWcikI9$Jmu;PlfCqa!37+S@V$zmv$+{LklPcZ$?E8NBh92dUoN8g|LE#7B!=_hKw zR)y(5#Y0gc+Wvj%9KJf6osLC`20DR@*NtHD$R6A^$AhE4>vP=13anpWh%SlQ*r~k< z$7gJWOYRoT3fYO!gM0DRKSz9rM{%On2|Rdk7T3k!*i9J3^`?)p`uB5WbQO^;(5XfU<_xj45!nUP~K@ikdGgH}`>`e1*_N?VUBsrad`(o*34V3Ja zTLhkel6%X!1<0Si3g@0>qSLh9_|-cHrI|;Ne)2zLc$`3FoOHJ;p2r;P%eZk_^0=FC zA#AE-j<-IBgYruZ{`U^k=caP%ngr3zrc<+IEcZ0@=c^)b$$3auR^CHg-kFK_(Fu5w zIT1a#hU3bj6cqTU!EN|vEIGOtS$Febf9(`*96E>9*DfMFUos^wHxO8G8_~Cm&?fjk zmdY&j*vY5Jy7K~ioy%eCBW(2!!jj34<36 za1x`!urz)q8l{Wz$NlyA&}T0eWlLZ5m`muGb{n;J_i=0BBRsh*jGif_XjFcI>Bg@x zR(Pss%-`TaS~<$5d_>UZ&v0A!LwJ%Y!dIBh!VMFJTR%uVZa(ZDr_BzhZy~(P8mNwr zKy8N*%z8W-&mPS|{*cw!B;4l}i;lzn%r(hIJ;Hvc*JvL29+9g);lb)kq)G>rz3o?I zwy(zJ+-gLceM1{z7^v3O!1-sL{M-$w?2$($h|ig)vdb97N#W_8tNYv-GY?UgPp z!wIlCJQ5#XPr{0%d6MzYz?s0*v*w;Trry*6Sa{G&}IZb&%I<-!Y7yd!H8V?&u zcE(hlDbiKn*Gl>Vg+J5fjSAcAsZvYy+Dn7fSw1R>+No37rnEonwVat;^9yU$_M`2Y zIK;$_L_yop(Eb?%uZd|WaoLH}wx{q{`f7^q$vbUqfN_nA{Onp>vRaoBqJ0+p)nl6c zsWnB9xBt_nt8m`@M3bx+{q=GyEjE=-H{%3ts+1@2@`mwj=huZ+%T4%cc`+P+twgzB z7+foc!Jy?tOu0A@ZRc!2X3}BQ+uXqD18=ag=Wn$377u8hcnaG|C$on1hRB(2#AQ=z zxXW29)Pzx8j5+&;q0B4|_)C74-b#A>Ua8BdFqs$riQ~70qxs{xFW2-{;-=k4k(QE- zUYR4%uOJAWK1E}1`!rMy$%cxBbkkHyC(Q3}&=3#*zYHCIQ!$aX#)6eY#M4z|&Hucu zImg+G7blD)KPeuegaFELhT9I$Z4dX|^3V zj<99LEgPn(He-vQmRz?~bOzy#jM`&F%})lrA~W`2={B%$9~7IqA+kV zn$HWy)`_As{EftknWAr$?tt5iNB1WB!bKzp7FSBployU>CiN899&V= zk(q%GJnL)6UnMsD=wZ$7t`^j4YRaX}jM(~-KJ9zz(I;4!o_^E$^k`@H@2<*#yRw%^ zuI$*G;mBPQf^Mf0;9?~lanFMoQgRhtCcQ;J%YUd8-TGC72_4dH*s!GqFEn+b!c!M= zk2BSVI5K9rbbl{w&MvOj^ourUx%o>e7C%^l ztwYA*$m1ZK_!xzUN0&iEI(hFK3!|;E6bnw&Vn?7ly=2bWUVI)4>K*ySt|il!dvHuU z589t{;}0cg#@?~#(9SmO-qV5+Z;W_woGu@C6FpRTr$b_dRXzV7=1&+;We0n@dtVp6 z>jJE~I2th>La|pf9xI=%lI~s6Lk+HCREM`vx$zI9nrpLG`d|)?wd3}aF3g(i!DESD zoS)^%{}!~QzpXQUJ?uH{vG9BUnexYf`W!M)lcPo{^L!@-wr^L5cALM+tTluI9}Jn_ zRn{TH2vkf8z{tokaIc($1*#j7*!w?N-@6B|yI&A0Gk%3e0}f1S#`PM`Opo#8{k6WL z^R=Z$d|PJhZq11+y=ng1gM;q7@_wKrM++b6lD8G3JxpnS%7B$}wmzjOY{tvOSk_gA z#c7)`QM%yf?i-4}g_Ch&z+$M^Zii9FGg#zbil%OL_}oc@oAgZC)=4^U`nvJf5O3BS zw53OId%ks$4DpjT>>|3^CnZlhj&>zlx8Sg)HoUUQoav>8EXgtcp%G* zexifY3N$}IQW)C9rITzrIw-HeiI;m&T6PJcj&HE7LVW6Pw7E6Jg1(k5sP)mED%*Uy zP^Ue|HtE2wQ~Vg4+m>&(`LOzp2lwrF;mH@mLrRfuo<}mz2-V{w$>L;fRHo;2;mZH& z%|qf{`Z;?6dX@!3$yYLE^J0;GVzuP>b1?qj4XluT^G9P7em$hPPb|0t!+W%&=R+TA9`Io94`JUD)8nxUTMt*{o2k;H zvaCDjH+hO_hmx^#P5?Bg1fzdk5<+b=&}zXEVO-wD)6Z3i^-yN~IRh@a+Ki#Kj7PMn=gAcpT^WgVsk|{o5OPBT*yf{NNkhhYfle>oMkiX~| z^c~SpyKvstd-!*KHjW<~j8j3Q5n`W;tNI)8vMdidn#EY3{S`ars>r<1kV+Lc)X@{a zc0bX%liKm;6n}mx>d2H;{tPi{&m~^IOkeE5tWM6H-`SP}R+{tteCbooQ|IwanPJt6 zKCtc!>UFv>wE7Ot%$O;8szK;_G8lWhrlKxl1CGZ22dg{9_&29on35_~pDB!hAkj(^ zWQKRxo7N@m*!hA#)$PG$ul%_tq&@9t`Etrt52__Nvv`&*lh2y-XO8sDJy7SXye90R zRgbr6!aL~Km3OV~;rhmz*fVwjhRhDa&^t*8Xt^F)$it)k_pp1;S1eqv!k<<~93+|M zj2^BmTk0)m>vlYx@6Sih;5cDOcV5??$FBKun4c##OkB9Pvz=s^EExUXfLA+d%6_Co z+j)P{&gVO>uIs_PEl+Upb1b$x_s6ok!!YDiEY^ptMI8@gws;xbgp*g&QJD+s4A|ew zhKpv%dhxgwkGQqt<<7!IDecIUrT*02-=216t-0Z@eCB?;u&5=RUD>p03to?FM*khAtoWwK zhou^PoTkj43W|K)X$S{Y)uCVMQmo!I5HnWyM{Me3L|<5nc@evW&2tgf!DR?|_YYq^ zbr@J~LFFqg#JlFf$unBBvaUTB>-clf&JOf4Z_go}TeI6$FWwZk--|>?>Sc&FKFNY+ zp^{@Q)8Sex4SrQoVcNox^a@kw0jrHD2^og`lHN$`9}1KB`5511D_q@AVTySP94l)v zVU7mJXUO?MeCj=959;jg!_%tm_-2|PQ$zeXwnIC%EcWG}J6`-YttIb@mt)W*J3jwp z#e*wNxjo;2){CTzT6}jMi^ed0soej4b8)6N3|&|Dk}TXX=!}iW{eA1O<<1f0b}xc{ zn=g3rMVbFa81hSV8@k_erejYpy3cM+hqmqbdT~3Rz0-!%@B8pd4=;X*Xi39%&dl6u zPqmxPILpj}Wv`6+yr%(GWRAFE`6OQ3=FF=qH*s)JB8qnR#kA)Gu=7+nc3xbLgR}Qw zx7B6gi@$+cvI1Sir+0Xj1rz^s;6Ir~ReQ+idO&MxDz;%`jxV!B_kN(@Nv~`-zN>X& z=y!YBA8j~a800;a%;a8g#90drm^&_#ojSGSZ1In`-na_gP6lH7#6H+*94Z>rJjAZZ z#Ovx4NOXUUj^Qp|Fzy^fQi@3AIZk?I9HRC#XBxhL$n+{T4d8(Z@Acn|tsbmz8rZtSz%h2Q=; z(#Ow%>29{nt+i$k$uhJbW5yfu+%+{ud?UKE=9(ewBqo*NC>NMhyNM$DRj+8FsT3FGT!?R?94$Z!!rJbO+#k z_mOxzEDi;iR%1`VK~(!(!)RfYJxpsvyt^hZzB3|Bt(dx{IkSb67&$?>7+SW>>utl? zldYK;Y{~Pc=KOL&^1!)9%yKZ~9zO&2yP;2K%S4WUDr}^)ovGhbKC3gY;EuyWtUNLd z?~;e$M8hQP_g;kUEw04xR3R3s&GxtNx{>FlOt?`ba*j0iIYc!v-6Ju^nd2W@0#C`pSA-h z)g$0!5+I#0!O-d*hdbh-e;>IQH-+IQ{#H%{3YBIOB7|IaW zwp?^rj}|$1CEu_HW0p+C%@^{n&rF8Vg1N%|lC0#x-8i@Sq;w+Oz~wcM#i#TN^)kyl zAstc^<#XqzTaNZ-@9?7QdpwK#fSoNqq3QcdoK~uam)}psU-%8@2$=&Pj%RF#XC!8_WH{F(O`c_IIy|CxoD({>_;jR`{w?HKGhvKSf9H(=11eONc-1j_U-NnZ0N zj%(aSgv_Y!{w>1X%f*Q7^ANo?p1|v42}TTjf&N8h*tq=zLdShULWiGl+?~o}lM=aS ztaNGXPhhX{L#W%n9gjyDv;Wl>h)>^vr1g@o4wXIXZ3O0O%|U#J6?hVw3HvGgpp>77 zyMqc)x9|*h=$}QM!g=UxFt5@bFKB&95lj>K>2e<%`fRcnz!%Z$;$| znU`J8!TZ05;q0AK4*)A79E!pt0r+uHLJ}Lg|D$E;Ii2 zJCfPOB91M$L{V{X7(0{>;r!?hbXZ`;wu#?`&zBFanw6N=ehw_!B;(Z41$gX`j(CNQ z@b4>`evLi2m?^oKeK}ZIaRjlY!m%qa#Oc5*c=_cv27Y~r9mii_^$$*#As&ysmq?zQBRr;6u$izJRgwW#-Lw)P znr(#s*X@XZv>Uh74!~3E5E`QYgI(4sEdF&-Sjo3ADoVJK<6ajjF6NL5a(Ct1Hx{F_B+D06d4)rnH zHo*DlHq;3R@qTy?&Q3UrhT!8Ed-06$5HI7q^wO-UdV)_=%Ai|Zfq}vTO&i*XMja(a z&z4Mv%)MW`rLx18B-;0m<7B;Prj4A!5U;mKNZfr*P-qT}HoOw=ny{LrlqjiOBzC<~#m7?*8`&!^Qu6%0iWW zcgUT4uPz_RoWEMWOTID5H2)hciMCY#8sg=|<%a?o@1N$lafRqU%uc z_+?zc>qlqMBl{dYvaZ0!^foj!q&HjtDPo7bf|2J(Op$d|_wOGp&TYbZvWBhhrpfRq z=_Z(M$O<)64i;S_bF;8=jZf*a<8?hA-(bWS5oW?Yv}C*bW=tux zWvht}ocq^NW+5)TD$nipB9+fAQW-C^O5HaJRP&EzkViD%F`OxtW0`a#kV)@)a7p(z z9PaJN{gchvC04j`+B)J-)MoVy9e#<|V|5pqi=`Ox)dW*&-Lc@(F3oswf-OTW9N1I5 zaDR%0wPWJOE$;3tljnv??zlHZi@lW0y;FtNU>wKZl3jf78p+7q@tnPVBm@8T<)|J0 zESl-gm>>V+dYMt_fe{}`C+#0OOB8%I=F73BJRtqJ1H_kYTGNcic6RJ0J%%G@Ik9ky z3*Beig8BOCy#ff~GI*Pj&58%It z5Iu)Wo0G9R1UV-Jf+5|D-+NY?t%3tQ#il)`$-KUUZcYo732I{8Vn0 zoOM>qB*y)VqvG5_9KNL|i&T86lkUW&0T!(LYrxedx(t+iz<^pE;f?DuyFrhyMV~Er zVMwzg6WSi`z^el-Sufs3qj@$w9^Z)%d)qVL%8_?u-DS@uBkCdV;UgIw+&7Ikf28nl z`j!=;NhgGcIcxr!be zM(MMvpFF#gJAS2T&I4h>s|vAZUr!qjckM(YGkfllb#)$z?yN38jzQ9+eIk{Al~Z`_ zS2A0)OA!8BG|x5b&%5)5|2fHn$u(ANHBg_sN62|jSj%P7jct0ZH9c$F@T}xQ%v!eP zq&3<+Ej=#d9viUofHC9SnsK(9IS2Q$l>4DIgI3sZpsc&8G>dI+2oL#II#*qvCi}=_ zRy`QUCkpY@UOR-#JbUw_x-Xl>$vs}q!!ah(sgs?U8dVctQ3nx5HVSJu#Cv)YExBdn%3A zOr*=>L>@XR46;=N=&2LJ18dxvC_e6$J%qh@wK=s1{DxE15A4bRDP2atF?37={F*oA z&e-NOTdl~TUM(5bs752vzWrFzA`^?qjhzJ9IZDY*6|AhU#O!nvR8GNsl#-Phn zIOW}VmOYE-$bUmvv$z+B$-isYXI4DfMw8A@8(~-Y2LDApM$p|VBwnjQL-QBnHGB;P z^$)l){4**>OWw@03HRByU|yjTUst!}SDAhFZmJ9;Sp3|Qy=0RLJ%7JkYwRuyg5zWU|35(+@K3}c^{m|NX5K2c2QM=#_CI(+ZLrobvey>2ygGaEieufQvhfCkTpz}WI5R^Tm!c{oT zr>2otCUBbbXlk~R9B!*HZu;s(f33F6iFu2X!-Y6CZMEc87NS6F0gmY{#Ur;|G!qTh zbk-Jhp1K_munXO0?SRnV&mn8FbipRiMrHFEu&csOHbNV7saS7dSm0;M8E9mpK6th2;;ljpp9LT*1*Uc4}Z1ez~ z-c+L|;RU*j&uq493cGEO=kej;?5*s@HlkakO25v5IU6v;aw?KyW1!wRO7z_f1fN`l zjJF$cvfBa7)jB84=Q33MxP#<>4*=^&s9yC5%Tp>*x}gfsCspIp>uNk*{20AEKSBFp z&xOJJ8XY`8V&Ej{KJc1EBb^vNa|&fzTk#TzmO9Go2-+*m#t7#TXsb3HWjV<(&zXy| z%WIIZVISV(Jd$c|WBi*Z_|^LzPO8+R@8CKp4Xs0ZLLHoz*C8;n4v9O2tE(n`r7^YA z&+{3@QQzP+^_S#xGw2yTp6MT=xORaz!^Jn2BtAjI4l7|hHUW=320{KdM>x&3l8j!A09b5%`wSfME&*f!zO z35_VIYeeG5Cj3028Q+M9u&r<>W@K?9B%6rh<^7GCZggkDz%V^+lkY%H9Q zr20)5J@c6KUEaXg>{rm19)P$nEf{6fk`JrIPrg!(7XsCow_7r=KU#4^^Hw}sEIf3X zd96)R;ukX|{>c^(Q-ZQ^OVWhX8^=ZK1Nd{6F5^U_Ki+9Mj9|1{f42r{@X4&Yu zU=@6#_e1IZWeghi6hHp`g2H^sdvulgop2Hd>S$9n|;86=?FTrevX%H04|~g${AyEbMD8zC>a8_g;wyse}(R^*2;&s{$nGFLj|PnVdps?nScEzM~nd4#}v6Ber()8?9i zurS-R%2+b8GOPPFxE(u5PG;TX3BolA|5grOZYsz zOc)()D4h=N+2cQ59&4vVdk695tx4j}@-BRr(1?*!^6*Xg;b)Bd!PRsaKDWxku0tzv z;(Y-+rfGAqxWp#do_39$C1@2E?WrQxZJ_^*T|IZB=z@wqN;Wi5U_;VBGrqD7)JLuNSf>j(!nDe6SK+mhj2 zBJ=k5W~|jT=J!~AcJ$U|-Pv~hBwWG&ibpW}stHdWK8Knq8FB~YM|AL8FVJ567p=0?xiqdl??syPifFK(*EzB0HdlK2yK%`=7hx?r@vNymD|gw@ z`-UZV{%>w6e!8-AdR+EfypC@)nLJ(Cf$N7**Gy)J@w?$QF99=R`(ety;TU8*12rMy zaaw*D>V~(V+rAdBzbf#!p%(kBF=Brc$+gL>`OXGcjtlhQ-YFjZu+fck=QxWVXV1jT zk}nJu&p>}+_gEOvMf~bp{f7sm&(LOL)uk4{Eh|8BxC?+?@Bsq#)( zfe)$&Fm!G?F3kT3jY0)pNfo~G5o7*ww4w4HXX@v8(Zvm{JL1dRe|$MJf;<-Ag>!Cr z^5_*;I&^d7lF9#!k_Y64S zUyC)*r2nn*9u_|O4mV*PljTB$Hc=~Wt zpcggnyNU+tz`$r5Hpw(6HW)Elm^~LF#INk!hUX=hv18#bEYj-_wf4v)bNwK@1S zbTiT{&f?S9YIqL#i{FwP%6h6t+hhyTVjP$g?!kBBah!70kLsrasGS+Wj1WKWx*&X% z8S-~HsRQsey-%2aD^$s74AQ1>$g5nn>kM7jrZ z+AP2|d6o@MT*C6Sm#}nb#)EU(aDA*Hl{Z-Px}mdVn7sJAm>hj1fM=%$@$lC`nS=SW z!(!=9e(%M+qvA8FvggzO)@(OWe#4S&uRN~Bh%zx*`ZCrmI9u`DTSxIIcuLrJp8$T z3HZjgGw%nuvBOmdem>Yy`jIR-o;dKi8g;F766n z>CW_FRIwLVPj#c~aYr7VV$0=~7F>Ur zY_m>+o%IsodhWs9+ESQZd57iaT5$0*O(lczp`A6m5FS!Rj73pVv^BZ^ag@2J9f@i~n z@%&pfX1$q%Sd}g46(e^#wI}%M*oagUb&hDS&ow@l42^eS^*nd>UFc2yNKo}9XgA%5 zG50!i)?yEqeRkpZ-wupkZ%YsH^2ZD?W2u%AE27%7$Vyltt%gdcl?JuaHX~B5l8v)x(}^}z^RT3`vKeQ27;#r!d*SEk(tO@XuH0(D2ALb2e>DO3 z_H>2I)gD;#Um}8EFT`H^T`&nL!N6$g_q*Mc*%w-~^$(einpjd>-=0M`U3kOJgI(M` z8Q|wZB_%grlFaw#PLjEPXv^kwE^*1);)1daWk~y2(@UdKI?kQf)j|j?L zg2-0k=x-MacYpDw&dNsXDeLg*-!WVrau2=Ee8bp1%F_MTjvus5_;|iGmu|M_us~