diff --git a/.buildkite/compare_regression.jl b/.buildkite/compare_regression.jl new file mode 100644 index 000000000..c85009bfd --- /dev/null +++ b/.buildkite/compare_regression.jl @@ -0,0 +1,116 @@ +# Compare benchmark harness results from the base branch and this commit. +# +# julia compare_regression.jl [--threshold=10] [--base=main] [--out=report.md] +# +# Each root holds the harness run directory for that side. Exits 1 on a +# slowdown above the threshold or a failed candidate run; base failures are +# only reported so a PR can fix them. + +using Printf +using Statistics +using TOML + +function load_runs(root) + runs = Dict{Tuple,Union{Float64,Nothing}}() + isdir(root) || return runs + for dir in readdir(root; join=true) + manifest_path = joinpath(dir, "manifest.toml") + isfile(manifest_path) || continue + for r in TOML.parsefile(manifest_path)["runs"] + key = (r["name"], r["T"], r["N"], r["M"]) + runs[key] = r["status"] == "complete" ? median_time(dir, r) : nothing + end + end + return runs +end + +# Median trial time in ms, or `nothing` when rows are missing or incorrect. +function median_time(dir, r) + # The harness writes `_.csv`, where save_as extends the model + # (e.g. `cunumeric_nofusion`, `cunumeric_struct`). + results = joinpath(dir, r["results_subdir"]) + isdir(results) || return nothing + prefix = "$(r["name"])_$(r["model"])" + files = filter(readdir(results)) do f + return f == "$prefix.csv" || (startswith(f, "$(prefix)_") && endswith(f, ".csv")) + end + times = Float64[] + for path in joinpath.(results, files), line in eachline(path) + f = split(strip(line), ',') + length(f) == 8 || continue + (parse(Int, f[3]), parse(Int, f[4])) == (r["N"], r["M"]) || continue + f[8] == "fail" && return nothing + push!(times, parse(Float64, f[6])) + end + return isempty(times) ? nothing : median(times) +end + +label(key) = "$(key[1]) ($(key[2]), $(key[3])×$(key[4]))" + +function report(base_root, candidate_root, threshold, base_name) + base = load_runs(base_root) + candidate = load_runs(candidate_root) + isempty(candidate) && error("no candidate results in $candidate_root") + + lines = [ + "## Benchmark regression vs. `$base_name`", "", + "| Benchmark | `$base_name` (ms) | PR (ms) | Change |", + "| --- | ---: | ---: | ---: |", + ] + regressions, failed, uncompared = String[], String[], String[] + for key in sort!(collect(keys(candidate))) + after = candidate[key] + before = get(base, key, nothing) + if after === nothing + push!(failed, label(key)) + elseif before === nothing + push!(uncompared, label(key)) + else + change = 100 * (after / before - 1) + flag = change > threshold ? " ⚠️" : "" + push!( + lines, + @sprintf("| %s | %.3f | %.3f | %+.1f%%%s |", + label(key), before, after, change, flag) + ) + change > threshold && + push!(regressions, @sprintf("%s: %.1f%% slower", label(key), change)) + end + end + for (title, items) in (("Slower than the threshold", regressions), + ("Failed on this PR", failed), + ("Not compared (no base result)", uncompared)) + isempty(items) && continue + append!(lines, ["", "**$title:**"], ["- $item" for item in items]) + end + push!(lines, "", @sprintf("Threshold: more than %.0f%% slower (median of trials).", threshold)) + return join(lines, '\n') * '\n', isempty(regressions) && isempty(failed) +end + +function main(args) + threshold, out, base_name = 10.0, nothing, "base" + positional = String[] + for arg in args + if startswith(arg, "--threshold=") + threshold = parse(Float64, split(arg, '='; limit=2)[2]) + elseif startswith(arg, "--base=") + base_name = split(arg, '='; limit=2)[2] + elseif startswith(arg, "--out=") + out = split(arg, '='; limit=2)[2] + else + push!(positional, arg) + end + end + length(positional) == 2 || error("usage: compare_regression.jl ") + text, ok = report(positional..., threshold, base_name) + print(text) + out === nothing || write(out, text) + return ok ? 0 : 1 +end + +try + exit(main(ARGS)) +catch e + println(stderr, "comparison failed: ", sprint(showerror, e)) + exit(2) +end diff --git a/.buildkite/developer.pipeline.yml b/.buildkite/developer.pipeline.yml index 340cf6644..38d6429c7 100644 --- a/.buildkite/developer.pipeline.yml +++ b/.buildkite/developer.pipeline.yml @@ -1,4 +1,13 @@ steps: + - label: ":lock: Enter developer CI" + command: "true" + concurrency: 1 + concurrency_group: "cunumeric/developer" + agents: + queue: "cuda" + + - wait: ~ + - group: ":hammer: Developer" key: "developer" steps: @@ -7,7 +16,7 @@ steps: - JuliaCI/julia#v1: version: "{{matrix.julia}}" # Per version AND fusion: dev wrappers are ABI-specific, and same-depot - # concurrent jobs contend on Pkg locks. Separate from the JLL cache. + # concurrent jobs contend on Pkg locks. Separate from the JLL cache. cache_dir: "${HOME}/.cache/julia-buildkite-plugin-developer-{{matrix.julia}}-{{matrix.fusion}}" command: ".buildkite/run_developer_ci.sh" artifact_paths: @@ -32,6 +41,17 @@ steps: - "1.10" - "1.11" - "1.12" + - "1.13" fusion: - "on" - "off" + + - wait: ~ + continue_on_failure: true + + - label: ":unlock: Leave developer CI" + command: "true" + concurrency: 1 + concurrency_group: "cunumeric/developer" + agents: + queue: "cuda" diff --git a/.buildkite/install_cmake.sh b/.buildkite/install_cmake.sh new file mode 100644 index 000000000..78ca58bb6 --- /dev/null +++ b/.buildkite/install_cmake.sh @@ -0,0 +1,11 @@ +# shellcheck shell=bash +# Source to put a pinned CMake first on PATH (developer wrapper builds need it). +CMAKE_VERSION="3.30.7" +CMAKE_ROOT="$(mktemp -d)" +CMAKE_INSTALLER="$CMAKE_ROOT/cmake-installer.sh" +curl --fail --silent --show-error --location \ + --output "$CMAKE_INSTALLER" \ + "https://github.com/Kitware/CMake/releases/download/v$CMAKE_VERSION/cmake-$CMAKE_VERSION-linux-x86_64.sh" +sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" +export PATH="$CMAKE_ROOT/bin:$PATH" +cmake --version diff --git a/.buildkite/jll.pipeline.yml b/.buildkite/jll.pipeline.yml index c76c5f8f3..ffe36a3f6 100644 --- a/.buildkite/jll.pipeline.yml +++ b/.buildkite/jll.pipeline.yml @@ -11,9 +11,9 @@ steps: plugins: - JuliaCI/julia#v1: version: "{{matrix.julia}}" - # Developer builds install local wrapper overrides into the depot. - # Keep JLL tests in a separate cache so those overrides cannot leak here. - cache_dir: "${HOME}/.cache/julia-buildkite-plugin-jll" + # Isolate Julia versions and compile-time fusion preferences, and keep + # JLL tests separate from developer wrapper overrides. + cache_dir: "${HOME}/.cache/julia-buildkite-plugin-jll-{{matrix.julia}}-{{matrix.fusion}}" - jquick/pre-hook#v1.2.0: command: | if [ "{{matrix.fusion}}" = "on" ]; then @@ -23,7 +23,7 @@ steps: fi julia --project -e 'using Pkg; Pkg.resolve(); Pkg.instantiate()' - JuliaCI/julia-test#v1: - test_args: "--quickfail --jobs=8 --verbose" + test_args: "--jobs=8 --verbose" - JuliaCI/julia-coverage#v1: dirs: - src @@ -50,6 +50,7 @@ steps: - "1.10" - "1.11" - "1.12" + - "1.13" fusion: - "on" - "off" diff --git a/.buildkite/pipeline.yml b/.buildkite/pipeline.yml index 8fab17e69..7fc8e4fef 100644 --- a/.buildkite/pipeline.yml +++ b/.buildkite/pipeline.yml @@ -10,6 +10,6 @@ steps: command: ".buildkite/upload_gpu_ci.sh" agents: queue: "cuda" - if: build.message !~ /\[skip tests\]/ + if: build.message !~ /\[skip ci\]/ if_changed: "{src/**,scripts/**,deps/build.jl,Project.toml,lib/CNPreferences/src/**,lib/cunumeric_jl_wrapper/**}" timeout_in_minutes: 5 diff --git a/.buildkite/regression.pipeline.yml b/.buildkite/regression.pipeline.yml new file mode 100644 index 000000000..8298f5f37 --- /dev/null +++ b/.buildkite/regression.pipeline.yml @@ -0,0 +1,21 @@ +steps: + - label: ":chart_with_upwards_trend: Benchmark regression vs. base branch" + key: "regression" + plugins: + - JuliaCI/julia#v1: + version: "1.12" + cache_dir: "${HOME}/.cache/julia-buildkite-plugin-regression" + command: ".buildkite/run_regression.sh" + artifact_paths: + - "regression/**/*" + agents: + queue: "cuda" + # One comparison at a time so runs do not share a GPU. + concurrency: 1 + concurrency_group: "cunumeric/regression" + timeout_in_minutes: 180 + env: + LD_LIBRARY_PATH: "" + LEGATE_AUTO_CONFIG: "0" + # nvidia-smi cannot read GPU memory in the CI container; cap the harness budget. + CUNUMERIC_BENCH_FBMEM_MB: "3072" diff --git a/.buildkite/regression.toml b/.buildkite/regression.toml new file mode 100644 index 000000000..fdc61a476 --- /dev/null +++ b/.buildkite/regression.toml @@ -0,0 +1,86 @@ +# Opt-in performance regression suite ([regression-ci]): cuNumeric on the base +# branch vs. this commit. Sizes are pinned so both sides run identical problems. +# gemm, montecarlo(_naive), grayscott_plain and cg_plain avoid @accelerate and +# newer APIs, so they are the entries comparable against main. +[Global] +models = ["cunumeric"] +n_warmup = 2 +n_iter = 5 +n_trial = 5 +check_correctness = true +auto_size = false +# Legate's pool is capped at CUNUMERIC_BENCH_FBMEM_MB (3 GB) on CI. +mem_frac = 0.9 +# cpus = 4: 8 CPU procs + GPU/util threads exceed the CI agent's cores. + +[[gemm]] +T = "Float32" +N = 4096 +M = 4096 +gpus = 1 +cpus = 4 + +[[montecarlo]] +T = "Float32" +N = 50_000_000 +M = 1 +gpus = 1 +cpus = 4 + +[[montecarlo_naive]] +T = "Float32" +N = 50_000_000 +M = 1 +gpus = 1 +cpus = 4 + +[[grayscott]] +T = "Float32" +N = 1024 +M = 1024 +gpus = 1 +cpus = 4 + +[[grayscott_plain]] +T = "Float32" +N = 1024 +M = 1024 +gpus = 1 +cpus = 4 + +[[cg]] +T = "Float64" +N = 65536 +M = 1 +gpus = 1 +cpus = 4 +kwargs = { check_every = 10, max_iter = 1000 } + +[[cg_plain]] +T = "Float64" +N = 65536 +M = 1 +gpus = 1 +cpus = 4 +kwargs = { check_every = 10, max_iter = 1000 } + +[[nas_ep]] +T = "Float64" +n_iter = 1 +gpus = 1 +cpus = 4 +kwargs = { class = "A" } + +[[nas_mg]] +T = "Float64" +n_iter = 1 +gpus = 1 +cpus = 4 +kwargs = { class = "A" } + +[[nas_ft]] +T = "Float64" +n_iter = 1 +gpus = 1 +cpus = 4 +kwargs = { class = "A" } diff --git a/.buildkite/regression_base_pins.toml b/.buildkite/regression_base_pins.toml new file mode 100644 index 000000000..6b1928aa1 --- /dev/null +++ b/.buildkite/regression_base_pins.toml @@ -0,0 +1,6 @@ +# Extra version pins for the base side's harness environment, keyed by base +# branch, for bases that no longer load against the current registry. + +[main] +# cuNumeric 0.2.0 calls LegatePreferences.has_cuda_gpu, removed in 0.1.7. +LegatePreferences = "0.1.6" diff --git a/.buildkite/run_developer_ci.sh b/.buildkite/run_developer_ci.sh index 537494b28..856f758c2 100755 --- a/.buildkite/run_developer_ci.sh +++ b/.buildkite/run_developer_ci.sh @@ -10,23 +10,22 @@ case "${CUNUMERIC_FUSION:-}" in ;; esac -CMAKE_VERSION="3.30.7" -CMAKE_ROOT="$(mktemp -d)" -CMAKE_INSTALLER="$CMAKE_ROOT/cmake-installer.sh" -curl --fail --silent --show-error --location \ - --output "$CMAKE_INSTALLER" \ - "https://github.com/Kitware/CMake/releases/download/v$CMAKE_VERSION/cmake-$CMAKE_VERSION-linux-x86_64.sh" -sh "$CMAKE_INSTALLER" --skip-license --prefix="$CMAKE_ROOT" -export PATH="$CMAKE_ROOT/bin:$PATH" -cmake --version +source .buildkite/install_cmake.sh + +# Exercise libcxxwrap cache validation separately from package tests, which run in +# JLL jobs where no build toolchain is installed. +julia --startup-file=no test/build_cxxwrap.jl # Clean slate so cached state doesn't leak across Julia versions. rm -f Manifest.toml test/Manifest.toml dev/Manifest.toml \ LocalPreferences.toml test/LocalPreferences.toml -# Dev mode rebuilds the wrapper .so each run, so drop stale overrides and the .ji that -# bake @wrapmodule bindings (cuNumeric/Legate + wrapper JLLs) — a cached .ji would -# mismatch the fresh .so and segfault. libcxxwrap override is kept. +# Build directly in the plugin's persistent cache so artifacts, precompile, and libcxxwrap all +# stay warm across runs. Developer builds are serialized across PRs, while each matrix job in a +# build uses a separate (Julia, fusion) cache. +# Dev mode rebuilds the wrapper .so each run, so drop stale overrides and the .ji that bake +# @wrapmodule bindings (cuNumeric/Legate + wrapper JLLs) — a cached .ji would mismatch the fresh +# .so and segfault. libcxxwrap's dev build is kept and reused. DEPOT="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" rm -rf "$DEPOT"/packages/*/*/override \ "$DEPOT"/compiled/v*/{cuNumeric,Legate,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} @@ -83,5 +82,6 @@ cp LocalPreferences.toml test/LocalPreferences.toml julia --color=yes --project=. -e ' using Pkg - Pkg.test("cuNumeric"; test_args = ["--quickfail", "--jobs=8", "--verbose"]) + Pkg.develop(PackageSpec(path = "lib/CNPreferences")) + Pkg.test("cuNumeric"; test_args = ["--jobs=8", "--verbose"]) ' diff --git a/.buildkite/run_regression.sh b/.buildkite/run_regression.sh new file mode 100755 index 000000000..33f213b0e --- /dev/null +++ b/.buildkite/run_regression.sh @@ -0,0 +1,119 @@ +#!/usr/bin/env bash +# Opt-in GPU performance comparison of this commit against a base branch: +# [regression-ci ], else REGRESSION_BASE_BRANCH, else the branch the PR +# targets, else main. Both sides run this commit's harness and regression.toml. + +set -euo pipefail + +requested="$(buildkite-agent meta-data get regression-base-branch --default "" 2>/dev/null || true)" +pr_base="${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-}" +readonly BASE_BRANCH="${requested:-${REGRESSION_BASE_BRANCH:-${pr_base:-main}}}" +readonly THRESHOLD="${REGRESSION_THRESHOLD:-10}" +# Bounds each base harness pass so a hang on the base cannot use up the step. +readonly BASE_TIMEOUT="${REGRESSION_BASE_TIMEOUT:-3600}" + +candidate="$PWD" +harness="$candidate/benchmark" +env_dir="$harness/environments/cunumeric" +out="$candidate/regression" +base="$(mktemp -d)/base" + +rm -rf "$out" +mkdir -p "$out" + +git fetch --no-tags origin "+refs/heads/$BASE_BRANCH:refs/remotes/origin/$BASE_BRANCH" +git worktree add --detach "$base" "origin/$BASE_BRANCH" +trap 'git -C "$candidate" worktree remove --force "$base"' EXIT +git submodule update --init benchmark + +julia --color=yes --project="$harness" -e 'using Pkg; Pkg.instantiate()' +mkdir -p "$harness/results" + +# Point the harness's cuNumeric environment at one checkout. Wrapper overrides +# and precompiled wrapper bindings from the other side must not leak across. +bind_checkout() { + local source=$1 mode=$2 pins=${3:-} + local depot + depot="$(julia --startup-file=no -e 'print(DEPOT_PATH[1])')" + rm -rf "$depot"/packages/*/*/override \ + "$depot"/compiled/v*/{cuNumeric,Legate,cunumeric_jl_wrapper_jll,legate_jl_wrapper_jll} + rm -f "$env_dir/Manifest.toml" "$env_dir/LocalPreferences.toml" + # The harness pins the current cuNumeric/CNPreferences; each side develops + # its own checkout, so drop those pins to let an older base resolve. + julia --color=yes --project="$env_dir" -e ' + using Pkg, TOML + project = Base.active_project() + toml = TOML.parsefile(project) + foreach(p -> delete!(get(toml, "compat", Dict()), p), ("cuNumeric", "CNPreferences")) + open(io -> TOML.print(io, toml), project, "w") + Pkg.develop([PackageSpec(path = ARGS[1]), PackageSpec(path = ARGS[2])]) + pins = isempty(ARGS[4]) ? Dict() : get(TOML.parsefile(ARGS[3]), ARGS[4], Dict()) + isempty(pins) || Pkg.add([PackageSpec(name = k, version = v) for (k, v) in pins]) + Pkg.instantiate() + ' "$source" "$source/lib/CNPreferences" "$candidate/.buildkite/regression_base_pins.toml" "$pins" + if [[ "$mode" == developer ]]; then + julia --color=yes --project="$env_dir" -e ' + using CNPreferences, Pkg + CNPreferences.use_developer_mode() + Pkg.build("cuNumeric") + ' + fi +} + +result_dirs() { find "$harness/results" -mindepth 1 -maxdepth 1 -type d -printf '%f\n'; } + +run_side() { + local side=$1 source=$2 mode=$3 + local limit=() + [[ "$side" == base ]] && limit=(timeout --signal=KILL "$BASE_TIMEOUT") + echo "--- :julia: $side ($mode wrapper)" + bind_checkout "$source" "$mode" "$([[ "$side" == base ]] && echo "$BASE_BRANCH")" + local before new + before="$(result_dirs)" + (cd "$harness" && "${limit[@]}" julia --color=yes --project=. run.jl \ + --config="$candidate/.buildkite/regression.toml" --fusion=on) || + echo "Harness reported failures ($side)." + new="$(comm -13 <(sort <<<"$before") <(result_dirs | sort) | head -1)" + if [[ -n "$new" ]]; then + mkdir -p "$out/$side" + mv "$harness/results/$new" "$out/$side/results" + fi +} + +# Build a side's wrapper from source when it differs from the release its own +# checkout records. A base without RELEASED_COMMIT is checked against ours. +wrapper_mode() { + local dir=$1 + [[ -f "$dir/scripts/wrapper_changed.sh" ]] || dir="$candidate" + if (cd "$dir" && scripts/wrapper_changed.sh "$2" >&2); then + echo jll + else + local status=$? + ((status == 1)) || exit "$status" + echo developer + fi +} + +base_mode="$(wrapper_mode "$base" "$(git -C "$base" rev-parse HEAD)")" || base_mode=developer +candidate_mode="$(wrapper_mode "$candidate" HEAD)" +if [[ "$base_mode" == developer || "$candidate_mode" == developer ]]; then + source .buildkite/install_cmake.sh +fi + +echo "Comparing against $BASE_BRANCH." +# Base failures only leave its results uncompared; the candidate must pass. +run_side base "$base" "$base_mode" || echo "Base side failed; its results are not compared." +run_side candidate "$candidate" "$candidate_mode" + +echo "--- :bar_chart: Compare" +status=0 +julia --startup-file=no "$candidate/.buildkite/compare_regression.jl" \ + "$out/base" "$out/candidate" --threshold="$THRESHOLD" --base="$BASE_BRANCH" \ + --out="$out/report.md" || + status=$? + +if [[ -f "$out/report.md" ]] && command -v buildkite-agent >/dev/null; then + style=$([[ $status == 0 ]] && echo success || echo error) + buildkite-agent annotate --context regression --style "$style" < "$out/report.md" +fi +exit "$status" diff --git a/.buildkite/upload_gpu_ci.sh b/.buildkite/upload_gpu_ci.sh index 3e8546d1e..abbe8b460 100755 --- a/.buildkite/upload_gpu_ci.sh +++ b/.buildkite/upload_gpu_ci.sh @@ -4,8 +4,6 @@ set -euo pipefail readonly JLL_PIPELINE=".buildkite/jll.pipeline.yml" readonly DEVELOPER_PIPELINE=".buildkite/developer.pipeline.yml" -readonly WRAPPER_PATH="lib/cunumeric_jl_wrapper" -readonly WRAPPER_BASE_BRANCH="main" branch="${BUILDKITE_BRANCH:-}" base_branch="${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-}" @@ -15,34 +13,59 @@ message="${BUILDKITE_MESSAGE:-}" run_jll=true run_developer=true -# Keep both suites for main and PRs into main. For non-main PRs, select the -# suite whose wrapper matches the code under test. +if [[ "$message" =~ \[skip[[:space:]]ci\] ]]; then + echo "Skipping all GPU CI because the build message requests it." + exit 0 +fi + +if [[ "$message" =~ \[skip[[:space:]]jll\] ]]; then + echo "Skipping JLL GPU CI because the build message contains [skip jll]." + run_jll=false +fi +if [[ "$message" =~ \[skip[[:space:]]dev\] ]]; then + echo "Skipping developer GPU CI because the build message contains [skip dev]." + run_developer=false +fi + +# Keep both suites for main and PRs into main. Otherwise use the JLL suite only +# when the wrapper matches the released JLL source. if [[ "$branch" != "main" && "$base_branch" != "main" ]]; then - if [[ "$message" =~ \[skip[[:space:]]jll\] ]]; then - echo "Skipping JLL GPU CI because the build message contains [skip jll]." - run_jll=false - elif [[ "$pull_request" != "false" && -n "$base_branch" ]]; then - base_ref="refs/remotes/origin/$WRAPPER_BASE_BRANCH" - # The published wrapper JLL tracks main, so compare against main even - # when the pull request targets develop. - git fetch --no-tags origin "+refs/heads/${WRAPPER_BASE_BRANCH}:${base_ref}" - - if git diff --quiet "${base_ref}...HEAD" -- "$WRAPPER_PATH"; then - echo "No wrapper changes detected against origin/$WRAPPER_BASE_BRANCH; using JLL GPU CI." - run_developer=false + if scripts/wrapper_changed.sh; then + echo "Wrapper matches the released JLL source; using JLL GPU CI." + run_developer=false + else + diff_status=$? + if ((diff_status == 1)); then + echo "Wrapper differs from the released JLL source; using developer GPU CI." + run_jll=false else - diff_status=$? - if ((diff_status == 1)); then - echo "Wrapper changes detected against origin/$WRAPPER_BASE_BRANCH; using developer GPU CI." - run_jll=false - else - echo "Could not determine whether the wrapper changed against origin/$WRAPPER_BASE_BRANCH." >&2 - exit "$diff_status" - fi + echo "Could not determine whether the wrapper matches the released JLL source." >&2 + exit "$diff_status" fi fi fi +# Opt-in performance comparison: [regression-ci] in the commit message or the +# pull request title/body compares against the PR's base branch, and +# [regression-ci ] against . +opt_in="$message" +if [[ "$pull_request" =~ ^[0-9]+$ ]]; then + opt_in+=$'\n'"$( + curl --fail --silent --show-error --location \ + --header "Accept: application/vnd.github+json" \ + "https://api.github.com/repos/JuliaLegate/cuNumeric.jl/pulls/$pull_request" | + python3 -c 'import json, sys; pr = json.load(sys.stdin); print(pr.get("title") or "", pr.get("body") or "")' + )" || true +fi +if [[ "$opt_in" =~ \[regression-ci([[:space:]]+([A-Za-z0-9._/-]+))?\] ]]; then + regression_base="${BASH_REMATCH[2]}" + if [[ -n "$regression_base" ]]; then + buildkite-agent meta-data set regression-base-branch "$regression_base" + fi + echo "Uploading benchmark regression CI (base: ${regression_base:-PR base branch})." + buildkite-agent pipeline upload .buildkite/regression.pipeline.yml +fi + # Each dynamic upload is inserted immediately after this job, so upload the # developer group first to keep the JLL group first when both suites run. if [[ "$run_developer" == "true" ]]; then diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 000000000..fd15a3363 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,47 @@ +# Local Julia environments and generated build configuration. +**/*Manifest.toml +**/*LocalPreferences.toml +dev/Project.toml +deps/deps.jl +build_wrapper.sh +compile_wrapper.sh + +# Local caches, logs, and editor configuration. +**/__pycache__ +**/*.pyc +**/*.log +**/*.err +**/*.prof +**/node_modules +.vscode +.envrc +.localenv +docs/.vitepress +docs/src/.vitepress/cache +docs/src/.vitepress/dist + +benchmark/results +benchmark/plots +# Host build products must not overwrite artifacts built inside the image. +**/build +**/*.o +**/*.obj +**/*.so +**/*.so.* +**/*.dylib +**/*.dll +**/*.a +**/*.lib +**/*.exe +deps/cupynumeric-* +deps/lapacke_build +libcupynumeric +install-libcupynumeric +install-cupynumeric +libcxxwrap-julia +lapacke + +# Keep root repository metadata intact, including files matching rules above. +**/.git +!.git +!.git/** diff --git a/.github/scripts/check_versions.py b/.github/scripts/check_versions.py index 0d8926aae..3f7e1dc99 100644 --- a/.github/scripts/check_versions.py +++ b/.github/scripts/check_versions.py @@ -2,9 +2,12 @@ """Version consistency check for pull requests targeting main.""" import argparse +import json import re import subprocess import sys +import urllib.parse +import urllib.request from pathlib import Path REPO_ROOT = Path(subprocess.run( @@ -18,6 +21,10 @@ WRAPPER_VERSION_FILE = "lib/cunumeric_jl_wrapper/VERSION" WRAPPER_JLL_COMPAT_KEY = "cunumeric_jl_wrapper_jll" +WRAPPER_RELEASED_COMMIT_FILE = "lib/cunumeric_jl_wrapper/RELEASED_COMMIT" +WRAPPER_JLL_REPO = "JuliaBinaryWrappers/cunumeric_jl_wrapper_jll.jl" +WRAPPER_JLL_TAG_PREFIX = "cunumeric_jl_wrapper-v" +WRAPPER_SOURCE_REPO = "https://github.com/JuliaLegate/cuNumeric.jl.git" WRAPPER_SRC_PREFIXES = ( "lib/cunumeric_jl_wrapper/src/", "lib/cunumeric_jl_wrapper/include/", @@ -149,6 +156,67 @@ def check_wrapper_compat_sync(pr_toml: str, errors: list): print(f"\t\tOK") +def http_get(url: str) -> str: + request = urllib.request.Request(url, headers={"User-Agent": "cuNumeric-version-check"}) + with urllib.request.urlopen(request, timeout=30) as response: + return response.read().decode() + + +def released_wrapper_commit(version: str) -> tuple: + """Return (tag, source revision) of the newest JLL build of `version`.""" + refs = json.loads(http_get( + f"https://api.github.com/repos/{WRAPPER_JLL_REPO}/git/matching-refs/tags/" + f"{WRAPPER_JLL_TAG_PREFIX}{version}+" + )) + builds = {} + for ref in refs: + tag = ref["ref"].removeprefix("refs/tags/") + m = re.fullmatch(re.escape(WRAPPER_JLL_TAG_PREFIX + version) + r"\+(\d+)", tag) + if m: + builds[int(m.group(1))] = tag + if not builds: + return None, None + tag = builds[max(builds)] + readme = http_get( + f"https://raw.githubusercontent.com/{WRAPPER_JLL_REPO}/{urllib.parse.quote(tag)}/README.md" + ) + m = re.search(re.escape(WRAPPER_SOURCE_REPO) + r" \(revision: `([0-9a-f]{40})`\)", readme) + return tag, m.group(1) if m else None + + +def check_wrapper_released_commit(pr_toml: str, errors: list): + compat_ver = parse_compat_section(pr_toml).get(WRAPPER_JLL_COMPAT_KEY) + recorded = (REPO_ROOT / WRAPPER_RELEASED_COMMIT_FILE).read_text().strip() + + print(f"\t[wrapper released commit]") + print(f"\t\t{WRAPPER_RELEASED_COMMIT_FILE} = {recorded}") + if compat_ver is None: + return # reported by check_wrapper_compat_sync + try: + tag, released = released_wrapper_commit(compat_ver) + except Exception as e: + errors.append( + f"Could not look up the source revision of {WRAPPER_JLL_COMPAT_KEY} {compat_ver}: {e}" + ) + return + + if tag is None: + errors.append( + f"{WRAPPER_JLL_COMPAT_KEY} {compat_ver} has no release in {WRAPPER_JLL_REPO}.\n" + f"\tRelease the wrapper JLL before merging into main." + ) + elif released is None: + errors.append(f"Could not find the source revision in the README of {WRAPPER_JLL_REPO} {tag}.") + elif released != recorded: + errors.append( + f"{WRAPPER_RELEASED_COMMIT_FILE} is stale.\n" + f"\t{tag} was built from {released}, but the file records {recorded}.\n" + f"\tSet {WRAPPER_RELEASED_COMMIT_FILE} to {released}." + ) + else: + print(f"\t\tOK (matches {tag})") + + def check_subpkg_version(base_ref: str, pr_toml: str, changed: list, errors: list): src_changed = [f for f in changed if any(f.startswith(p) for p in SUBPKG_SRC_PREFIXES)] if not src_changed: @@ -210,6 +278,7 @@ def main(): check_package_version(base_ref, pr_toml, errors) check_wrapper_version(base_ref, changed, errors) check_wrapper_compat_sync(pr_toml, errors) + check_wrapper_released_commit(pr_toml, errors) check_subpkg_version(base_ref, pr_toml, changed, errors) print("─" * 60) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 72d381fbb..ffa9f02ec 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,4 +1,4 @@ -name: CI CPU +name: CI on: workflow_dispatch: @@ -15,10 +15,14 @@ on: push: paths: - 'src/**' + - 'ext/**' - 'scripts/**' - 'deps/build.jl' - 'Project.toml' + - 'lib/cunumeric_jl_wrapper/**' - 'lib/CNPreferences/src/**' + - '.github/workflows/ci.yml' + - '.github/workflows/developer.yml' tags: - 'v*' branches: @@ -26,30 +30,47 @@ on: pull_request: paths: - 'src/**' + - 'ext/**' - 'scripts/**' - 'deps/build.jl' - 'Project.toml' + - 'lib/cunumeric_jl_wrapper/**' - 'lib/CNPreferences/src/**' + - '.github/workflows/ci.yml' + - '.github/workflows/developer.yml' jobs: - pkg_resolve: + resolve: + name: Package resolution + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} uses: ./.github/workflows/pkg_resolve.yml - check_changes: - name: Check for wrapper changes + wrapper_changes: + name: Wrapper change detection + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} runs-on: ubuntu-latest outputs: wrapper_changed: ${{ steps.wrapper-changes.outputs.changed }} + skip_dev: ${{ steps.skip-flags.outputs.skip_dev }} + skip_jll: ${{ steps.skip-flags.outputs.skip_jll }} steps: - uses: actions/checkout@v4 - with: - fetch-depth: 0 - - name: Compare wrapper against main + # pull_request payloads carry no commit messages, so read the PR head commit directly. + - name: Read skip flags from head commit message + id: skip-flags + shell: bash + env: + HEAD_SHA: ${{ github.event.pull_request.head.sha || github.sha }} + run: | + git fetch --depth=1 origin "$HEAD_SHA" + message="$(git log -1 --format=%B FETCH_HEAD)" + [[ "$message" == *"[skip dev]"* ]] && echo "skip_dev=true" >> "$GITHUB_OUTPUT" + [[ "$message" == *"[skip jll]"* ]] && echo "skip_jll=true" >> "$GITHUB_OUTPUT" + true + - name: Compare wrapper against the released JLL source id: wrapper-changes shell: bash run: | - git fetch --no-tags origin +refs/heads/main:refs/remotes/origin/main - - if git diff --quiet origin/main...HEAD -- lib/cunumeric_jl_wrapper; then + if scripts/wrapper_changed.sh; then echo "changed=false" >> "$GITHUB_OUTPUT" else status=$? @@ -60,10 +81,22 @@ jobs: fi fi - test: - name: Julia ${{ matrix.julia }} - ${{ matrix.os }} - needs: [pkg_resolve, check_changes] - if: ${{ github.base_ref == 'main' || needs.check_changes.outputs.wrapper_changed != 'true' }} + developer_tests: + name: Developer wrapper tests + needs: [resolve, wrapper_changes] + if: ${{ !contains(toJSON(github.event), '[skip ci]') && !contains(toJSON(github.event), '[skip dev]') && needs.wrapper_changes.outputs.skip_dev != 'true' && (github.event_name != 'pull_request' || github.base_ref == 'main' || needs.wrapper_changes.outputs.wrapper_changed == 'true') }} + permissions: + contents: read + packages: write + attestations: write + id-token: write + actions: write + uses: ./.github/workflows/developer.yml + + jll_tests: + name: JLL wrapper tests - Julia ${{ matrix.julia }} - ${{ matrix.os }} + needs: [resolve, wrapper_changes] + if: ${{ !contains(toJSON(github.event), '[skip ci]') && !contains(toJSON(github.event), '[skip jll]') && needs.wrapper_changes.outputs.skip_jll != 'true' && (github.event_name != 'pull_request' || github.base_ref == 'main' || needs.wrapper_changes.outputs.wrapper_changed != 'true') }} runs-on: ${{ matrix.os }} strategy: fail-fast: false @@ -72,6 +105,7 @@ jobs: - '1.10' - '1.11' - '1.12' + - '1.13' os: - ubuntu-latest # include: @@ -125,4 +159,4 @@ jobs: LEGATE_AUTO_CONFIG: "0" LEGATE_SKIP_RUNTIME: "true" LEGATE_CONFIG: "--cpus 1 --utility 1 --sysmem 500" - run: julia --project -e 'using Pkg; Pkg.test(test_args=["--quickfail", "--jobs=2", "--verbose"])' + run: julia --project -e 'using Pkg; Pkg.test(test_args=["--jobs=2", "--verbose"])' diff --git a/.github/workflows/container.yml b/.github/workflows/container.yml index 07a68c927..36d75fa74 100644 --- a/.github/workflows/container.yml +++ b/.github/workflows/container.yml @@ -11,14 +11,23 @@ on: type: boolean required: false default: false - workflow_run: - workflows: ['CI CPU'] - types: [completed] - branches: - - main + # Build the image associated with a published GitHub release. + release: + types: [published] + # GitHub Actions cannot filter pushes by commit message at trigger time, so + # the job-level condition below implements the [container] opt-in. + push: jobs: push_to_registry: - if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }} + if: >- + ${{ + !contains(toJSON(github.event), '[skip ci]') && + ( + github.event_name == 'workflow_dispatch' || + github.event_name == 'release' || + (github.event_name == 'push' && contains(github.event.head_commit.message, '[container]')) + ) + }} name: Container for ${{ matrix.platform }} - Julia ${{ matrix.julia }} - CUDA ${{ matrix.cuda }} permissions: contents: read @@ -27,8 +36,7 @@ jobs: id-token: write strategy: matrix: - # julia: ["1.10", "1.11"] - julia: ["1.11"] # 1.10 will break + julia: ["1.13"] cuda: ["13.0"] platform: ["linux/amd64"] os: ["ubuntu-22.04"] @@ -40,6 +48,44 @@ jobs: steps: - name: Check out the repo uses: actions/checkout@v4 + with: + ref: ${{ inputs.tag || github.event.release.tag_name || github.sha }} + fetch-depth: 0 + # The image includes .git; do not store the checkout token there. + persist-credentials: false + + - name: Select wrapper build + id: wrapper-build + env: + EVENT_NAME: ${{ github.event_name }} + INPUT_TAG: ${{ inputs.tag }} + run: | + # Published releases and explicitly selected tags use their released + # wrapper. Branch builds use the in-tree wrapper when it differs from + # the wrapper published from main. + if [[ "$EVENT_NAME" == "release" || -n "$INPUT_TAG" ]]; then + changed=false + else + git fetch --no-tags origin +refs/heads/main:refs/remotes/origin/main + if git diff --quiet origin/main...HEAD -- lib/cunumeric_jl_wrapper; then + changed=false + else + status=$? + if [[ "$status" -eq 1 ]]; then + changed=true + else + exit "$status" + fi + fi + fi + + if [[ "$changed" == "true" ]]; then + echo "Wrapper changes detected; building docker/Dockerfile.developer" + echo "dockerfile=docker/Dockerfile.developer" >> "$GITHUB_OUTPUT" + else + echo "No wrapper changes detected; building docker/Dockerfile" + echo "dockerfile=docker/Dockerfile" >> "$GITHUB_OUTPUT" + fi - name: Get package spec id: pkg @@ -47,6 +93,9 @@ jobs: if [[ -n "${{ inputs.tag }}" ]]; then echo "ref=${{ inputs.tag }}" >> $GITHUB_OUTPUT echo "name=${{ inputs.tag }}" >> $GITHUB_OUTPUT + elif [[ "${{ github.event_name }}" == "release" ]]; then + echo "ref=${{ github.event.release.tag_name }}" >> $GITHUB_OUTPUT + echo "name=${{ github.event.release.tag_name }}" >> $GITHUB_OUTPUT elif [[ "${{ github.ref_type }}" == "tag" ]]; then echo "ref=${{ github.ref_name }}" >> $GITHUB_OUTPUT echo "name=${{ github.ref_name }}" >> $GITHUB_OUTPUT @@ -64,6 +113,10 @@ jobs: VERSION=$(grep "^version = " Project.toml | cut -d'"' -f2) echo "version=$VERSION" >> $GITHUB_OUTPUT + # Include the checked-out source revision in every image tag. + COMMIT_SHA=$(git rev-parse --short=12 HEAD) + echo "commit=$COMMIT_SHA" >> $GITHUB_OUTPUT + - name: Get CUDA major version id: cuda run: | @@ -97,22 +150,30 @@ jobs: with: images: ghcr.io/${{ github.repository }} tags: | - type=raw,value=${{ steps.pkg.outputs.name }}-julia${{ matrix.julia }}-cuda${{ steps.cuda.outputs.major }}.${{ steps.cuda.outputs.minor }} - type=raw,value=${{ steps.pkg.outputs.name }},enable=${{ matrix.default == true && (github.ref_type == 'tag' || inputs.tag != '') }} - type=raw,value=latest,enable=${{ matrix.default == true && (github.ref_type == 'tag' || (inputs.tag != '' && inputs.mark_as_latest)) }} - type=raw,value=dev,enable=${{ matrix.default == true && github.ref_type == 'branch' && inputs.tag == '' }} + type=raw,value=${{ steps.pkg.outputs.commit }}-julia${{ matrix.julia }}-cuda${{ steps.cuda.outputs.major }}.${{ steps.cuda.outputs.minor }} + type=raw,value=${{ steps.pkg.outputs.name }},enable=${{ github.ref_type == 'tag' || inputs.tag != '' }} + type=raw,value=latest,enable=${{ github.ref_type == 'tag' || (inputs.tag != '' && inputs.mark_as_latest) }} + type=raw,value=dev,enable=${{ github.ref_type == 'branch' && inputs.tag == '' }} labels: | org.opencontainers.image.version=${{ steps.pkg.outputs.version }} - - name: Save tag to file + - name: Select immutable image tags + id: image-tags + env: + BASE_TAGS: ${{ steps.meta.outputs.tags }} + run: | + echo "base=$(printf '%s\n' "$BASE_TAGS" | head -n1)" >> "$GITHUB_OUTPUT" + + - name: Save tags to files run: | - echo "${{ steps.meta.outputs.tags }}" | cut -d',' -f1 > image_tag.txt + echo "${{ steps.image-tags.outputs.base }}" > image_tag.txt - name: Upload tag artifact uses: actions/upload-artifact@v4 with: name: docker-tag - path: image_tag.txt + path: | + image_tag.txt - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 @@ -130,19 +191,21 @@ jobs: - name: Build image uses: docker/build-push-action@v6 with: - file: Dockerfile + context: . + file: ${{ steps.wrapper-build.outputs.dockerfile }} load: true push: false - provenance: false # the build fetches the repo again, so provenance tracking is not useful + provenance: false # the image is loaded and pushed explicitly below platforms: ${{ matrix.platform }} tags: ${{ steps.meta.outputs.tags }} labels: ${{ steps.meta.outputs.labels }} + cache-from: type=local,src=/tmp/.buildx-cache/base + cache-to: type=local,dest=/tmp/.buildx-cache/base-new,mode=max build-args: | JULIA_VERSION=${{ matrix.julia }} CUDA_MAJOR=${{ steps.cuda.outputs.major }} CUDA_MINOR=${{ steps.cuda.outputs.minor }} JULIA_CPU_TARGET=${{ steps.cpu_target.outputs.target }} - REF=${{ steps.pkg.outputs.ref }} # - name: Run tests in built image # run: | @@ -156,16 +219,26 @@ jobs: - name: Push image if: success() + env: + IMAGE_TAGS: ${{ steps.meta.outputs.tags }} run: | - docker push ${{ steps.meta.outputs.tags }} - docker image rm ${{ steps.meta.outputs.tags }} + while IFS= read -r image; do + docker push "$image" + done <<< "$IMAGE_TAGS" + + - name: Update base layer cache + run: | + rm -rf /tmp/.buildx-cache/base + mv /tmp/.buildx-cache/base-new /tmp/.buildx-cache/base # can happen if there is a failure to push - name: Ensure image is removed (safety) if: always() + env: + BASE_TAGS: ${{ steps.meta.outputs.tags }} run: | - if docker image inspect ${{ steps.meta.outputs.tags }} > /dev/null 2>&1; then - docker stop ${{ steps.pkg.outputs.ref }} || true - docker rm ${{ steps.pkg.outputs.ref }} || true - docker image rm ${{ steps.meta.outputs.tags }} || true - fi + while IFS= read -r image; do + if [[ -n "$image" ]] && docker image inspect "$image" > /dev/null 2>&1; then + docker image rm "$image" || true + fi + done <<< "$BASE_TAGS" diff --git a/.github/workflows/developer.yml b/.github/workflows/developer.yml index 93cd090ce..827e4fdae 100644 --- a/.github/workflows/developer.yml +++ b/.github/workflows/developer.yml @@ -1,55 +1,12 @@ -# Develeper CI test. This will build the workflow using Jlls and building wrappers from SRC -name: Develeper CI test +# Developer wrapper tests build the wrappers from source instead of using JLLs. +name: Developer Wrapper Tests on: - workflow_dispatch: - inputs: - tag: - description: 'Tag to build instead' - required: false - default: '' - mark_as_latest: - description: 'Mark as latest' - type: boolean - required: false - default: false - push: - paths: - - 'src/**' - - 'scripts/**' - - 'deps/build.jl' - - 'Project.toml' - - 'lib/cunumeric_jl_wrapper/src/**' - - 'lib/cunumeric_jl_wrapper/include/**' - - 'lib/CNPreferences/src/**' - - '.github/workflows/developer.yml' - tags: - - 'v*' - branches: - - main - pull_request: - paths: - - 'src/**' - - 'scripts/**' - - 'deps/build.jl' - - 'Project.toml' - - 'lib/cunumeric_jl_wrapper/src/**' - - 'lib/cunumeric_jl_wrapper/include/**' - - 'lib/CNPreferences/src/**' - - '.github/workflows/developer.yml' -jobs: - pkg_resolve: - uses: ./.github/workflows/pkg_resolve.yml + workflow_call: - docs: - name: Developer CI test - Julia ${{ matrix.julia }} - needs: pkg_resolve - permissions: - contents: read - packages: write - attestations: write - id-token: write - actions: write +jobs: + test: + name: Julia ${{ matrix.julia }} strategy: fail-fast: false matrix: @@ -57,6 +14,7 @@ jobs: - '1.10' - '1.11' - '1.12' + - '1.13' # runs-on: [self-hosted, linux, x64] runs-on: ubuntu-latest container: @@ -157,4 +115,4 @@ jobs: cp LocalPreferences.toml test/LocalPreferences.toml - julia --color=yes --project=. -e 'using Pkg; Pkg.test("cuNumeric"; test_args=["--quickfail", "--jobs=2", "--verbose"])' + julia --color=yes --project=. -e 'using Pkg; Pkg.develop(PackageSpec(path = "lib/CNPreferences")); Pkg.test("cuNumeric"; test_args=["--jobs=2", "--verbose"])' diff --git a/.github/workflows/docs-tags.yml b/.github/workflows/docs-tags.yml index ead34e481..612bbd32a 100644 --- a/.github/workflows/docs-tags.yml +++ b/.github/workflows/docs-tags.yml @@ -6,9 +6,12 @@ on: push: tags: - 'v*' + # gh workflow run "Docs (tags)" --ref vX.Y.Z + workflow_dispatch: jobs: docs: name: Documentation + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} permissions: actions: write contents: write @@ -19,6 +22,9 @@ jobs: LEGATE_AUTO_CONFIG: 0 steps: - uses: actions/checkout@v4 + with: + # all tags needed to pick `stable` + fetch-depth: 0 - uses: julia-actions/setup-julia@v2 with: version: '1.11' diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index 3317cea6a..84da51106 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -16,6 +16,7 @@ on: jobs: docs: name: Documentation + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} permissions: actions: write contents: write diff --git a/.github/workflows/pkg_resolve.yml b/.github/workflows/pkg_resolve.yml index 47c5302d1..6eb18b0bf 100644 --- a/.github/workflows/pkg_resolve.yml +++ b/.github/workflows/pkg_resolve.yml @@ -1,11 +1,11 @@ -name: Pkg Resolve +name: Package Resolution on: workflow_call: jobs: resolve: - name: Pkg.resolve + name: Resolve dependencies runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/version_check.yml b/.github/workflows/version_check.yml index 346bc7d6b..a738ada5d 100644 --- a/.github/workflows/version_check.yml +++ b/.github/workflows/version_check.yml @@ -8,6 +8,7 @@ on: jobs: version-check: name: Version Check + if: ${{ !contains(toJSON(github.event), '[skip ci]') }} runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 diff --git a/.gitignore b/.gitignore index 0311f9e68..045e0e303 100644 --- a/.gitignore +++ b/.gitignore @@ -16,6 +16,16 @@ logging logging/* debug debug/* +!benchmark/debug/ +!benchmark/debug/grayscott_accelerate.jl + +# example outputs (examples/data and the docs copy of gray-scott.gif are tracked) +examples/*.h5 +examples/*.gif +/gray-scott.h5 +/dmd-mode.h5 +# h5write emits a virtual dataset backed by this sidecar directory +*_legate_vds/ # benchmark outputs benchmark/results** diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 000000000..267133871 --- /dev/null +++ b/.gitmodules @@ -0,0 +1,3 @@ +[submodule "benchmark"] + path = benchmark + url = https://github.com/JuliaLegate/benchmarking.git diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 03d022894..938d9539b 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -14,6 +14,11 @@ repos: - id: clang-format types_or: [c++, c] + - repo: https://github.com/reteps/dockerfmt + rev: v0.5.4 + hooks: + - id: dockerfmt + - repo: local hooks: - id: julia-formatter diff --git a/Project.toml b/Project.toml index 8104b9c40..eff1a1ad0 100644 --- a/Project.toml +++ b/Project.toml @@ -1,8 +1,12 @@ name = "cuNumeric" uuid = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" -version = "0.2.0" +version = "0.3.0" + +[workspace] +projects = ["test", "dev"] [deps] +AbstractFFTs = "621f4979-c628-5d54-868e-fcf4e3e8185c" CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" CUDACore = "bd0ed864-bdfe-4181-a5ed-ce625a5fdea2" CUDATools = "9ec180c6-1c07-47c7-9e6e-ebefa4d1f6d0" @@ -19,30 +23,50 @@ OpenBLAS32_jll = "656ef2d0-ae68-5445-9ca0-591084a874a2" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" Preferences = "21216c6a-2e73-6563-6e65-726566657250" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" +StaticArrays = "90137ffa-7385-5640-81b9-e52037218182" StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" cunumeric_jl_wrapper_jll = "49048992-29d2-5fd1-994f-9cecf112d624" cupynumeric_jll = "2862d674-414d-5b0b-a494-b21f8deca547" libcxxwrap_julia_jll = "3eaa8342-bff7-56a5-9981-c04077f7cee7" +[weakdeps] +StructArrays = "09ab397b-f2b6-538f-b94a-2f83cf4a842a" +Krylov = "ba0b0d4f-ebba-5204-a429-3ac8c609bfb7" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" + +[extensions] +cuNumericStructArraysExt = "StructArrays" +cuNumericKrylovExt = "Krylov" +cuNumericTensorOperationsExt = "TensorOperations" + [compat] -CNPreferences = "0.1.3" -CUDACore = "6.2" -CUDATools = "6.2" -CxxWrap = "0.17" +AbstractFFTs = "1.5" +CNPreferences = "0.1.4" +CUDACore = "6.4" +CUDATools = "6.4" +CxxWrap = "0.17.5" ExpressionExplorer = "1.1.4" JuliaFormatter = "2.3.0" KernelAbstractions = "0.9.41" -Legate = "0.2.1" -LegatePreferences = "0.1.6" +Krylov = "0.10.10" +Legate = "0.2.4" +LegatePreferences = "0.1.7" MacroTools = "0.5.16" OpenBLAS32_jll = "0.3" Pkg = "1" Preferences = "1" Random = "1" +StaticArrays = "1" StatsBase = "0.34" -cunumeric_jl_wrapper_jll = "26.6.0" -cupynumeric_jll = "26.6.0" +StructArrays = "0.7" +TensorOperations = "5.8" +cunumeric_jl_wrapper_jll = "26.6.4" +cupynumeric_jll = "26.6.1" julia = "1.10" -[workspace] -projects = ["test", "dev"] +[extras] +CUDA_Runtime_jll = "76a88914-d11a-5bdc-97e0-2f5a05c973a2" +CUDA_Compiler_jll = "d1e2174e-dfdc-576e-b43e-73b79eb1aca8" + +[targets] +build = ["CUDA_Runtime_jll", "CUDA_Compiler_jll"] diff --git a/README.md b/README.md index b51c551b3..531ddaf1b 100644 --- a/README.md +++ b/README.md @@ -1,9 +1,9 @@

cuNumeric.jl - cuNumeric.jl + cuNumeric.jl

-[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) +[![Documentation stable](https://img.shields.io/badge/docs-stable-blue.svg)](https://julialegate.github.io/cuNumeric.jl/stable) [![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. @@ -18,7 +18,7 @@ using Pkg Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") ``` -The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). +The first installation can take a while because it includes several large dependencies, such as the CUDA SDK. To use a local cupynumeric build, see [Build Modes](https://julialegate.github.io/cuNumeric.jl/stable/install). ```julia using cuNumeric @@ -28,26 +28,32 @@ cuNumeric.versioninfo() > [!WARNING] > Starting more than one instance of cuNumeric.jl can lead to a hard-crash. The default hardware configuration reserves all available resources. -For more details, see [Hardware](./configuration/hardware.md). +For more details, see [Hardware](https://julialegate.github.io/cuNumeric.jl/stable/configuration/hardware). ### How `NDArray`s work The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it is a drop-in replacement. You can use the same constructors (i.e., `zeros`, `ones`, `rand`), broadcasting, slicing, and linear algebra. Under the hood a few details differ from Base, and knowing them can help you write fast code. -**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be partitioned over GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. +**Data may live across many devices.** An `NDArray` is a logical array whose physical buffers can be stored across multiple GPUs and CPUs by the Legate runtime. You write ordinary array code and Legate decides where the data lives and how/when it is communicated between devices. As a result, elementwise indexing (i.e. `arr[1]`) is slow (and is prevented by default). Scalar indexing like this forces synchronization and blocks other tasks from executing. -**Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. +**Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutating an `NDArray` slice mutates the parent and all other aliases of the underlying data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communite with the Julia runtime. When you need a plain Julia number, call `unwrap`: +**Scalar reductions return device scalars.** Full reductions such as `sum(A)`, `dot(x,y)`, and `norm(x)` return numeric wrappers (`CNFloat`, `CNInt`, `CNUInt`, `CNBool`, or `CNComplex`) backed by 0D NDArrays. Each subtypes the corresponding abstract numeric category; `CNReal` and `CNScalar` group the wrapper families. Dimension-preserving reductions still return NDArrays. Arithmetic stays on the backend; use `fetch` for explicit host extraction: ```julia -s = sum(A) # NDArray{T,0} -x = unwrap(s) # T, e.g. Float32 +s = sum(A) # CNScalar +x = fetch(s) # native Julia scalar, e.g. Float32 ``` -**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. +Use `allowautofetch() do ... end` or `@allowautofetch` to permit host comparisons and numeric conversions of device scalars. Permission is task-local and disabled by default; arithmetic remains on the backend. See [Device scalars](https://julialegate.github.io/cuNumeric.jl/stable/api_cnscalar). -For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). +**0D arrays support scalar-shaped arithmetic.** Use `+`, `-`, `*`, `/`, and `^` +with two 0D NDArrays or with a 0D NDArray and a Julia number. Results stay as +0D NDArrays on the backend, following the existing promotion policy. + +**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into a task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them [for example `println`, `fetch`, or communicating with the Julia runtime (i.e., `Array(A)`)]. Hiding latency enables performant code. + +For API details see [Initialization](https://julialegate.github.io/cuNumeric.jl/stable/api_initialization) and [NDArray Reference](https://julialegate.github.io/cuNumeric.jl/stable/api). For common performance pitfalls, see [Patterns to Avoid](https://julialegate.github.io/cuNumeric.jl/stable/perf/patterns_to_avoid). ### Kernel Fusion @@ -57,46 +63,58 @@ Nested broadcast expressions fuse into a single kernel by default when on GPU. P y .= @. -a + b * c ``` -See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. +See [Kernel Fusion](https://julialegate.github.io/cuNumeric.jl/stable/perf/kernel_fusion) and [Debugging](https://julialegate.github.io/cuNumeric.jl/stable/debugging) for controls and diagnostics. -### Helping the Garbage Collector +### The `@accelerate` macro -Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. +`@accelerate` fuses eligible GPU broadcasts within and across statements, then releases materialized temporary `NDArray`s after their last use on CPU or GPU. See [The `@accelerate` Macro](https://julialegate.github.io/cuNumeric.jl/stable/perf/reduce_allocations) for usage guidance. -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. +### Benchmarks -```julia -@analyze_lifetimes begin - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 -end -``` +Results and reproduction instructions live under [Benchmark Results](https://julialegate.github.io/cuNumeric.jl/stable/benchmarks/results) and [How to Benchmark](https://julialegate.github.io/cuNumeric.jl/stable/benchmarks/howto). + +### TensorOperations.jl Integration + +We implement a package extension for [TensorOperations.jl](https://github.com/QuantumKitHub/TensorOperations.jl) to enable clean syntax and optimal contraction order for tensor contractions. Simply load TensorOperations alongside cuNumeric to take advantage! Some useful macros if you are unfamiliar are [@tensor](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#The-@tensor-macro), [@tensoropt](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@tensoropt) and [@notensor](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@notensor). -### Performance at a glance +```julia +using TensorOperations +using cuNumeric -A representative benchmark figure will go here (add something like `docs/src/images/benchmarks-overview.png` when ready). +α = randn() # prefer a Julia Number when you already have one +A = cuNumeric.randn(5, 5, 5, 5, 5, 5) +B = cuNumeric.randn(5, 5, 5) +C = cuNumeric.randn(5, 5, 5) +D = cuNumeric.zeros(5, 5, 5) -Numbers, plots, and how to reproduce them live under [Benchmark Results](./benchmarks/results.md) and [How to Benchmark](./benchmarks/howto.md). +@tensor begin + D[a, b, c] = A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + E[a, b, c] := A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] +end +``` ### Try an example ```julia using cuNumeric -integrand = (x) -> @. exp(-x^2) +integrand(x) = exp(-x^2) + +@accelerate function monte_carlo(N, x_max) + Ω = 2 * x_max + raw_samples = cuNumeric.rand(N) + samples = @. Ω * raw_samples - x_max + return (Ω / N) * sum(integrand.(samples)) +end N = 1_000_000 x_max = 10.0f0 -Ω = 2 * x_max - -samples = Ω .* cuNumeric.rand(N) -samples = samples .- x_max -estimate = (Ω / N) .* sum(integrand(samples)) +estimate = monte_carlo(N, x_max) -println("Monte-Carlo Estimate: $(estimate)") +println("Monte-Carlo Estimate: $(estimate)") # Should be ~ sqrt(pi) ``` More worked examples (initialization, Gray-Scott, …) are in the documentation sidebar under **Examples**. ### Known Limitations -- There is no support for `Float16` or `ComplexF16` +- There is no support for `Float16` or `ComplexF16` or `Complex{<:Integer}` diff --git a/TODO.md b/TODO.md index 23b4be98b..9248f7253 100644 --- a/TODO.md +++ b/TODO.md @@ -4,7 +4,84 @@ - Replace `as_type` with `Base.convert` - Support Ints on methods that takes floats - Programatic manipulation of Legate hardware config (not currently possible) -- Float32 random number generation (not possible in current C++ API) -- Normal random numbers (not possible in current C++ API) - Add Aqua.jl to CI to ensure we didn't pirate any types -- Fix CodeCov reports + +## Base + +Easy `Base` / `AbstractArray` gaps for `NDArray`. Module helpers +(`cuNumeric.reshape` / `transpose` / `unique` / …) often exist, but the +corresponding `Base` methods are missing, so calls fall through to +`AbstractArray` and may scalar-index. Wire up `Base.*` when convenient. + +**P0** +- `Base.reshape`, `Base.vec` +- 1-D range `setindex!` (`A[2:3] = B`); only 2-D range assignment exists +- `fill!` (convertible eltypes) +- `collect` + +**P1** +- `transpose` / `adjoint` +- `unique` +- `ones_like` +- `dropdims` + +**P2** +- `extrema`, `mean` +- `count` / nonzero +- `abs2` +- numeric `all` / `any` +- `sum(f, A)` specials + +**P3** +- `copyto!(NDArray, AbstractArray)` +- `floor` / `ceil` / `clamp` +- 2D `permutedims` +- `diff` + +## Broadcast fusion + +- Fused runtime scalars are promoted to one common type before launch, so + mixing e.g. `UInt64` and `Float32` scalars loses precision. Keeping each + scalar's type would also let struct `fill!` / `setindex!` use the fused + kernel instead of Legate's `issue_fill`. + +## LinearAlgebra + +Starter list of easy/medium LA gaps. Prefer wiring `LinearAlgebra` entry +points so they do not fall through to scalar-indexing Base paths. + +**BLAS-1 style (dense `NDArray`)** +- Additional BLAS-style indexed updates. + +**Reductions / traces** +- `LinearAlgebra.tr` for dense 2D `NDArray` — `cuNumeric.trace` exists; `tr` is + already wired for `Diagonal{<:NDArray}` + +**Diagonal vs fallthrough (context)** +- Still fallthrough / unsupported on `Diagonal` (e.g. `svd`, `pinv`, + `cholesky`, host `AbstractArray` RHS): leave alone unless fixing is cheap; + densify intentionally when needed + +**Decompositions** +- `eigh` / Hermitian eigen — the `SYEV` task is already wrapped, but there is no + entry point until `Hermitian` / `Symmetric` work on `NDArray` +- `PosDefException` from `cholesky` — a non-positive-definite input now raises a + catchable `ErrorException` (since the launchers call `task_throws_exception`), + but mapping it onto `LinearAlgebra.PosDefException` needs the failing pivot + index, which the task message does not carry +- The decomposition launchers must keep calling `task_throws_exception`, as + cupynumeric's own launchers do. Besides propagating task exceptions it grows + each leaf allocation pool by `--max-exception-size` (4096 bytes), which the + batched GPU kernels depend on: `CuPyNumericMapper::allocation_pool_size` only + declares one 16-byte-aligned `int32` of ZCMEM for `SOLVE` / `GEEV`, while + `solve.cu` allocates `batchsize` pointer arrays plus `batchsize` infos and + `geev.cu` allocates an info per matrix. Without it, `b > 1` aborts on GPU +- `adjoint(::NDArray)` (already on the P1 list) would unblock `Cholesky.U`, + `SVD.V`, destructuring a `Cholesky` as `L, U`, and `ldiv!` / least-squares `\` + on `NDArrayQR` (which needs `Q' * b`) +- More than one batch dimension — needs `domain_from_shape` in Legate.jl to + handle more than three dimensions, plus a cupynumeric `POTRF` instantiated for + `DIM >= 4` +- Multi-GPU cuSolverMp paths (`MP_POTRF`, `MP_SOLVE`) for large single matrices; + `cupynumeric_has_cusolvermp()` gates them and they need an NCCL communicator +- Batched `svd` / `qr` diff --git a/benchmark b/benchmark new file mode 160000 index 000000000..05f003c52 --- /dev/null +++ b/benchmark @@ -0,0 +1 @@ +Subproject commit 05f003c528dc25faf1e87def5c4884e6c80bbd68 diff --git a/benchmark/Project.toml b/benchmark/Project.toml deleted file mode 100644 index 805db8aad..000000000 --- a/benchmark/Project.toml +++ /dev/null @@ -1,12 +0,0 @@ -[deps] -BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf" -CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" -CUDACore = "bd0ed864-bdfe-4181-a5ed-ce625a5fdea2" -Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" -Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" -Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" -TOML = "fa267f1f-6049-4f14-aa54-33bafae1ed76" -cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" - -[extras] -LegatePreferences = "8028f36a-2b64-49e9-aa04-2d0933fd2ed9" diff --git a/benchmark/README.md b/benchmark/README.md deleted file mode 100644 index 205cd08ca..000000000 --- a/benchmark/README.md +++ /dev/null @@ -1,84 +0,0 @@ -# Benchmark configuration - -Benchmarks are declared in `benchmarks.toml`. `run.jl` parses it. - -## Running - -```bash -julia --project run.jl # runs whatever benchmarks.toml configures -``` - -`run.jl` runs each (benchmark, backend) pair in its own process via -`run_benchmark.sh`, so backends never share a GPU/runtime within a measurement. -cuNumeric always runs; extra comparison backends are toggled in `[Global]`: - -- `cuda = true` → also run under CUDA.jl (single-GPU configs only; CUDA.jl is - single-device). -- `cupynumeric = true` → also run under cupynumeric (see below). - -### Comparing against cupynumeric - -cupynumeric runs in a conda env whose major.minor matches this project's -resolved `cupynumeric_jll`. Build it once: - -```bash -./install_cupynumeric.sh # creates env cupynumeric-bench- -``` - -`run.jl` derives the env name automatically; override it with `CUPYNUMERIC_ENV`. - -## Layout - -```toml -[Global] -n_warmup = 5 -n_iter = 1000 -n_trial = 5 - -[[gemm]] # name registered in src/benchmarks.jl -T = "Float32" # element type -gpus = 1 -cpus = 2 -N = 150 -M = 150 # optional, defaults to 1 -fusion = true # optional, defaults to true; toggles cuNumeric broadcast fusion -``` - -Repeat a `[[name]]` block to add independent configs. - -## Lists - -Any of `T`, `fusion`, `gpus`, `cpus`, `N`, `M` may be a list. They expand along -two axes: -- **`T` and `fusion` multiply.** The whole sweep runs once per type and once per - fusion setting (`fusion = [true, false]` sweeps both). -- **`gpus`, `cpus`, `N`, `M` zip** into a single lockstep sweep — element `i` - of each is paired together. - -`fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, -default `true`); it only affects cuNumeric, so comparison backends run once, not -per variant. - -Each zipped field must be one of: - -- a scalar or single-element list (`cpus = 2` or `[2]`) -> broadcast to every config -- a list whose length equals the sweep length - -Any other length mismatch is an error. - -```toml -[[sgemm]] -T = ["Float64", "Float32"] # multiplies -gpus = [1, 2, 4] # -cpus = 2 # zip -> (1,2,150,150), (2,2,300,300), (4,2,600,600) -N = [150, 300, 600] # -M = [150, 300, 600] # -``` - --> 2 types * 3 sweep points = **6 runs**. - -### Gotcha - -When `T = ["Float32", "Float64"]` and a length-2 `N`/`M` sweep you get all **4** -combinations, not a paired `Float32 -> N[1], Float64 -> N[2]`. To pin a type -to a specific size, use separate `[[name]]` blocks. diff --git a/benchmark/benchmarks.toml b/benchmark/benchmarks.toml deleted file mode 100644 index 70d5a47a6..000000000 --- a/benchmark/benchmarks.toml +++ /dev/null @@ -1,59 +0,0 @@ -[Global] -n_warmup = 5 -n_iter = 1000 -n_trial = 5 -cupynumeric = true # (needs install_cupynumeric.sh) -cuda = false # compare against CUDA.jl (single-GPU configs only) -# One CPU-reference check per config (not per timed iter). Written to CSV. -check_correctness = true -n_correctness_iter = 5 - -#################################### -# GEMM # -# Weak scaling, restricted to NxN. # -# Work ~ 2*N^2*M = 2*N^3. # -# N^3 / P --> constant. # -# N = baseline * P^(1/3). # -#################################### - -[[gemm]] -T = ["Float32"] -gpus = [1, 2, 4, 8] -cpus = 16 -N = [20000, 25200, 31752, 40000] -M = [20000, 25200, 31752, 40000] - -################################# -# Gray-Scott # -# Weak scaling, square NxN grid.# -# Work ~ N*M = N^2. # -# N^2 / P --> constant. # -# N = baseline * P^(1/2). # -################################# - -[[grayscott_baseline]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 16 -fusion = [true, false] -N = [2000, 2832, 4000, 5656] -M = [2000, 2832, 4000, 5656] - -[[grayscott_lifetimes]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 16 -fusion = false -N = [2000, 2832, 4000, 5656] -M = [2000, 2832, 4000, 5656] - -################################# -# Monte-Carlo Integration # -# Work ~ N. Scale N linearly # -################################# - -[[montecarlo]] -T = "Float32" -gpus = [1, 2, 4, 8] -cpus = 16 -N = [1_000_000, 2_000_000, 4_000_000, 8_000_000] diff --git a/benchmark/cpp_matmul/CMakeLists.txt b/benchmark/cpp_matmul/CMakeLists.txt deleted file mode 100644 index d547f62dd..000000000 --- a/benchmark/cpp_matmul/CMakeLists.txt +++ /dev/null @@ -1,17 +0,0 @@ - -cmake_minimum_required(VERSION 3.22.1 FATAL_ERROR) - -project(cuNumericWrapper VERSION 0.01 LANGUAGES C CXX) - -# Specify C++ standard -set(CMAKE_CXX_STANDARD_REQUIRED True) - -if (NOT CMAKE_CXX_STANDARD) - set(CMAKE_CXX_STANDARD 20) -endif() - -find_package(cupynumeric REQUIRED) - -add_executable(matmulfp32_test main.cpp) -target_link_libraries(matmulfp32_test PRIVATE cupynumeric::cupynumeric) -install(TARGETS matmulfp32_test DESTINATION "${CMAKE_CURRENT_BINARY_DIR}/cmake-install") diff --git a/benchmark/cpp_matmul/build.sh b/benchmark/cpp_matmul/build.sh deleted file mode 100755 index d7463c345..000000000 --- a/benchmark/cpp_matmul/build.sh +++ /dev/null @@ -1,7 +0,0 @@ -legate_root=`python -c 'import legate.install_info as i; from pathlib import Path; print(Path(i.libpath).parent.resolve())'` -echo "Using Legate at $legate_root" -cupynumeric_root=`python -c 'import cupynumeric.install_info as i; from pathlib import Path; print(Path(i.libpath).parent.resolve())'` -echo "Using cuPyNumeric at $cupynumeric_root" - -cmake -S . -B build -D legate_ROOT="$legate_root" -D cupynumeric_ROOT="$cupynumeric_root" -D CMAKE_BUILD_TYPE=Debug -cmake --build build --parallel 8 --verbose diff --git a/benchmark/cpp_matmul/main.cpp b/benchmark/cpp_matmul/main.cpp deleted file mode 100644 index a549e02ec..000000000 --- a/benchmark/cpp_matmul/main.cpp +++ /dev/null @@ -1,50 +0,0 @@ -/* Copyright 2025 Northwestern University, - * Carnegie Mellon University University - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - * - * Author(s): David Krasowska - * Ethan Meitz - */ - -#include - -#include "cupynumeric.h" -#include "legate.h" -// #include - -// using cupynumeric::slice; - -void matmul_fp32(size_t N) { - std::vector dims = {N, N}; - auto A = cupynumeric::random(dims).as_type(legate::float32()); - auto B = cupynumeric::random(dims).as_type(legate::float32()); - std::optional T = legate::float32(); - auto C = cupynumeric::zeros(dims, T); - - C.dot(A, B); - return; - // std::cout << C[{slice(0,0), slice(0,0)}] << std::endl; -} - -int main(int argc, char** argv) { - auto result = legate::start(argc, argv); - // assert(result == 0); - - cupynumeric::initialize(argc, argv); - - const size_t N = 10000; - matmul_fp32(N); - - return legate::finish(); -} diff --git a/benchmark/install_cupynumeric.sh b/benchmark/install_cupynumeric.sh deleted file mode 100755 index 6140cf64a..000000000 --- a/benchmark/install_cupynumeric.sh +++ /dev/null @@ -1,76 +0,0 @@ -#!/bin/bash -# Install a cupynumeric conda env matching the cupynumeric_jll our project resolves. -# The conda package and the JLL share the calendar-versioning scheme (e.g. 25.10), -# so we pin major.minor (patch ignored) and install from the legate channel. -# -# Usage: -# ./install_cupynumeric.sh # create a fresh env named cupynumeric-bench- -# ./install_cupynumeric.sh --name myenv # override the env name -# ./install_cupynumeric.sh --into existing # install into an existing env instead of creating one -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" - -ENV_NAME="" -INTO_ENV="" - -while [[ $# -gt 0 ]]; do - case $1 in - --name) - ENV_NAME=$2 - shift 2 - ;; - --into) - INTO_ENV=$2 - shift 2 - ;; - *) - echo "Unknown argument: $1" - echo "Usage: $0 [--name ] [--into ]" - exit 1 - ;; - esac -done - -# Resolve the JLL version Julia actually instantiated for this project, then keep -# major.minor only — conda packages are not published per patch. -echo "Detecting cupynumeric_jll version from the benchmark project..." -VER=$(cd "$SCRIPT_DIR" && julia --project -e ' -using Pkg -for (_, info) in Pkg.dependencies() - info.name == "cupynumeric_jll" || continue - v = info.version - isnothing(v) && continue - println("$(v.major).$(v.minor)") -end' | tail -1) - -if [[ -z "$VER" ]]; then - echo "Error: could not detect cupynumeric_jll version. Has the project been instantiated?" - exit 1 -fi - -echo "cupynumeric_jll major.minor: $VER" -SPEC="cupynumeric=$VER.*" - -# numpy 2.3 dropped the private numpy.linalg.linalg path that cupynumeric 25.10 imports. -NUMPY_SPEC="numpy<2.3" - -if [[ -n "$INTO_ENV" ]]; then - echo "Installing $SPEC into existing env '$INTO_ENV'..." - conda install -y -n "$INTO_ENV" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" - echo "Done. Activate with: conda activate $INTO_ENV" - exit 0 -fi - -[[ -z "$ENV_NAME" ]] && ENV_NAME="cupynumeric-bench-$VER" - -if conda env list | awk '{print $1}' | grep -qx "$ENV_NAME"; then - echo "Env '$ENV_NAME' already exists with $SPEC; nothing to do." - echo "Activate with: conda activate $ENV_NAME" - exit 0 -fi - -echo "Creating env '$ENV_NAME' with $SPEC..." -conda create -y -n "$ENV_NAME" -c conda-forge -c legate "$SPEC" "$NUMPY_SPEC" - -echo "Done. Activate with: conda activate $ENV_NAME" diff --git a/benchmark/run.jl b/benchmark/run.jl deleted file mode 100644 index cb748ccde..000000000 --- a/benchmark/run.jl +++ /dev/null @@ -1,139 +0,0 @@ -# run.jl: orchestrator. Builds one run_benchmark.sh command per benchmark and -# dispatches it; the script sets LEGATE_CONFIG (from --gpus/--cpus) before -# launching the worker (single.jl) that actually runs the benchmark. -# no args -> one command per benchmarks.toml entry -# with args -> one command from [fusion] - -# Orchestrator stays off the GPU: it only needs GlobalSettings + parse_config, -# both cuNumeric-free. The worker (single.jl) loads cuNumeric and the kernels. - -using Pkg - -include("src/core.jl") -include("src/parse_benchmarks.jl") - -const RUNNER = joinpath(@__DIR__, "run_benchmark.sh") -const WORKER = joinpath(@__DIR__, "src/single.jl") -const PY_WORKER = joinpath(@__DIR__, "src_py/single.py") - -# quiet by default (banner + results only); -v/--verbose shows the plumbing -const VERBOSE_FLAGS = ("-v", "--verbose") -const VERBOSE = any(in(ARGS), VERBOSE_FLAGS) -const POSARGS = filter(a -> a ∉ VERBOSE_FLAGS, ARGS) - -banner(msg) = println("\n", "="^128, "\n", msg, "\n", "="^128) - -# `_lifetimes` is a cuNumeric-only code-path variant (@analyze_lifetimes) -cunumeric_only(name) = endswith(name, "_lifetimes") - -const LAST_FUSION_TOGGLE = Ref{Union{Nothing,Bool}}(nothing) - -# dev CNPreferences && cuNumeric -function ensure_project_ready() - Pkg.develop([ - Pkg.PackageSpec(; path=joinpath(@__DIR__, "..", "lib", "CNPreferences")), - Pkg.PackageSpec(; path=joinpath(@__DIR__, "..")), - ]) - return Pkg.instantiate() -end - -# default env name mirrors install_cupynumeric.sh: cupynumeric-bench-. -# CUPYNUMERIC_ENV overrides it. -function cupynumeric_env_name() - haskey(ENV, "CUPYNUMERIC_ENV") && return ENV["CUPYNUMERIC_ENV"] - for (_, info) in Pkg.dependencies() - info.name == "cupynumeric_jll" || continue - info.version === nothing && continue - return "cupynumeric-bench-$(info.version.major).$(info.version.minor)" - end - return error("could not resolve cupynumeric_jll version; set CUPYNUMERIC_ENV explicitly") -end - -function dispatch(; gpus, cpus, name, T, N, M, n_iter, n_warmup, n_trial, - fusion=true, cupynumeric=false, cudajl=false, - check_correctness=false, n_correctness_iter=5) - fstr = fusion ? "enabled" : "disabled" - banner( - "$(name): T=$(T) gpus=$(gpus) cpus=$(cpus) N=$(N) M=$(M) fusion=$(fstr) " * - "n_iter=$(n_iter) n_warmup=$(n_warmup) n_trial=$(n_trial)", - ) - - # precompile in the orchestrator so the worker loads a warm cache quietly - CNPreferences.set_broadcast_fusion!(fusion) - if LAST_FUSION_TOGGLE[] != fusion - VERBOSE && println("Precompiling cuNumeric (fusion=$(fstr))") - Pkg.precompile("cuNumeric"; io=devnull) - LAST_FUSION_TOGGLE[] = fusion - end - - # each backend runs in its own worker process - vflag = VERBOSE ? `--verbose` : `` - args = `--gpus $gpus --cpus $cpus $name $T $N $M $n_iter $n_warmup $n_trial` - # trailing: backend check_correctness n_correctness_iter - corr_args = `$check_correctness $n_correctness_iter` - cmds = [`bash $RUNNER $WORKER $vflag $args cunumeric $corr_args`] - - # comparison backends have no fusion knob, so run them once instead of per - # fusion variant; the fused pass (the default) is that single run - run_comparison_backends = fusion - if run_comparison_backends - # CUDA.jl is single-GPU only - if cudajl && gpus == 1 && !cunumeric_only(name) - push!(cmds, `bash $RUNNER $WORKER $vflag $args cudajl $corr_args`) - end - if cupynumeric && !cunumeric_only(name) - push!(cmds, `bash $RUNNER $PY_WORKER $vflag --pyenv $(cupynumeric_env_name()) $args`) - end - end - - for cmd in cmds - try - run(cmd) - catch e - @error "Benchmark '$(name)' failed; continuing." exception = e - end - end -end - -function run_all_benchmarks(config="benchmarks.toml") - gs, specs = parse_config(joinpath(@__DIR__, config)) - for spec in specs - N, M = spec.args - dispatch(; - gpus=spec.gpus, - cpus=spec.cpus, - name=spec.name, - T=spec.T, - N=N, M=M, - fusion=spec.fusion, - n_iter=gs.n_iter, - n_warmup=gs.n_warmup, - n_trial=gs.n_trial, - cupynumeric=gs.cupynumeric, - cudajl=gs.cuda, - check_correctness=gs.check_correctness, - n_correctness_iter=gs.n_correctness_iter, - ) - end -end - -ensure_project_ready() -using CNPreferences: CNPreferences -if isempty(POSARGS) - run_all_benchmarks() -else # dispatch on args - dispatch(; - gpus=parse(Int, POSARGS[1]), - cpus=parse(Int, POSARGS[2]), - name=POSARGS[3], - T=POSARGS[4], - N=parse(Int, POSARGS[5]), - M=parse(Int, POSARGS[6]), - n_iter=parse(Int, POSARGS[7]), - n_warmup=parse(Int, POSARGS[8]), - n_trial=parse(Int, POSARGS[9]), - fusion=length(POSARGS) >= 10 ? parse_fusion(POSARGS[10]) : true, - check_correctness=length(POSARGS) >= 11 ? parse(Bool, POSARGS[11]) : false, - n_correctness_iter=length(POSARGS) >= 12 ? parse(Int, POSARGS[12]) : 5, - ) -end diff --git a/benchmark/run_benchmark.sh b/benchmark/run_benchmark.sh deleted file mode 100755 index 8fc47aa8d..000000000 --- a/benchmark/run_benchmark.sh +++ /dev/null @@ -1,81 +0,0 @@ -#!/bin/bash - -if [[ $# -lt 1 ]]; then - echo "Usage: $0 [--gpus ] [--cpus ] [extra_args...]" - exit 1 -fi - -# Parse arguments -FILENAME=$1 -shift - -GPUS=0 -CPUS=1 -PYENV="" -VERBOSE=0 - -while [[ $# -gt 0 ]]; do - case $1 in - --gpus) - GPUS=$2 - shift 2 - ;; - --cpus) - CPUS=$2 - shift 2 - ;; - --pyenv) - PYENV=$2 - shift 2 - ;; - --verbose) - VERBOSE=1 - shift - ;; - *) - # Collect all other arguments as extra arguments - EXTRA_ARGS+=("$1") - shift - ;; - esac -done - -# Validate the filename exists -if [[ ! -f $FILENAME ]]; then - echo "Error: File $FILENAME does not exist." - exit 1 -fi - -# Inform user of the configuration -if [[ $GPUS -lt 0 ]]; then - echo "GPUs invalid, using gpus = 0" - exit -fi - -if [[ $CPUS -lt 0 ]]; then - echo "CPUs invalid, using cpus = 1" - exit -fi - -export LEGATE_AUTO_CONFIG=1 -export LEGATE_CONFIG="--cpus=$CPUS --gpus=$GPUS" -export LEGATE_SHOW_CONFIG=$VERBOSE - -export LD_LIBRARY_PATH="" - -[[ $VERBOSE == 1 ]] && echo "Running $FILENAME with $CPUS CPUs and $GPUS GPUs" - -# Python (cupynumeric) workers run in the conda env built by install_cupynumeric.sh; -# Julia (cuNumeric) workers run against the local project. -if [[ $FILENAME == *.py ]]; then - if [[ -z $PYENV ]]; then - echo "Error: running a .py worker requires --pyenv (run install_cupynumeric.sh first)." - exit 1 - fi - CMD="conda run --no-capture-output -n $PYENV python $FILENAME $GPUS ${EXTRA_ARGS[@]}" -else - CMD="julia --project $FILENAME $GPUS ${EXTRA_ARGS[@]}" -fi - -[[ $VERBOSE == 1 ]] && printf "Running: %s\n" "$CMD" -eval "$CMD" diff --git a/benchmark/src/benchmarks/gemm.jl b/benchmark/src/benchmarks/gemm.jl deleted file mode 100644 index 4f85d3e24..000000000 --- a/benchmark/src/benchmarks/gemm.jl +++ /dev/null @@ -1,27 +0,0 @@ -Base.@kwdef struct GEMM{T} <: AbstractBenchmark{T} - N::Int - M::Int -end - -name(::GEMM) = "gemm" -dims(g::GEMM) = (g.N, g.M) -data(g::GEMM{T}) where {T} = "GEMM with T=$(T), N=$(g.N), M=$(g.M)" - -function allowed_types(::Type{GEMM}) - return Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_INT_TYPES} -end - -total_flops(s::GEMM) = s.N * s.N * ((2*s.M) - 1) -total_space(s::GEMM{T}) where {T} = 2 * ((s.N*s.M) * sizeof(T)) + ((s.N*s.N) * sizeof(T)) - -function initialize(s::GEMM{T}; mod=cuNumeric) where {T} - A = mod.rand(T, s.N, s.M) - B = mod.rand(T, s.M, s.N) - C = mod.zeros(T, s.N, s.N) - GC.gc() - return C, A, B -end - -run!(::GEMM, C, A, B) = mul!(C, A, B) - -register_benchmark("gemm", GEMM) diff --git a/benchmark/src/benchmarks/grayscott.jl b/benchmark/src/benchmarks/grayscott.jl deleted file mode 100644 index 3ba6e6398..000000000 --- a/benchmark/src/benchmarks/grayscott.jl +++ /dev/null @@ -1,166 +0,0 @@ -struct GSParams{T} - dx::T - dt::T - c_u::T - c_v::T - f::T - k::T -end - -function GSParams{T}(; dx=1, c_u=1.0, c_v=0.3, f=0.03, k=0.06) where {T} - return GSParams{T}(T(dx), T(dx / 5), T(c_u), T(c_v), T(f), T(k)) -end - -abstract type AbstractGrayScott{T} <: AbstractBenchmark{T} end - -Base.@kwdef struct GrayScottBaseline{T} <: AbstractGrayScott{T} - N::Int - M::Int -end - -Base.@kwdef struct GrayScottLifetimes{T} <: AbstractGrayScott{T} - N::Int - M::Int -end - -name(::AbstractGrayScott) = "grayscott" -dims(b::AbstractGrayScott) = (b.N, b.M) -data(b::AbstractGrayScott{T}) where {T} = "GrayScott with T=$(T), N=$(b.N), M=$(b.M)" -allowed_types(::Type{AbstractGrayScott}) = cuNumeric.SUPPORTED_FLOAT_TYPES -total_flops(b::AbstractGrayScott) = b.N * b.M # grid points updated per step - -function build_benchmark(::Type{A}, ::Type{T}, N, M) where {A<:AbstractGrayScott,T} - return A{T}(; N=N, M=M) -end - -mutable struct GrayScottState{A,P} - u::A - v::A - u_new::A - v_new::A - params::P -end - -function initialize(b::AbstractGrayScott{T}; mod=cuNumeric, deterministic::Bool=false) where {T} - d = (b.N, b.M) - u = mod.ones(T, d) - v = mod.zeros(T, d) - u_new = mod.zeros(T, d) - v_new = mod.zeros(T, d) - - seed = min(150, b.N, b.M) - if deterministic - # Fixed host pattern so CPU and GPU (any GPU count) share the same IC. - # Avoids Random streams differing across array backends. - host_u = T[ - T(0.5) + T(0.5) * sin(T(i)) * cos(T(j)) for i in 1:seed, j in 1:seed - ] - host_v = T[ - T(0.25) + T(0.25) * cos(T(i)) * sin(T(j)) for i in 1:seed, j in 1:seed - ] - u[1:seed, 1:seed] = mod === cuNumeric ? NDArray(host_u) : host_u - v[1:seed, 1:seed] = mod === cuNumeric ? NDArray(host_v) : host_v - else - u[1:seed, 1:seed] = mod.rand(T, (seed, seed)) - v[1:seed, 1:seed] = mod.rand(T, (seed, seed)) - end - - return (GrayScottState(u, v, u_new, v_new, GSParams{T}()),) -end - -correctness_supported(::AbstractGrayScott) = true - -function check_benchmark_correctness( - b::AbstractGrayScott{T}, gs::GlobalSettings; mod=cuNumeric, atol=1e-4, rtol=1e-4 -) where {T} - # CPU reference compares via cuNumeric.compare (scalar gather). Other backends skip. - mod === cuNumeric || return "skipped" - - n = gs.n_correctness_iter - st_gpu = only(initialize(b; mod=mod, deterministic=true)) - st_cpu = only(initialize(b; mod=Base, deterministic=true)) - - for _ in 1:n - run!(b, st_gpu) - run!(b, st_cpu) - end - - # Element-wise NDArray indexing gathers across tiles — do not use Array(NDArray) - # for multi-GPU (get_ptr is local-tile only). - u_ok = @allowscalar cuNumeric.compare(st_cpu.u, st_gpu.u, atol, rtol) - v_ok = @allowscalar cuNumeric.compare(st_cpu.v, st_gpu.v, atol, rtol) - return (u_ok && v_ok) ? "pass" : "fail" -end - -# VARIANT DESCRIPTION -# baseline: as written -# lifetimes: step wrapped in @analyze_lifetimes -let body = quote - # currently we don't have NDArray^x working yet. every operator is dotted - # so each rhs fuses into a single broadcast kernel rather than shattering - # into bare +/-/* binary tasks. - F_u = ( - ( - .-u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) .+ args.f .* (1.0f0 .- u[2:(end - 1), 2:(end - 1)]) - ) - F_v = ( - ( - u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) .- (args.f + args.k) .* v[2:(end - 1), 2:(end - 1)] - ) - # 2-D Laplacian via slicing, excluding boundaries - u_lap = ( - ( - u[3:end, 2:(end - 1)] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ - u[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 .+ - ( - u[2:(end - 1), 3:end] .- 2 .* u[2:(end - 1), 2:(end - 1)] .+ - u[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - v_lap = ( - ( - v[3:end, 2:(end - 1)] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ - v[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 .+ - ( - v[2:(end - 1), 3:end] .- 2 .* v[2:(end - 1), 2:(end - 1)] .+ - v[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - - # Forward-Euler step for all interior points - u_new[2:(end - 1), 2:(end - 1)] = - ((args.c_u .* u_lap) .+ F_u) .* args.dt .+ u[2:(end - 1), 2:(end - 1)] - v_new[2:(end - 1), 2:(end - 1)] = - ((args.c_v .* v_lap) .+ F_v) .* args.dt .+ v[2:(end - 1), 2:(end - 1)] - - # Periodic boundary conditions - u_new[:, 1] = u[:, end - 1] - u_new[:, end] = u[:, 2] - u_new[1, :] = u[end - 1, :] - u_new[end, :] = u[2, :] - v_new[:, 1] = v[:, end - 1] - v_new[:, end] = v[:, 2] - v_new[1, :] = v[end - 1, :] - v_new[end, :] = v[2, :] - end - @eval _gs_step!(b::GrayScottBaseline, u, v, u_new, v_new, args::GSParams) = $body - @eval _gs_step!(b::GrayScottLifetimes, u, v, u_new, v_new, args::GSParams) = - @analyze_lifetimes $body -end - -function run!(b::AbstractGrayScott, st::GrayScottState) - _gs_step!(b, st.u, st.v, st.u_new, st.v_new, st.params) - # swap references rather than copy - st.u, st.u_new = st.u_new, st.u - st.v, st.v_new = st.v_new, st.v - return nothing -end - -register_benchmark("grayscott_baseline", GrayScottBaseline) -register_benchmark("grayscott_lifetimes", GrayScottLifetimes) diff --git a/benchmark/src/benchmarks/montecarlo.jl b/benchmark/src/benchmarks/montecarlo.jl deleted file mode 100644 index 978e5906f..000000000 --- a/benchmark/src/benchmarks/montecarlo.jl +++ /dev/null @@ -1,31 +0,0 @@ -Base.@kwdef struct MonteCarloIntegration{T} <: AbstractBenchmark{T} - n_samples::Int -end - -name(::MonteCarloIntegration) = "montecarlo" -dims(mci::MonteCarloIntegration) = (mci.n_samples, 1) -function data(mci::MonteCarloIntegration{T}) where {T} - return "Monte Carlo Integration with T=$(T), n_samples=$(mci.n_samples)" -end - -allowed_types(::Type{MonteCarloIntegration}) = cuNumeric.SUPPORTED_FLOAT_TYPES - -total_space(s::MonteCarloIntegration{T}) where {T} = s.n_samples * sizeof(T) -total_flops(s::MonteCarloIntegration) = s.n_samples - -function initialize(mci::MonteCarloIntegration{T}; mod=cuNumeric) where {T} - # Uniform samples over the integration domain [0, 10]. - x = T(10) .* mod.rand(T, mci.n_samples) - GC.gc() - return (x,) -end - -_domain_volume(mci::MonteCarloIntegration{T}) where {T} = T(10) / mci.n_samples -run!(mci::MonteCarloIntegration, x) = _domain_volume(mci) * sum(exp.(-x .^ 2)) - -# n_samples comes in as N; M is unused. -function build_benchmark(::Type{MonteCarloIntegration}, ::Type{T}, N, M) where {T} - return MonteCarloIntegration{T}(; n_samples=N) -end - -register_benchmark("montecarlo", MonteCarloIntegration) diff --git a/benchmark/src/core.jl b/benchmark/src/core.jl deleted file mode 100644 index 526ce4eb1..000000000 --- a/benchmark/src/core.jl +++ /dev/null @@ -1,155 +0,0 @@ -using Printf -using Statistics - -""" -- `n_warmup::Int` : Number of warmup steps. These are not timed. Intended - to avoid pre-compilation cost being timed. -- `n_iter::Int` : Number of iterations to run per trial. Should be large enough - to build up queue depth of tasks such that latency is hidden. -- `n_trial::Int` : Number of independent trials to run. Timing is restarted and - legate in between each trial. Sets number of datapoints used to estimated - standard deviations/errors. -- `n_gpu::Int` : The number of GPUs used by legate. Set through the LEGATE_CONFIG, - this value is just bookkeeping. -- `check_correctness::Bool` : If true, run one CPU-reference check per config - (not per timed iteration) before timing; result is recorded in the CSV. -- `n_correctness_iter::Int` : Steps to run for that single correctness check. -""" -Base.@kwdef struct GlobalSettings - n_warmup::Int # Number of warmup steps, where timing is not done. - n_iter::Int # Number of iterations to run per trial - n_trial::Int = 1 # Number of independent trials to run. Benchmark - n_gpu::Int = 0 - cupynumeric::Bool = false # also run baselines under cupynumeric for comparison - cuda::Bool = false # also run under CUDA.jl for comparison (single-GPU only) - check_correctness::Bool = false - n_correctness_iter::Int = 5 -end - -######################################### - -abstract type AbstractBenchmark{T} end - -# Interface each benchmark implements (see benchmarks/gemm.jl for a template). -function name end -function dims end -function data end -function allowed_types end -function total_flops end -function initialize end -function run! end - -# Maps a benchmarks.toml table name to its benchmark type. Each benchmark file -# registers itself via `register_benchmark`. -const BENCHMARKS = Dict{String,Type}() -function register_benchmark(key::AbstractString, ::Type{B}) where {B<:AbstractBenchmark} - return BENCHMARKS[key] = B -end - -function build_benchmark(::Type{B}, ::Type{T}, N, M) where {B<:AbstractBenchmark,T} - return B{T}(; N=N, M=M) -end - -######################################### - -# Per-trial timings for one benchmark. `times_ms[i]`/`gflops[i]` are the mean -# over `n_iter` iterations for trial `i`; the spread across trials gives stddev. -# `correctness` is one of "pass", "fail", "skipped" — checked once per config. -struct BenchmarkResult{B<:AbstractBenchmark} - times_ms::Vector{Float64} - gflops::Vector{Float64} - benchmark::B - correctness::String -end - -# Optional per-benchmark correctness vs a CPU/`Array` reference. -# Return "pass", "fail", or "skipped". Default: no check implemented. -correctness_supported(::AbstractBenchmark) = false -function check_benchmark_correctness(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) - return "skipped" -end - -# One timed trial: warmup, then time `n_iter` iterations of `run!`. -function _trial(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) - GC.gc(true) - state = initialize(b; mod=mod) - - start_time = nothing - for idx in 1:(gs.n_warmup + gs.n_iter) - if idx == gs.n_warmup + 1 - start_time = get_time_microseconds() - end - run!(b, state...) - end - total_time_μs = get_time_microseconds() - start_time - - mean_time_ms = total_time_μs / (gs.n_iter * 1e3) - gflops = total_flops(b) / (mean_time_ms * 1e6) - return mean_time_ms, gflops -end - -# Run `n_trial` independent trials and collect their per-trial measurements. -# Correctness (if enabled) runs once before timing, not per trial/iteration. -function run_benchmark(b::AbstractBenchmark, gs::GlobalSettings; mod=cuNumeric) - correctness = "skipped" - if gs.check_correctness - if correctness_supported(b) - correctness = check_benchmark_correctness(b, gs; mod=mod) - else - correctness = "skipped" - end - end - - times_ms = Float64[] - gflops = Float64[] - for _ in 1:gs.n_trial - t, g = _trial(b, gs; mod=mod) - push!(times_ms, t) - push!(gflops, g) - end - return BenchmarkResult(times_ms, gflops, b, correctness) -end - -_std(x) = length(x) > 1 ? std(x) : 0.0 - -function save_result(br::BenchmarkResult, gpus; mod::String="cunumeric") - N, M = dims(br.benchmark) - path = joinpath(@__DIR__, "..", "results", "$(name(br.benchmark))_$(mod).csv") - mkpath(dirname(path)) - open(path, "a") do io - for trial in eachindex(br.times_ms) - # correctness is per-config; repeated on each trial row for CSV joins - @printf( - io, "%s,%d,%d,%d,%d,%.6f,%.6f,%s\n", - mod, gpus, N, M, trial, - br.times_ms[trial], br.gflops[trial], br.correctness, - ) - end - end -end - -######################################### - -# `setup` runs in the worker before the benchmark is built (e.g. flip a runtime -# preference); code-path variants leave it a no-op. -# struct Variant -# name::String -# setup::Function -# end - -# const VARIANTS = Dict{String,Variant}() - -# function register_variant(name, setup=() -> nothing) -# VARIANTS[name] = Variant(name, setup) -# end - -# function variant_setup(name) -# if haskey(VARIANTS, name) -# return VARIANTS[name].setup -# end -# return () -> nothing -# end - -# register_variant("baseline") -# register_variant("fusion_off", cuNumeric.disable_broadcast_fusion!) -# register_variant("fusion_on", cuNumeric.enable_broadcast_fusion!) diff --git a/benchmark/src/parse_benchmarks.jl b/benchmark/src/parse_benchmarks.jl deleted file mode 100644 index 28cad96d4..000000000 --- a/benchmark/src/parse_benchmarks.jl +++ /dev/null @@ -1,99 +0,0 @@ -using TOML - -""" -One benchmark invocation parsed from `benchmarks.toml`. `name` selects the -benchmark type from `BENCHMARKS`; `T` is the element type (e.g. "Float32"); -`args` are the sizes (currently `N M`). -""" -struct BenchmarkSpec - name::String - T::String - gpus::Int - cpus::Int - fusion::Bool - args::Vector{Int} -end - -# A field may be a scalar or a list. -aslist(x) = x isa AbstractVector ? collect(x) : [x] - -# `fusion` accepts a bool or "on"/"off" (or a list of these). -function parse_fusion(x) - x isa Bool && return x - s = lowercase(string(x)) - s in ("on", "true") && return true - s in ("off", "false") && return false - return error("fusion must be on/off (or true/false); got $(repr(x))") -end - -# Value of a zipped field for sweep position `i`. length==1 field broadcasts. -sweep_value(field, i) = length(field) == 1 ? field[1] : field[i] - -# Number of positions in the sweep. Every multi-element field must agree on length; -# length==1 fields broadcast and don't constrain it. -function sweep_length(name, fields) - lengths = [length(field) for (_, field) in fields if length(field) > 1] - isempty(lengths) && return 1 - allequal(lengths) || error( - "benchmark '$(name)': zipped fields gpus/cpus/N/M must share one length " * - "or be scalar; got " * join(("$k=$(length(v))" for (k, v) in fields), ", "), - ) - return first(lengths) -end - -# Names of the `[[name]]` blocks in the order they appear in the file. TOML.jl -# parses into an unordered Dict, so we scan the source to preserve run order. -function declared_order(path) - order = String[] - for line in eachline(path) - header = strip(line) - startswith(header, "[[") && endswith(header, "]]") || continue - name = strip(header[3:(end - 2)]) - name in order || push!(order, name) # if not in list, push to ordered list - end - return order -end - -function parse_config(path) - raw = TOML.parsefile(path) - - g = raw["Global"] - global_settings = GlobalSettings(; - n_warmup=g["n_warmup"], n_iter=g["n_iter"], n_trial=get(g, "n_trial", 1), - cupynumeric=get(g, "cupynumeric", false), - cuda=get(g, "cuda", false), - check_correctness=get(g, "check_correctness", false), - n_correctness_iter=get(g, "n_correctness_iter", 5), - ) - - specs = BenchmarkSpec[] - for name in declared_order(path) - entries = raw[name] - for e in entries - types = aslist(get(e, "T", "Float32")) - gpus = aslist(e["gpus"]) - cpus = aslist(e["cpus"]) - fusion = aslist(get(e, "fusion", true)) - N = aslist(e["N"]) - M = aslist(get(e, "M", 1)) - - n = sweep_length(name, ["gpus" => gpus, "cpus" => cpus, "N" => N, "M" => M]) - - for T in types, fuse in fusion, i in 1:n - push!( - specs, - BenchmarkSpec( - name, - T, - sweep_value(gpus, i), - sweep_value(cpus, i), - parse_fusion(fuse), - [sweep_value(N, i), sweep_value(M, i)], - ), - ) - end - end - end - - return global_settings, specs -end diff --git a/benchmark/src/single.jl b/benchmark/src/single.jl deleted file mode 100644 index 56ecd171b..000000000 --- a/benchmark/src/single.jl +++ /dev/null @@ -1,77 +0,0 @@ -# single.jl: worker that runs exactly one benchmark under one backend. Launched by -# run_benchmark.sh (dispatched from run.jl), which sets LEGATE_CONFIG before julia starts. -# Args: -# [check_correctness] [n_correctness_iter] -# backend is "cunumeric" or "cudajl"; run.jl launches one worker per backend. -# run.jl sets the compile-time fusion pref before launch; we read it back to label results. - -using cuNumeric -using CUDACore -using LinearAlgebra - -include("core.jl") -const BENCHMARK_DIR = joinpath(@__DIR__, "benchmarks") -include.(filter(contains(r".jl$"), readdir(BENCHMARK_DIR; join=true))) - -# Resolve a TOML type string like "Float32" to the actual Julia type. -parse_type(s) = getfield(Base, Symbol(s))::DataType - -# mod runs the kernels; label tags stdout; save_as names the results CSV. -const BACKENDS = Dict( - "cunumeric" => (mod=cuNumeric, label="cuNumeric", save_as="cunumeric"), - "cudajl" => (mod=CUDACore, label="CUDA.jl", save_as="CUDA.jl"), -) - -function run_single( - gpus, name, T_str, N, M, n_iter, n_warmup, n_trial, backend; - check_correctness=false, n_correctness_iter=5, -) - haskey(BENCHMARKS, name) || error( - "No benchmark registered for '$(name)'. Known: $(join(sort(collect(keys(BENCHMARKS))), ", "))" - ) - haskey(BACKENDS, backend) || error( - "Unknown backend '$(backend)'. Known: $(join(sort(collect(keys(BACKENDS))), ", "))" - ) - bk = BACKENDS[backend] - - # unfused cuNumeric runs land in their own CSV so they stay a distinct series - fused = cuNumeric.FUSE_BROADCAST_EXPRS - save_as = fused ? bk.save_as : "$(bk.save_as)_nofusion" - label = fused ? bk.label : "$(bk.label) (no fusion)" - - T = parse_type(T_str) - b = build_benchmark(BENCHMARKS[name], T, N, M) - gs = GlobalSettings(; - n_warmup=n_warmup, - n_iter=n_iter, - n_trial=n_trial, - check_correctness=check_correctness, - n_correctness_iter=n_correctness_iter, - ) - - println( - "[$(label)] $(name) benchmark ($(T)) on $(N)x$(M) for $(n_iter) " * - "iterations ($(n_warmup) warmup) x $(n_trial) trials", - ) - br = run_benchmark(b, gs; mod=bk.mod) - @printf("[%s] Mean Run Time: %.5f ± %.5f ms\n", label, mean(br.times_ms), _std(br.times_ms)) - @printf("[%s] FLOPS: %.5f ± %.5f GFLOPS\n", label, mean(br.gflops), _std(br.gflops)) - println("[$(label)] Correctness: $(br.correctness)") - return save_result(br, gpus; mod=save_as) -end - -gpus = parse(Int, ARGS[1]) -bench_name = ARGS[2] -T_str = ARGS[3] -N = parse(Int, ARGS[4]) -M = parse(Int, ARGS[5]) -n_iter = parse(Int, ARGS[6]) -n_warmup = parse(Int, ARGS[7]) -n_trial = parse(Int, ARGS[8]) -backend = ARGS[9] -check_correctness = length(ARGS) >= 10 ? parse(Bool, ARGS[10]) : false -n_correctness_iter = length(ARGS) >= 11 ? parse(Int, ARGS[11]) : 5 -run_single( - gpus, bench_name, T_str, N, M, n_iter, n_warmup, n_trial, backend; - check_correctness=check_correctness, n_correctness_iter=n_correctness_iter, -) diff --git a/benchmark/src_py/benchmarks/__init__.py b/benchmark/src_py/benchmarks/__init__.py deleted file mode 100644 index 2eee44770..000000000 --- a/benchmark/src_py/benchmarks/__init__.py +++ /dev/null @@ -1,8 +0,0 @@ -import importlib -import pkgutil - -from core import BENCHMARKS - -# Import each module so it self-registers into BENCHMARKS. -for _info in pkgutil.iter_modules(__path__): - importlib.import_module(f"{__name__}.{_info.name}") diff --git a/benchmark/src_py/benchmarks/gemm.py b/benchmark/src_py/benchmarks/gemm.py deleted file mode 100644 index b5d1a4b3d..000000000 --- a/benchmark/src_py/benchmarks/gemm.py +++ /dev/null @@ -1,29 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark - - -class GEMM: - name = "gemm" - - def __init__(self, T, N, M): - self.T, self.N, self.M = T, N, M - - def dims(self): - return self.N, self.M - - def total_flops(self): - return self.N * self.N * (2 * self.M - 1) - - def initialize(self): - A = np.random.rand(self.N, self.M).astype(self.T) - B = np.random.rand(self.M, self.N).astype(self.T) - C = np.zeros((self.N, self.N), dtype=self.T) - return (C, A, B) - - def run(self, state): - C, A, B = state - np.matmul(A, B, out=C) - - -register_benchmark("gemm", GEMM) diff --git a/benchmark/src_py/benchmarks/grayscott.py b/benchmark/src_py/benchmarks/grayscott.py deleted file mode 100644 index a1a89e739..000000000 --- a/benchmark/src_py/benchmarks/grayscott.py +++ /dev/null @@ -1,71 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark - - -class GrayScott: - name = "grayscott" - - # dt = dx/5; c_u, c_v, f, k as in grayscott.jl's GSParams defaults. - def __init__(self, T, N, M, dx=1.0, c_u=1.0, c_v=0.3, f=0.03, k=0.06): - self.T, self.N, self.M = T, N, M - self.dx = T(dx) - self.dt = T(dx / 5) - self.c_u, self.c_v, self.f, self.k = T(c_u), T(c_v), T(f), T(k) - - def dims(self): - return self.N, self.M - - def total_flops(self): - return self.N * self.M - - def initialize(self): - d = (self.N, self.M) - u = np.ones(d, dtype=self.T) - v = np.zeros(d, dtype=self.T) - u_new = np.zeros(d, dtype=self.T) - v_new = np.zeros(d, dtype=self.T) - - seed = min(150, self.N, self.M) - u[:seed, :seed] = np.random.rand(seed, seed).astype(self.T) - v[:seed, :seed] = np.random.rand(seed, seed).astype(self.T) - # mutable list so run() can swap buffers in place - return [u, v, u_new, v_new] - - def run(self, state): - u, v, u_new, v_new = state - ui = u[1:-1, 1:-1] - vi = v[1:-1, 1:-1] - - F_u = (-ui * (vi * vi)) + self.f * (1 - ui) - F_v = (ui * (vi * vi)) - (self.f + self.k) * vi - - dx2 = self.dx * self.dx - u_lap = ( - (u[2:, 1:-1] - 2 * ui + u[:-2, 1:-1]) / dx2 - + (u[1:-1, 2:] - 2 * ui + u[1:-1, :-2]) / dx2 - ) - v_lap = ( - (v[2:, 1:-1] - 2 * vi + v[:-2, 1:-1]) / dx2 - + (v[1:-1, 2:] - 2 * vi + v[1:-1, :-2]) / dx2 - ) - - u_new[1:-1, 1:-1] = (self.c_u * u_lap + F_u) * self.dt + ui - v_new[1:-1, 1:-1] = (self.c_v * v_lap + F_v) * self.dt + vi - - # periodic boundary conditions - u_new[:, 0] = u[:, -2] - u_new[:, -1] = u[:, 1] - u_new[0, :] = u[-2, :] - u_new[-1, :] = u[1, :] - v_new[:, 0] = v[:, -2] - v_new[:, -1] = v[:, 1] - v_new[0, :] = v[-2, :] - v_new[-1, :] = v[1, :] - - # swap references rather than copy - state[0], state[2] = u_new, u - state[1], state[3] = v_new, v - - -register_benchmark("grayscott_baseline", GrayScott) diff --git a/benchmark/src_py/benchmarks/montecarlo.py b/benchmark/src_py/benchmarks/montecarlo.py deleted file mode 100644 index 370fc7b98..000000000 --- a/benchmark/src_py/benchmarks/montecarlo.py +++ /dev/null @@ -1,28 +0,0 @@ -import cupynumeric as np - -from core import register_benchmark - - -class MonteCarlo: - name = "montecarlo" - - def __init__(self, T, N, M): - self.T = T - self.n_samples = N - - def dims(self): - return self.n_samples, 1 - - def total_flops(self): - return self.n_samples - - def initialize(self): - x = (self.T(10) * np.random.rand(self.n_samples)).astype(self.T) - return (x,) - - def run(self, state): - (x,) = state - return (self.T(10) / self.n_samples) * np.sum(np.exp(-(x * x))) - - -register_benchmark("montecarlo", MonteCarlo) diff --git a/benchmark/src_py/core.py b/benchmark/src_py/core.py deleted file mode 100644 index 1632e5999..000000000 --- a/benchmark/src_py/core.py +++ /dev/null @@ -1,57 +0,0 @@ -import os -import math - -import cupynumeric as np -from legate.timing import time # blocks on preceding legate ops; returns microseconds - -MOD = "cupynumeric" -RESULTS_DIR = os.path.join(os.path.dirname(__file__), "..", "results") - -DTYPES = {"Float32": np.float32, "Float64": np.float64} - - -def parse_type(s): - if s not in DTYPES: - raise ValueError(f"Unsupported type '{s}'. Known: {', '.join(DTYPES)}") - return DTYPES[s] - - -BENCHMARKS = {} - - -def register_benchmark(key, cls): - BENCHMARKS[key] = cls - - -def trial(bench, n_warmup, n_iter): - state = bench.initialize() - start = None - for idx in range(n_warmup + n_iter): - if idx == n_warmup: - start = time() - bench.run(state) - total_us = time() - start - - mean_time_ms = total_us / (n_iter * 1e3) - gflops = bench.total_flops() / (mean_time_ms * 1e6) - return mean_time_ms, gflops - - -def _mean(x): - return sum(x) / len(x) - - -def _std(x): - if len(x) < 2: - return 0.0 - m = _mean(x) - return math.sqrt(sum((v - m) ** 2 for v in x) / (len(x) - 1)) - - -def save_result(name, dims, gpus, times_ms, gflops): - os.makedirs(RESULTS_DIR, exist_ok=True) - N, M = dims - path = os.path.join(RESULTS_DIR, f"{name}_{MOD}.csv") - with open(path, "a") as io: - for i, (t, g) in enumerate(zip(times_ms, gflops), start=1): - io.write(f"{MOD},{gpus},{N},{M},{i},{t:.6f},{g:.6f},skipped\n") diff --git a/benchmark/src_py/single.py b/benchmark/src_py/single.py deleted file mode 100644 index 005cda31b..000000000 --- a/benchmark/src_py/single.py +++ /dev/null @@ -1,48 +0,0 @@ -# cupynumeric worker, run by run_benchmark.sh which sets LEGATE_CONFIG first. -# Args: -import os -import sys - -# Make `core` and the `benchmarks` package importable when run as a script. -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) - -from core import MOD, parse_type, trial, save_result, _mean, _std -from benchmarks import BENCHMARKS # import populates BENCHMARKS - - -def main(): - gpus = int(sys.argv[1]) - name = sys.argv[2] - T_str = sys.argv[3] - N = int(sys.argv[4]) - M = int(sys.argv[5]) - n_iter = int(sys.argv[6]) - n_warmup = int(sys.argv[7]) - n_trial = int(sys.argv[8]) - - if name not in BENCHMARKS: - raise ValueError( - f"No benchmark registered for '{name}'. Known: {', '.join(sorted(BENCHMARKS))}" - ) - T = parse_type(T_str) - bench = BENCHMARKS[name](T, N, M) - - print( - f"[{MOD}] {name} benchmark ({T_str}) on {N}x{M} for {n_iter} " - f"iterations ({n_warmup} warmup) x {n_trial} trials" - ) - - times_ms, gflops = [], [] - for _ in range(n_trial): - t, g = trial(bench, n_warmup, n_iter) - times_ms.append(t) - gflops.append(g) - - print(f"[{MOD}] Mean Run Time: {_mean(times_ms):.5f} ± {_std(times_ms):.5f} ms") - print(f"[{MOD}] FLOPS: {_mean(gflops):.5f} ± {_std(gflops):.5f} GFLOPS") - - save_result(bench.name, bench.dims(), gpus, times_ms, gflops) - - -if __name__ == "__main__": - main() diff --git a/deps/build.jl b/deps/build.jl index efcc23683..e9e965557 100644 --- a/deps/build.jl +++ b/deps/build.jl @@ -19,6 +19,7 @@ using Pkg using Preferences +using Libdl: dlext # The build only needs Legate's paths/tooling, not a running runtime. # Setting this env prevents a segfault on Julia 1.12 @@ -35,6 +36,7 @@ using OpenBLAS32_jll: OpenBLAS32_jll const BuildTools = Legate.BuildTools include("version.jl") +include("cxxwrap.jl") function build_cpp_wrapper( repo_root, cupynumeric_loc, legate_loc, blas_loc, install_root; @@ -42,7 +44,7 @@ function build_cpp_wrapper( ) @info "libcunumeric_jl_wrapper: Building C++ Wrapper Library" isdir(install_root) && (rm(install_root; recursive=true); mkdir(install_root)) - bld_command = `$(joinpath(repo_root, "scripts/build_cpp_wrapper.sh")) $repo_root $cupynumeric_loc $legate_loc $blas_loc $install_root 8` + bld_command = `$(joinpath(repo_root, "scripts/build_cpp_wrapper.sh")) $repo_root $cupynumeric_loc $legate_loc $blas_loc $install_root $(Threads.nthreads())` return BuildTools.run_build_wrapper_script( repo_root, bld_command; cuda_root, cuda_enabled, log_dir=@__DIR__ ) @@ -60,15 +62,19 @@ function build_deps(pkg_root, cupynumeric_root, blas_root; cuda_root=nothing, cu ) end - BuildTools.build_jlcxxwrap( + ensure_cxxwrap( pkg_root, get_cupynumeric_version(cupynumeric_root); - log_dir=@__DIR__, is_compatible=is_supported_version, + log_dir=@__DIR__, ) build_cpp_wrapper( pkg_root, cupynumeric_root, up_dir(legate_lib), blas_root, install_lib; cuda_root, cuda_enabled, ) + for name in ("cunumeric_jl_wrapper", "cunumeric_c_wrapper") + library = joinpath(install_lib, "lib", "lib$name.$dlext") + isfile(library) || error("Wrapper build did not produce $library; see deps/cpp_wrapper.err. JLL override was not updated.") + end return BuildTools.set_jll_artifact_override(:cunumeric_jl_wrapper_jll, install_lib) end diff --git a/deps/cxxwrap.jl b/deps/cxxwrap.jl new file mode 100644 index 000000000..695e37a17 --- /dev/null +++ b/deps/cxxwrap.jl @@ -0,0 +1,54 @@ +# A version marker alone cannot validate a build-tree CMake export: its headers +# can live in a checkout that has since been deleted or moved. +cxxwrap_julia_identity() = string( + VERSION, '\n', realpath(joinpath(Sys.BINDIR, Base.julia_exename())), +) + +function cxxwrap_usable(override_dir; log_dir) + mktempdir() do build_dir + probe = joinpath(@__DIR__, "cxxwrap_probe") + julia = joinpath(Sys.BINDIR, Base.julia_exename()) + cmd = `cmake -S $probe -B $build_dir -DJlCxx_DIR=$override_dir -DJulia_EXECUTABLE=$julia` + open(joinpath(log_dir, "libcxxwrap_check.log"), "w") do io + return success(pipeline(cmd; stdout=io, stderr=io)) + end + end +end + +function ensure_cxxwrap(repo_root, package_version; log_dir) + override_dir = joinpath(DEPOT_PATH[1], "dev", "libcxxwrap_julia_jll", "override") + version_path = joinpath(override_dir, "LEGATE_INSTALL.txt") + julia_path = joinpath(override_dir, "JULIA_INSTALL.txt") + julia_identity = cxxwrap_julia_identity() + cached = isfile(version_path) ? tryparse(VersionNumber, strip(read(version_path, String))) : nothing + # LEGATE_INSTALL.txt records the provider version, not Julia's ABI. Legacy + # builds without a Julia identity must be rebuilt once, even if paths exist. + same_julia = isfile(julia_path) && read(julia_path, String) == julia_identity + if !isnothing(cached) && is_supported_version(cached) && + same_julia && cxxwrap_usable(override_dir; log_dir) + @info "libcxxwrap: Up to date (provider $cached, Julia $VERSION)" + return nothing + end + + @info "libcxxwrap: Missing, incompatible, or stale build. Rebuilding..." + # Invalidate before attempting the build, and propagate failures. The shared + # run_sh helper catches errors, so it cannot establish a successful build. + rm(version_path; force=true) + rm(julia_path; force=true) + script = joinpath(repo_root, "scripts", "install_cxxwrap.sh") + cmd = addenv(`bash $script $repo_root`, "JULIA" => joinpath(Sys.BINDIR, Base.julia_exename())) + open(joinpath(log_dir, "libcxxwrap.log"), "w") do out + open(joinpath(log_dir, "libcxxwrap.err"), "w") do err + try + run(pipeline(cmd; stdout=out, stderr=err)) + catch + error("libcxxwrap build failed; see $(joinpath(log_dir, "libcxxwrap.err"))") + end + end + end + cxxwrap_usable(override_dir; log_dir) || + error("libcxxwrap build produced an unusable CMake package; see $(joinpath(log_dir, "libcxxwrap_check.log"))") + write(version_path, string(package_version)) + write(julia_path, julia_identity) + return nothing +end diff --git a/deps/cxxwrap_probe/CMakeLists.txt b/deps/cxxwrap_probe/CMakeLists.txt new file mode 100644 index 000000000..054f77bec --- /dev/null +++ b/deps/cxxwrap_probe/CMakeLists.txt @@ -0,0 +1,42 @@ +cmake_minimum_required(VERSION 3.16) +project(CheckCxxWrap LANGUAGES NONE) + +# Inspect exactly the override the wrapper will use, without falling back to +# another installation or requiring a compiler/CUDA for this check. +find_package(JlCxx REQUIRED CONFIG PATHS "${JlCxx_DIR}" NO_DEFAULT_PATH) +foreach(target JlCxx::cxxwrap_julia JlCxx::cxxwrap_julia_stl) + if(NOT TARGET ${target}) + message(FATAL_ERROR "Missing target ${target}") + endif() + get_target_property(includes ${target} INTERFACE_INCLUDE_DIRECTORIES) + if(NOT includes) + message(FATAL_ERROR "Missing include directories for ${target}") + endif() + foreach(path IN LISTS includes) + if(path STREQUAL "") + continue() + endif() + if(NOT IS_DIRECTORY "${path}") + message(FATAL_ERROR "Missing include directory: ${path}") + endif() + endforeach() + get_target_property(configs ${target} IMPORTED_CONFIGURATIONS) + set(properties IMPORTED_LOCATION) + foreach(config IN LISTS configs) + string(TOUPPER "${config}" config) + list(APPEND properties "IMPORTED_LOCATION_${config}") + endforeach() + set(found_location FALSE) + foreach(property IN LISTS properties) + get_target_property(path ${target} ${property}) + if(path) + set(found_location TRUE) + if(NOT EXISTS "${path}") + message(FATAL_ERROR "Missing library: ${path}") + endif() + endif() + endforeach() + if(NOT found_location) + message(FATAL_ERROR "Missing library location for ${target}") + endif() +endforeach() diff --git a/Dockerfile b/docker/Dockerfile similarity index 66% rename from Dockerfile rename to docker/Dockerfile index 00f919780..56cdebdc1 100644 --- a/Dockerfile +++ b/docker/Dockerfile @@ -1,12 +1,10 @@ -ARG JULIA_VERSION=1.11 +ARG JULIA_VERSION=1.13 FROM julia:${JULIA_VERSION} ARG CUDA_MAJOR=13 ARG CUDA_MINOR=0 ENV CUDA_VERSION_MAJOR_MINOR="${CUDA_MAJOR}.${CUDA_MINOR}" -ARG REF=main -ENV REF=${REF} # using bash SHELL ["/bin/bash", "-c"] ENV DEBIAN_FRONTEND=noninteractive @@ -27,11 +25,11 @@ ENV JULIA_NUM_THREADS=auto ARG CUNUMERIC_VERSION=25.10.00 ARG PACKAGE_SPEC_CUDA=CUDA LABEL org.opencontainers.image.authors="David Krasowska , Ethan Meitz " \ - org.opencontainers.image.description="A cuNumeric.jl container with CUDA ${CUDA_VERSION_MAJOR_MINOR}, Julia ${JULIA_VERSION}, and cuNumeric ${CUNUMERIC_VERSION}" \ - org.opencontainers.image.title="cuNumeric.jl" \ + org.opencontainers.image.description="A cuNumeric.jl container with CUDA ${CUDA_VERSION_MAJOR_MINOR}, Julia ${JULIA_VERSION}, and cuNumeric ${CUNUMERIC_VERSION}" \ + org.opencontainers.image.title="cuNumeric.jl" \ # org.opencontainers.image.url="https://juliagpu.org/cuda/" \ - org.opencontainers.image.source="https://github.com/JuliaLegate/cuNumeric.jl" \ - org.opencontainers.image.licenses="MIT" + org.opencontainers.image.source="https://github.com/JuliaLegate/cuNumeric.jl" \ + org.opencontainers.image.licenses="MIT" COPY scripts/test_container.sh /workspace/test_container.sh RUN chmod +x /workspace/test_container.sh @@ -39,8 +37,8 @@ RUN cat /workspace/test_container.sh # # system-wide packages RUN apt-get update && apt-get install -y \ - wget curl git build-essential && \ - rm -rf /var/lib/apt/lists/* + wget curl git build-essential \ + && rm -rf /var/lib/apt/lists/* ENV JULIA_DEPOT_PATH=/usr/local/share/julia @@ -48,26 +46,38 @@ ENV PATH="/usr/local/.juliaup/bin:/usr/local/bin:$PATH" # install CUDA.jl itself. RUN julia --color=yes -e 'using Pkg; Pkg.add("CUDA"); using CUDA; CUDA.set_runtime_version!(VersionNumber(ENV["CUDA_VERSION_MAJOR_MINOR"]))' -RUN julia -e 'using Pkg; Pkg.add(name = "CUDA_Driver_jll", version = "13.0.0"); Pkg.add("CUDA_Runtime_jll")' +RUN julia -e 'using Pkg; Pkg.add(["CUDA_Driver_jll", "CUDA_Runtime_jll"])' RUN echo "export LD_LIBRARY_PATH=\$(julia -e 'print(Sys.BINDIR * \"/../lib\")'):\$(julia -e 'using CUDA_Driver_jll; print(joinpath(CUDA_Driver_jll.artifact_dir, \"lib\"))'):\$(julia -e 'using CUDA_Runtime_jll; print(joinpath(CUDA_Runtime_jll.artifact_dir, \"lib\"))'):\$LD_LIBRARY_PATH" >> /etc/.env RUN chmod +x /etc/.env RUN cat /etc/.env RUN echo "Install Legate and cuNumeric.jl" -# Install Legate.jl and cuNumeric.jl RUN source /etc/.env && source /etc/.env && julia --color=yes -e ' \ using Pkg; \ Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/Legate.jl", rev = "main")) \ ' +# The workflow checks out the requested revision before building. Copy only the +# package source so benchmark-only changes leave this image unchanged. +COPY Project.toml /opt/cuNumeric.jl/Project.toml +COPY deps /opt/cuNumeric.jl/deps +COPY ext /opt/cuNumeric.jl/ext +COPY lib /opt/cuNumeric.jl/lib +COPY scripts /opt/cuNumeric.jl/scripts +COPY src /opt/cuNumeric.jl/src +COPY test /opt/cuNumeric.jl/test + RUN source /etc/.env && source /etc/.env && julia --color=yes -e ' \ using Pkg; \ - Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = ENV["REF"])) \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl/lib/CNPreferences")); \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl")); \ + Pkg.precompile() \ ' -RUN #= remove useless stuff =# \ - cd /usr/local/share/julia && \ - rm -rf registries scratchspaces logs +RUN \ + #= remove useless stuff =# \ + cd /usr/local/share/julia \ + && rm -rf registries scratchspaces logs # user environment @@ -98,10 +108,19 @@ end pushfirst!(DEPOT_PATH, "/depot") EOF -RUN apt-get clean && \ - rm -rf /var/lib/apt/lists/* +RUN apt-get clean \ + && rm -rf /var/lib/apt/lists/* ENV LEGATE_AUTO_CONFIG=1 -ENTRYPOINT source /etc/.env && exec /bin/bash +# Include the complete checkout and history after package installation so +# documentation and history changes do not invalidate the expensive layers. +COPY . /opt/cuNumeric.jl/ +RUN git -C /opt/cuNumeric.jl rev-parse --verify HEAD + +COPY docker/profile.sh /etc/profile.d/cunumeric.sh +COPY docker/entrypoint.sh /usr/local/bin/cunumeric-entrypoint +RUN chmod +x /usr/local/bin/cunumeric-entrypoint +ENTRYPOINT ["/usr/local/bin/cunumeric-entrypoint"] +CMD ["/bin/bash"] WORKDIR /workspace diff --git a/docker/Dockerfile.developer b/docker/Dockerfile.developer new file mode 100644 index 000000000..0a4f08525 --- /dev/null +++ b/docker/Dockerfile.developer @@ -0,0 +1,134 @@ +ARG JULIA_VERSION=1.13 +FROM julia:${JULIA_VERSION} + +ARG CUDA_MAJOR=13 +ARG CUDA_MINOR=0 +ENV CUDA_VERSION_MAJOR_MINOR="${CUDA_MAJOR}.${CUDA_MINOR}" + +# using bash +SHELL ["/bin/bash", "-c"] +ENV DEBIAN_FRONTEND=noninteractive + +# force turn off legate auto config for precompilation. +ENV LEGATE_AUTO_CONFIG=0 + +# much of the CUDA.jl setup is from Tim Besard +# CUDA.jl Dockerfile https://github.com/JuliaGPU/CUDA.jl/blob/master/Dockerfile +# Thank you Tim for the reccomendation. + +ARG JULIA_CPU_TARGET=native +ENV JULIA_CPU_TARGET=${JULIA_CPU_TARGET} + +ENV JULIA_NUM_THREADS=auto + +ARG CUNUMERIC_VERSION=25.10.00 +ARG PACKAGE_SPEC_CUDA=CUDA +LABEL org.opencontainers.image.authors="David Krasowska , Ethan Meitz " \ + org.opencontainers.image.description="A cuNumeric.jl developer container with CUDA ${CUDA_VERSION_MAJOR_MINOR}, Julia ${JULIA_VERSION}, and cuNumeric ${CUNUMERIC_VERSION}" \ + org.opencontainers.image.title="cuNumeric.jl" \ + org.opencontainers.image.source="https://github.com/JuliaLegate/cuNumeric.jl" \ + org.opencontainers.image.licenses="MIT" + +COPY scripts/test_container.sh /workspace/test_container.sh +RUN chmod +x /workspace/test_container.sh +RUN cat /workspace/test_container.sh + +# system-wide packages +RUN apt-get update && apt-get install -y \ + wget curl git build-essential \ + && rm -rf /var/lib/apt/lists/* + +# Developer mode needs the same CMake version used by developer CI. +ARG CMAKE_VERSION=3.30.7 +RUN wget --no-verbose \ + "https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-linux-x86_64.sh" \ + -O /tmp/cmake-installer.sh \ + && sh /tmp/cmake-installer.sh --skip-license --prefix=/usr/local \ + && rm /tmp/cmake-installer.sh + +ENV JULIA_DEPOT_PATH=/usr/local/share/julia +ENV PATH="/usr/local/.juliaup/bin:/usr/local/bin:$PATH" + +# install CUDA.jl itself. +RUN julia --color=yes -e 'using Pkg; Pkg.add("CUDA"); using CUDA; CUDA.set_runtime_version!(VersionNumber(ENV["CUDA_VERSION_MAJOR_MINOR"]))' +RUN julia -e 'using Pkg; Pkg.add(["CUDA_Driver_jll", "CUDA_Runtime_jll"])' +RUN echo "export LD_LIBRARY_PATH=\$(julia -e 'print(Sys.BINDIR * \"/../lib\")'):\$(julia -e 'using CUDA_Driver_jll; print(joinpath(CUDA_Driver_jll.artifact_dir, \"lib\"))'):\$(julia -e 'using CUDA_Runtime_jll; print(joinpath(CUDA_Runtime_jll.artifact_dir, \"lib\"))'):\$LD_LIBRARY_PATH" >> /etc/.env +RUN chmod +x /etc/.env +RUN cat /etc/.env + +RUN echo "Install Legate and cuNumeric.jl with the in-tree wrapper" +RUN source /etc/.env && julia --color=yes -e ' \ + using Pkg; \ + Pkg.add(PackageSpec(url = "https://github.com/JuliaLegate/Legate.jl", rev = "main")) \ +' + +# The workflow checks out the requested ref before starting the Docker build, +# so this source tree is the exact package revision represented by the image. +COPY Project.toml /opt/cuNumeric.jl/Project.toml +COPY deps /opt/cuNumeric.jl/deps +COPY ext /opt/cuNumeric.jl/ext +COPY lib /opt/cuNumeric.jl/lib +COPY scripts /opt/cuNumeric.jl/scripts +COPY src /opt/cuNumeric.jl/src +COPY test /opt/cuNumeric.jl/test + +RUN source /etc/.env && julia --color=yes -e ' \ + using Pkg; \ + ENV["JULIA_PKG_PRECOMPILE_AUTO"] = "0"; \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl/lib/CNPreferences")); \ + Pkg.develop(PackageSpec(path = "/opt/cuNumeric.jl")); \ + using CNPreferences; \ + CNPreferences.use_developer_mode(); \ + Pkg.build("cuNumeric"); \ + Pkg.precompile() \ +' +RUN \ + #= remove useless stuff =# \ + cd /usr/local/share/julia \ + && rm -rf registries scratchspaces logs + +# user environment + +# we hard-code the primary depot regardless of the actual user, i.e., we do not let it +# default to `$HOME/.julia`. this is for compatibility with `docker run --user`, in which +# case there might not be a (writable) home directory. + +RUN mkdir -m 0777 /depot +# we add the user environment from a start-up script +# so that the user can mount `/depot` for persistency +COPY < "examples/initialization.md", "Monte-Carlo" => "examples/montecarlo.md", "Gray-Scott" => "examples/grayscott.md", + "Dynamic Mode Decomposition" => "examples/dmd.md", + "Periodic Poisson (FFT)" => "examples/poisson_fft.md", + "Conjugate Gradient" => "examples/cg.md", + "Tensor Network Contraction" => "examples/tensor_network.md", ], "Performance Tips" => [ "Kernel Fusion" => "perf/kernel_fusion.md", - "Reduce Allocations" => "perf/reduce_allocations.md", + "@accelerate" => "perf/reduce_allocations.md", "Patterns to Avoid" => "perf/patterns_to_avoid.md", ], "Configuration" => [ @@ -60,9 +64,14 @@ makedocs(; ], "Public API" => [ "Initialization" => "api_initialization.md", + "Random" => "api_random.md", "Unary Operations" => "api_unary.md", + "Mapped Reductions" => "api_mapreduce.md", + "Device Scalars" => "api_cnscalar.md", "Binary Operations" => "api_binary.md", "Linear Algebra" => "linalg.md", + "Tensor Contractions" => "api_tensor.md", + "FFT" => "fft.md", "HDF5" => "api_hdf5.md", "NDArray Reference" => "api.md", "CUDA.jl Tasking" => "api_cuda.md", diff --git a/docs/src/api.md b/docs/src/api.md index 82345f271..b998aaf10 100644 --- a/docs/src/api.md +++ b/docs/src/api.md @@ -1,9 +1,12 @@ # NDArray Reference -Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). +Indexing, reshaping, reductions, comparisons, memory helpers, lifetime macros, and related utilities. For constructors (`zeros`, `ones`, `rand`, …) see [Initialization](./api_initialization.md). For RNG engines and `default_rng`, see [Random](./api_random.md). For `fft` / `ifft` / `fft!` / `ifft!` and `batched_fft`, see [FFT](./fft.md). There is no `plan_fft`: cupynumeric does not expose a cuFFT handle. + +For `mapreduce` and the mapped forms of `sum`, `prod`, `minimum`, and `maximum`, +see [Mapped Reductions](./api_mapreduce.md). ```@autodocs Modules = [cuNumeric] -Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl"] -Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!)) +Pages = ["ndarray/ndarray.jl", "ndarray/linalg.jl", "ndarray/batched_linalg.jl", "ndarray/sort.jl", "cuNumeric.jl", "warnings.jl", "util.jl", "memory.jl", "scoping/scoping.jl", "scoping/accelerate.jl"] +Filter = t -> !(t isa Function && nameof(t) in (:zeros, :ones, :fill, :trues, :falses, :eye, :rand, :rand!, :randn, :randn!, :randexp, :randexp!, :default_rng, :random, :random!)) ``` diff --git a/docs/src/api_binary.md b/docs/src/api_binary.md index 7285ae15e..ae6d72e82 100644 --- a/docs/src/api_binary.md +++ b/docs/src/api_binary.md @@ -6,7 +6,13 @@ The following binary operations are supported and can be applied elementwise to pairs of `NDArray` values: -- `+`, `-`, `*`, `/`, `^`, `<`, `<=`, `>`, `>=`, `==`, `!=`, `atan`, `hypot`, `max`, `min`, `lcm`, `gcd` +- `+`, `-`, `*`, `/`, `^`, `%`, `<`, `<=`, `>`, `>=`, `==`, `!=`, `&`, `|`, `⊻`, `<<`, `>>`, `atan`, `hypot`, `max`, `min`, `lcm`, `gcd`, `fld`, `mod`, `rem`, `copysign` + +## Differences from Base Julia + +- `div` / `÷` are not provided. cuPyNumeric floor-divide matches Julia `fld` (toward `-Inf`), not truncated `div`. For example `-7 ÷ 2` is `-3` in Julia and `-4` for `fld`. +- `&`, `|`, `⊻` are integer and `Bool` bitwise ops. `<<` and `>>` are integers excluding `Bool`. +- `copysign` is float-only. ```@autodocs Modules = [cuNumeric] diff --git a/docs/src/api_cnscalar.md b/docs/src/api_cnscalar.md new file mode 100644 index 000000000..6ffb8d7ef --- /dev/null +++ b/docs/src/api_cnscalar.md @@ -0,0 +1,97 @@ +# Device scalars and automatic fetching + +Device scalars let you use reduction results in further calculations without +first copying their values to the host. They also participate in Julia's numeric +type hierarchy, so they can be passed to compatible numeric methods and stored +in structs with abstract numeric type constraints. Extract a host value explicitly +when you need one, or enable scoped automatic fetching for comparisons and conversions. + +Full reductions such as `sum(A)`, `dot(x, y)`, and `norm(x)` return a +`CNScalar`. Concrete wrappers mirror the numeric category of their storage: +`CNFloat <: AbstractFloat`, `CNInt <: Signed`, `CNUInt <: Unsigned`, +`CNBool <: Integer`, and `CNComplex <: Number`. `CNReal` is the union of the +four real wrapper families, and `CNScalar` also includes `CNComplex`. +Each owns a reference to a 0D NDArray, with no copy or host +extraction when it is wrapped. Reductions that retain dimensions still return +NDArrays. Explicit 0D array construction and array broadcasts remain arrays. + +`DeviceScalar{T}` is the shared dispatch alias for a raw `NDArray{T,0}` or a +`CNScalar{T}` wrapper. Use `DeviceScalar` when either representation is accepted: + +```julia +twice(x::DeviceScalar) = x .* 2 +``` + +Numeric data arguments such as search needles, fill values, and +linear algebra coefficients use this shared storage path. `Ref(s)` in broadcast +also preserves device storage for either representation. + +Host control parameters, including a norm's `p`, random distribution parameters, +and `isapprox` tolerances, require allowautofetch permission for either representation. +The `searchsorted` convenience function explicitly extracts its result indices +to construct a Julia range; +use `searchsortedfirst`/`searchsortedlast` to retain device results. + +```julia +using cuNumeric, LinearAlgebra +x = cuNumeric.NDArray([1.0, 2.0, 3.0]) +s = sum(x) # CNFloat{Float64}, accepted in <:AbstractFloat fields +t = s^2 / 2 # another device scalar +y = x .* t # backend broadcast; no implicit host extraction +value = fetch(t) # explicit synchronization, always permitted +``` + +`cnscalar(a)` wraps an existing 0D NDArray; `s.value` accesses its backend array +without synchronizing. Use `fetch(s)` for explicit host extraction. +This extends Julia's `Base.fetch`; ordinary Julia values retain their existing +`fetch(x) == x` behavior. Explicit `fetch` does not require `allowautofetch`. +Arithmetic, `min`/`max`, and promotion between numeric types keep results on the +backend. Showing a wrapper uses the backing 0D NDArray's display and extracts +its value, including in the REPL, without requiring allowautofetch permission. End +an expression with `;` to suppress REPL display and that synchronization. +The existing promotion policy still applies. + +Device scalars can also be coefficients in `contract!`, `mul!`, `axpy!`, +`axpby!`, and TensorOperations calls without enabling allowautofetch. + +Value-dependent conversions, such as floating point to integer or complex to +real, require allowautofetch permission even when the target is another CNScalar. +They preserve Julia's `InexactError` checks instead of silently truncating values. + +## Scoped permission + +Implicit host extraction is disabled by default. Comparisons, scalar predicates, +and conversion to supported host numeric types require `allowautofetch` permission: + +```julia +@allowautofetch s > 0 # Julia Bool +allowautofetch() do + Float64(s) # Julia Float64 +end +``` + +The scope returns the body's result and restores the previous permission even +when the body throws. Nested `allowautofetch(false) do ... end` disables extraction +temporarily. `allowautofetch(true)` / `allowautofetch(false)` set the calling task's +permission until changed again. Independent tasks do not inherit permission. +This permission is separate from `allowscalar` and `allowpromotion`. + +The permission also applies inside ordinary functions called within the scope; +the macro does not need to inspect their bodies: + +```julia +function host_residual(x) + r = norm(x) # device scalar + return Float64(r) # permission checked here, inside this function +end + +residual = @allowautofetch host_residual(x) +# host_residual(x) # errors outside the scope +``` + +Arithmetic remains asynchronous even inside an allowautofetch scope. There is no +general fallback that fetches arguments when Julia cannot find a method. +For example, a function accepting only `Float64` still needs `f(Float64(s))`. +Julia also requires an actual Bool in `if`: use `Bool(all(A))` within the scope, +or a comparison that returns a host Bool. Automatic fetching does not change Julia's +dispatch or condition evaluation rules. diff --git a/docs/src/api_cuda.md b/docs/src/api_cuda.md index 6e912cd8a..eedb9806c 100644 --- a/docs/src/api_cuda.md +++ b/docs/src/api_cuda.md @@ -58,7 +58,7 @@ allowscalar() do end ``` -See `examples/custom_cuda.jl` for a more complete example with multiple kernels. +See `examples/custom_cuda.jl` for a runnable two-kernel example. ## API Reference diff --git a/docs/src/api_hdf5.md b/docs/src/api_hdf5.md index c60787935..5ad2aa34b 100644 --- a/docs/src/api_hdf5.md +++ b/docs/src/api_hdf5.md @@ -22,6 +22,13 @@ restored = cuNumeric.h5read("checkpoint.h5", "field"; layout=:row) Synchronize before accessing the file outside the runtime, moving or deleting it, or exiting immediately after the write. +`h5write` removes a leftover empty or truncated `.h5` before launching the +write, which is what otherwise aborts `HDF5CombineVDS`. A valid HDF5 file is +left for Legate to overwrite. The `*_legate_vds` sidecar is left alone: after +a Legate write that directory is the data, and the `.h5` is only an index. +`h5read` rejects a missing path, a directory, or a file that does not start +with the HDF5 signature, instead of opening it in a Legate task. + ## Dataset layout ```julia diff --git a/docs/src/api_initialization.md b/docs/src/api_initialization.md index b6b08ef05..96d62c179 100644 --- a/docs/src/api_initialization.md +++ b/docs/src/api_initialization.md @@ -2,52 +2,105 @@ Constructors for new `NDArray`s. Default floating-point type is `Float32`. -## zeros +## Basic Initialization + +### Uninitialized arrays + +Use `NDArray{T}(undef, dims...)` or `NDArray{T}(undef, dims::Tuple)` when the +next operation writes every element. `similar(A)` and `similar(A, T, dims)` +also return uninitialized arrays. Assign all elements before reading them; +use `cuNumeric.zeros` when the initial zero values are needed. + +```julia +A = NDArray{Float32}(undef, 2, 3) +fill!(A, 1f0) +``` + +### zeros ```@docs cuNumeric.zeros ``` -## ones +### ones ```@docs cuNumeric.ones ``` -## fill +### fill ```@docs cuNumeric.fill ``` -## trues +### trues ```@docs cuNumeric.trues ``` -## falses +### falses ```@docs cuNumeric.falses ``` -## eye +## Special Matrices -```@docs -cuNumeric.eye +### Diagonal + +Construct a `Diagonal` matrix whose elements (diagonal only) are stored in an `NDArray`. +`LinearAlgebra.I` can be used to construct dense identity matrices as well. + +```julia +using LinearAlgebra +using cuNumeric + +D = Diagonal(cuNumeric.ones(Float32, 5)) # preferred for diagonal work +I32 = NDArray{Float32}(I, 5, 5) # dense Float32 identity +Ib = NDArray(I, 5, 5) # Bool identity ``` -## rand +See [Linear Algebra](./linalg.md#diagonal-and-identity) for preferred patterns. + +## Random Numbers + +### rand ```@docs cuNumeric.rand ``` -## rand! +### rand! + +```@docs +Random.rand!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) +``` + +## randn + +```@docs +cuNumeric.randn +``` + +## randn! + +```@docs +Random.randn!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) +``` + +## randexp + +```@docs +cuNumeric.randexp +``` + +## randexp! ```@docs -Random.rand!(::NDArray{Float64}) +Random.randexp!(::NDArray{<:cuNumeric.SUPPORTED_FLOAT_TYPES}) ``` -The backend currently draws `Float64` uniforms. `cuNumeric.rand(Float32, dims...)` converts for you. `rand!` on `NDArray` currently requires `Float64` storage. +See [Random](./api_random.md) for BitGenerators (`XORWOW`, `MRG32k3a`, +`PHILOX4_32_10`), `Generator`, and `default_rng`. diff --git a/docs/src/api_mapreduce.md b/docs/src/api_mapreduce.md new file mode 100644 index 000000000..bdb731265 --- /dev/null +++ b/docs/src/api_mapreduce.md @@ -0,0 +1,53 @@ +# Mapped Reductions + +`mapreduce(f, op, A; dims=:, init=...)` maps and reduces one NDArray without +allocating a mapped copy. Supported operators are `+`, `*`, `min`, and `max`. +Mapped `sum`, `prod`, `minimum`, and `maximum` use the same implementation. + +```julia +A = cuNumeric.ones(Float32, 1024, 512) +energy = sum(abs2, A) # CNScalar +columns = mapreduce(abs2, +, A; dims=1) # 1 × 512 +α = 0.5f0 +distance = sum(x -> abs2(x - α), A) +largest = maximum(abs, A; init=0f0) +``` + +Full reductions return a device-resident CNScalar. Dimensional reductions +keep reduced axes with size one. Duplicate dimensions are ignored; positive +out-of-rank dimensions have no effect. `dims=()` still applies the mapping. + +## Types and initialization + +`sum` and `prod` use Base's integer widening rules; `mapreduce` with `+` or `*` +uses ordinary scalar arithmetic. Implicit widening, including within the mapping, +is subject to `allowpromotion`. + +Supply a neutral scalar `init`; it is applied once. Full empty reductions return +`init` when supplied. Otherwise, Base's empty-input rules apply: dimensional +sums/products return their identities, while empty extrema error except for +special cases such as `maximum(abs2, A)`. Arbitrary mappings may require `init`. + +For full reductions, combining with `init` determines the output type. For +dimensional reductions, `typeof(init)` determines it and must accommodate the +accumulator without narrowing. + +## Current limitations + +- GPU execution only, independent of broadcast-fusion settings. Existing unmapped + reductions retain CPU support. Participating GPUs must support the compilation + target; heterogeneous target selection is unsupported. +- One input NDArray and a type-stable, GPU-compilable mapping with immutable scalar + captures. Captured arrays/pointers, host allocation or side effects, and custom + reducers are unsupported. +- Results may be Bool, the supported integer widths, Float32/Float64, or + ComplexF32/ComplexF64. Complex extrema and ComplexF64 product accumulators + (including those selected by dimensional `init`) are unsupported. +- Floating-point reassociation can change rounding, overflow, and arithmetic + signed zeros. Bitwise reproducibility is not guaranteed. Floating-point extrema + preserve NaN propagation and signed-zero ordering, but not NaN payloads. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/mapreduce.jl"] +``` diff --git a/docs/src/api_preferences.md b/docs/src/api_preferences.md index 12892687a..3d44a580a 100644 --- a/docs/src/api_preferences.md +++ b/docs/src/api_preferences.md @@ -8,10 +8,34 @@ Out of the box (no `LocalPreferences.toml` changes): |---|---| | Binary / build mode | **JLL** prebuilt binaries | | Broadcast fusion | **on** | -| `FUSE_BROADCAST_MIN_OPS` | **2** (single-op broadcasts stay unfused) | +| `FUSE_BROADCAST_MIN_OPS` | **2** (single native operations stay unfused) | | Task scope names | **off** | +| `MIN_SOLVE_MATRIX_SIZE` | **2048** rows | +| `MIN_SOLVE_TILE_SIZE` | **512** | +| `MIN_CHOLESKY_MATRIX_SIZE` | **8192** rows | +| `MIN_CHOLESKY_TILE_SIZE` | **2048** | +| `MIN_QR_MATRIX_SIZE` | **1048576** elements | +| `QR_TILE_SIZE` | **128** | +| `MAX_CHOLESKY_TILES_PER_PROC` | **4** | -Build-mode setup (JLL / conda / developer) is documented under [Build Modes](./install.md). Fusion usage tips live under [Kernel Fusion](./perf/kernel_fusion.md). +Build-mode setup (JLL / conda / developer) is documented under [Build Modes](./install.md). + +## Linear algebra + +`set_linalg!` accepts the constant names above as keywords. All values must be +positive integers; unspecified settings are unchanged. Restart Julia to load +the new constants. `cuNumeric.versioninfo()` shows the effective values. + +```julia +CNPreferences.set_linalg!(; MIN_SOLVE_MATRIX_SIZE=4096, MIN_SOLVE_TILE_SIZE=512) +``` + +See [Distributed solves and factorizations](./linalg.md#distributed-solves-and-factorizations) +for selection rules and tuning details. + +```@docs +CNPreferences.set_linalg! +``` ## Build mode @@ -36,7 +60,7 @@ CNPreferences.set_broadcast_fusion_min_ops!(1) # also fuse single-ops `set_broadcast_fusion_min_ops!` counts `Broadcasted` nodes (ops) in the tree: -- **`2` (default):** fuse multi-op trees such as `y .= @. a * b + c`. Single-ops like `y .= cos.(x)` stay on the unfused C-API path. +- **`2` (default):** fuse multi-op trees such as `y .= @. a * b + c`. Single native operations like `y .= cos.(x)` use the C-API path. Functions without a native implementation are fused on the GPU when their broadcast arguments are eligible. - **`1`:** fuse every eligible expression, including single-ops. Set the preference in one Julia process, then start a fresh process to use it. diff --git a/docs/src/api_random.md b/docs/src/api_random.md new file mode 100644 index 000000000..9d6775710 --- /dev/null +++ b/docs/src/api_random.md @@ -0,0 +1,44 @@ +# Random + +Module-level [`rand`](@ref cuNumeric.rand), [`randn`](@ref cuNumeric.randn), +and [`randexp`](@ref cuNumeric.randexp) are documented under +[Initialization](./api_initialization.md). This page covers the cuPyNumeric RNG +stack: BitGenerators, `Generator`, and `default_rng`. + +Random draws use cuRAND with the [`XORWOW`](@ref cuNumeric.XORWOW) random +number generator by default. We support `Float32` and `Float64` uniforms, +normals, and exponentials (`randexp`), `ComplexF32` / `ComplexF64` uniforms +and normals (independent real/imag parts; no `randexp`), `Bool` coin flips, +and signed `Int16` / `Int32` / `Int64`. There is no native Bool or complex +distribution, so `rand(Bool, …)` draws `Int16` values in `{0,1}` and compares +them to zero, and complex draws two real arrays packed as `re + i*imag`. + +`Random.seed!` is not hooked. Use [`default_rng`](@ref cuNumeric.default_rng) +with an explicit seed (or a specific BitGenerator) when you need a private +stream. Module-level `rand` / `randn` / `randexp` always use a process-global XORWOW +generator. + +## BitGenerators + +```@docs +cuNumeric.BitGenerator +cuNumeric.XORWOW +cuNumeric.MRG32k3a +cuNumeric.PHILOX4_32_10 +``` + +## Generator + +```@docs +cuNumeric.Generator +cuNumeric.default_rng +``` + +```julia +g = cuNumeric.default_rng() # XORWOW, fresh seed +g = cuNumeric.default_rng(1234) # XORWOW, fixed seed +g = cuNumeric.default_rng(cuNumeric.PHILOX4_32_10, 1) +A = cuNumeric.random(g, Float32, (8, 8)) # U[0, 1) +cuNumeric.randn!(g, A) # N(0, 1) +cuNumeric.randexp!(g, A; scale=1) # Exp(scale) +``` diff --git a/docs/src/api_tensor.md b/docs/src/api_tensor.md new file mode 100644 index 000000000..b2b0e603e --- /dev/null +++ b/docs/src/api_tensor.md @@ -0,0 +1,92 @@ +# Tensor Contractions + +We extend [TensorOperations.jl](https://quantumkithub.github.io/TensorOperations.jl/stable/) to work with `NDArray`s. Loading TensorOperations alongside cuNumeric activates the package extension. + +Useful macros if you are new to TensorOperations are +[`@tensor`](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#The-@tensor-macro), +[`@tensoropt`](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@tensoropt), +and +[`@notensor`](https://quantumkithub.github.io/TensorOperations.jl/stable/man/indexnotation/#TensorOperations.@notensor). + +The extension supports additions and permutations, traces, pairwise +contractions, complex conjugation, scale factors, accumulation into an existing +tensor, and multi-step `@tensor` blocks. Prefer `@tensor` (with `opt=true` for +larger networks) over the low-level `contract` / `contract!` API at the bottom +of this page. + +```julia +using cuNumeric +using TensorOperations + +α = randn() # prefer a Julia Number when you already have one +A = cuNumeric.randn(5, 5, 5, 5, 5, 5) +B = cuNumeric.randn(5, 5, 5) +C = cuNumeric.randn(5, 5, 5) +D = cuNumeric.zeros(5, 5, 5) + +@tensor begin + D[a, b, c] = A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + E[a, b, c] := A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] +end +``` + +## Scalars + +Prefer a Julia `Number` for scale factors when possible (i.e., `2.0f0`, `randn()`, …). This allows TensorOperations.jl to perform optimizatoins for special values like zero and one. + +A `CNScalar`, such as `sum(C)`, or a raw 0D `NDArray`, such as a +fully contracted `@tensor` result, is also accepted as a scale. Keep these results in runtime-managed storage when reusing them in tensor contractions. `fetch` retrieves a native Julia scalar and waits for the result, so only fetch +when you truly need a Julia `Number` (i.e. printing of host-side if-else). + +```julia +A = cuNumeric.rand(Float64, 64, 32) +B = cuNumeric.rand(Float64, 32, 16) + +@tensor opt=true C[i, j] := A[i, k] * B[k, j] # NDArray + +α = 2.0 +@tensor D[i, j] := α * C[i, j] # Julia Number, no sync + +@tensor s = conj(C[i, j]) * C[i, j] # 0D NDArray, stays asynchronous +@tensor E[i, j] := s * C[i, j] # reuse 0D as a scale, still no fetch +x = fetch(s) # host Number; blocks +``` + +## Low-level `contract` / `contract!` + +> [!WARNING] +> Prefer `@tensor`. `contract` and `contract!` are the pairwise primitive the +> extension calls. They do not parse einsum strings, pick a contraction order, +> or free intermediates for you. + +Mode labels are not einsum strings: `"ik"` with `"kj"` is a GEMM; `"ijk"` with +`"ikl"` and explicit output `"ijl"` is a batched product. Labels may be ASCII +strings, `Char` tuples, or integers (`1` maps to `'a'`). + +```julia +A = cuNumeric.rand(Float32, 64, 32) +B = cuNumeric.rand(Float32, 32, 16) +C = contract(A, "ik", B, "kj") # allocates, Einstein output order +contract!(similar(C), "ij", A, "ik", B, "kj"; α=2, β=0) + +# batched: keep the shared 'i' on the output +AA = cuNumeric.rand(Float32, 8, 16, 32) +BB = cuNumeric.rand(Float32, 8, 32, 4) +CC = cuNumeric.zeros(Float32, 8, 16, 4) +contract!(CC, "ijl", AA, "ijk", BB, "ikl") + +tensordot(AA, BB, ([3], [2])) # same contraction, axes form +``` + +`α` and `β` implement `C = β*C + α*(A ⋆ B)` in Julia. Prefer a Julia `Number`; +a 0-d `NDArray` is also accepted. The C++ kernel always writes the unscaled +product; `β ≠ 0` uses a temporary. Duplicate labels inside one array are not +allowed — use `cuNumeric.diagonal` first. + +This path is multi-GPU via Legate tiling (per-tile cuTENSOR or TBLIS), not +cuTensorMp. Integer and `Bool` inputs promote to `Float64` like other linalg. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/contract.jl"] +``` diff --git a/docs/src/api_unary.md b/docs/src/api_unary.md index e2099a911..6dcc75c35 100644 --- a/docs/src/api_unary.md +++ b/docs/src/api_unary.md @@ -1,16 +1,25 @@ # Unary Operations >[!NOTE] -> Prefer `@.` for multi-op elementwise expressions so every operator is dotted (especially unary negation). This ensures broadcast operations are fused. See [Kernel Fusion](./perf/kernel_fusion.md). +> Prefer `@.` for multi-op elementwise expressions so every operator is dotted (especially unary negation). This makes eligible CUDA broadcasts fusion-friendly. See [Kernel Fusion](./perf/kernel_fusion.md). The following unary operations are supported and can be broadcast over `NDArray`: -- `-`, `!`, `abs`, `acos`, `acosh`, `asin`, `asinh`, `atan`, `atanh`, `cbrt`, `conj`, `cos`, `cosh`, `deg2rad`, `exp`, `exp2`, `expm1`, `floor`, `imag`, `isfinite`, `log`, `log10`, `log1p`, `log2`, `rad2deg`, `real`, `sign`, `signbit`, `sin`, `sinh`, `sqrt`, `tan`, `tanh`, `^2`, `^-1` or `inv` +- `-`, `!`, `~`, `abs`, `acos`, `acosh`, `asin`, `asinh`, `atan`, `atanh`, `cbrt`, `ceil`, `conj`, `cos`, `cosh`, `deg2rad`, `exp`, `exp2`, `expm1`, `floor`, `imag`, `isfinite`, `isinf`, `isnan`, `log`, `log10`, `log1p`, `log2`, `rad2deg`, `real`, `round`, `sign`, `signbit`, `sin`, `sinh`, `sqrt`, `tan`, `tanh`, `trunc`, `^2`, `^-1` or `inv` ## Differences from Base Julia - The `acosh` function in Julia will error on inputs outside of the domain (`x >= 1`), but cuNumeric.jl will return `NaN`. +- `round.(A)` uses IEEE round-to-nearest-even (the `RINT` kernel), matching Julia `round(x)` / `RoundNearest`. Only that 1-arg path is wired. `digits`, `sigdigits`, and `RoundingMode` are not supported (`round.(A; digits=n)` errors). +- `floor`, `ceil`, `trunc`, and `signbit` are float-only kernels. `round` also supports complex values. Bool and integer inputs are not accepted (Julia's `floor`/`ceil`/`trunc`/`round` on integers are identity). +- `~` is bitwise not on integers. On `Bool` it matches `!` (the invert kernel rejects `Bool`, so we use logical not). +- Full reductions (`sum`, `mean`, `var`, `std`, …) return a `CNScalar` backed by a 0D NDArray. Use `fetch` to read a native Julia scalar. +- `var` / `std` match Julia / StatsBase sample statistics (`corrected=true`, divisor `n-1`). Complex is not supported. +- `argmax` / `argmin` are 1-d fetch (matching Base's `Int` return, not `CartesianIndex`). The result is a 0-d `NDArray{Int64}` of the 1-based index. Complex is not supported. + +For `mapreduce` and mapped `sum`, `prod`, `minimum`, and `maximum`, see +[Mapped Reductions](./api_mapreduce.md). ```@autodocs Modules = [cuNumeric] diff --git a/docs/src/benchmarks/howto.md b/docs/src/benchmarks/howto.md index 6e6e8e6e4..f1850c7b3 100644 --- a/docs/src/benchmarks/howto.md +++ b/docs/src/benchmarks/howto.md @@ -1,6 +1,6 @@ # How to Benchmark -The benchmark harness lives in `benchmark/` at the repo root. Configs are declared in `benchmarks.toml`. `run.jl` expands those configs and launches one worker process per run. Workers never share a GPU runtime within a measurement. +The benchmark harness lives in the `benchmark/` submodule, maintained in [JuliaLegate/benchmarking](https://github.com/JuliaLegate/benchmarking). Configs are declared in `benchmarks.toml`. `run.jl` expands those configs and launches one worker process per run. Workers never share a GPU runtime within a measurement. > [!WARNING] > We do not commit to maintaining the benchmark scripts forever. The harness evolves with the package. The ideas here (declare configs in TOML, one process per run, time with Legate fences) should still apply even if file names move. @@ -14,17 +14,19 @@ cuNumeric ops are asynchronous. A Julia call usually returns before the GPU work From the repo: ```bash +git submodule update --init --recursive cd benchmark +./instantiate_projects.sh julia --project=. run.jl ``` -With no extra args, `run.jl` reads `benchmarks.toml` and runs every expanded config. It develops `CNPreferences` and `cuNumeric` from the parent checkout, then for each config: +The setup script develops `CNPreferences` and `cuNumeric` from the parent checkout. See the harness README for cuPyNumeric's Conda setup. With no extra args, `run.jl` reads `benchmarks.toml` and runs every expanded config: 1. Sets the broadcast-fusion preference if needed and precompiles 2. Calls `run_benchmark.sh`, which exports `LEGATE_CONFIG` from `--gpus` / `--cpus` **before** Julia starts 3. Launches `src/single.jl` for the cuNumeric backend (and optional comparison backends) -Add `-v` / `--verbose` for more plumbing output. +After the sweep, it launches `plot_results.jl` on the result CSVs. One-off CLI runs skip plotting. Add `-v` / `--verbose` for more plumbing output. ## `benchmarks.toml` @@ -36,26 +38,45 @@ n_warmup = 5 n_iter = 1000 n_trial = 5 cupynumeric = true # also run Python cupynumeric (needs install_cupynumeric.sh) -cuda = false # also run CUDA.jl (single-GPU configs only) +cuda = true # also run CUDA.jl (single-GPU configs only) check_correctness = true n_correctness_iter = 5 +auto_size = true +mem_frac = 0.5 # fraction of the smallest visible GPU's total RAM ``` - `n_warmup`: untimed iterations (hide compile / first-touch cost) - `n_iter`: timed iterations per trial (build task queue depth) - `n_trial`: independent trials; mean ± stddev across trials is what gets printed / saved -- `cupynumeric` / `cuda`: optional comparison backends -- `check_correctness`: one CPU-reference check per config (not per timed iter), recorded in the CSV - -Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, `grayscott_baseline`, `grayscott_lifetimes`, …). Names must match what `src/benchmarks/*.jl` registers. +- `cupynumeric` / `cuda`: optional comparison backends. `cuda = true` still + **times** CUDA.jl on configs with `gpus == 1` (GEMM, Monte Carlo, Gray-Scott, + DMD, Poisson FFT, and the TensorOperations kernels). +- `check_correctness`: when `gpus == 1`, the cuNumeric worker also compares a + tiny problem against CUDA.jl (not CPU). The CUDA.jl timing run is unchanged + and records `skipped` for correctness. Multi-GPU and cupynumeric skip. +- `auto_size` / `mem_frac`: when `auto_size` is true and a block omits `N` + (or sets `N = "auto"`), the harness RAM-fits the 1-GPU problem then scales + with GPU count. `mem_frac` is a fraction of the *smallest* visible GPU's + total RAM; override it with `CUNUMERIC_BENCH_MEM_FRAC`. Peak bytes are + `total_space` on each Julia benchmark type (live arrays including tensor + intermediates), not the seed array. Legion / cuSOLVER scratch is the rest of + `mem_frac`. Pin `N = [20000, …]` on a block to keep paper sizes. + +Each `[[name]]` block is a registered benchmark (`gemm`, `montecarlo`, +`dmd_baseline`, `dmd_accelerated`, `grayscott_baseline`, the Gray-Scott +`@accelerate` forms, `poisson_fft`, `tensor_projection3`, …). Names must match +what `src/benchmarks/*.jl` registers. + +DMD's `N` is the number of spatial degrees of freedom (rows of the snapshot matrix), not a grid side length. The SVD is of the tall-skinny `N × (M-1)` matrix `X1`. Thin SVD plus the rank-`r` lift is `Θ(N)` when `M` and `r` are fixed, so weak scaling is `N ∝ P` (same idea as Monte Carlo, not GEMM's `N ∝ P^{1/3}`). Auto-sizing only scales to `P>1` when the 1-GPU `N` is at least `10 M`; otherwise the 1-GPU run still happens and larger `P` is skipped. `M` stays in the toml (intensity knob). The flop count is in `src/benchmarks/dmd.jl`. + +`poisson_fft` solves ``M`` independent periodic Poisson problems on an ``N \times N`` grid (FFT, divide by ``-|k|^2``, inverse FFT). The transform is over the last two axes, so the leading batch axis can split across GPUs. Weak scaling is ``M \propto P`` with ``N`` **held constant across the GPU sweep** — growing `N` would change the FFT size. `N` itself may be RAM-fitted (one grid filling the 1-GPU budget, then `M = P`) or pinned (`N = 1024` and `M = "auto"` to search the batch). A single all-axes 2-d `fft` of one grid is single-GPU and would not scale that way. ```toml [[gemm]] T = ["Float32"] gpus = [1, 2, 4, 8] cpus = 16 -N = [20000, 25200, 31752, 40000] -M = [20000, 25200, 31752, 40000] +# omit N → RAM-fit, then N ∝ P^{1/3} ``` ### How lists expand @@ -76,7 +97,12 @@ M = [150, 300, 600] That is 2 types × 3 sweep points = **6 runs**. -`fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, default `true`). Comparison backends ignore fusion and run once (on the fused pass), not per variant. Names ending in `_lifetimes` are cuNumeric-only code-path variants. +`fusion` toggles cuNumeric broadcast fusion (`true`/`false` or `"on"`/`"off"`, +default `true`). Comparison backends ignore fusion and run once (on the fused +pass), not per variant. Entries ending in `_accelerated` are cuNumeric-only. +The Gray-Scott function, `begin`, `let`, and expression entries compare the +four `@accelerate` scope contracts on the same step; `dmd_accelerated` applies +the recommended function form to the DMD projection. Gotcha: when `T = ["Float32", "Float64"]` and a length-2 `N`/`M` sweep you get all **4** combinations, not a paired `Float32 -> N[1]`. To pin a type to a size, use separate `[[name]]` blocks. @@ -88,17 +114,22 @@ You can dispatch a single config without editing the TOML: julia --project=. run.jl [fusion] ``` -Example: +`N` and `M` may be integers or `auto`: ```bash julia --project=. run.jl 1 16 gemm Float32 20000 20000 1000 5 5 true +CUNUMERIC_BENCH_MEM_FRAC=0.1 julia --project=. run.jl 8 8 gemm Float32 auto auto 3 1 1 true ``` `run.jl` still goes through `run_benchmark.sh` so Legate sees the right GPU/CPU count at process start. ## Comparison backends -- **CUDA.jl:** set `cuda = true` in `[Global]`. Only runs when `gpus == 1`. +- **CUDA.jl:** set `cuda = true` in `[Global]`. Timed on configs with `gpus == 1`. + Every registered kernel has a `CuArray` path (GEMM / broadcast / FFT / + SVD via cuBLAS, cuFFT, cuSOLVER; tensor contractions use TensorOperations' + cuTENSOR extension). That timed run is separate from the tiny cuNumeric vs + CUDA.jl correctness check. - **cupynumeric (Python):** set `cupynumeric = true`, then build a matching conda env once: ```bash @@ -109,10 +140,34 @@ julia --project=. run.jl 1 16 gemm Float32 20000 20000 1000 5 5 true ## Results and timing -Each worker prints mean ± stddev run time (ms) and GFLOPS, plus a correctness tag (`pass` / `fail` / `skipped`). CSVs append under `benchmark/results/`. +Each worker prints mean ± stddev run time (ms) and GFLOPS, plus a correctness tag (`pass` / `fail` / `skipped`). On one GPU the cuNumeric tag is the tiny CUDA.jl compare; CUDA.jl and cupynumeric record `skipped`. CSVs append under `benchmark/results/`. Unfused cuNumeric runs are labeled and saved separately (for example `cunumeric_nofusion`) so they stay a distinct series from fused runs. +## Plotting + +`[plot.groups]` in `benchmarks.toml` names the figures. Related kernels share one PNG: + +```toml +[plot.groups] +grayscott = [ + "grayscott_baseline", + "grayscott_function_accelerated", + "grayscott_begin_accelerated", + "grayscott_let_accelerated", + "grayscott_expression_accelerated", +] +dmd = ["dmd_baseline", "dmd_accelerated"] +``` + +Any `[[benchmark]]` table not listed (GEMM, Poisson, Monte Carlo, tensors) is its own figure. CUDA.jl (single GPU) and cupynumeric overlay from the group's baseline CSV — the member ending in `_baseline`, or the only / first name. Accelerated variants are cuNumeric-only; their CUDA.jl and Python points still come from the baseline. + +Fused and unfused cuNumeric share a color (solid vs dashed). A full `run.jl` pass writes `plots/_weak_scaling.png` at the end. To plot existing CSVs: + +```bash +julia --project=. plot_results.jl +``` + ## Hardware notes `LEGATE_CONFIG` must be set before Julia / Legate starts. The harness does that for you via `run_benchmark.sh`. For manual REPL experiments, see [Hardware Configuration](../configuration/hardware.md). Do not expect to change GPU count mid-session without restarting Julia. diff --git a/docs/src/benchmarks/results.md b/docs/src/benchmarks/results.md index fd62aedcf..08cc9e3c2 100644 --- a/docs/src/benchmarks/results.md +++ b/docs/src/benchmarks/results.md @@ -1,6 +1,6 @@ # Benchmark Results -For JuliaCon2025 we benchmarks cuNumeric.jl on 8 A100 GPUs (single-node) and compared it to the Python library cuPyNumeric and other relevant benchmarks depending on the problem. All results shown are weak scaling. We hope to have multi-node benchmarks soon! +These historical JuliaCon 2025 results compare cuNumeric.jl with cuPyNumeric and problem-specific alternatives on one node with eight A100 GPUs. All plots show weak scaling. See [How to Benchmark](./howto.md) for the current harness and its baseline, `@accelerate`, fused, and unfused variants. ## SGEMM @@ -24,7 +24,7 @@ mul!(C, A, B) ## Monte-Carlo Integration -Monte-Carlo integration is embaressingly parallel and should scale perfectly. We do not know the exact number of operations in `exp` so the GFLOPs is off by a constant factor. +Monte-Carlo integration is embarrassingly parallel. Because the exact operation count of `exp` is implementation-dependent, the plotted operation rate is scaled by an approximate constant. Code Outline: ```julia diff --git a/docs/src/configuration/hardware.md b/docs/src/configuration/hardware.md index 0102ee183..4ddacc4c2 100644 --- a/docs/src/configuration/hardware.md +++ b/docs/src/configuration/hardware.md @@ -1,6 +1,6 @@ # Hardware Configuration -There is no programmatic way to set the hardware configuration used by CuPyNumeric (as of 26.01). By default, the hardware configuration is set automatically by Legate. This configuration can be manipulated through the following environment variables: +Legate chooses the hardware configuration automatically by default. Set these environment variables before starting Julia to override it: - `LEGATE_SHOW_CONFIG` : When set to 1, the Legate config is printed to stdout - `LEGATE_AUTO_CONFIG`: When set to 1, Legate will automatically choose the hardware configuration diff --git a/docs/src/debugging.md b/docs/src/debugging.md index 6b25941d5..3e868c907 100644 --- a/docs/src/debugging.md +++ b/docs/src/debugging.md @@ -6,7 +6,7 @@ Debug the layer that matches the problem: |---|---| | Which operations did Legate submit, and when did they run? | [Legate logs and profiles](#trace-legate-runtime-work) | | How were broadcasts fused? | [`BCAST_FUSION_DEBUG`](#inspect-fused-broadcasts-with-bcast_fusion_debug) | -| Where does `@analyze_lifetimes` free temporaries? | [`@show_lifetimes`](#inspect-lifetime-rewrites-with-show_lifetimes) | +| How does `@accelerate` rewrite code and free temporaries? | [`@show_lifetimes`](#inspect-lifetime-rewrites-with-show_lifetimes) | ## Trace Legate runtime work @@ -78,7 +78,7 @@ cuNumeric already supplies names for individual operations when task-scope namin When broadcast fusion is on, set `cuNumeric.BCAST_FUSION_DEBUG[] = true` to print inter-statement rewrites and each fused kernel's expression tree, arguments, and launch geometry. Inter-statement rewrites are reported when -`@analyze_lifetimes` expands, so enable the flag before defining or evaluating +`@accelerate` expands, so enable the flag before defining or evaluating the expression you want to inspect. Kernel details are reported at runtime. ```julia @@ -91,15 +91,16 @@ A = cuNumeric.ones(Float32, N, N) B = cuNumeric.ones(Float32, N, N) C = cuNumeric.zeros(Float32, N, N) -@analyze_lifetimes begin +@accelerate function combine!(C, A, B) product = A[2:end-1, 2:end-1] .* B[2:end-1, 2:end-1] C[2:end-1, 2:end-1] = product .+ 2.0f0 + return C end cuNumeric.BCAST_FUSION_DEBUG[] = false ``` -For example, a single-use producer inside `@analyze_lifetimes` is reported as: +For example, a single-use producer inside `@accelerate` is reported as: ```text ======================================== inter-broadcast fusion rewrite @@ -147,38 +148,23 @@ See [Kernel Fusion](./perf/kernel_fusion.md) for `@.` / fusion usage, and [Inter ## Inspect lifetime rewrites with `@show_lifetimes` -`@analyze_lifetimes` rewrites a block so temps are freed after their last use. `@show_lifetimes` prints the re-written code (without execution). It is pure AST work, so it works even without a GPU. +`@accelerate` rewrites straight-line code so eligible broadcasts combine and non-returned temporaries are freed after their final use. `@show_lifetimes` prints the exact expansion without executing it, so it works without a GPU. ```julia using cuNumeric -@show_lifetimes begin +@show_lifetimes function update!(C, A, B) result = A[1:end, :] .+ B[1:end, :] C .= result .* 2.0 + return C end ``` -Example output when broadcast fusion is enabled (fusion-aware analysis): - -```text -@analyze_lifetimes expansion (fusion-aware analysis) ------------------------------------------------------------- - 1 tmp1 = A[1:end, :] - 2 tmp2 = B[1:end, :] - 3 tmp3 = tmp1 .+ tmp2 - ✗ free tmp1 - ✗ free tmp2 - 4 result = tmp3 - 5 res3 = (C .= result .* 2.0) - ✗ free tmp3 - 6 res3 ------------------------------------------------------------- -``` - How to read it: -- Numbered lines are the rewritten statements. +- The header identifies the exact function, `let`, block, or expression form expanded. +- Numbered lines are rewritten statements. - Red `✗ free tmpN` lines are the inserted `maybe_insert_delete` calls. -- With fusion enabled, dotted intermediates stay as broadcast expressions instead of being treated as many separate allocations. With fusion disabled, the header says `plain analysis` and more call sites are hoisted. +- With fusion enabled, dotted intermediates stay as broadcast expressions instead of being treated as separate allocations. With fusion disabled, the header says `plain` and more call sites are hoisted. Use this when a hot loop still looks allocation-heavy, or when you want to confirm that a value is freed before it escapes the block. diff --git a/docs/src/developer_mode.md b/docs/src/developer_mode.md index 5cac1631d..bf76a0821 100644 --- a/docs/src/developer_mode.md +++ b/docs/src/developer_mode.md @@ -44,6 +44,16 @@ Then restart Julia (or at least reload cuNumeric) so the new shared library is p If the build fails, check CMake / g++ (C++20) / CUDA toolkit availability as described on [Build Modes](./install.md). Build logs from the helper scripts are written under `deps/`. +Recreating or moving the repository can leave a libcxxwrap CMake export in the +Julia depot pointing at deleted headers. `Pkg.build("cuNumeric")` checks the +exported header and library paths before reusing that build and rebuilds it when +stale. It also records the Julia version and executable path: changing either +forces a rebuild. Older builds without this Julia marker are rebuilt once. +No manual checkout or depot cleanup is needed. The installer retains your +manifest and developed JLL checkout, recreating only its generated `override/` +directory. Check `deps/libcxxwrap_check.log`, `deps/libcxxwrap.log`, and +`deps/libcxxwrap.err` if this repair fails. + ### Typical edit loop ```text @@ -69,4 +79,4 @@ Restart Julia. You do not need `Pkg.build` for pure JLL mode (the build script e - [Build Modes](./install.md): JLL, developer, and conda providers - [CNPreferences](./api_preferences.md): preference defaults and function reference - [Debugging](./debugging.md): fusion and lifetime printers while developing -- [Internals](./internals.md): how fusion and `@analyze_lifetimes` work +- [Internals](./internals.md): how fusion and `@accelerate` work diff --git a/docs/src/examples/cg.md b/docs/src/examples/cg.md new file mode 100644 index 000000000..5cf92294f --- /dev/null +++ b/docs/src/examples/cg.md @@ -0,0 +1,50 @@ +# Conjugate gradient + +Conjugate gradient solves ``Ax=b`` for a real symmetric positive-definite +matrix using matrix-vector products, reductions and vector updates. The complete recurrence is shown below. + +```julia +using cuNumeric, LinearAlgebra + +function cg!(x, A, b; rtol=1e-8, check_every=10, max_iter=1000) + r = b - A*x + p, Ap = copy(r), similar(r) + rho = sum(r .* r) + target = rtol^2 * only(sum(b .* b)) + only(rho) <= target && return x + + for k in 1:max_iter + mul!(Ap, A, p) + # Protect zero denominators if convergence occurs between checks. + alpha = rho ./ max.(sum(p .* Ap), floatmin(eltype(x))) + x .+= alpha .* p + r .-= alpha .* Ap + next = sum(r .* r) + beta = next ./ max.(rho, floatmin(eltype(x))) + p .= r .+ beta .* p + rho = next + + if k % check_every == 0 || k == max_iter + only(rho) <= target && return x + end + end + error("CG did not converge within max_iter") +end + +A = NDArray([4.0 1.0 1.0; 1.0 3.0 0.5; 1.0 0.5 2.0]) +# cuNumeric's matrix multiplication uses single-column matrices for vectors. +b = cuNumeric.ones(Float64, 3, 1) +x = cuNumeric.zeros(Float64, 3, 1) +cg!(x, A, b; check_every=5, max_iter=100) + +# Compare the iterative result with the direct solve API. +x_direct = cuNumeric.solve(A, b) +println(Array(x)) +@assert isapprox(Array(x), Array(x_direct); rtol=1e-8) +``` + +`rho` holds the squared residual norm. Reductions and the coefficients `alpha` +and `beta` remain in cuNumeric's computation graph; `only(rho)` reads a value +back to the host at a convergence check. Increasing `check_every` lets the host +submit more iterations ahead, at the cost of potentially doing extra work +before observing convergence. A check also occurs at the iteration limit. diff --git a/docs/src/examples/dmd.md b/docs/src/examples/dmd.md new file mode 100644 index 000000000..9677cc9b2 --- /dev/null +++ b/docs/src/examples/dmd.md @@ -0,0 +1,138 @@ +# Dynamic Mode Decomposition + +Dynamic mode decomposition takes a sequence of states from a simulation or an +experiment and finds the best linear operator that advances one state to the +next: + +```math +x_{k+1} \approx A x_k +``` + +Its eigenvectors are spatial patterns and its eigenvalues say what each pattern +does per step: ``|\lambda|`` is growth or decay and ``\arg(\lambda)`` is +rotation. For a field on an ``N \times N`` grid, ``A`` is ``N^2 \times N^2`` and +far too large to form. DMD never does. Stack the states as columns of +``X = [x_1 \; \dots \; x_n]``, split it into + +```math +X_1 = [x_1 \; \dots \; x_{n-1}], \qquad X_2 = [x_2 \; \dots \; x_n] +``` + +and take the rank-``r`` SVD ``X_1 = U \Sigma V^*``. Projecting ``A`` onto the +columns of ``U`` gives an ``r \times r`` matrix whose eigendecomposition carries +the dynamics: + +```math +\tilde{A} = U^* X_2 V \Sigma^{-1}, \qquad +\Phi = X_2 V \Sigma^{-1} W +``` + +where ``W`` holds the eigenvectors of ``\tilde{A}``. The columns of ``\Phi`` are +the exact DMD modes, back in the full ``N^2``-dimensional space. + +## Snapshots from Gray-Scott + +The [Gray-Scott](./grayscott.md) example writes its `u` field to +`gray-scott.h5` every 20 steps, one flattened frame per column: + +```julia +snapshots = cuNumeric.zeros(Float32, N * N, n_steps ÷ snapshot_interval) + +# inside the time loop +if n%snapshot_interval == 0 + snapshots[:, n ÷ snapshot_interval] = cuNumeric.reshape(u, (N * N, 1)) +end + +cuNumeric.h5write(SNAPSHOT_FILE, "u", snapshots) +# h5write is asynchronous, so flush before another process opens the file. +cuNumeric.Legate.runtime_sync() +``` + +The snapshots never touch the host: they are assembled into a device array and +handed straight to [HDF5](../api_hdf5.md). + +## Decomposition + +Run `examples/gray-scott.jl` first to produce the file, then: + +```julia +# found in examples/dmd.jl +using cuNumeric +using LinearAlgebra +using Printf + +const SNAPSHOT_FILE = "gray-scott.h5" + +function dmd(X::NDArray{Float32,2}, r::Int) + n = size(X, 2) + X1 = X[:, 1:(n - 1)] # states x_1 … x_{n-1} + X2 = X[:, 2:n] # the same states advanced one snapshot + + F = svd(X1) + r = min(r, length(F.S)) + U = F.U[:, 1:r] + Vt = F.Vt[1:r, :] + + # Σ⁻¹ stays a Diagonal instead of a dense r×r matrix. + Sinv = Diagonal(1.0f0 ./ F.S[1:r]) + + # X2 V Σ⁻¹ appears in both the projected operator and the exact modes. + B = X2 * cuNumeric.transpose(Vt) * Sinv + à = cuNumeric.transpose(U) * B + + E = eigen(Ã) # always complex, even for a real à + Φ = cuNumeric.as_type(B, ComplexF32) * E.vectors + + return E.values, Φ +end + +X = cuNumeric.h5read(SNAPSHOT_FILE, "u") +n_points, n_snapshots = size(X) +N = isqrt(n_points) + +λ, Φ = dmd(X, 20) + +# Only r eigenvalues, so ranking them on the host costs nothing. +vals = Array(λ) +order = sortperm(abs.(vals); rev=true) + +println("mode |λ| cycles/snapshot") +for i in order[1:min(5, end)] + @printf("%4d %6.4f %+8.4f\n", i, abs(vals[i]), angle(vals[i]) / 2π) +end + +# The slowest-decaying mode is the pattern the simulation settles into. +lead = order[1] +mode = cuNumeric.reshape(abs.(Φ[:, lead:lead]), (N, N)) +cuNumeric.h5write("dmd-mode.h5", "leading", mode) +cuNumeric.Legate.runtime_sync() +``` + +On 100 snapshots of a ``100 \times 100`` grid this prints something like: + +``` +100 snapshots of 10000 points + +mode |λ| cycles/snapshot + 17 0.9945 +0.0091 + 18 0.9945 -0.0091 + 14 0.9928 +0.0000 + 19 0.9825 +0.0202 + 20 0.9825 -0.0202 +``` + +Every ``|\lambda|`` sits just below 1, which is the pattern settling rather than +growing, and the oscillatory modes come in the conjugate pairs a real operator +must produce. + +## Notes + +- `svd`, `eigen`, and matrix multiply all run on device; see + [Linear Algebra](../linalg.md). +- `Diagonal(1.0f0 ./ F.S[1:r])` scales by ``\Sigma^{-1}`` without building a + dense ``r \times r`` matrix. Note the `1.0f0` — an `Int` literal would try to + widen the `Float32` singular values. +- `eigen` is always complex, so `B` is converted with `cuNumeric.as_type` before + multiplying by the eigenvectors. +- The eigenvalues are only `r` numbers, so sorting and printing them on the host + is cheap. The modes stay on device. diff --git a/docs/src/examples/grayscott.md b/docs/src/examples/grayscott.md index dc0fca426..7c230143b 100644 --- a/docs/src/examples/grayscott.md +++ b/docs/src/examples/grayscott.md @@ -1,56 +1,27 @@ # Gray-Scott Reaction Diffusion -```julia -# found in examples/gray-scott.jl -using cuNumeric -using Plots - -struct Params{T} - dx::T - dt::T - c_u::T - c_v::T - f::T - k::T - - function Params(dx=1.0f0, c_u=1.0f0, c_v=0.3f0, f=0.03f0, k=0.06f0) - new{Float32}(dx, dx/5, c_u, c_v, f, k) - end -end - -function bc!(u_new, v_new, u, v) - u_new[:,1] = u[:,end-1] - u_new[:,end] = u[:,2] - u_new[1,:] = u[end-1,:] - u_new[end,:] = u[2,:] - v_new[:,1] = v[:,end-1] - v_new[:,end] = v[:,2] - v_new[1,:] = v[end-1,:] - v_new[end,:] = v[2,:] -end - -function step!(u, v, u_new, v_new, args::Params) - @analyze_lifetimes begin - # Prefer @. so every op is dotted and the tree can fuse - F_u = @. -u[2:end-1, 2:end-1] * (v[2:end-1, 2:end-1]^2) + - args.f * (1.0f0 - u[2:end-1, 2:end-1]) - F_v = @. u[2:end-1, 2:end-1] * (v[2:end-1, 2:end-1]^2) - - (args.f + args.k) * v[2:end-1, 2:end-1] - - u_lap = @. ( - (u[3:end, 2:end-1] - 2 * u[2:end-1, 2:end-1] + u[1:end-2, 2:end-1]) / args.dx^2 + - (u[2:end-1, 3:end] - 2 * u[2:end-1, 2:end-1] + u[2:end-1, 1:end-2]) / args.dx^2 - ) - v_lap = @. ( - (v[3:end, 2:end-1] - 2 * v[2:end-1, 2:end-1] + v[1:end-2, 2:end-1]) / args.dx^2 + - (v[2:end-1, 3:end] - 2 * v[2:end-1, 2:end-1] + v[2:end-1, 1:end-2]) / args.dx^2 - ) - - u_new[2:end-1, 2:end-1] = @. (args.c_u * u_lap + F_u) * args.dt + u[2:end-1, 2:end-1] - v_new[2:end-1, 2:end-1] = @. (args.c_v * v_lap + F_v) * args.dt + v[2:end-1, 2:end-1] - end +The runnable example in `examples/gray-scott.jl` evolves two chemical fields with periodic boundaries. Its update is a straight-line function, so the recommended function form of `@accelerate` can release reaction and Laplacian temporaries after their final use and fuse eligible CUDA broadcasts. +```julia +@accelerate function step!(u, v, u_new, v_new, args::Params) + F_u = @. -u[2:end-1, 2:end-1] * v[2:end-1, 2:end-1]^2 + + args.f * (1.0f0 - u[2:end-1, 2:end-1]) + F_v = @. u[2:end-1, 2:end-1] * v[2:end-1, 2:end-1]^2 - + (args.f + args.k) * v[2:end-1, 2:end-1] + + u_lap = @. ( + (u[3:end, 2:end-1] - 2u[2:end-1, 2:end-1] + u[1:end-2, 2:end-1]) / args.dx^2 + + (u[2:end-1, 3:end] - 2u[2:end-1, 2:end-1] + u[2:end-1, 1:end-2]) / args.dx^2 + ) + v_lap = @. ( + (v[3:end, 2:end-1] - 2v[2:end-1, 2:end-1] + v[1:end-2, 2:end-1]) / args.dx^2 + + (v[2:end-1, 3:end] - 2v[2:end-1, 2:end-1] + v[2:end-1, 1:end-2]) / args.dx^2 + ) + + u_new[2:end-1, 2:end-1] = @. (args.c_u * u_lap + F_u) * args.dt + u[2:end-1, 2:end-1] + v_new[2:end-1, 2:end-1] = @. (args.c_v * v_lap + F_v) * args.dt + v[2:end-1, 2:end-1] bc!(u_new, v_new, u, v) + return nothing end function gray_scott() @@ -63,12 +34,16 @@ function gray_scott() n_steps = 2000 # number of steps to take frame_interval = 200 # steps to take between making plots + snapshot_interval = 20 # steps to take between saved snapshots u = cuNumeric.ones(dims) v = cuNumeric.zeros(dims) u_new = cuNumeric.zeros(dims) v_new = cuNumeric.zeros(dims) + # One flattened frame per column, the layout DMD wants. + snapshots = cuNumeric.zeros(Float32, N * N, n_steps ÷ snapshot_interval) + u[1:15,1:15] = cuNumeric.rand(15,15) v[1:15,1:15] = cuNumeric.rand(15,15) @@ -79,6 +54,10 @@ function gray_scott() u, u_new = u_new, u v, v_new = v_new, v + if n%snapshot_interval == 0 + snapshots[:, n ÷ snapshot_interval] = cuNumeric.reshape(u, (N * N, 1)) + end + if n%frame_interval == 0 u_cpu = u[:, :] heatmap(u_cpu, clims=(0, 1)) @@ -86,10 +65,16 @@ function gray_scott() end end gif(anim, "gray-scott.gif", fps=10) - return u, v -end + cuNumeric.h5write(SNAPSHOT_FILE, "u", snapshots) + # h5write is asynchronous, so flush before another process opens the file. + cuNumeric.Legate.runtime_sync() -u, v = gray_scott() + return u, v + end ``` + ![Simulation Output](../gray-scott.gif) + +The snapshots written to `gray-scott.h5` are the input to +[Dynamic Mode Decomposition](./dmd.md). diff --git a/docs/src/examples/initialization.md b/docs/src/examples/initialization.md index 6a1ddbe4a..d187828dc 100644 --- a/docs/src/examples/initialization.md +++ b/docs/src/examples/initialization.md @@ -3,6 +3,7 @@ Create `NDArray`s with the usual Julia-style constructors. The default element type is `Float32` unless you pass one. ```julia +using LinearAlgebra using cuNumeric # Zeros / ones / fill @@ -15,14 +16,24 @@ F = cuNumeric.fill(7.5f0, (2, 3)) T = cuNumeric.trues(2, 3) Fbool = cuNumeric.falses(2, 3) -# Identity -I = cuNumeric.eye(5) -I16 = cuNumeric.eye(Float32, 5) +# Identity: prefer Diagonal / I; densify only when needed (no eye) +D = Diagonal(cuNumeric.ones(Float32, 5)) +I32 = NDArray{Float32}(I, 5, 5) # dense identity +Ib = NDArray(I, 5, 5) # Bool identity -# Uniform random values (default Float32; backend draws Float64 then converts) +# Uniform / normal random values (native Float32 and Float64) R = cuNumeric.rand(4, 4) R64 = cuNumeric.rand(Float64, 1000) cuNumeric.rand!(R64) # fill an existing Float64 array +N = cuNumeric.randn(Float32, 8, 8) +C = cuNumeric.rand(ComplexF32, 8, 8) # independent real/imag uniforms +E = cuNumeric.randexp(Float64, 1000) # exponential, scale 1 (real only) +I = cuNumeric.rand(0:9, 4, 4) # Int64 in 0:9 (inclusive) +Coin = cuNumeric.rand(Bool, 8) # fair coin flips + +# Private stream / non-default engine (see Random in the Public API) +g = cuNumeric.default_rng(cuNumeric.PHILOX4_32_10, 1234) +P = cuNumeric.random(g, Float32, (4, 4)) ``` Shapes can be passed as separate `Int`s or as a `Tuple` / `Dims`: @@ -33,4 +44,4 @@ cuNumeric.zeros((2, 3)) cuNumeric.ones(Float64, (10, 10)) ``` -For signatures and more detail, see [Initialization](../api_initialization.md) in the Public API. +For signatures and more detail, see [Initialization](../api_initialization.md) and [Random](../api_random.md) in the Public API. diff --git a/docs/src/examples/montecarlo.md b/docs/src/examples/montecarlo.md index 5a66572db..49c366d3f 100644 --- a/docs/src/examples/montecarlo.md +++ b/docs/src/examples/montecarlo.md @@ -1,40 +1,29 @@ # Monte-Carlo Integration -Most integrals can be estimated with a basic Monte-Carlo estimator: +For uniformly sampled points `x_i` in a domain of volume `\Omega`, a basic Monte-Carlo estimator is ```math -\hat{I}_N = \frac{\Omega}{N}\sum_{i=1}^Nf(x_i) +\hat{I}_N = \frac{\Omega}{N}\sum_{i=1}^N f(x_i). ``` -where `N` is the number of samples, ``\Omega`` is the volume of the domain and ``x_i`` are sampled indpendently and uniformly at random from the domain. This estimator is guranteed to converge (subject to some minor constraints) at a rate independent of the dimension and is embaressingly parallel to compute! -In the example below, we estimate the integral: -```math -I = \int_{-\infty}^{\infty}e^{-x^2}. -``` +This example estimates `\int_{-\infty}^{\infty} e^{-x^2}\,dx` by sampling the finite interval `[-10, 10]`. `@accelerate` frees the non-returned sample arrays after their final use; CUDA may also fuse eligible broadcasts. -Since we cannot uniformly sample form negative to positive infinity, we truncate the domain between -5 and 5. This is ok since the integrand exponentially decays and we won't be off by much in the end. ```julia -# found in examples/integrate.jl +# examples/integrate.jl using cuNumeric -# Note that we do not yet support broadcasting -# custom functions over NDArray, so the broadcasting MUST -# be done inside the function -integrand = (x) -> @. exp(-x^2) - -N = 1_000_000 +integrand(x) = @. exp(-x^2) -x_max = 10.0f0 -domain = [-x_max, x_max] -Ω = domain[2] - domain[1] +@accelerate function monte_carlo(N, x_max) + Ω = 2 * x_max + raw_samples = cuNumeric.rand(N) + samples = @. Ω * raw_samples - x_max + return (Ω / N) * sum(integrand(samples)) +end -samples = Ω * cuNumeric.rand(N) -samples = @. samples - x_max - -# Reductions return 0D NDArrays instead -# of a scalar to avoid blocking runtime -estimate = (Ω / N) * sum(integrand(samples)) - -println("Monte-Carlo Estimate: $(estimate)") -println("Analytical: $(sqrt(pi))") +estimate = monte_carlo(1_000_000, 10.0f0) +println("Monte-Carlo estimate: $(estimate)") +println("Analytical value: $(sqrt(pi))") ``` + +The result is a `CNScalar` backed by a 0D NDArray, which keeps the reduction asynchronous. Use `fetch(estimate)` only when a Julia scalar is required. diff --git a/docs/src/examples/poisson_fft.md b/docs/src/examples/poisson_fft.md new file mode 100644 index 000000000..a466103ec --- /dev/null +++ b/docs/src/examples/poisson_fft.md @@ -0,0 +1,63 @@ +# Periodic Poisson (FFT) + +The Poisson equation + +```math +\nabla^2 u = f +``` + +on the periodic unit square is a pointwise divide in Fourier space. With +``k = 2\pi (m_x, m_y)`` the wavevector of each DFT mode, + +```math +\hat{u}[k] = \frac{\hat{f}[k]}{-|k|^2}, \qquad \hat{u}[0] = 0. +``` + +The zero mode is dropped so ``u`` has mean zero (the potential is only defined +up to a constant). One forward [`fft`](@ref), a broadcasted divide, and one +[`ifft`](@ref) is the whole solve. That is the electrostatics / gravitational +potential of a periodic charge density, not an FFT round-trip. There is no +reusable plan to hold across the two transforms; each call launches +`CUPYNUMERIC_FFT` and cuFFT planning stays inside that task. + +A manufactured solution ``u = \sin(2\pi x)\sin(2\pi y)`` has +``\nabla^2 u = -8\pi^2 u``, which is what the example checks. + +```julia +# found in examples/poisson_fft.jl +using cuNumeric + +function integer_fftfreq(n::Int) + n2 = n ÷ 2 + return iseven(n) ? vcat(0:(n2 - 1), (-n2):-1) : vcat(0:n2, (-n2):-1) +end + +function poisson_inv_laplacian(::Type{T}, n::Int) where {T} + freq = integer_fftfreq(n) + invk = Matrix{T}(undef, n, n) + s = T(4 * π^2) + for j in 1:n, i in 1:n + k2 = s * T(freq[i]^2 + freq[j]^2) + invk[i, j] = k2 == 0 ? zero(T) : -inv(k2) + end + return invk +end + +n = 64 +xs = range(0, 1; length=n + 1)[1:(end - 1)] +u_true = Float32[sin(2π * x) * sin(2π * y) for x in xs, y in xs] +f = NDArray((-8 * Float32(π)^2) .* u_true) + +invk = NDArray(poisson_inv_laplacian(Float32, n)) +u = real(ifft(fft(f) .* invk)) +``` + +A stack of right-hand sides (several charge distributions, several snapshots) +uses [`batched_fft`](@ref). The leading axis is the batch and is the one that +can split across GPUs; a single all-axes `fft` of one grid cannot. + +```julia +f_batch = NDArray(repeat(reshape(Array(f), 1, n, n), 4, 1, 1)) +invk3 = reshape(invk, 1, n, n) +u_batch = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f_batch) .* invk3)) +``` diff --git a/docs/src/examples/special_mat.md b/docs/src/examples/special_mat.md new file mode 100644 index 000000000..56e64ce64 --- /dev/null +++ b/docs/src/examples/special_mat.md @@ -0,0 +1,23 @@ +# Special Matrices + +Currently cuNumeric only supports `LinearAlgebra.Diagonal`. Other special matrix types like `Tridiagonal` and `Symmetric` will follow. + +Diagonal matrices are common, require only storage of the diagonal elements and are often simple to compute operations on (i.e. `LinearAlgebra.inv`). `LinearAlgebra.Diagonal` matrices can be constructed from 1D or 2D `NDArray`s and have certain operations implemented (i.e., `eigen` and `inv`). + + +```julia +using cuNumeric +using LinearAlgebra + +one_dim = cuNumeric.NDArray([1,2,3,4,5]) +two_dim = cuNumeric.rand(5,5) + +D1 = Diagonal(one_dim) +D2 = Diagonal(two_dim) + +evals, evecs = eigen(D1) +D1_inv = inv(D1) + +D1 ./= 2 # stays diagonal +arr = D2 .+ two_dim # densifies because `two_dim` is not guranteed to be diagonal +``` diff --git a/docs/src/examples/tensor_network.md b/docs/src/examples/tensor_network.md new file mode 100644 index 000000000..5d87358eb --- /dev/null +++ b/docs/src/examples/tensor_network.md @@ -0,0 +1,71 @@ +# Tensor Network Contraction + +[TensorOperations.jl](https://quantumkithub.github.io/TensorOperations.jl/stable/) +provides Einstein-index notation through `@tensor`. Loading it together with +cuNumeric activates the `NDArray` extension, so contractions and their +intermediate tensors remain in Legate-managed memory. + +The two-site spin-1/2 Heisenberg interaction is + +```math +H = S^x_1 S^x_2 + S^y_1 S^y_2 + S^z_1 S^z_2. +``` + +The examples below apply it to an MPS-like state with tensors ``A^{(1)}``, +``A^{(2)}`` and boundary environments ``L``, ``R``. The first contractions keep +open bond indices and stay `NDArray`s. The second contracts every index; the +ratio stays on device until `fetch` for printing. + +```julia +# found in examples/tensor_network.jl +using cuNumeric +using LinearAlgebra +using Random +using TensorOperations + +Random.seed!(1234) # for Random.randn, cuNumeric expects seed via default_rng + +physical_dim = 2 +bond_dim = 16 + +σx = ComplexF64[0 1; 1 0] +σy = ComplexF64[0 -im; im 0] +σz = ComplexF64[1 0; 0 -1] +Hmatrix = (kron(σx, σx) + kron(σy, σy) + kron(σz, σz)) / 4 +H = NDArray(reshape(Hmatrix, 2, 2, 2, 2)) + +A1 = NDArray(randn(ComplexF64, bond_dim, physical_dim, bond_dim)) +A2 = NDArray(randn(ComplexF64, bond_dim, physical_dim, bond_dim)) +L = NDArray(randn(ComplexF64, bond_dim, bond_dim)) +R = NDArray(randn(ComplexF64, bond_dim, bond_dim)) +``` + +## Apply a two-site operator + +These contractions have free indices ``a, s_1, s_2, b``, so the results are +`NDArray`s. TensorOperations lowers the network to pairwise contractions and +releases the intermediate `NDArray`s. + +```julia +@tensor ψ[a, s1, s2, b] := + L[a, ap] * A1[ap, s1, c] * A2[c, s2, bp] * R[bp, b] +@tensor Hψ[a, s1, s2, b] := H[s1, s2, t1, t2] * ψ[a, t1, t2, b] + +println("ψ is a ", typeof(ψ), " of size ", size(ψ)) +println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) +``` + +## Expectation value + +A fully contracted `@tensor` assignment is a 0D `NDArray`, like `sum`. +Prefer to keep it that way and do device arithmetic (`./`) until you +need a host `Number`. `fetch` (here, only to print) **blocks**. See +[Scalars](../api_tensor.md#Scalars). + +```julia +@tensor energy = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] +@tensor norm² = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] + +println("two-site energy = ", real(fetch(energy ./ norm²))) +``` +See [Tensor Contractions](../api_tensor.md) documentation for more details. diff --git a/docs/src/fft.md b/docs/src/fft.md new file mode 100644 index 000000000..bcd207110 --- /dev/null +++ b/docs/src/fft.md @@ -0,0 +1,58 @@ +# FFT + +cuNumeric.jl exposes GPU-only [`fft`](@ref) / [`ifft`](@ref) (and in-place +[`fft!`](@ref) / [`ifft!`](@ref)) on `NDArray`. The functions follow +[AbstractFFTs.jl](https://github.com/JuliaMath/AbstractFFTs.jl) / FFTW +conventions, not NumPy's defaults: + +- every dimension is transformed unless you pass `dims` +- `fft` is unnormalized; `ifft` divides by the product of the transformed lengths +- real input is promoted to complex (`Float32` → `ComplexF32`, otherwise `ComplexF64`) + +There is no `plan_fft`, and we cannot control the cuFFT plan. cupynumeric never +returns a `cufftHandle` (or any other plan object). Each `fft` / `ifft` call +launches the `CUPYNUMERIC_FFT` task; cuFFT planning and any internal plan cache +live entirely inside that GPU task. A Julia `Plan` that only stored sizes and +re-called `fft` would not be a real plan, so `plan_fft` is not implemented. +`using AbstractFFTs; fft(A)` dispatches, but `plan_fft` will not. + +```julia +using AbstractFFTs +using cuNumeric + +A = cuNumeric.rand(ComplexF32, 64, 64) +Y = fft(A) # all dimensions +Z = fft(A, 1) # first dimension only +ifft(Y) # ≈ A + +fft!(copy(A)) # overwrites a complex array +``` + +`fft!` / `ifft!` require `ComplexF32` or `ComplexF64`. Real arrays must go +through out-of-place `fft`. + +A stack of independent transforms uses [`batched_fft`](@ref): the leading +dimension is the batch, and every trailing dimension is transformed. That is +the same task as `fft(A, 2:ndims(A))`. The batch axis is the one that can +split across GPUs; an all-axes `fft(A)` cannot. + +```julia +signals = cuNumeric.rand(ComplexF32, 32, 1024) # 32 length-1024 traces +batched_fft(signals) # FFT along dim 2 + +fields = cuNumeric.rand(ComplexF32, 8, 64, 64) # 8 images +batched_fft(fields) # 2-d FFT of each +``` + +FFT is GPU-only. A CPU-only runtime raises an error. Multi-GPU use is limited +to batching over dimensions that are not transformed (the same restriction as +cupynumeric). + +Awkward lengths whose prime factors exceed 131 make cuFFT take the Bluestein +path; a warning is emitted. Padding to a nearby highly composite size avoids +that. + +```@autodocs +Modules = [cuNumeric] +Pages = ["ndarray/fft.jl"] +``` diff --git a/docs/src/index.md b/docs/src/index.md index 932cc5b52..4bd2d9d3b 100644 --- a/docs/src/index.md +++ b/docs/src/index.md @@ -5,7 +5,7 @@ ``` -[![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev/) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) +[![Documentation stable](https://img.shields.io/badge/docs-stable-blue.svg)](https://julialegate.github.io/cuNumeric.jl/stable) [![Documentation dev](https://img.shields.io/badge/docs-dev-blue.svg)](https://julialegate.github.io/cuNumeric.jl/dev) [![codecov](https://codecov.io/github/julialegate/cuNumeric.jl/branch/main/graph/badge.svg)](https://app.codecov.io/github/JuliaLegate/cuNumeric.jl) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](https://opensource.org/licenses/MIT) cuNumeric.jl wraps and extends the [cuPyNumeric](https://github.com/nv-legate/cupynumeric) library from NVIDIA to bring distributed array computing on GPUs and CPUs to Julia. The central type is `NDArray`, which behaves like Julia's `Array` or the `CuArray` from [CUDA.jl](https://github.com/juliagpu/cuda.jl), but executes across multiple GPUs/CPUs. We implement array-level operations on `NDArray` which can be composed into larger programs without the need for explicit MPI calls or writing CUDA kernels. @@ -20,7 +20,7 @@ using Pkg Pkg.add(url = "https://github.com/JuliaLegate/cuNumeric.jl", rev = "main") ``` -The first time might take awhile as it has to install multiple large dependencies such as the CUDA SDK (if you have an NVIDIA GPU). To use a local build of cupynumeric.so, see [Build Modes](./install.md). +The first installation can take a while because it includes several large dependencies, such as the CUDA SDK. To use a local cupynumeric build, see [Build Modes](install.md). ```julia using cuNumeric @@ -30,7 +30,7 @@ cuNumeric.versioninfo() > [!WARNING] > Starting more than one instance of cuNumeric.jl can lead to a hard-crash. The default hardware configuration reserves all available resources. -For more details, see [Hardware](./configuration/hardware.md). +For more details, see [Hardware](configuration/hardware.md). ### How `NDArray`s work @@ -40,16 +40,16 @@ The semantics of `NDArray` closely mirror Julia's `Array`, and in most cases it **Slices are views.** Indexing an `NDArray` with ranges returns a view onto the same store, not a copy. That differs from Base Julia, where `A[1:n]` allocates a new `Array`. Mutations through an `NDArray` slice are visible through other aliases of the same data. -**Reductions return arrays, not Julia scalars.** Reductions such as `sum(A)` produce a **0D or 1D** `NDArray` (axis reductions produce a lower-rank `NDArray`), rather than a bare `Float64` / `Float32`. That keeps the Legate task graph asynchronous instead of forcing synchronization to communite with the Julia runtime. When you need a plain Julia number, call `unwrap`: +**Scalar reductions return device scalars.** Full reductions such as `sum(A)` return a `CNScalar` backed by a 0D NDArray. Reductions that retain dimensions still return NDArrays. This keeps the Legate task graph asynchronous. When you need a native Julia scalar, call `fetch`: ```julia -s = sum(A) # NDArray{T,0} -x = unwrap(s) # T, e.g. Float32 +s = sum(A) # CNScalar +x = fetch(s) # native Julia scalar, e.g. Float32 ``` -**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `unwrap`, or converting with `Array(A)`). Hiding latency enables performant code. +**The Legate runtime builds a DAG asynchronously.** Calling `cuNumeric.zeros` or `A .+ B` records work into Legate's task graph rather than blocking until every GPU kernel finishes. Results are materialized when you need them (for example `println`, `fetch`, or converting with `Array(A)`). Hiding latency enables performant code. -For API details see [Initialization](./api_initialization.md) and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). +For API details see [Initialization](./api_initialization.md), [Random](./api_random.md), [FFT](./fft.md), and [NDArray Reference](./api.md). For anti-patterns that kill performance, see [Patterns to Avoid](./perf/patterns_to_avoid.md). ### Kernel Fusion @@ -59,41 +59,33 @@ Nested broadcast expressions fuse into a single kernel by default when on GPU. P y .= @. -a + b * c ``` -See [Kernel Fusion](./perf/kernel_fusion.md) and [Debugging](./debugging.md) for controls and pretty printers. +See [Kernel Fusion](perf/kernel_fusion.md) and [Debugging](debugging.md) for controls and diagnostics. -### Helping the Garbage Collector +### The `@accelerate` macro -Many calls such as array slicing and un-fused broadcasts allocate a new `NDArray`. The Legate runtime keeps track of all references to the underlying data and will not free the memory until Julia's GC frees the `NDArray` handles. Because Julia's GC runs on memory pressure and an `NDArray` only stores a pointer (i.e., Julia's GC does not know the true size), many dead buffers accumulate and can cause out-of-memory errors. +`@accelerate` fuses eligible GPU broadcasts within and across statements, then releases materialized temporary `NDArray`s after their last use on CPU or GPU. See [The `@accelerate` Macro](perf/reduce_allocations.md) for usage guidance. -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager calls to immediately free unused `NDArrays`. These buffers can then be reused by legate later for same-sized allocations. +### Benchmarks -```julia -@analyze_lifetimes begin - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 -end -``` - -### Performance at a glance - -A representative benchmark figure will go here (add something like `docs/src/images/benchmarks-overview.png` when ready). - -Numbers, plots, and how to reproduce them live under [Benchmark Results](./benchmarks/results.md) and [How to Benchmark](./benchmarks/howto.md). +Results and reproduction instructions live under [Benchmark Results](benchmarks/results.md) and [How to Benchmark](benchmarks/howto.md). ### Try an example ```julia using cuNumeric -integrand = (x) -> @. exp(-x^2) +integrand(x) = @. exp(-x^2) + +@accelerate function monte_carlo(N, x_max) + Ω = 2 * x_max + raw_samples = cuNumeric.rand(N) + samples = @. Ω * raw_samples - x_max + return (Ω / N) * sum(integrand(samples)) +end N = 1_000_000 x_max = 10.0f0 -Ω = 2 * x_max - -samples = Ω .* cuNumeric.rand(N) -samples = samples .- x_max -estimate = (Ω / N) .* sum(integrand(samples)) +estimate = monte_carlo(N, x_max) println("Monte-Carlo Estimate: $(estimate)") ``` diff --git a/docs/src/install.md b/docs/src/install.md index a4dc9ef37..429d92cc5 100644 --- a/docs/src/install.md +++ b/docs/src/install.md @@ -1,6 +1,6 @@ # Build Modes -cuNumeric.jl gets its cupynumeric / Legate binaries from one of three providers, chosen through `CNPreferences` (writes `LocalPreferences.toml`; **restart Julia** after changing mode): +cuNumeric.jl gets its cupynumeric / Legate binaries from one of three providers, chosen through `CNPreferences` (writes `LocalPreferences.toml`, **restart Julia** after changing mode): | Mode | When to use | |---|---| diff --git a/docs/src/internals.md b/docs/src/internals.md index 1afa20c03..652433251 100644 --- a/docs/src/internals.md +++ b/docs/src/internals.md @@ -1,72 +1,30 @@ # Internals -This page describes the implementation details of kernel fusion, manual memory management via `@analyze_lifetimes` and automatic memory management via GC heuristics. For docs on how to use these features, see [Kernel Fusion](./perf/kernel_fusion.md) and [Reduce Allocations](./perf/reduce_allocations.md). +This page summarizes the machinery behind [`@accelerate`](./perf/reduce_allocations.md), [kernel fusion](./perf/kernel_fusion.md), and allocation-driven garbage collection. ## Broadcast fusion -We compile nested Julia broadcast expressions on `NDArray` to a single CUDA kernel instead of launching one kernel per operation. +Julia first builds a nested `Broadcasted` tree for a dotted expression. When CUDA fusion is enabled and the tree is eligible, cuNumeric flattens it, generates a CUDA.jl kernel, and launches it through Legate. Otherwise `unravel_broadcast_tree` evaluates the operations individually. CPU execution always uses the unfused path. -### Pipeline +`@accelerate` adds an inter-statement syntax pass. It can merge a single-use broadcast producer into its consumer when doing so preserves mutation and aliasing semantics. The soft-scope `begin` form instead keeps named values materialized and may lower an eligible chain to a multi-output CUDA kernel. -- **Expression:** You write a dotted or `@.` expression such as `y .= @. a * b + c`. -- **Broadcast tree:** Julia builds the `Broadcasted` tree. -- **Fused path:** Flatten the tree, build a kernel with CUDA.jl, and launch it with Legate. -- **Unfused path:** `unravel_broadcast_tree` recursively unravels the tree and executes each operation one at a time. +Relevant source: `src/ndarray/broadcast_fusion.jl` and `src/scoping/`. -## Lifetimes and GC +## Lifetime analysis -Julia's GC sees an `NDArray` as a small handle. The actual data is owned by the Legate runtime. This means that to Julia's GC `NDArrays` do not create memory pressure and GC is never executed. To avoid out-of-memory errors we created the `@analyze_lifetimes` macro so users can manually manage the lifetimes of a code block and also manually track device memory to automatically invoke GC. +An `NDArray` is a small Julia handle to storage owned by Legate, so Julia's heap pressure understates the size of live array data. `@accelerate` therefore performs static last-use analysis: -### Eager last-use freeing with `@analyze_lifetimes` +1. Expand nested `@.` macros and reject non-straight-line code. +2. Apply conservative inter-statement fusion where the selected macro form permits it. +3. Hoist materialized temporary values and find their final uses. +4. Insert `maybe_insert_delete` calls after those uses while protecting caller-owned arguments, named soft-scope bindings, and returned values. -`@analyze_lifetimes` rewrites a block at macro-expansion time: +With fusion enabled, nested dotted nodes remain lazy and are not counted as separate array allocations. With fusion disabled, allocating calls are analyzed individually. `@show_lifetimes` prints the exact expansion without running it. -- Hoist temporary allocations into named temps. -- Find each temp's static last use. -- Insert calls to free temporary `NDArrays` after last-use. +The analysis is statement-linear rather than a control-flow graph pass. Apply it to straight-line function or loop bodies, not to control flow itself. -Under broadcast fusion, intermediate dotted nodes are **not** real `NDArray` allocations. The macro switches to a fusion-aware hoist that keeps dotted trees lazy and only treats slices, broadcast roots, and non-broadcast calls as real allocations. When fusion is off, every call (including dotted ops) is treated as a real allocation. +## Allocation-driven GC -```julia -@analyze_lifetimes begin - result = A[1:end, :] .+ B[1:end, :] - C .= result .* 2 -end -``` - -Use `@show_lifetimes` to print the rewritten block and the free sites without running the code. That is pure AST work and works without a GPU. - -The implementation uses separate passes for inter-statement broadcast fusion, -allocation hoisting, and finalizer insertion. The top-level scoping pass selects -the appropriate lifetime analysis based on whether broadcast fusion is enabled. - -- **Inter-statement broadcast fusion** (`rewrite_scope`): - Merges single-use broadcast statements into their consumer, e.g.: - `p = A .* B; C .= p .+ 1` → `C .= A .* B .+ 1` - This pass operates only on syntax and does not depend on cuNumeric types. - -- **Lifetime analysis**: - - With fusion enabled (`rewrite_broadcast_lifetimes`): keeps dotted trees lazy - and hoists only materialized values (slices, non-broadcast calls, etc.) - - With fusion disabled (`rewrite_eager_lifetimes`): hoists all allocating calls - Both passes rename temps and insert free calls after static last uses. - -- **Finalizer insertion** (`insert_finalizers`): - Traverses the AST generated by either lifetime pass and inserts `delete` calls - at the computed last-use sites. It preserves the final return value of the - block by not freeing it. - -Limits to keep in mind: - -- Analysis is statement-linear. It is not a full control-flow graph pass. Wrap hot loop bodies, not entire programs. -- Some paths free eagerly outside the macro (for example LHS slice views created during indexed assignment). - -Relevant source: `src/scoping/`. - -### Allocation-driven GC heuristics - -Every `NDArray` registers its byte size on construct and free. When predicted live bytes cross soft (~80%) or hard (~90%) fractions of available memory, and enough new growth has accumulated since the last collection, cuNumeric.jl triggers Julia `GC.gc`. - -`@analyze_lifetimes` reduces peak live temps. The heuristics catch cases the macro cannot see. +Each `NDArray` reports its byte size when constructed and freed. When predicted live bytes cross soft and hard fractions of available memory—and enough new growth has accumulated—cuNumeric asks Julia to collect garbage. Eager last-use freeing reduces peak live storage; the GC heuristic covers allocations the static pass cannot prove dead. Relevant source: `src/memory.jl`. diff --git a/docs/src/linalg.md b/docs/src/linalg.md index 103c2c40b..9bc782bc6 100644 --- a/docs/src/linalg.md +++ b/docs/src/linalg.md @@ -1,11 +1,79 @@ # Linear Algebra -cuNumeric.jl provides matrix multiplication, batched solves, SVD, QR, and related -helpers for `NDArray`. +cuNumeric.jl provides matrix multiplication, solves, Cholesky, eigen, SVD, QR, +and related helpers for `NDArray`. For Einstein-index contractions, see +[Tensor Contractions](./api_tensor.md). -`solve`, `svd`, and `qr` accept `Float32`, `Float64`, `ComplexF32`, and -`ComplexF64`. Integer and `Bool` inputs require `@allowpromotion` or -`allowpromotion` and produce `Float64` outputs. +All of the decompositions accept `Float32`, `Float64`, `ComplexF32`, and +`ComplexF64`. Integer and `Bool` inputs are converted to `Float64`. As +everywhere else in the package, that conversion needs `@allowpromotion` (or +`allowpromotion`) only when it widens the element type, so `Int64` and `UInt64` +pass through silently while `Int32`, smaller integers, and `Bool` do not. + +## Which entry point to use + +Single matrices go through the standard `LinearAlgebra` functions and return the +standard factorization objects. Stacks of matrices have no `LinearAlgebra` +equivalent, so they get their own `batched_*` names. + +| Single matrix (2D) | Returns | Stack of matrices (3D) | +| --- | --- | --- | +| `LinearAlgebra.cholesky(A)` | `Cholesky` | `cuNumeric.batched_cholesky(A)` | +| `LinearAlgebra.eigen(A)` | `Eigen` | `cuNumeric.batched_eigen(A)` | +| `LinearAlgebra.eigvals(A)` | `NDArray` | `cuNumeric.batched_eigvals(A)` | +| `LinearAlgebra.svd(A)` | `SVD` | not supported | +| `LinearAlgebra.qr(A)` | `NDArrayQR` | not supported | +| `cuNumeric.solve(A, b)`, `A \ b` | `NDArray` | `cuNumeric.batched_solve(A, B)` | + +## Distributed solves and factorizations + +The existing `A \ b`, `cuNumeric.solve`, `cholesky`, and `qr` APIs select +distributed tasks automatically. Selection follows cuPyNumeric 26.06: + +| Operation | cuSolverMp cutoff | Internal block size | +| --- | --- | --- | +| Square solve | dimension ≥ 2048 | 512 | +| Lower Cholesky | dimension ≥ 8192 | 2048 | +| Reduced QR | matrix contains ≥ 1048576 elements | 128 | + +These are the default values of preference-backed module constants. The loaded library +must support cuSolverMp and Legate must have more than one active GPU. The +library selects the algorithm; Legate handles placement, communication, and +redistribution. Configure resources before starting Julia, as described in +[Hardware Configuration](./configuration/hardware.md). + +Ordinary solve and QR tasks remain the fallback. Cholesky also uses the tiled +POTRF/TRSM/SYRK/GEMM algorithm for eligible multi-processor configurations; +below its partitioning cutoff this is a single tile. Batched operations still +distribute independent matrices, each of which must fit on one processor. +SVD and general eigen do not acquire distributed factorization paths. + +Results retain their existing Julia types, element promotion, and shapes. +Inputs are preserved. Kernel failures propagate when the runtime reports them; +failed collectives are not retried using another algorithm. No new factor-reuse, +triangular-solve, or CG API is introduced by this change. + +This requires the new C++ wrapper; see [Developer Mode](./developer_mode.md#cusolvermp-wrapper-update). + +### Tuning + +Use [`CNPreferences.set_linalg!`](./api_preferences.md#linear-algebra) to set any +of the seven documented constants. `MIN_*_MATRIX_SIZE` controls when an operation +can select cuSolverMp. Solve and Cholesky use the row count; QR uses the number +of matrix elements. `MIN_SOLVE_TILE_SIZE`, `MIN_CHOLESKY_TILE_SIZE`, and +`QR_TILE_SIZE` set the solver block sizes. QR uses the same block size on both axes. + +For the tiled Cholesky fallback, `MIN_CHOLESKY_MATRIX_SIZE` also sets the +single-tile cutoff. `MIN_CHOLESKY_TILE_SIZE` guides tile subdivision, and +`MAX_CHOLESKY_TILES_PER_PROC` limits the number of tiles per matrix axis relative +to the processor count. These are tuning heuristics, not memory limits. + +The constants are loaded in `cuNumeric.jl` through `load_preference(CNPreferences, ...)`. +Changing them requires a fresh Julia process. Library capability, configured +GPU/processor counts, and MP eligibility are cached once during runtime startup +in a typed `const Ref`. Solves reuse that configuration; size checks still happen +per operation. This assumes the configured machine stays fixed for the runtime's +lifetime. Scoped processor subsets would require revisiting this cache. ## Matrix multiply @@ -29,63 +97,300 @@ Pages = ["ndarray/binary.jl"] Filter = t -> t isa Function && nameof(t) === :mul! ``` -## Solve (batched) +## Vector operations + +Supported `LinearAlgebra` operations include: + +- `mul!(y, A, x)` and `mul!(y, A, x, α, β)` for matrix-vector products; `A * x` is also supported. +- `mul!(C, A, B, α, β)` for matrix-matrix products. +- `dot(x, y)` and entrywise `norm(A, p)` (dense-array norms currently require a GPU). +- `axpy!`, `axpby!`, `lmul!`, and `rmul!`. +- `ldiv!(y, D, x)` and `ldiv!(D, x)` for an NDArray-backed `Diagonal`. -`cuNumeric.solve(A, b)` solves linear systems and returns an array with the same -shape as `b`. `A` has shape `(..., m, m)` and `b` has shape `(..., m)` or -`(..., m, n)`. +`dot` and dense-array `norm` return [device scalars](api_cnscalar.md). + +## Solve + +`cuNumeric.solve(A, b)` solves a linear system and returns an array with the same +shape as `b`. `A` is a square `m × m` matrix and `b` has shape `(m,)` or +`(m, n)`. `A \ b` is equivalent. ```julia A = cuNumeric.rand(Float32, 64, 64) b = cuNumeric.rand(Float32, 64) -x = cuNumeric.solve(A, b) +x = cuNumeric.solve(A, b) # or: A \ b B = cuNumeric.rand(Float32, 64, 4) -X = cuNumeric.solve(A, B) +X = A \ B +``` -As = cuNumeric.rand(Float32, 8, 32, 32) -Bs = cuNumeric.rand(Float32, 8, 32, 2) -Xs = cuNumeric.solve(As, Bs) +## Cholesky + +`LinearAlgebra.cholesky(A)` returns a `Cholesky` object holding the lower factor +`L`, with `A ≈ L * L'`. + +```julia +using LinearAlgebra + +A = cuNumeric.rand(Float32, 64, 64) +A = A * cuNumeric.transpose(A) + 64 * NDArray{Float32}(I, 64, 64) # make it SPD + +F = cholesky(A) +L = F.L # LowerTriangular view of F.factors ``` -## Singular value decomposition +Three things differ from Base: -`cuNumeric.svd(A, full_matrices=true)` returns `(U, S, Vh)` for a 2D `m × n` -array. With `k = min(m, n)`, the output shapes are: +- The input is **not** checked for being Hermitian; only its lower triangle is + read. There are no `Hermitian` / `Symmetric` overloads. +- A non-positive-definite input raises an `ErrorException` carrying the task's + `"Matrix is not positive definite"` message, not a `PosDefException`. The + `check` keyword is therefore not supported. +- `F.U` (and hence destructuring as `L, U = F`) needs `copy(F.factors')`, which + falls back to scalar indexing until `adjoint(::NDArray)` is implemented. Use + `F.L` or `F.factors`. -- Full: `U` is `m × m`, `S` has length `k`, and `Vh` is `n × n`. -- Thin: `U` is `m × k`, `S` has length `k`, and `Vh` is `k × n`. +## Eigendecomposition + +`LinearAlgebra.eigen(A)` returns an `Eigen` object; `eigvals` and `eigvecs` +return the pieces individually. Column `j` of the eigenvector matrix is the +eigenvector for eigenvalue `j`. + +```julia +A = cuNumeric.rand(Float64, 64, 64) + +F = eigen(A) +values, vectors = F # or: eigvals(A), eigvecs(A) +``` + +Eigenvalues and eigenvectors are **always complex**, even when `A` is real with +real eigenvalues: `ComplexF32` for `Float32` / `ComplexF32` input and +`ComplexF64` otherwise. This follows the underlying LAPACK `geev` path and +differs from Base, which returns real factors for such input. CUDA.jl's +non-symmetric branch behaves the same way. + +On a GPU this requires `cusolverDnXgeev`, added in CUDA 12.6.2. When it is +missing, the eigen entry points throw an error rather than silently falling back +to host execution. + +The symmetric/Hermitian path (`eigh`, backed by the `SYEV` task) is not wired up +yet; it needs `Hermitian` / `Symmetric` support on `NDArray` first. + +## Singular value decomposition + +`LinearAlgebra.svd(A; full=false)` returns an `SVD` object with +`A ≈ F.U * Diagonal(F.S) * F.Vt`. ```julia A = cuNumeric.rand(Float32, 128, 64) -U, S, Vh = cuNumeric.svd(A, false) +F = svd(A) +U, S, Vt = F.U, F.S, F.Vt ``` +With `k = min(m, n)`, the output shapes are: + +- Thin (`full=false`, the default): `U` is `m × k`, `S` has length `k`, and `Vt` + is `k × n`. +- Full (`full=true`): `U` is `m × m`, `S` has length `k`, and `Vt` is `n × n`. + +Note that `full=false` is the Julia default, whereas numpy's +`full_matrices=True` is not. Destructuring as `U, S, V = F` also gives you the +*adjoint* of `F.Vt`, and `F.V` is a lazy `Adjoint` wrapper, so operating on it +falls back to scalar indexing until `adjoint(::NDArray)` is implemented. Prefer +`F.Vt`. + `S` is real-valued for both real and complex inputs. +The backend only factors tall or square matrices (`m >= n`), matching +cupynumeric. A wide input throws `ArgumentError`. + ## QR decomposition -`cuNumeric.qr(A)` returns the economy-size factors `(Q, R)` for a 2D `m × n` -array. With `k = min(m, n)`, `Q` is `m × k` and `R` is `k × n`. +`LinearAlgebra.qr(A)` returns an `NDArrayQR`, holding the economy-size factors +with `A ≈ F.Q * F.R`. For a 2D `m × n` input and `k = min(m, n)`, `Q` is `m × k` +and `R` is `k × n`. ```julia A = cuNumeric.rand(Float32, 128, 64) -Q, R = cuNumeric.qr(A) +F = qr(A) +Q, R = F # or: F.Q, F.R ``` -SVD and QR currently accept only 2D arrays; batched decompositions are not -supported. +`NDArrayQR` is a `LinearAlgebra.Factorization` but not one of Base's QR types. +Base's default `QRCompactWY` stores a blocked Householder representation and +`QR` stores `factors` plus `τ`; the backend runs `geqrf` followed by `orgqr` and +discards `τ`, so neither is constructible. The practical difference is that +`F.Q` is a materialized `NDArray` rather than a lazy `QRCompactWYQ`. + +SVD and QR accept only 2D arrays; there are no batched versions. + +## Batched decompositions + +The `batched_*` functions apply an operation to every matrix in a stack. They +take exactly one batch dimension, i.e. shape `(b, m, m)`, and return plain +tuples or arrays rather than factorization objects. + +```julia +As = cuNumeric.rand(Float32, 8, 32, 32) +Bs = cuNumeric.rand(Float32, 8, 32, 2) + +Xs = cuNumeric.batched_solve(As, Bs) # (8, 32, 2) +Ls = cuNumeric.batched_cholesky(As) # (8, 32, 32), needs SPD blocks +values, vectors = cuNumeric.batched_eigen(As) # (8, 32) and (8, 32, 32) +``` + +Each matrix is factored on a single processor, so this is a good fit for many +small matrices and a poor one for a few large ones. Two or more batch dimensions +are rejected: the `POTRF` task body is only instantiated up to three dimensions, +and Legate.jl builds launch domains for at most three dimensions. + +Every caveat listed under Cholesky and Eigendecomposition applies here too. ## Helpers These helpers live on `NDArray` and are also listed in the Public API: - `cuNumeric.transpose` -- `cuNumeric.eye` - `cuNumeric.diag` (2D to 1D) - `cuNumeric.trace` +## Diagonal and identity + +Prefer structured `LinearAlgebra` types over materializing a full matrix. + +Wrap a 1D `NDArray` in `Diagonal` for scale / solve / inverse along a diagonal. +Use `LinearAlgebra.I` (`UniformScaling`) for `A + I`, `D + I`, and `A * I`. +`D + I` stays a `Diagonal`; `A + I` returns a dense `NDArray`. + +```julia +using LinearAlgebra +using cuNumeric + +d = cuNumeric.ones(Float32, 64) +D = Diagonal(d) # preferred: keep diagonal structure +A = cuNumeric.rand(Float32, 64, 64) +v = cuNumeric.rand(Float32, 64) + +y = D * v # scale a vector +B = D * A # scale rows +X = D \ A # scale columns by 1 ./ d +Di = D + I # still Diagonal +C = A + I # dense NDArray +``` + +### Supported `Diagonal{<:NDArray}` APIs + +These paths stay on-device (no host densify for the math). RHS / other operands +must be `NDArray` unless noted. + +**Construction / display** + +- `Diagonal(v::NDArray{<:Any,1})` — wrap without copying +- `Diagonal(A::NDArray{<:Any,2})` — `Diagonal(diag(A))` +- `Matrix(D)` / `Matrix{T}(D)` — densify to a host `Matrix` (conversion only) +- `show` — densifies `.diag` for printing only + +**Multiply / divide / inverse** + +- `D * A`, `A * D`, `D * v` for 2D / 1D `NDArray` +- `mul!`, `lmul!`, `rmul!` with `NDArray` +- `D \ B`, `A / D`, `ldiv!`, `rdiv!` with `NDArray` +- `inv(D)` — reciprocal on-device; zeros become Inf (no `SingularException`) +- `det(D)` — `CNScalar` product of the diagonal + +**`NDArray` ± `Diagonal`** + +- `A + D`, `D + A`, `A - D`, `D - A` for square 2D `NDArray` + +**UniformScaling (`I`)** + +- `NDArray{T}(I, m, n)` / `NDArray(I, …)`, `copyto!(A, I)`, `one(A)`, `oneunit(A)` +- `A ± I`, `A * I`, `I * A` +- `D ± I`, `D * I`, `I * D`, `copyto!(D, I)` — `D ± I` stays `Diagonal` + +**Broadcast** + +- Structure- or zero-preserving broadcasts on `Diagonal` (e.g. `D .* c`, + `D .*= c`, `D .+ D`) lower to 1D broadcast on `.diag` +- Densifying out-of-place broadcasts (e.g. `D .+ 1`, `D .+ A`) materialize a + dense `NDArray`, matching Base’s densify-to-`Matrix` behavior +- In-place densifying writes into `Diagonal` (e.g. `D .+= 1`, `D .+= A`) still + throw `ArgumentError` (off-diagonal / densify), matching Base + +**Eigen / reductions / predicates / norms** + +- `eigvals(D)`, `eigen(D)`, `eigvecs(D)` — unsorted; values are a copy of the + diagonal (`NDArray`), vectors are `NDArray` identity. Keyword `sortby` is not + supported on this method. +- `tr`, `sum`, `prod`, `maximum`, `minimum` — `CNScalar` (not a Julia scalar) +- `iszero`, `isone`, `istriu`, `istril`, `ishermitian`, `issymmetric`, `isposdef` — 0-dimensional `NDArray{Bool}` +- `opnorm(D)` / `opnorm(D, p)` for `p ∈ {1, 2, Inf}` — `CNScalar` +- `norm(D)` / `norm(D, p)` for finite `p` (including `±Inf`); off-diagonals are zero — `CNScalar` +- `cond(D)` / `cond(D, p)` for `p ∈ {1, 2, Inf}` — `CNScalar` +- `logdet(D)` for real `Diagonal` only — `CNScalar` + +**Helpers on dense `NDArray`** + +- `cuNumeric.diag` / `LinearAlgebra.diag` (2D → 1D), `cuNumeric.diagonal` (two + axes of an `N`-D array → rank `N-1`), `cuNumeric.trace` / `LinearAlgebra.tr` + (2D square → 0D) +- `iszero(A)` — all elements `== zero(T)` → 0-dimensional `NDArray{Bool}` +- `isone(A)` — square 2D vs `_eye(T, n)` → 0-dimensional `NDArray{Bool}` (non-square → `false`) + +### Unsupported / fallthrough + +Other `LinearAlgebra` operations on `Diagonal{<:NDArray}` (for example `svd`, +`svdvals`, `pinv`, `logabsdet`, complex `logdet`, `kron`, `cholesky`, host +`AbstractArray` RHS for `\` / `/` / `ldiv!` / `rdiv!`, or `eigen(...; sortby)`) +are **not** specially implemented. They fall through to Base and typically fail +with the package’s scalar-indexing error (NDArray does not support scalar +indexing without `@allowscalar`). There are no “not implemented” `ArgumentError` +stubs for these. + +Only densify when you truly need a full identity matrix: + +```julia +E = NDArray{Float32}(I, 64, 64) # dense identity +copyto!(A, I) # fill an existing array +E2 = one(A) # same shape / eltype as A +``` + +Avoid building a dense identity (or densifying `D` with `Matrix(D)`) just to +scale or shift; prefer `Diagonal` and `I` instead. There is no public `eye`. + +## Krylov.jl CG + +Loading `Krylov` activates the cuNumeric extension for runtime-backed scalar +coefficients and CG workspace allocation. Allow automatic fetching for the +solver's convergence decisions: + +```julia +using cuNumeric, Krylov + +A = NDArray([4.0 1.0; 1.0 3.0]) +b = NDArray([1.0, 2.0]) +x, stats = @allowautofetch Krylov.cg(A, b; rtol=1e-8) +``` + +For complex systems, use `@allowpromotion @allowautofetch Krylov.cg(...)`: +CG computes real scalar coefficients that must promote when scaling complex +vectors. + +The extension forwards `kaxpy!` and `kaxpby!` with `DeviceScalar` coefficients +to cuNumeric's `LinearAlgebra` methods. See Krylov's +[documented custom-vector helpers](https://jso.dev/Krylov.jl/stable/custom_workspaces/#Methods-to-overload-for-compatibility-with-Krylov.jl). +The `HaloVector` in that example illustrates a custom vector type; this +extension defines the corresponding methods for `NDArray`. + +To reuse allocations, construct `workspace = Krylov.CgWorkspace(A, b)` and call +`@allowautofetch Krylov.cg!(workspace, A, b)`. Other Krylov solvers may require +additional integration methods. + ## Not available yet -There is no public `cholesky`, `eig`, `lu`, matrix `inv`, or `ldiv!` yet. -Elementwise `inv` / `^-1` are unary operations, not matrix inverse. +There is no public dense-matrix `lu`, matrix `inv`, or `ldiv!` yet (beyond the +`Diagonal` / `NDArray` paths listed above). Elementwise `inv` / `^-1` are unary +operations, not matrix inverse. + +Also missing: `eigh` / Hermitian eigen (needs `Hermitian` and `Symmetric` +support on `NDArray`), and batched SVD and QR. diff --git a/docs/src/perf/kernel_fusion.md b/docs/src/perf/kernel_fusion.md index 04c437742..8e2fb2d59 100644 --- a/docs/src/perf/kernel_fusion.md +++ b/docs/src/perf/kernel_fusion.md @@ -1,76 +1,62 @@ # Kernel Fusion -On CUDA, nested broadcast expressions are fused into a single PTX kernel when fusion is enabled (the default). There is no separate `@fuse` macro. You write ordinary Julia broadcast code, and cuNumeric compiles eligible trees into one kernel instead of launching one op at a time. +When CUDA is available cuNumeric.jl provides several methods to fuse operations into a single CUDA kernel. This can greatly improve performance and should be used whenever possible. -Prefer Julia's `@.` macro for multi-op elementwise expressions. Placing `.` on every operator by hand is easy to get wrong: missing a dot on unary negation or addition silently changes the meaning, and it can also break fusion by splitting work into the wrong ops. +## Automatic Broadcast Fusion + +Eligible broadcast expressions (i.e., `z .= cos.(x) .+ y`) are automatically fused into a single CUDA kernel. + +We reccomend using Julia's `@.` macro to ensure the entire expression gets fused. A missing dot can result in poor performance! ```julia -# Easy to miss the dot on negation +# A missing dot on unary negation changes the expression and can prevent fusion. y .= .-a .+ b .* c -# Prefer: @. dots every operator, which is clearer and fusion-friendly +# Prefer this form. y .= @. -a + b * c ``` -## Avoid preallocated intermediate broadcast buffers +Automatic kernel fusion requires that: -Preallocation is useful for a final output or a buffer that must persist across -iterations. It can be counterproductive for a single-use intermediate inside -`@analyze_lifetimes`, however. An in-place `.=` assignment is an observable -mutation, so inter-statement broadcast fusion treats it as a kernel boundary: +```@raw html +
    +
  1. Arrays have the same shape. Shape-mismatched broadcasts such as matrix .+ vector use the unfused path.
  2. +
  3. Expressions have at least two broadcast operations. Expressions like y .= cos.(x) with a single operation are unfused to reduce compilation overhead. This setting can be modified by calling CNPreferences.set_broadcast_fusion_min_ops!(x::Int) before launching cuNumeric. The default is x = 2.
  4. +
  5. A CUDA device is available.
  6. +
+``` +Broadcast fusion can be disabled or re-enabled with CNPreferences as well: ```julia -tmp = cuNumeric.zeros(Float32, N, N) -result = cuNumeric.zeros(Float32, N, N) - -@analyze_lifetimes begin - tmp .= @. A + B - result .= @. tmp * C + 1.0f0 -end +CNPreferences.enable_broadcast_fusion!() # default +CNPreferences.disable_broadcast_fusion!() ``` -This materializes `tmp` before the second expression and requires separate -kernel launches. Instead, use an ordinary assignment for a single-use -intermediate and keep `.=` for the final destination: +For more information on setting preferences see the CNPreferences [CNPreferences documentation](../api_preferences.md). -```julia -result = cuNumeric.zeros(Float32, N, N) +## Fuse Multiple Expressions with `@accelerate` -@analyze_lifetimes begin +One function of the `@accelerate` macro is to analyze code and find temporary variables which can be elided via kernel fusion. In the example below `tmp` is unused outside of the `update!` function, so `@accelerate` will merge the two lines together. + +```julia +@accelerate function update!(result, A, B, C) tmp = @. A + B result .= @. tmp * C + 1.0f0 + return result end ``` -The inter-statement pass can substitute `tmp` into its only consumer, producing -the equivalent of `result .= @. (A + B) * C + 1.0f0`. The intermediate is never -materialized, so the full expression can run as one fused kernel. - -This rewrite is intentionally conservative: the intermediate must have one -use, no intervening statement may invalidate its inputs, and all normal fusion -requirements still apply. Keep preallocation when an intermediate is reused, -must preserve mutation semantics, or cannot be fused. Use -[`BCAST_FUSION_DEBUG`](../debugging.md#inspect-fused-broadcasts-with-bcast_fusion_debug) -to confirm whether the rewrite occurred. - -Fusion applies when CUDA is available, the array leaves share the same shape, and the expression has at least `FUSE_BROADCAST_MIN_OPS` ops (default 2). Otherwise cuNumeric falls back to evaluating one op at a time. Shape-mismatched broadcasts such as `matrix .+ vector` use the unfused path. +The equivalent code is `result .= @. (A + B) * C + 1.0f0`. The rewrite is conservative: the producer must have one use, no intervening statement may invalidate its inputs, and the normal fusion requirements still apply. -Toggle fusion through `CNPreferences` (restart Julia after changing these): +This pattern requires the intermediate object to be temporary, and not pre-allocated. For example, if `tmp` was externally managed storage whose values were modified with `.=`, `@accelerate` would not be able to combine the expressions. For example, the following code would remain as two kernels. ```julia -using CNPreferences - -CNPreferences.enable_broadcast_fusion!() # default -CNPreferences.disable_broadcast_fusion!() -CNPreferences.set_broadcast_fusion_min_ops!(2) # default -CNPreferences.set_broadcast_fusion_min_ops!(1) # also fuse single-ops +tmp .= @. A + B +result .= @. tmp * C + 1.0f0 ``` -What `set_broadcast_fusion_min_ops!` controls: - -- **`2` (default):** only trees with two or more ops fuse. Example: `y .= @. a * b + c` can fuse; `y .= cos.(x)` does not. Keeping single-ops on the unfused C-API path avoids PTX compile overhead when there is little to gain. -- **`1`:** every eligible broadcast can fuse, including unary / single-op forms. Prefer this when you want uniform fused behavior (for example in tests) rather than for typical apps. +More details on `@accelerate` can be found in the [memory management](./reduce_allocations.md) docs. -The threshold counts `Broadcasted` nodes in the expression tree. Set it through `CNPreferences`, then restart Julia. See [CNPreferences](../api_preferences.md). +## Other -To inspect a fused launch or a lifetime rewrite, see [Debugging](../debugging.md). For the implementation pipeline, see [Internals](../internals.md). +To inspect the kernels emitted by broadcast fusion see these docs: [`BCAST_FUSION_DEBUG`](../debugging.md#inspect-fused-broadcasts-with-bcast_fusion_debug). diff --git a/docs/src/perf/patterns_to_avoid.md b/docs/src/perf/patterns_to_avoid.md index 3fe864acc..28f6fb2c9 100644 --- a/docs/src/perf/patterns_to_avoid.md +++ b/docs/src/perf/patterns_to_avoid.md @@ -4,6 +4,12 @@ Accessing elements of an NDArray one at a time (e.g., `arr[5]`) is slow and should be avoided. Indexing like this requires data to be transferred between device and host and maybe even communicated across nodes. Scalar indexing will emit an error which can be opted out of with `@allowscalar` or `allowscalar() do ... end`. Several functions in the existing API invoke scalar indexing and are intended for testing (e.g., the `==` operator). +`fetch` retrieves a native Julia scalar from a `CNScalar` or a single-element +`NDArray`, waiting for the result as needed. Prefer a native Julia number when +you already have one, and leave reductions / fully contracted `@tensor` results +in runtime-managed storage until you need the host value. +See [Scalars](../api_tensor.md#Scalars). + ## Implicit promotion Mixing integral types of different size (e.g., `Float64` and `Float32`) will result in implicit promotion of the smaller type to the larger types. This creates a copy of the data and hurts performance. Implicit promotion from a smaller integral type to a larger integral type will emit an error which can be opted out of with `@allowpromotion` or `allowpromotion() do ... end`. This error is common when mixing literals with `NDArrays`. By default a floating point literal (i.e., 1.0) is `Float64` but the default type of an `NDArray` is `Float32`. diff --git a/docs/src/perf/reduce_allocations.md b/docs/src/perf/reduce_allocations.md index 37b38ca50..a36eabb86 100644 --- a/docs/src/perf/reduce_allocations.md +++ b/docs/src/perf/reduce_allocations.md @@ -1,31 +1,93 @@ -# Reduce Allocations +# `@accelerate` -Every intermediate `NDArray` (from a slice, broadcast, or function call) allocates a fresh buffer and waits for the Julia GC to free it. Because the GC runs on memory pressure, many dead buffers accumulate and pressure cuNumeric's allocator. +`@accelerate` optimizes array operations with no branches, loops, jumps, `try`, or nested functions (i.e., straight-line code). Ordinary function calls are opaque boundaries, general control flow is not rewritten. -`@analyze_lifetimes` performs a **static last-use analysis** at macro-expansion time and inserts eager `maybe_insert_delete` calls immediately after each temporary's final use. Freed buffers can then be reused by later same-sized allocations instead of waiting on GC. +Given those contraints, `@accelerate` will: -When broadcast fusion is on, intermediate dotted nodes in a broadcast tree are not real `NDArray` allocations. The macro accounts for that automatically. +```@raw html +
    +
  1. Fuse broadcast expressions. On CUDA, an eligible dotted expression such as @. A + B * C can run as one kernel. CPU execution uses the normal unfused path.
  2. +
  3. Fuse across broadcast statements. A single-use broadcast result can be substituted into its consumer, producing fewer GPU kernel launches.
  4. +
  5. Temporary lifetime analysis. After rewriting the code, the macro releases materialized, non-returned NDArrays after their final use on CPU or GPU.
  6. +
+``` +These jobs must happen together: an intermediate that fuses into its consumer is never allocated, while an intermediate that cannot fuse is materialized and then released after its last use. + +## `@accelerate` on Functions + +We reccomend using `@accelerate` on function definitions as they naturally separate inputs/ouputs (i.e., what should be kept) from everything else (i.e., something we can eagerly delete). We provide other forms (see below), but find this the most natural. ```julia -T = Float32 -A = cuNumeric.ones(T, (N, N)) -B = cuNumeric.ones(T, (N, N)) -C = cuNumeric.zeros(T, (N, N)) - -@analyze_lifetimes begin - result = @. A[1:end, :] + B[1:end, :] - C .= @. result * 2.0f0 +@accelerate function update!(C, A, B) + combined = @. A + B + C .= @. 2.0f0 * combined + return C end ``` -**Benchmark** (Gray-Scott reaction-diffusion, 512×512, 10 000 steps): +On an eligible GPU path, `combined` will be folded into the second broadcast so the chain runs as one kernel. On CPU, or when fusion is ineligible, `combined` is materialized and *immediately* released after the update (we do not wait for Julia GC). Function arguments belong to the caller and are never released by `@accelerate`. Returned values also remain valid. + +## Other Forms + +The other forms of `@accelerate` provide other mechanisms to indicate which values must remain available, which in turn determines how aggressively the macro may fuse or release intermediates. + +| Form | Use it when | Fusion and lifetime behavior | +| :--- | :--- | :--- | +| `@accelerate function ... end` | Defining reusable array code. This is the recommended default. | Arguments and returned values are protected. Non-returned locals may fuse into consumers or be released after their last use. | +| `@accelerate begin ... end` | Named results must remain in the current scope. | `begin` creates no new Julia scope, so every named binding is protected. An eligible same-shape CUDA chain may still use one multi-output kernel, but each named result is materialized. | +| `@accelerate let ... end` | Writing a one-off multi-statement calculation when only its result is needed. | `let` creates a local scope. Only the result escapes; other locals may fuse away or be released after their last use. | +| `@accelerate expr` | Evaluating one expression without named intermediates. | The result is materialized and returned. Eligible operations fuse within the expression, and transient temporaries are released. | + +For example, nested scope lets a private intermediate feed a value that is also +used by the outer block: + +```julia +@accelerate begin + shifted = let + product = @. A * B + @. product + 1 + end + x = @. shifted * C +end + +consume(shifted, x) +``` + +Choose `let` when only the final result should escape: +```julia +result = @accelerate let + product = @. A * B + @. product + 1 +end ``` - user system elapsed CPU max RSS -without 106.50 s 23.87 s 58.66 s 222% 3786 MB -with 61.74 s 13.66 s 27.84 s 270% 2999 MB + +For a single unnamed expression, use: + +```julia +result = @accelerate (@. A + B * C) ``` -~2× wall-clock speedup and ~800 MB lower peak memory with no algorithmic changes. +## Writing an accelerated body + +- Apply `@.` to each elementwise right-hand side. Applying it to the entire body would change `x = ...` into `x .= ...` and `f(...)` into `f.(...)`. +- Use ordinary `=` for a disposable intermediate. This allows a single-use producer to fuse into its consumer. +- Use `.=` when the mutation must be visible. The destination write is preserved, although an eligible producer may fuse into it. +- Keep control flow outside the accelerated body. Loops, conditionals, `try`, short-circuit operators, and nested functions are rejected. +- Ordinary function calls run in program order and form rewrite boundaries. Annotate the called function separately if its body should also be accelerated. + +The `@accelerate` macro is great for fusing hot-loops where GC or kernel launch overhead should be minimized. + +```julia +@accelerate function update!(C, A, B) + combined = @. A + B + C .= @. 2.0f0 * combined + return C +end + +for _ in 1:nsteps + update!(C, A, B) +end +``` -Use `@show_lifetimes` to print the rewrite without running it ([Debugging](../debugging.md)). For how the rewriter and GC heuristics work, see [Internals](../internals.md). +See [Kernel Fusion](./kernel_fusion.md) for CUDA fusion requirements and [`@show_lifetimes`](../debugging.md#inspect-lifetime-rewrites-with-show_lifetimes) to inspect the exact rewrite without executing it. diff --git a/examples/Project.toml b/examples/Project.toml index 9153064ac..fa6e87e75 100644 --- a/examples/Project.toml +++ b/examples/Project.toml @@ -1,7 +1,10 @@ [deps] CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" Legate = "1238f2cf-6593-4d60-9aca-2f5364e49909" +LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" Plots = "91a5bcdd-55d7-5caf-9e0b-520d859cae80" +Printf = "de0858da-6303-5e67-8744-51eddeeeb8d7" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" [extras] diff --git a/examples/custom_cuda.jl b/examples/custom_cuda.jl index 6092642bb..c968b29ae 100644 --- a/examples/custom_cuda.jl +++ b/examples/custom_cuda.jl @@ -3,6 +3,8 @@ using cuNumeric using CUDA import CUDA: i32 +cuNumeric.Experimental(true) + function kernel_add(a, b, c, N) i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x if i <= N @@ -19,25 +21,26 @@ function kernel_sin(a, b, N) return nothing end -N = 1024 -threads = 256 -blocks = cld(N, threads) +function run_custom_cuda(N=1024) + threads = 256 + blocks = cld(N, threads) + a = cuNumeric.fill(1.0f0, N) + b = cuNumeric.fill(2.0f0, N) + c = cuNumeric.zeros(Float32, N) + n_scalar = UInt32(N) -a = cuNumeric.fill(1.0f0, N) -b = cuNumeric.fill(2.0f0, N) -c = cuNumeric.ones(Float32, N) + add_task = cuNumeric.@cuda_task kernel_add(a, b, c, n_scalar) + cuNumeric.@launch task=add_task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=n_scalar -# task = cuNumeric.@cuda_task kernel_add(a, b, c, UInt32(1)) -# cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=UInt32(N) -# allowscalar() do -# c_cpu = c[:] -# println("Result of c after kenel launch: ", c_cpu[1]) -# end + sin_task = cuNumeric.@cuda_task kernel_sin(c, b, n_scalar) + cuNumeric.@launch task=sin_task threads=threads blocks=blocks inputs=c outputs=b scalars=n_scalar -task = cuNumeric.@cuda_task kernel_sin(a, b, UInt32(1)) -cuNumeric.@launch task=task threads=threads blocks=blocks inputs=a outputs=b scalars=UInt32(N) + allowscalar() do + return println("sin(1 + 2) = ", b[1]) + end + return b +end -allowscalar() do - b_cpu = b[:] - println("Result of b after kenel launch: ", b_cpu[1]) +if abspath(PROGRAM_FILE) == @__FILE__ + run_custom_cuda() end diff --git a/examples/data/gray-scott.h5 b/examples/data/gray-scott.h5 new file mode 100644 index 000000000..09bf0a657 Binary files /dev/null and b/examples/data/gray-scott.h5 differ diff --git a/examples/daxpy.jl b/examples/daxpy.jl deleted file mode 100644 index f7983fff0..000000000 --- a/examples/daxpy.jl +++ /dev/null @@ -1,11 +0,0 @@ -# found in examples/daxpy.jl -using cuNumeric - -arr = cuNumeric.rand(20) - -α = 1.32f0 -b = 2.0f0 - -arr2 = @. α * arr + b - -println(arr2) diff --git a/examples/dmd.jl b/examples/dmd.jl new file mode 100644 index 000000000..03b20f855 --- /dev/null +++ b/examples/dmd.jl @@ -0,0 +1,80 @@ +#= Dynamic mode decomposition of the Gray-Scott snapshots. + +Reads gray-scott.h5 from the working directory if examples/gray-scott.jl has been +run, otherwise the smaller sample committed under examples/data (a 64x64 grid +sampled every 100 of 3000 steps). + +DMD fits the best linear operator A with x_{k+1} ≈ A x_k across the snapshots. +A itself is N²×N² and never formed: the SVD of the snapshot matrix gives a rank-r +subspace, A is projected onto it, and the eigendecomposition of that small r×r +matrix carries the dynamics. Each eigenvalue λ says what its spatial mode does +from one snapshot to the next — |λ| is growth or decay, arg(λ) is rotation. +=# + +using cuNumeric +using LinearAlgebra +using Printf + +const SNAPSHOT_FILE = "gray-scott.h5" +const SAMPLE_FILE = joinpath(@__DIR__, "data", "gray-scott.h5") + +""" + dmd(X, r) + +Eigenvalues and exact DMD modes of the snapshot matrix `X`, truncated to rank `r`. +Columns of `X` are successive states. +""" +function dmd(X::NDArray{Float32,2}, r::Int) + n = size(X, 2) + X1 = X[:, 1:(n - 1)] # states x_1 … x_{n-1} + X2 = X[:, 2:n] # the same states advanced one snapshot + + F = svd(X1) + r = min(r, length(F.S)) + U = F.U[:, 1:r] + Vt = F.Vt[1:r, :] + + # Σ⁻¹ stays a Diagonal instead of a dense r×r matrix. + Sinv = Diagonal(1.0f0 ./ F.S[1:r]) + + # X2 V Σ⁻¹ appears in both the projected operator and the exact modes. + B = X2 * cuNumeric.transpose(Vt) * Sinv + A_tilde = cuNumeric.transpose(U) * B + + E = eigen(A_tilde) # always complex, even for a real à + Φ = cuNumeric.as_type(B, ComplexF32) * E.vectors + + return E.values, Φ +end + +function main() + path = isfile(SNAPSHOT_FILE) ? SNAPSHOT_FILE : SAMPLE_FILE + println("reading $path") + + X = cuNumeric.h5read(path, "u") + n_points, n_snapshots = size(X) + N = isqrt(n_points) + println("$n_snapshots snapshots of $n_points points") + + λ, Φ = dmd(X, 20) + + # Only r eigenvalues, so ranking them on the host costs nothing. + vals = Array(λ) + order = sortperm(abs.(vals); rev=true) + + println("\nmode |λ| cycles/snapshot") + for i in order[1:min(5, end)] + @printf("%4d %6.4f %+8.4f\n", i, abs(vals[i]), angle(vals[i]) / 2π) + end + + # The slowest-decaying mode is the pattern the simulation settles into. + lead = order[1] + mode = cuNumeric.reshape(abs.(Φ[:, lead:lead]), (N, N)) + cuNumeric.h5write("dmd-mode.h5", "leading", mode) + cuNumeric.Legate.runtime_sync() + println("\nwrote leading mode $lead to dmd-mode.h5") + + return λ, Φ +end + +λ, Φ = main() diff --git a/examples/gray-scott.jl b/examples/gray-scott.jl index b7eae81c6..9d2a757e3 100644 --- a/examples/gray-scott.jl +++ b/examples/gray-scott.jl @@ -1,6 +1,9 @@ using cuNumeric using Plots +# Flattened u snapshots land here for examples/dmd.jl to analyze. +const SNAPSHOT_FILE = "gray-scott.h5" + struct Params{T} dx::T dt::T @@ -10,7 +13,7 @@ struct Params{T} k::T function Params(dx=1.0f0, c_u=1.0f0, c_v=0.3f0, f=0.03f0, k=0.06f0) - new{Float32}(dx, dx/5, c_u, c_v, f, k) + return new{Float32}(dx, dx/5, c_u, c_v, f, k) end end @@ -22,10 +25,10 @@ function bc!(u_new, v_new, u, v) v_new[:, 1] = v[:, end - 1] v_new[:, end] = v[:, 2] v_new[1, :] = v[end - 1, :] - v_new[end, :] = v[2, :] + return v_new[end, :] = v[2, :] end -function step!(u, v, u_new, v_new, args::Params) +@accelerate function step!(u, v, u_new, v_new, args::Params) # calculate F_u and F_v functions F_u = ( (-u[2:(end - 1), 2:(end - 1)] .* (v[2:(end - 1), 2:(end - 1)] .^ 2)) .+ @@ -59,7 +62,7 @@ function step!(u, v, u_new, v_new, args::Params) ((args.c_v * v_lap) + F_v) * args.dt + v[2:(end - 1), 2:(end - 1)] # Apply periodic boundary conditions - bc!(u_new, v_new, u, v) + return bc!(u_new, v_new, u, v) end function gray_scott() @@ -72,12 +75,16 @@ function gray_scott() n_steps = 2000 # number of steps to take frame_interval = 200 # steps to take between making plots + snapshot_interval = 20 # steps to take between saved snapshots u = cuNumeric.ones(dims) v = cuNumeric.zeros(dims) u_new = cuNumeric.zeros(dims) v_new = cuNumeric.zeros(dims) + # One flattened frame per column, the layout DMD wants. + snapshots = cuNumeric.zeros(Float32, N * N, n_steps ÷ snapshot_interval) + u[1:15, 1:15] = cuNumeric.rand(Float32, 15, 15) v[1:15, 1:15] = cuNumeric.rand(Float32, 15, 15) @@ -88,12 +95,22 @@ function gray_scott() u, u_new = u_new, u v, v_new = v_new, v + if n%snapshot_interval == 0 + snapshots[:, n ÷ snapshot_interval] = cuNumeric.reshape(u, (N * N, 1)) + end + if n%frame_interval == 0 heatmap(Array(u); clims=(0, 1)) frame(anim) end end gif(anim, "gray-scott.gif"; fps=10) + + cuNumeric.h5write(SNAPSHOT_FILE, "u", snapshots) + # h5write is asynchronous, so flush before another process opens the file. + cuNumeric.Legate.runtime_sync() + println("wrote $(size(snapshots, 2)) snapshots to $SNAPSHOT_FILE") + return u, v end diff --git a/examples/gray-scott.py b/examples/gray-scott.py index dce4e6cab..5eab0b8c8 100644 --- a/examples/gray-scott.py +++ b/examples/gray-scott.py @@ -1,82 +1,54 @@ -# python equivalent of gray-scott.jl to test the GC problem +"""cuPyNumeric equivalent of examples/gray-scott.jl.""" import cupynumeric as np -# import matplotlib.animation as animation -# from IPython.display import HTML -# import matplotlib.pyplot as plt - -def greyScottSys(u, v, dx, dt, c_u, c_v, f, k): - # u,v are arrays - # dx,dt are space and time steps - # c_u, c_v, f, k are constant paramaters - - #create new u array +def step(u, v, dx, dt, c_u, c_v, feed, kill): u_new = np.zeros_like(u) v_new = np.zeros_like(v) - #calculate F_u and F_v functions - F_u = (-u[1:-1,1:-1]*(v[1:-1,1:-1]**2)) + f*(1-u[1:-1,1:-1]) - F_v = (u[1:-1,1:-1]*(v[1:-1,1:-1]**2)) - (f+k)*v[1:-1,1:-1] - - # 2-D Laplacian of f using array slicing, excluding boundaries - # For an N x N array f, f_lap is the N-1 x N-1 array in the "middle" - u_lap = (u[2:,1:-1] - 2*u[1:-1,1:-1] + u[:-2,1:-1]) / dx**2\ - + (u[1:-1,2:] - 2*u[1:-1,1:-1] + u[1:-1,:-2]) / dx**2 - v_lap = (v[2:,1:-1] - 2*v[1:-1,1:-1] + v[:-2,1:-1]) / dx**2\ - + (v[1:-1,2:] - 2*v[1:-1,1:-1] + v[1:-1,:-2]) / dx**2 - - # Forward-Euler time step for all points except the boundaries - u_new[1:-1,1:-1] = ((c_u * u_lap) + F_u)*dt + u[1:-1,1:-1] - v_new[1:-1,1:-1] = ((c_v * v_lap) + F_v)*dt + v[1:-1,1:-1] - - # Apply periodic boundary conditions - u_new[:,0] = u[:,-2] - u_new[:,-1] = u[:,1] - u_new[0,:] = u[-2,:] - u_new[-1,:] = u[1,:] - v_new[:,0] = v[:,-2] - v_new[:,-1] = v[:,1] - v_new[0,:] = v[-2,:] - v_new[-1,:] = v[1,:] - + u_mid = u[1:-1, 1:-1] + v_mid = v[1:-1, 1:-1] + reaction = u_mid * v_mid**2 + f_u = -reaction + feed * (1 - u_mid) + f_v = reaction - (feed + kill) * v_mid + + u_lap = ( + u[2:, 1:-1] - 2 * u_mid + u[:-2, 1:-1] + + u[1:-1, 2:] - 2 * u_mid + u[1:-1, :-2] + ) / dx**2 + v_lap = ( + v[2:, 1:-1] - 2 * v_mid + v[:-2, 1:-1] + + v[1:-1, 2:] - 2 * v_mid + v[1:-1, :-2] + ) / dx**2 + + u_new[1:-1, 1:-1] = (c_u * u_lap + f_u) * dt + u_mid + v_new[1:-1, 1:-1] = (c_v * v_lap + f_v) * dt + v_mid + + u_new[:, 0] = u[:, -2] + u_new[:, -1] = u[:, 1] + u_new[0, :] = u[-2, :] + u_new[-1, :] = u[1, :] + v_new[:, 0] = v[:, -2] + v_new[:, -1] = v[:, 1] + v_new[0, :] = v[-2, :] + v_new[-1, :] = v[1, :] return u_new, v_new +def gray_scott(n=4000, n_steps=100): + dx = 1.0 + dt = dx / 5 + u = np.ones((n, n)) + v = np.zeros((n, n)) + seed = min(150, n) + u[:seed, :seed] = np.random.rand(seed, seed) + v[:seed, :seed] = np.random.rand(seed, seed) -# initial conditions and discretizaiton -dx = 1 -dt = dx/5 -u = np.ones((4000,4000)) -v = np.zeros((4000,4000)) -u[:150,:150] = np.random.rand(150,150) -v[:150,:150] = np.random.rand(150,150) - - -# fig = plt.figure() - -c_u = 1 -c_v = 0.3 -f = 0.03 -k = 0.06 - -# t_final = 1000 - -# ims = [] -n_steps = 100 # number of steps to take -frame_interval = 200 # steps to take between making plots - -# build a list of images -for n in range(n_steps) : - - ## This may need to be changed. - u,v = greyScottSys(u, v, dx, dt, c_u, c_v, f, k) + for _ in range(n_steps): + u, v = step(u, v, dx, dt, 1.0, 0.3, 0.03, 0.06) + return u, v - # ## Store frames when n is a multiple of frame_interval - # if n%frame_interval == 0: - # im = plt.imshow(u, vmin=0, vmax=1) # Show a plot of u. - # ims.append([im]) # append single image to the list of images -# anim = animation.ArtistAnimation(fig, ims, interval=100, repeat=False) -# HTML(anim.to_jshtml()) +if __name__ == "__main__": + gray_scott() diff --git a/examples/integrate.jl b/examples/integrate.jl index f27dae031..8d4e9653d 100644 --- a/examples/integrate.jl +++ b/examples/integrate.jl @@ -1,9 +1,4 @@ -using cuNumeric - -# Note that we do not yet support broadcasting -# custom functions over NDArray, so the broadcasting MUST -# be done inside the function -integrand = (x) -> @. exp(-x^2) +integrand = (x) -> exp(-x^2) N = 1_000_000 @@ -11,12 +6,13 @@ x_max = 10.0f0 domain = [-x_max, x_max] Ω = domain[2] - domain[1] -samples = Ω * cuNumeric.rand(N) -samples = @. samples - x_max +estimate = @accelerate begin + samples = @. Ω * cuNumeric.rand(N) - x_max -# Reductions return 0D NDArrays instead -# of a scalar to avoid blocking runtime -estimate = (Ω / N) * sum(integrand(samples)) + # Reductions return 0D NDArrays instead + # of a scalar to avoid blocking runtime + return (Ω / N) * sum(integrand.(samples)) +end println("Monte-Carlo Estimate: $(estimate)") println("Analytical: $(sqrt(pi))") diff --git a/examples/poisson_fft.jl b/examples/poisson_fft.jl new file mode 100644 index 000000000..3cdd26d7c --- /dev/null +++ b/examples/poisson_fft.jl @@ -0,0 +1,84 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +#= Periodic Poisson via FFT. + +Solve ∇²u = f on the unit square with periodic boundaries. In Fourier space +that is a pointwise divide: û[k] = f̂[k] / (−|k|²), with the k = 0 mode set +to zero so the potential has mean zero. + +A manufactured solution u = sin(2πx) sin(2πy) has ∇²u = −8π² u, so we can +check the residual after one forward FFT, the divide, and one inverse FFT. + +The same kernel on a stack of right-hand sides is `batched_fft` — each slice +is an independent electrostatics / gravity solve, and the batch axis is what +can split across GPUs. +=# + +using cuNumeric +using Printf + +function integer_fftfreq(n::Int) + n2 = n ÷ 2 + return iseven(n) ? vcat(0:(n2 - 1), (-n2):-1) : vcat(0:n2, (-n2):-1) +end + +# Multipliers for û = f̂ / (−|k|²). DC is 0 (mean-zero potential). +function poisson_inv_laplacian(::Type{T}, n::Int) where {T} + freq = integer_fftfreq(n) + invk = Matrix{T}(undef, n, n) + s = T(4 * π^2) + for j in 1:n, i in 1:n + k2 = s * T(freq[i]^2 + freq[j]^2) + invk[i, j] = k2 == 0 ? zero(T) : -inv(k2) + end + return invk +end + +function poisson_solve(f::NDArray) + n, m = size(f) + n == m || throw(ArgumentError("poisson_solve expects a square grid")) + invk = NDArray(poisson_inv_laplacian(eltype(f), n)) + uhat = fft(f) + return real(ifft(uhat .* invk)) +end + +function main() + n = 64 + xs = range(0, 1; length=n + 1)[1:(end - 1)] + u_true = Float32[sin(2π * x) * sin(2π * y) for x in xs, y in xs] + f_h = (-8 * Float32(π)^2) .* u_true + + f = NDArray(f_h) + u = poisson_solve(f) + err = maximum(abs.(u - NDArray(u_true))) + @printf("single grid %dx%d max |u − u_true| = %.3e\n", n, n, fetch(err)) + + # Four independent charge distributions, one FFT task, batch axis first. + b = 4 + f_batch = NDArray(repeat(reshape(f_h, 1, n, n), b, 1, 1)) + invk = reshape(NDArray(poisson_inv_laplacian(Float32, n)), 1, n, n) + u_batch = real(cuNumeric.batched_ifft(cuNumeric.batched_fft(f_batch) .* invk)) + err_b = maximum(abs.(u_batch - NDArray(repeat(reshape(u_true, 1, n, n), b, 1, 1)))) + @printf("batched %d×%dx%d max |u − u_true| = %.3e\n", b, n, n, fetch(err_b)) + + return u +end + +main() diff --git a/examples/stencil.jl b/examples/stencil.jl index b8d8e3586..ee8c3e6a9 100644 --- a/examples/stencil.jl +++ b/examples/stencil.jl @@ -1,4 +1,3 @@ - using cuNumeric: cuNumeric function initialize(N) @@ -13,7 +12,6 @@ end function run_stencil(N, I, warmup) grid = initialize(N) - println("Running Jacobi stencil...") center = grid[2:(N + 1), 2:(N + 1)] @@ -22,11 +20,14 @@ function run_stencil(N, I, warmup) west = grid[2:(N + 1), 1:N] south = grid[3:(N + 2), 2:(N + 1)] - for i in 1:(I + warmup) + for _ in 1:(I + warmup) average = center .+ north .+ east .+ west .+ south work = 0.2 .* average - center = work + center .= work end + return grid end -run_stencil(1000, 100, 5) +if abspath(PROGRAM_FILE) == @__FILE__ + run_stencil(1000, 100, 5) +end diff --git a/examples/tensor_network.jl b/examples/tensor_network.jl new file mode 100644 index 000000000..74c731814 --- /dev/null +++ b/examples/tensor_network.jl @@ -0,0 +1,42 @@ +#= Contract a two-site tensor network, then print a host scalar. + +The first contractions build an MPS-like two-site state and apply a Heisenberg +operator. Those results stay on device as NDArrays. The fully contracted +⟨ψ|H|ψ⟩ / ⟨ψ|ψ⟩ ratio is computed on device; fetch only to print (it blocks). +=# + +using cuNumeric +using LinearAlgebra +using Random +using TensorOperations + +Random.seed!(1234) + +const PHYSICAL_DIM = 2 +const BOND_DIM = 16 + +σx = ComplexF64[0 1; 1 0] +σy = ComplexF64[0 -im; im 0] +σz = ComplexF64[1 0; 0 -1] +Hmatrix = (kron(σx, σx) + kron(σy, σy) + kron(σz, σz)) / 4 +H = NDArray(reshape(Hmatrix, PHYSICAL_DIM, PHYSICAL_DIM, PHYSICAL_DIM, PHYSICAL_DIM)) + +# Initialize Random Tensors +A1 = NDArray(randn(ComplexF64, BOND_DIM, PHYSICAL_DIM, BOND_DIM)) +A2 = NDArray(randn(ComplexF64, BOND_DIM, PHYSICAL_DIM, BOND_DIM)) +L = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) +R = NDArray(randn(ComplexF64, BOND_DIM, BOND_DIM)) + +# Perform Initial Tensor Contraction +@tensor ψ[a, s1, s2, b] := L[a, ap] * A1[ap, s1, c] * A2[c, s2, bp] * R[bp, b] +@tensor Hψ[a, s1, s2, b] := H[s1, s2, t1, t2] * ψ[a, t1, t2, b] + +println("ψ is a ", typeof(ψ), " of size ", size(ψ)) +println("Hψ is a ", typeof(Hψ), " of size ", size(Hψ)) + +# Fully contracted @tensor results are 0D NDArrays. Keep them on device +# until a host Number is required; fetch blocks. +@tensor energy = conj(ψ[a, s1, s2, b]) * Hψ[a, s1, s2, b] +@tensor norm² = conj(ψ[a, s1, s2, b]) * ψ[a, s1, s2, b] + +println("two-site energy = ", real(fetch(energy ./ norm²))) diff --git a/ext/cuNumericKrylovExt.jl b/ext/cuNumericKrylovExt.jl new file mode 100644 index 000000000..26428d138 --- /dev/null +++ b/ext/cuNumericKrylovExt.jl @@ -0,0 +1,39 @@ +module cuNumericKrylovExt + +using cuNumeric: NDArray, DeviceScalar +using LinearAlgebra: axpy!, axpby! +import Krylov + +# Allocate through similar, preserving the NDArray storage type without needing +# S(undef, n). Dagger uses the same workspace-constructor integration point: +# https://github.com/JuliaParallel/Dagger.jl/blob/master/ext/KrylovExt.jl +function Krylov.CgWorkspace(A, b::NDArray{T,1}) where {T} + return Krylov.CgWorkspace(Krylov.KrylovConstructor(similar(b))) +end + +function Krylov.BicgstabWorkspace(A, b::NDArray{T,1}) where {T} + return Krylov.BicgstabWorkspace(Krylov.KrylovConstructor(similar(b))) +end + +# Krylov's documented custom-vector hooks: +# https://jso.dev/Krylov.jl/stable/custom_workspaces/#Methods-to-overload-for-compatibility-with-Krylov.jl +# Its AbstractVector fallbacks operate on whole vectors (the n argument is +# unused), but require coefficients to match the vector element type. These +# methods keep runtime-backed coefficients in the existing cuNumeric operations. +function Krylov.kaxpy!(n::Integer, α::DeviceScalar, x::NDArray{T,1}, y::NDArray{T,1}) where {T} + return axpy!(α, x, y) +end + +function Krylov.kaxpby!(n::Integer, α::DeviceScalar, x::NDArray{T,1}, β::Number, y::NDArray{T,1}) where {T} + return axpby!(α, x, β, y) +end + +function Krylov.kaxpby!(n::Integer, α::Number, x::NDArray{T,1}, β::DeviceScalar, y::NDArray{T,1}) where {T} + return axpby!(α, x, β, y) +end + +function Krylov.kaxpby!(n::Integer, α::DeviceScalar, x::NDArray{T,1}, β::DeviceScalar, y::NDArray{T,1}) where {T} + return axpby!(α, x, β, y) +end + +end diff --git a/ext/cuNumericStructArraysExt.jl b/ext/cuNumericStructArraysExt.jl new file mode 100644 index 000000000..8d11fb548 --- /dev/null +++ b/ext/cuNumericStructArraysExt.jl @@ -0,0 +1,64 @@ +module cuNumericStructArraysExt + +using cuNumeric +using StructArrays + +const Broadcasted = Base.Broadcast.Broadcasted +const Extruded = Base.Broadcast.Extruded + +# Project a struct-valued scalar function directly to one field. This keeps the +# temporary struct inside the GPU kernel instead of allocating an NDArray of it. +struct FieldFunction{field,F} + f::F +end +@inline (p::FieldFunction{field})(args...) where {field} = getfield(p.f(args...), field) + +_uses_component(x, components) = any(c -> x === c, components) +_uses_component(x::Extruded, components) = _uses_component(x.x, components) +_uses_component(x::Broadcasted, components) = + any(arg -> _uses_component(arg, components), x.args) + +function Base.copyto!( + dest::StructArray{T}, bc::Broadcasted{<:cuNumeric.NDArrayStyle} +) where {T} + cuNumeric.assert_experimental() + components = Tuple(StructArrays.components(dest)) + all(c -> c isa cuNumeric.NDArray, components) || + throw(ArgumentError("StructArray broadcast requires NDArray field storage")) + axes(dest) == axes(bc) || Base.Broadcast.throwdm(axes(dest), axes(bc)) + isempty(dest) && return dest + # Each field is computed by the fused kernel; there is no unfused fallback. + cuNumeric._struct_kernel_available() || throw( + ArgumentError( + "StructArray broadcast requires GPU broadcast fusion " * + "(fusion enabled: $(cuNumeric.FUSE_BROADCAST_EXPRS), " * + "GPU available: $(cuNumeric._has_gpu_target()))", + ), + ) + + # A field may read another field of dest. Stage results before replacing any + # component so in-place broadcasts retain their usual simultaneous semantics. + aliases_dest = _uses_component(bc, components) + outputs = aliases_dest ? map(similar, components) : components + try + for (name, out) in zip(fieldnames(T), outputs) + projected = Broadcasted{cuNumeric.NDArrayStyle{ndims(dest)}}( + FieldFunction{name,typeof(bc.f)}(bc.f), bc.args, bc.axes + ) + cuNumeric.can_fuse_linear_broadcast(out, projected) || throw( + ArgumentError("StructArray assignment requires same-shaped NDArray inputs") + ) + cuNumeric.fuse_broadcast_tree!(out, projected) + end + if aliases_dest + for (component, output) in zip(components, outputs) + copyto!(component, output) + end + end + finally + aliases_dest && foreach(cuNumeric.destroy!, outputs) + end + return dest +end + +end diff --git a/ext/cuNumericTensorOperationsExt.jl b/ext/cuNumericTensorOperationsExt.jl new file mode 100644 index 000000000..8cfef747c --- /dev/null +++ b/ext/cuNumericTensorOperationsExt.jl @@ -0,0 +1,677 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): Ethan Meitz +=# + +module cuNumericTensorOperationsExt + +using cuNumeric +using TensorOperations + +const CN = cuNumeric +const TO = TensorOperations +const One = TO.One +const Zero = TO.Zero + +struct CuNumericBackend <: TO.AbstractBackend end + +TO.select_backend(::typeof(TO.tensoradd!), C::CN.NDArray, A::CN.NDArray) = + CuNumericBackend() +TO.select_backend(::typeof(TO.tensortrace!), C::CN.NDArray, A::CN.NDArray) = + CuNumericBackend() +function TO.select_backend( + ::typeof(TO.tensorcontract!), C::CN.NDArray, A::CN.NDArray, B::CN.NDArray +) + return CuNumericBackend() +end + +# Promotion with an CNScalar coefficient describes the scalar wrapper, but the +# tensor result stores its native backend element type. +_tensor_eltype(::Type{T}) where {T} = T +_tensor_eltype(::Type{T}) where {T<:CN.CNScalar} = CN._scalar_eltype(T) + +function TO.tensoradd_type( + TC, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool +) + return CN.NDArray{_tensor_eltype(TC),TO.numind(pA)} +end + +function TO.tensorcontract_type( + TC, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, +) + T = _tensor_eltype(TC) + Tout = CN._contract_eltype(T) + return CN.NDArray{Tout,TO.numind(pAB)} +end + +function TO.tensoralloc( + ::Type{<:CN.NDArray{T,N}}, + structure, + ::Val{istemp}=Val(false), + allocator=TO.DefaultAllocator(), +) where {T,N,istemp} + return CN.zeros(T, Tuple(structure)) +end + +function TO.tensorfree!(C::CN.NDArray, allocator=TO.DefaultAllocator()) + CN.destroy!(C) + return nothing +end + +# ------------------------------------------------------------------------------------------ +# Scale factors: One/Zero dispatch, Number wraps to 0D and re-dispatches +# ------------------------------------------------------------------------------------------ + +function _require_0d_scale(x::CN.NDArray) + ndims(x) == 0 || throw( + DimensionMismatch("tensor scale factor must be a Julia Number or a 0-d NDArray") + ) + return x +end + +_as_scale(x::One, ::Type) = x +_as_scale(x::Zero, ::Type) = x +_as_scale(x::CN.CNScalar, ::Type{T}) where {T} = _as_scale(x.value, T) +function _as_scale(x::CN.NDArray, ::Type{T}) where {T} + _require_0d_scale(x) + return _convert_eltype(x, T) +end +function _as_scale(x::Number, ::Type{T}) where {T} + isone(x) && return One() + iszero(x) && return Zero() + return CN.NDArray(convert(T, x)) +end + +function _free_scale!(orig, scaled) + if scaled isa CN.NDArray && !(orig isa CN.NDArray && orig === scaled) + CN.destroy!(scaled) + end + return nothing +end + +_free_scale!(orig::CN.CNScalar, scaled) = _free_scale!(orig.value, scaled) + +_accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, ::Zero) = (C.=A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, ::Zero) = (C.=α .* A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, ::One) = (C.=C .+ A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, ::One) = (C.=C .+ α .* A; C) +_accumulate!(C::CN.NDArray, A::CN.NDArray, ::One, β::CN.NDArray) = (C.=β .* C .+ A; C) +function _accumulate!(C::CN.NDArray, A::CN.NDArray, α::CN.NDArray, β::CN.NDArray) + C .= β .* C .+ α .* A + return C +end +# α = 0: product does not contribute. Strong-zero for β = 0 does not read C. +_accumulate!(C::CN.NDArray, ::CN.NDArray, ::Zero, ::Zero) = (C.=zero(eltype(C)); C) +_accumulate!(C::CN.NDArray, ::CN.NDArray, ::Zero, ::One) = C +_accumulate!(C::CN.NDArray, ::CN.NDArray, ::Zero, β::CN.NDArray) = (C.=β .* C; C) + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α::Number, β::Number) where {T} + α′ = _as_scale(α, T) + β′ = _as_scale(β, T) + try + return _accumulate!(C, A, α′, β′) + finally + _free_scale!(α, α′) + _free_scale!(β, β′) + end +end + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α::CN.NDArray, β::Number) where {T} + α′ = _as_scale(α, T) + β′ = _as_scale(β, T) + try + return _accumulate!(C, A, α′, β′) + finally + _free_scale!(α, α′) + _free_scale!(β, β′) + end +end + +function _accumulate!(C::CN.NDArray{T}, A::CN.NDArray, α::Number, β::CN.NDArray) where {T} + α′ = _as_scale(α, T) + β′ = _as_scale(β, T) + try + return _accumulate!(C, A, α′, β′) + finally + _free_scale!(α, α′) + _free_scale!(β, β′) + end +end + +function _convert_eltype(A::CN.NDArray, ::Type{T}) where {T} + return eltype(A) === T ? A : CN.as_type(A, T) +end + +function _ensure_backend(op, backend, C, select_args...) + if backend isa TO.DefaultBackend + return TO.select_backend(op, C, select_args...) + elseif backend isa CuNumericBackend + return backend + else + throw(ArgumentError("Unknown backend $backend for $op and NDArray")) + end +end + +# ------------------------------------------------------------------------------------------ +# tensoradd! +# ------------------------------------------------------------------------------------------ + +function _tensoradd_impl!(C::CN.NDArray, A::CN.NDArray, pA, conjA, α, β) + TO.argcheck_tensoradd(C, A, pA) + TO.dimcheck_tensoradd(C, A, pA) + if C.ptr === A.ptr && !TO.istrivialpermutation(pA) + throw(ArgumentError("output tensor must not alias a permuted input tensor")) + end + + opA = conjA ? conj(A) : A + converted = _convert_eltype(opA, eltype(C)) + permutation = TO.linearize(pA) + permuted = + TO.istrivialpermutation(permutation) ? converted : + permutedims(converted, permutation) + + _accumulate!(C, permuted, α, β) + + permuted !== converted && CN.destroy!(permuted) + converted !== opA && CN.destroy!(converted) + opA !== A && CN.destroy!(opA) + return C +end + +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + return _tensoradd_impl!(C, A, pA, conjA, α, β) +end + +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β::Union{Number,CN.NDArray}, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(α) + β isa CN.NDArray && _require_0d_scale(β) + return _tensoradd_impl!(C, A, pA, conjA, α, β) +end + +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(β) + return _tensoradd_impl!(C, A, pA, conjA, α, β) +end + +function TO.tensoradd!( + C::CN.NDArray, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool, α::CN.NDArray, β +) + return TO.tensoradd!(C, A, pA, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensoradd!( + C::CN.NDArray, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool, α::Number, β::CN.NDArray +) + return TO.tensoradd!(C, A, pA, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensoradd!( + C::CN.NDArray, A::CN.NDArray, pA::TO.Index2Tuple, conjA::Bool, α::CN.NDArray, β, backend +) + return TO.tensoradd!(C, A, pA, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, +) + return TO.tensoradd!(C, A, pA, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, + backend, + allocator, +) + b = _ensure_backend(TO.tensoradd!, backend, C, A) + return TO.tensoradd!(C, A, pA, conjA, α, β, b, allocator) +end +function TO.tensoradd!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, + allocator, +) + b = _ensure_backend(TO.tensoradd!, backend, C, A) + return TO.tensoradd!(C, A, pA, conjA, α, β, b, allocator) +end + +# ------------------------------------------------------------------------------------------ +# tensortrace! +# ------------------------------------------------------------------------------------------ + +function _trace_pairs(A::CN.NDArray, q::TO.Index2Tuple) + current = A + current_owned = false + labels = collect(1:ndims(A)) + + for (left, right) in zip(q[1], q[2]) + left_position = findfirst(==(left), labels)::Int + right_position = findfirst(==(right), labels)::Int + diagonal = CN.diagonal(current; dims=(left_position, right_position)) + current_owned && CN.destroy!(current) + + reduced = sum(diagonal; dims=ndims(diagonal)) + CN.destroy!(diagonal) + current = dropdims(reduced; dims=ndims(reduced)) + CN.destroy!(reduced) + current_owned = true + + deleteat!(labels, max(left_position, right_position)) + deleteat!(labels, min(left_position, right_position)) + end + return current, current_owned, labels +end + +function _tensortrace_impl!(C::CN.NDArray, A::CN.NDArray, p, q, conjA, α, β) + TO.argcheck_tensortrace(C, A, p, q) + TO.dimcheck_tensortrace(C, A, p, q) + C.ptr === A.ptr && throw(ArgumentError("output tensor must not alias input tensor")) + + opA = conjA ? conj(A) : A + traced, traced_owned, labels = _trace_pairs(opA, q) + output_labels = TO.linearize(p) + permutation = Tuple(findfirst(==(label), labels) for label in output_labels) + ordered = TO.istrivialpermutation(permutation) ? traced : + permutedims(traced, permutation) + converted = _convert_eltype(ordered, eltype(C)) + + _accumulate!(C, converted, α, β) + + converted !== ordered && CN.destroy!(converted) + ordered !== traced && CN.destroy!(ordered) + traced_owned && CN.destroy!(traced) + opA !== A && CN.destroy!(opA) + return C +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + return _tensortrace_impl!(C, A, p, q, conjA, α, β) +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β::Union{Number,CN.NDArray}, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(α) + β isa CN.NDArray && _require_0d_scale(β) + return _tensortrace_impl!(C, A, p, q, conjA, α, β) +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(β) + return _tensortrace_impl!(C, A, p, q, conjA, α, β) +end + +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, TO.DefaultBackend()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, + backend, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, +) + return TO.tensortrace!(C, A, p, q, conjA, α, β, backend, TO.DefaultAllocator()) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::CN.NDArray, + β, + backend, + allocator, +) + b = _ensure_backend(TO.tensortrace!, backend, C, A) + return TO.tensortrace!(C, A, p, q, conjA, α, β, b, allocator) +end +function TO.tensortrace!( + C::CN.NDArray, + A::CN.NDArray, + p::TO.Index2Tuple, + q::TO.Index2Tuple, + conjA::Bool, + α::Number, + β::CN.NDArray, + backend, + allocator, +) + b = _ensure_backend(TO.tensortrace!, backend, C, A) + return TO.tensortrace!(C, A, p, q, conjA, α, β, b, allocator) +end + +# ------------------------------------------------------------------------------------------ +# tensorcontract! +# ------------------------------------------------------------------------------------------ + +function _contract_modes( + A::CN.NDArray, + pA::TO.Index2Tuple, + B::CN.NDArray, + pB::TO.Index2Tuple, + pAB::TO.Index2Tuple, +) + nlabels = ndims(A) + length(pB[2]) + nlabels <= typemax(UInt8) || + throw(ArgumentError("tensor contraction requires $nlabels distinct mode labels")) + + Amodes = collect(UInt8, 1:ndims(A)) + Bmodes = Vector{UInt8}(undef, ndims(B)) + for (a, b) in zip(pA[2], pB[1]) + Bmodes[b] = Amodes[a] + end + nextlabel = ndims(A) + for b in pB[2] + nextlabel += 1 + Bmodes[b] = UInt8(nextlabel) + end + + free_modes = (map(i -> Amodes[i], pA[1])..., map(i -> Bmodes[i], pB[2])...) + Cmodes = UInt8[free_modes[i] for i in TO.linearize(pAB)] + return Cmodes, Amodes, Bmodes +end + +function _tensorcontract_impl!( + C::CN.NDArray, A::CN.NDArray, pA, conjA, B::CN.NDArray, pB, conjB, pAB, α, β +) + TO.argcheck_tensorcontract(C, A, pA, B, pB, pAB) + TO.dimcheck_tensorcontract(C, A, pA, B, pB, pAB) + + opA = conjA ? conj(A) : A + opB = conjB ? conj(B) : B + convertedA = _convert_eltype(opA, eltype(C)) + convertedB = _convert_eltype(opB, eltype(C)) + Cmodes, Amodes, Bmodes = _contract_modes(convertedA, pA, convertedB, pB, pAB) + + CN.contract!(C, Cmodes, convertedA, Amodes, convertedB, Bmodes; α=α, β=β) + + convertedB !== opB && CN.destroy!(convertedB) + convertedA !== opA && CN.destroy!(convertedA) + opB !== B && CN.destroy!(opB) + opA !== A && CN.destroy!(opA) + return C +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::Number, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + return _tensorcontract_impl!(C, A, pA, conjA, B, pB, conjB, pAB, α, β) +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β::Union{Number,CN.NDArray}, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(α) + β isa CN.NDArray && _require_0d_scale(β) + return _tensorcontract_impl!(C, A, pA, conjA, B, pB, conjB, pAB, α, β) +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, + ::CuNumericBackend, + allocator=TO.DefaultAllocator(), +) + _require_0d_scale(β) + return _tensorcontract_impl!(C, A, pA, conjA, B, pB, conjB, pAB, α, β) +end + +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, TO.DefaultBackend() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, TO.DefaultBackend() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β, + backend, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, backend, TO.DefaultAllocator() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, + backend, +) + return TO.tensorcontract!( + C, A, pA, conjA, B, pB, conjB, pAB, α, β, backend, TO.DefaultAllocator() + ) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::CN.NDArray, + β, + backend, + allocator, +) + b = _ensure_backend(TO.tensorcontract!, backend, C, A, B) + return TO.tensorcontract!(C, A, pA, conjA, B, pB, conjB, pAB, α, β, b, allocator) +end +function TO.tensorcontract!( + C::CN.NDArray, + A::CN.NDArray, + pA::TO.Index2Tuple, + conjA::Bool, + B::CN.NDArray, + pB::TO.Index2Tuple, + conjB::Bool, + pAB::TO.Index2Tuple, + α::Number, + β::CN.NDArray, + backend, + allocator, +) + b = _ensure_backend(TO.tensorcontract!, backend, C, A, B) + return TO.tensorcontract!(C, A, pA, conjA, B, pB, conjB, pAB, α, β, b, allocator) +end + +function TO.tensorscalar(C::CN.NDArray) + ndims(C) == 0 || throw(DimensionMismatch("tensorscalar requires a rank-zero tensor")) + return copy(C) +end + +end diff --git a/lib/CNPreferences/Project.toml b/lib/CNPreferences/Project.toml index 376ace094..d1fa6276f 100644 --- a/lib/CNPreferences/Project.toml +++ b/lib/CNPreferences/Project.toml @@ -1,13 +1,13 @@ name = "CNPreferences" uuid = "3e078157-ea10-49d5-bf32-908f777cd46f" authors = ["""David Krasowska and Ethan Meitz """] -version = "0.1.3" +version = "0.1.4" [deps] LegatePreferences = "8028f36a-2b64-49e9-aa04-2d0933fd2ed9" Preferences = "21216c6a-2e73-6563-6e65-726566657250" [compat] -LegatePreferences = "0.1.6" +LegatePreferences = "0.1.7" Preferences = "1.4.3" julia = "1.10" diff --git a/lib/CNPreferences/src/CNPreferences.jl b/lib/CNPreferences/src/CNPreferences.jl index 558e47713..ba9663ee8 100644 --- a/lib/CNPreferences/src/CNPreferences.jl +++ b/lib/CNPreferences/src/CNPreferences.jl @@ -84,4 +84,34 @@ Disable named Legate task scopes. This is the default. """ disable_task_scope_names!(; kwargs...) = set_task_scope_names!(false; kwargs...) +""" + set_linalg!(; export_prefs=false, force=true, settings...) + +Set linear algebra tuning preferences using their constant names as keywords: +`MIN_SOLVE_MATRIX_SIZE` (2048), `MIN_SOLVE_TILE_SIZE` (512), +`MIN_CHOLESKY_MATRIX_SIZE` (8192), `MIN_CHOLESKY_TILE_SIZE` (2048), +`MIN_QR_MATRIX_SIZE` (1048576 elements), `QR_TILE_SIZE` (128), and +`MAX_CHOLESKY_TILES_PER_PROC` (4). All values must be positive integers. +Unspecified settings are unchanged. Restart Julia after changing preferences. + +```julia +CNPreferences.set_linalg!(; MIN_SOLVE_MATRIX_SIZE=4096, MIN_SOLVE_TILE_SIZE=512) +``` +""" +function set_linalg!(; export_prefs=false, force=true, settings...) + valid = ( + :MIN_SOLVE_MATRIX_SIZE, :MIN_SOLVE_TILE_SIZE, :MIN_CHOLESKY_MATRIX_SIZE, + :MIN_CHOLESKY_TILE_SIZE, :MIN_QR_MATRIX_SIZE, :QR_TILE_SIZE, :MAX_CHOLESKY_TILES_PER_PROC, + ) + pairs = Pair{String,Int}[] + for (key, value) in settings + key in valid || throw(ArgumentError("Unknown linear algebra setting: $key")) + value isa Integer && !(value isa Bool) && 0 < value <= typemax(Int) || + throw(ArgumentError("$key must be a positive Int")) + push!(pairs, string(key) => Int(value)) + end + isempty(pairs) && return nothing + return set_preferences!(@__MODULE__, pairs...; export_prefs, force) +end + end # module CNPreferences diff --git a/lib/cunumeric_jl_wrapper/CMakeLists.txt b/lib/cunumeric_jl_wrapper/CMakeLists.txt index b4eb3c551..104e93e96 100644 --- a/lib/cunumeric_jl_wrapper/CMakeLists.txt +++ b/lib/cunumeric_jl_wrapper/CMakeLists.txt @@ -47,7 +47,7 @@ set(SOURCES if(LEGATE_WRAPPER_ENABLE_CUDA) find_package(CUDAToolkit 13.0 REQUIRED) - list(APPEND SOURCES src/cuda.cpp) + list(APPEND SOURCES src/cuda.cpp src/mapreduce.cpp) message(STATUS "LEGATE_WRAPPER_ENABLE_CUDA=ON: adding src/cuda.cpp") else() # only disables find_package requirement for CUDAToolkit. @@ -71,6 +71,7 @@ target_link_libraries(${CXX_CUNUMERICJL_WRAPPER} PRIVATE target_include_directories(${CXX_CUNUMERICJL_WRAPPER} PRIVATE include) if(LEGATE_WRAPPER_ENABLE_CUDA) target_include_directories(${CXX_CUNUMERICJL_WRAPPER} PRIVATE ${CUDAToolkit_INCLUDE_DIRS}) + target_link_libraries(${CXX_CUNUMERICJL_WRAPPER} PRIVATE CUDA::cuda_driver CUDA::cudart) endif() install(TARGETS ${CXX_CUNUMERICJL_WRAPPER} DESTINATION lib) diff --git a/lib/cunumeric_jl_wrapper/RELEASED_COMMIT b/lib/cunumeric_jl_wrapper/RELEASED_COMMIT new file mode 100644 index 000000000..113f5b469 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/RELEASED_COMMIT @@ -0,0 +1 @@ +50e3f8ed4b92bb2e4b44a90d156f2852f9b086cd diff --git a/lib/cunumeric_jl_wrapper/VERSION b/lib/cunumeric_jl_wrapper/VERSION index 2e6fb0f27..0b6d792e1 100644 --- a/lib/cunumeric_jl_wrapper/VERSION +++ b/lib/cunumeric_jl_wrapper/VERSION @@ -1 +1 @@ -26.6.0 +26.6.4 diff --git a/lib/cunumeric_jl_wrapper/include/accessors.h b/lib/cunumeric_jl_wrapper/include/accessors.h index baf27c6ac..1ce4c5c1d 100644 --- a/lib/cunumeric_jl_wrapper/include/accessors.h +++ b/lib/cunumeric_jl_wrapper/include/accessors.h @@ -26,9 +26,12 @@ #include "cupynumeric.h" #include "jlcxx/jlcxx.hpp" +#include "legate.h" #include "legion.h" -using coord_t = long long; +// Match Legate coordinates; large scalar indices must never narrow to int. +static_assert(sizeof(legate::coord_t) >= 8, + "NDArray accessors require 64-bit coordinates"); // To auto-magically generate templated classes and their // respective member functions you must define a `BuildParameterList` @@ -49,7 +52,7 @@ class NDArrayAccessor { ~NDArrayAccessor() {} // static T read(void* arr, const std::vector& dims) { - auto p = Realm::Point(0); + auto p = legate::Point(0); // Realm::Point defaults to 32-bit int. for (int i = 0; i < n_dims; ++i) { p[i] = dims[i]; } @@ -59,7 +62,7 @@ class NDArrayAccessor { // static void write(void* arr, const std::vector& dims, T val) { - auto p = Realm::Point(0); + auto p = legate::Point(0); // Preserve indices beyond INT32_MAX. for (int i = 0; i < n_dims; ++i) { p[i] = dims[i]; } diff --git a/lib/cunumeric_jl_wrapper/include/cuda_macros.h b/lib/cunumeric_jl_wrapper/include/cuda_macros.h index de9a9f2fa..596fffb7f 100644 --- a/lib/cunumeric_jl_wrapper/include/cuda_macros.h +++ b/lib/cunumeric_jl_wrapper/include/cuda_macros.h @@ -45,30 +45,31 @@ } while (0) #endif -#define CUDA_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ - template < \ - typename T, int D, \ - typename std::enable_if<(D >= 1 && D <= REALM_MAX_DIM), int>::type = 0> \ - void cuda_device_array_arg_##MODE(char *&p, \ - const legate::PhysicalArray &rf) { \ - auto shp = rf.shape(); \ - auto acc = rf.data().ACCESSOR_CALL(); \ - CUDA_DEBUG_PRINT(std::cerr << "[RunPTXTask] " #MODE " accessor shape: " \ - << shp.lo << " - " << shp.hi << ", dim: " << D \ - << std::endl; \ - std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ - << acc.accessor.strides << std::endl;); \ - void *dev_ptr = const_cast(/*.lo to ensure multiple GPU support*/ \ - static_cast( \ - acc.ptr(Realm::Point(shp.lo)))); \ - auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ - CuDeviceArray desc; \ - desc.ptr = dev_ptr; \ - desc.maxsize = shp.volume() * sizeof(T); \ - for (size_t i = 0; i < D; ++i) { \ - desc.dims[i] = extents[i]; \ - } \ - desc.length = shp.volume(); \ - memcpy(p, &desc, sizeof(CuDeviceArray)); \ - p += sizeof(CuDeviceArray); \ +#define CUDA_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ + template < \ + typename T, int D, \ + typename std::enable_if<(D >= 1 && D <= REALM_MAX_DIM), int>::type = 0> \ + void cuda_device_array_arg_##MODE(char *&p, \ + const legate::PhysicalArray &rf) { \ + auto shp = rf.shape(); \ + auto acc = rf.data().ACCESSOR_CALL(); \ + CUDA_DEBUG_PRINT(std::cerr << "[RunPTXTask] " #MODE " accessor shape: " \ + << shp.lo << " - " << shp.hi << ", dim: " << D \ + << std::endl; \ + std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ + << acc.accessor.strides << std::endl;); \ + /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ + void *dev_ptr = \ + const_cast(/*.lo to ensure multiple GPU support*/ \ + static_cast(acc.ptr(shp.lo))); \ + auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ + CuDeviceArray desc; \ + desc.ptr = dev_ptr; \ + desc.maxsize = shp.volume() * sizeof(T); \ + for (size_t i = 0; i < D; ++i) { \ + desc.dims[i] = extents[i]; \ + } \ + desc.length = shp.volume(); \ + memcpy(p, &desc, sizeof(CuDeviceArray)); \ + p += sizeof(CuDeviceArray); \ } diff --git a/lib/cunumeric_jl_wrapper/include/mapreduce.h b/lib/cunumeric_jl_wrapper/include/mapreduce.h new file mode 100644 index 000000000..1c1f09027 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/include/mapreduce.h @@ -0,0 +1,39 @@ +#pragma once + +#include +#include + +#include "legate.h" + +namespace ufi { +// The wire values come from Legate. Julia receives this enum through CxxWrap +// and never defines a parallel table of integer operation codes. +enum class MapReduceOp : int32_t { + ADD = static_cast(legate::ReductionOpKind::ADD), + MUL = static_cast(legate::ReductionOpKind::MUL), + MIN = static_cast(legate::ReductionOpKind::MIN), + MAX = static_cast(legate::ReductionOpKind::MAX), +}; +using MapReduceOpValue = std::underlying_type_t; + +#if LEGATE_DEFINED(LEGATE_USE_CUDA) +legate::Library get_mapreduce_library(); +void register_mapreduce_tasks(); + +class RunPTXMapReduceTask : public legate::LegateTask { + public: + static inline const auto TASK_CONFIG = + legate::TaskConfig{legate::LocalTaskID{0}}.with_variant_options( + legate::VariantOptions{}.with_has_allocations(true).with_elide_device_ctx_sync(true)); + static void gpu_variant(legate::TaskContext context); +}; + +class RunPTXReduceFinishTask : public legate::LegateTask { + public: + static inline const auto TASK_CONFIG = + legate::TaskConfig{legate::LocalTaskID{1}}.with_variant_options( + legate::VariantOptions{}.with_elide_device_ctx_sync(true)); + static void gpu_variant(legate::TaskContext context); +}; +#endif +} // namespace ufi diff --git a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h index b13ddd1ac..290822970 100644 --- a/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h +++ b/lib/cunumeric_jl_wrapper/include/ndarray_c_api.h @@ -52,6 +52,14 @@ uint64_t nda_query_device_memory(); // CN_Type : Legate type of object CN_NDArray* nda_zeros_array(int32_t dim, const uint64_t* shape, CN_Type type); +// Internal allocation without a fill; every element must be written before use. +CN_NDArray* nda_empty_array(int32_t dim, const uint64_t* shape, CN_Type type); +CN_NDArray* nda_empty_struct_array(int32_t dim, const uint64_t* shape, + int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets); +bool nda_struct_layout_matches(int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets); + // full(shape, value) // dim : number of dimensions // shape : pointer to array[length=dim] @@ -64,6 +72,7 @@ CN_NDArray* nda_reshape_array(CN_NDArray* arr, int32_t dim, const uint64_t* shape); CN_NDArray* nda_astype(CN_NDArray* arr, CN_Type type); void nda_fill_array(CN_NDArray* arr, CN_Type type, const void* value); +void nda_fill_struct_array(CN_NDArray* arr, const void* value, uint64_t size); void nda_multiply(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out); void nda_add(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out); @@ -72,6 +81,15 @@ CN_NDArray* nda_multiply_scalar(CN_NDArray* rhs1, CN_Type type, CN_NDArray* nda_add_scalar(CN_NDArray* rhs1, CN_Type type, const void* value); CN_NDArray* nda_dot(CN_NDArray* rhs1, CN_NDArray* rhs2); void nda_three_dot_arg(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out); +CN_NDArray* nda_transpose_axes(CN_NDArray* arr, const int32_t* axes, int32_t n); +CN_NDArray* nda_squeeze(CN_NDArray* arr, const int32_t* axes, int32_t n); +CN_NDArray* nda_diagonal(CN_NDArray* arr, int32_t offset, int32_t axis1, + int32_t axis2); +void nda_contract(CN_NDArray* out, const char* lhs_modes, int32_t n_lhs, + CN_NDArray* rhs1, const char* rhs1_modes, int32_t n_rhs1, + CN_NDArray* rhs2, const char* rhs2_modes, int32_t n_rhs2, + const char* extent_keys, const int32_t* extents, + int32_t n_extents); CN_NDArray* nda_copy(CN_NDArray* arr); void nda_assign(CN_NDArray* arr, CN_NDArray* other); @@ -87,15 +105,31 @@ uint64_t nda_nbytes(CN_NDArray* arr); void nda_binary_op(CN_NDArray* out, CuPyNumericBinaryOpCode op_code, const CN_NDArray* rhs1, const CN_NDArray* rhs2); +void nda_binary_reduction(CN_NDArray* out, CuPyNumericBinaryOpCode op_code, + const CN_NDArray* rhs1, const CN_NDArray* rhs2); +CN_NDArray* nda_array_equal(const CN_NDArray* rhs1, const CN_NDArray* rhs2); void nda_unary_op(CN_NDArray* out, CuPyNumericUnaryOpCode op_code, CN_NDArray* input); void nda_unary_reduction(CN_NDArray* out, CuPyNumericUnaryRedCode op_code, CN_NDArray* input); +CN_NDArray* nda_unary_reduction_axes(CuPyNumericUnaryRedCode op_code, + CN_NDArray* input, const int32_t* axes, + int32_t num_axes, bool keepdims); CN_NDArray* nda_get_slice(CN_NDArray* arr, const CN_Slice* slices, int32_t ndim); +bool nda_overlaps(CN_NDArray* lhs, CN_NDArray* rhs); CN_NDArray* nda_attach_external(const void* ptr, size_t size, int dim, const uint64_t* shape, CN_Type type); +// axis is 0-based. stable=true maps to cupynumeric kind="stable". +CN_NDArray* nda_sort(CN_NDArray* arr, int32_t axis, bool stable); +void nda_sort_inplace(CN_NDArray* arr, int32_t axis, bool stable); +CN_NDArray* nda_argsort(CN_NDArray* arr, int32_t axis, bool stable); +// a is 1-D sorted; v is the needle array (any rank, same dtype). +// left=true is NumPy side='left'; left=false is side='right'. +// Returns int64 indices with v's shape (0-based). +CN_NDArray* nda_searchsorted(CN_NDArray* a, CN_NDArray* v, bool left); + #ifdef __cplusplus } #endif diff --git a/lib/cunumeric_jl_wrapper/include/ptx.h b/lib/cunumeric_jl_wrapper/include/ptx.h new file mode 100644 index 000000000..f51530dd2 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/include/ptx.h @@ -0,0 +1,27 @@ +#pragma once + +#include "legate.h" + +#if LEGATE_DEFINED(LEGATE_USE_CUDA) +#include +#include +#include +#include +#include + +namespace ufi { +// ABI matches Julia CuStridedDeviceArray; strides count elements, not bytes. +template +struct CuStridedDeviceArray { + void* ptr; + int64_t maxsize; + std::array dims; + std::array strides; + int64_t length; +}; + +// Shared by all PTX task launchers; modules remain in the existing +// processor-local cache populated by LoadPTXTask. +CUfunction lookup_ptx(const std::string& name, cudaStream_t stream); +} // namespace ufi +#endif diff --git a/lib/cunumeric_jl_wrapper/include/types.h b/lib/cunumeric_jl_wrapper/include/types.h index 88a18dc91..6271b60e2 100644 --- a/lib/cunumeric_jl_wrapper/include/types.h +++ b/lib/cunumeric_jl_wrapper/include/types.h @@ -68,3 +68,9 @@ void wrap_binary_ops(jlcxx::Module&); // Linear algebra op codes void wrap_linalg_ops(jlcxx::Module& mod); + +// FFT task id and type/direction enums (mirror cupynumeric_c.h) +void wrap_fft_ops(jlcxx::Module& mod); + +// BitGenerator op codes / enums (mirror cupynumeric_c.h) +void wrap_bitgenerator_ops(jlcxx::Module& mod); diff --git a/lib/cunumeric_jl_wrapper/src/cuda.cpp b/lib/cunumeric_jl_wrapper/src/cuda.cpp index 4eec2525b..c8e71c9ca 100644 --- a/lib/cunumeric_jl_wrapper/src/cuda.cpp +++ b/lib/cunumeric_jl_wrapper/src/cuda.cpp @@ -21,16 +21,20 @@ #include "cuda.h" #include +#include #include #include +#include #include "legate.h" #include "legate/utilities/proc_local_storage.h" #include "legion.h" +#include "ptx.h" #include "types.h" #include "ufi.h" // #define CUDA_DEBUG +#include "cuda_macros.h" // Shared error/debug and dense argument-packing macros. #define BLOCK_START 1 #define THREAD_START 4 @@ -39,51 +43,6 @@ // global padding for CUDA.jl kernel state std::size_t padded_bytes_kernel_state = 16; -#define ERROR_CHECK(x) \ - { \ - cudaError_t status = x; \ - if (status != cudaSuccess) { \ - fprintf(stderr, "CUDA Error at %s:%d: %s\n", __FILE__, __LINE__, \ - cudaGetErrorString(status)); \ - if (stream_) cudaStreamDestroy(stream_); \ - exit(-1); \ - } \ - } - -#define DRIVER_ERROR_CHECK(x) \ - { \ - CUresult status = x; \ - if (status != CUDA_SUCCESS) { \ - const char *err_str = nullptr; \ - cuGetErrorString(status, &err_str); \ - fprintf(stderr, "CUDA Driver Error at %s:%d: %s\n", __FILE__, __LINE__, \ - err_str); \ - if (stream_) cudaStreamDestroy(stream_); \ - exit(-1); \ - } \ - } - -#define TEST_PRINT_DEBUG(dev_ptr, N, T, format, stream, message) \ - { \ - std::vector host_arr(N); \ - ERROR_CHECK(cudaMemcpy(host_arr.data(), \ - reinterpret_cast(dev_ptr), \ - sizeof(T) * N, cudaMemcpyDeviceToHost)); \ - ERROR_CHECK(cudaStreamSynchronize(stream)); \ - fprintf(stderr, "[TEST_PRINT] %s: " format "\n", message, host_arr[0]); \ - } - -#ifdef CUDA_DEBUG -#define CUDA_DEBUG_PRINT(x) \ - do { \ - x; \ - } while (0) -#else -#define CUDA_DEBUG_PRINT(x) \ - do { \ - } while (0) -#endif - namespace ufi { using namespace Legion; // TODO CUcontext key hashing is redundant. ProcLocalStorage is local to the @@ -108,6 +67,19 @@ using FunctionMap = std::unordered_map cufunction_ptr{}; +CUfunction lookup_ptx(const std::string &name, cudaStream_t stream) { + CUcontext ctx; + if (cuStreamGetCtx(stream, &ctx) != CUDA_SUCCESS) + throw std::runtime_error("PTX: could not get the task CUDA context"); + if (!cufunction_ptr.has_value()) + throw std::runtime_error("PTX: no modules loaded on this processor"); + auto &functions = cufunction_ptr.get(); + auto it = functions.find({ctx, name}); + if (it == functions.end()) + throw std::runtime_error("PTX: missing kernel " + name); + return it->second; +} + #ifdef CUDA_DEBUG std::string context_to_string(CUcontext ctx) { std::ostringstream oss; @@ -136,46 +108,6 @@ struct CuDeviceArray { uint64_t length; // Number of elements (at the end) }; -// Strided — matches Julia cuNumeric.CuStridedDeviceArray (RunPTXBroadcastTask -// only). -template -struct CuStridedDeviceArray { - void *ptr; - uint64_t maxsize; - std::array dims; - std::array - strides; // element strides (byte strides / sizeof(T)) - uint64_t length; -}; - -#define CUDA_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ - template < \ - typename T, int D, \ - typename std::enable_if<(D >= 1 && D <= REALM_MAX_DIM), int>::type = 0> \ - void cuda_device_array_arg_##MODE(char *&p, \ - const legate::PhysicalArray &rf) { \ - auto shp = rf.shape(); \ - auto acc = rf.data().ACCESSOR_CALL(); \ - CUDA_DEBUG_PRINT(std::cerr << "[RunPTXTask] " #MODE " accessor shape: " \ - << shp.lo << " - " << shp.hi << ", dim: " << D \ - << std::endl; \ - std::cerr << "[RunPTXTask] " #MODE " accessor strides: " \ - << acc.accessor.strides << std::endl;); \ - void *dev_ptr = const_cast(/*.lo to ensure multiple GPU support*/ \ - static_cast( \ - acc.ptr(Realm::Point(shp.lo)))); \ - auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ - CuDeviceArray desc; \ - desc.ptr = dev_ptr; \ - desc.maxsize = shp.volume() * sizeof(T); \ - for (size_t i = 0; i < D; ++i) { \ - desc.dims[i] = extents[i]; \ - } \ - desc.length = shp.volume(); \ - memcpy(p, &desc, sizeof(CuDeviceArray)); \ - p += sizeof(CuDeviceArray); \ - } - #define CUDA_STRIDED_DEVICE_ARRAY_ARG(MODE, ACCESSOR_CALL) \ template < \ typename T, int D, \ @@ -190,8 +122,9 @@ struct CuStridedDeviceArray { std::cerr << "[RunPTXBroadcastTask] " #MODE \ << " accessor byte strides: " << acc.accessor.strides \ << std::endl;); \ - void *dev_ptr = const_cast( \ - static_cast(acc.ptr(Realm::Point(shp.lo)))); \ + /* Preserve 64-bit coordinates; Realm::Point defaults to int. */ \ + void *dev_ptr = \ + const_cast(static_cast(acc.ptr(shp.lo))); \ auto extents = shp.hi - shp.lo + legate::Point::ONES(); \ CuStridedDeviceArray desc; \ desc.ptr = dev_ptr; \ @@ -241,6 +174,52 @@ struct ufiStridedFunctor { } }; +template +void pack_struct(char *&p, const legate::PhysicalArray &array, + AccessMode mode) { + const auto elem_size = array.type().size(); + const auto shape = array.shape(); + const auto extents = shape.hi - shape.lo + legate::Point::ONES(); + // Legate permits byte views with a separate logical element size. The mdspan + // mapping reports strides in logical elements, as CUDA.jl's descriptor needs. + auto store = array.data(); + CuStridedDeviceArray desc{}; + if (mode == AccessMode::WRITE) { + auto span = store.span_write_accessor(elem_size); + desc.ptr = span.data_handle(); + for (int i = 0; i < D; ++i) { + desc.strides[i] = span.mapping().stride(i); + } + } else { + auto span = store.span_read_accessor(elem_size); + desc.ptr = const_cast(span.data_handle()); + for (int i = 0; i < D; ++i) { + desc.strides[i] = span.mapping().stride(i); + } + } + desc.maxsize = shape.volume() * elem_size; + for (int i = 0; i < D; ++i) { + desc.dims[i] = extents[i]; + } + desc.length = shape.volume(); + memcpy(p, &desc, sizeof(desc)); + p += sizeof(desc); +} + +struct PackStructDispatch { + template + void operator()(char *&p, const legate::PhysicalArray &array, + AccessMode mode) const { + pack_struct(p, array, mode); + } +}; + +void pack_struct(char *&p, const legate::PhysicalArray &array, + AccessMode mode) { + // Numeric stores reach every rank through double_dispatch; match that here. + legate::dim_dispatch(array.dim(), PackStructDispatch{}, p, array, mode); +} + struct PTXLaunchParams { cudaStream_t stream; CUstream custream; @@ -265,27 +244,7 @@ static PTXLaunchParams read_launch_params(legate::TaskContext &context) { p.ty = context.scalar(THREAD_START + 1).value(); p.tz = context.scalar(THREAD_START + 2).value(); - CUcontext ctx; - cuStreamGetCtx(p.stream, &ctx); - - FunctionKey key = {ctx, p.kernel_name}; - assert(cufunction_ptr.has_value()); - FunctionMap &fmap = cufunction_ptr.get(); - auto it = fmap.find(key); - -#ifdef CUDA_DEBUG - if (it == fmap.end()) { - std::cerr << "[RunPTXTask] Could not find key: " << key_to_string(key) - << std::endl; - for (const auto &[k, v] : fmap) { - std::cerr << "[RunPTXTask] Map key: " << key_to_string(k) << std::endl; - } - assert(0 && "[RunPTXTask] key is not found in hashmap"); - } -#endif - - assert(it != fmap.end()); - p.func = it->second; + p.func = lookup_ptx(p.kernel_name, p.stream); p.custream = reinterpret_cast(p.stream); return p; } @@ -380,6 +339,11 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const legate::PhysicalArray &out) { const std::uint32_t budget = std::max(lp.tx, 1u); const int dim = out.dim(); + // Cap before narrowing; Julia grid-stride loops cover the remaining elements. + const auto blocks = [](std::uint64_t n, std::uint32_t t, + std::uint64_t limit) { + return static_cast(std::min((n - 1) / t + 1, limit)); + }; assert(dim > 0); @@ -395,8 +359,8 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, lp.ty = static_cast( std::min(budget / lp.tx, rows)); lp.tz = 1; - lp.bx = static_cast((cols + lp.tx - 1) / lp.tx); - lp.by = static_cast((rows + lp.ty - 1) / lp.ty); + lp.bx = blocks(cols, lp.tx, 2147483647); + lp.by = blocks(rows, lp.ty, 65535); lp.bz = 1; #ifdef CUDA_DEBUG @@ -423,9 +387,9 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const std::uint32_t z_budget = yz_budget / lp.ty; lp.tz = static_cast( std::min({z_budget, dim1, 64})); - lp.bx = static_cast((dim3 + lp.tx - 1) / lp.tx); - lp.by = static_cast((dim2 + lp.ty - 1) / lp.ty); - lp.bz = static_cast((dim1 + lp.tz - 1) / lp.tz); + lp.bx = blocks(dim3, lp.tx, 2147483647); + lp.by = blocks(dim2, lp.ty, 65535); + lp.bz = blocks(dim1, lp.tz, 65535); #ifdef CUDA_DEBUG std::cerr << "[RunPTXBroadcastTask] local shape=" << dim1 << "x" << dim2 @@ -466,10 +430,7 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, const std::uint32_t threads = static_cast(std::min(budget, volume)); - const std::uint32_t blocks = - static_cast((volume + threads - 1) / threads); - - lp.bx = blocks; + lp.bx = blocks(volume, threads, 2147483647); lp.by = 1; lp.bz = 1; lp.tx = threads; @@ -526,13 +487,21 @@ static void broadcast_launch_dims_from_tile(PTXLaunchParams &lp, if (val >= 0 && val < static_cast(num_outputs)) { align8(p); auto ps = context.output(val); - legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, - ufi::AccessMode::WRITE, p, ps); + if (ps.type().code() == legate::Type::Code::STRUCT) { + pack_struct(p, ps, ufi::AccessMode::WRITE); + } else { + legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, + ufi::AccessMode::WRITE, p, ps); + } } else if (val >= static_cast(num_outputs)) { align8(p); auto ps = context.input(val - num_outputs); - legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, - ufi::AccessMode::READ, p, ps); + if (ps.type().code() == legate::Type::Code::STRUCT) { + pack_struct(p, ps, ufi::AccessMode::READ); + } else { + legate::double_dispatch(ps.dim(), ps.type().code(), ufiStridedFunctor{}, + ufi::AccessMode::READ, p, ps); + } } else { std::size_t scalar_idx = static_cast(-(val + 1)); const auto &scalar = context.scalar(scalar_values_start + scalar_idx); diff --git a/lib/cunumeric_jl_wrapper/src/mapreduce.cpp b/lib/cunumeric_jl_wrapper/src/mapreduce.cpp new file mode 100644 index 000000000..808b70438 --- /dev/null +++ b/lib/cunumeric_jl_wrapper/src/mapreduce.cpp @@ -0,0 +1,222 @@ +#include "mapreduce.h" +#include "ptx.h" + +#include +#include +#include +#include +#include +#include +#include + +#include "legate/data/buffer.h" +#include "legate/mapping/mapping.h" +#include "legate/redop/redop.h" + +extern std::size_t padded_bytes_kernel_state; + +namespace ufi { +namespace { +constexpr int THREADS = 256; +constexpr int64_t MAX_PARTIALS = 4096; + +// These tasks need their own mapper: cuPyNumeric's mapper cannot account for +// scratch allocated by tasks registered outside its built-in task table. +class MapReduceMapper : public legate::mapping::Mapper { + public: + std::vector store_mappings( + const legate::mapping::Task&, + const std::vector&) override { return {}; } + + std::optional allocation_pool_size( + const legate::mapping::Task&, legate::mapping::StoreTarget target) override { + // One partial buffer; ComplexF64 is the largest supported accumulator. + return target == legate::mapping::StoreTarget::FBMEM + ? MAX_PARTIALS * sizeof(legate::type_of_t) : 0; + } + + legate::Scalar tunable_value(legate::TunableID) override { + throw std::invalid_argument("mapreduce has no tunables"); + } +}; + +template +using ArrayArg = CuStridedDeviceArray; + +template +constexpr bool supported = C == legate::Type::Code::BOOL || + C == legate::Type::Code::INT8 || C == legate::Type::Code::INT16 || + C == legate::Type::Code::INT32 || C == legate::Type::Code::INT64 || + C == legate::Type::Code::UINT8 || C == legate::Type::Code::UINT16 || + C == legate::Type::Code::UINT32 || C == legate::Type::Code::UINT64 || + C == legate::Type::Code::FLOAT32 || C == legate::Type::Code::FLOAT64 || + C == legate::Type::Code::COMPLEX64 || C == legate::Type::Code::COMPLEX128; + +template +struct PackArray { + template + ArrayArg operator()(const legate::PhysicalStore& store) const { + if constexpr (supported) { + using T = legate::type_of_t; + auto rect = store.shape(); + auto acc = [&]() { + if constexpr (WRITE) return store.write_accessor(); + else return store.read_accessor(); + }(); + ArrayArg arg{}; + arg.ptr = const_cast(static_cast(acc.ptr(rect.lo))); + arg.length = rect.volume(); + arg.maxsize = arg.length * sizeof(T); + for (int d = 0; d < D; ++d) { + arg.dims[d] = rect.hi[d] - rect.lo[d] + 1; + arg.strides[d] = acc.accessor.strides[d] / sizeof(T); + } + return arg; + } else { + throw std::invalid_argument("mapreduce: unsupported physical element type"); + } + } +}; + +static void launch(CUfunction kernel, int blocks, cudaStream_t stream, + void** args) { + auto status = cuLaunchKernel(kernel, blocks, 1, 1, THREADS, 1, 1, 0, + reinterpret_cast(stream), args, nullptr); + if (status != CUDA_SUCCESS) { + const char* message = nullptr; + cuGetErrorString(status, &message); + throw std::runtime_error(std::string("mapreduce PTX launch: ") + (message ? message : "unknown CUDA error")); + } +} + +template +ArrayArg reduction_arg(const legate::PhysicalStore& store, + const legate::Rect& bounds, uint64_t axes, bool single) { + using T = typename OP::RHS; + size_t strides[D]; + ArrayArg arg{}; + arg.ptr = single ? store.write_accessor().ptr(bounds, strides) + : store.reduce_accessor().ptr(bounds, strides); + arg.length = 1; + for (int d = 0; d < D; ++d) { + // Promoted axes alias the same destination. Exclude those coordinates + // so exactly one Julia thread updates each physical output element. + arg.dims[d] = ((axes >> d) & 1) ? 1 : bounds.hi[d] - bounds.lo[d] + 1; + arg.strides[d] = strides[d]; // ptr(rect, strides) returns element strides + arg.length *= arg.dims[d]; + } + arg.maxsize = arg.length * sizeof(T); + return arg; +} + +template +void contribute(CUfunction kernel, cudaStream_t stream, void* state, + ArrayArg<1>& scratch, ArrayArg& dest, + bool single, int64_t start, int64_t count, int64_t chunks) { + std::array args{state, &scratch, &dest, &single, &start, &count, &chunks}; + launch(kernel, static_cast((count + THREADS - 1) / THREADS), stream, args.data()); +} + +template +void run_reduction(legate::TaskContext& context, CUfunction kernel) { + using T = typename OP::RHS; + auto input = context.input(0).data(); + auto single = context.scalar(6).value(); + auto red = single ? context.output(0).data() : context.reduction(0).data(); + const auto rect = input.shape(); + if (rect.empty()) return; + auto stream = context.get_task_stream(); + auto axes = context.scalar(1).value(); + auto full = context.scalar(2).value(); + auto combine = lookup_ptx(context.scalar(5).value(), stream); + auto src = legate::type_dispatch(input.type().code(), PackArray{}, input); + int64_t reduced = 1, retained = 1; + for (int d = 0; d < D; ++d) + (((axes >> d) & 1) ? reduced : retained) *= src.dims[d]; + + // Deferred buffers belong to the task; kernels may still be queued when + // their C++ handles leave scope. Reuse is ordered on the task's stream. + auto buffer = legate::create_buffer(MAX_PARTIALS, legate::Memory::Kind::GPU_FB_MEM); + ArrayArg<1> scratch{buffer.ptr(0), MAX_PARTIALS * int64_t(sizeof(T)), + {MAX_PARTIALS}, {1}, MAX_PARTIALS}; + std::vector state(padded_bytes_kernel_state, 0); + for (int64_t start = 0; start < retained;) { + int64_t count = std::min(MAX_PARTIALS, retained - start); + int64_t chunks = std::min(MAX_PARTIALS / count, 1 + (reduced - 1) / 1024); + std::array args{state.data(), &src, &scratch}; + std::size_t nargs = 3; + // Zero-size Julia singleton arguments are absent from the PTX signature. + if (context.scalar(4).size()) args[nargs++] = const_cast(context.scalar(4).ptr()); + args[nargs++] = &start; + args[nargs++] = &chunks; + launch(kernel, static_cast(count * chunks), stream, args.data()); + if (full) { + auto dest = reduction_arg(red, red.shape<1>(), 1, single); + contribute(combine, stream, state.data(), scratch, dest, single, start, count, chunks); + } else { + auto dest = reduction_arg(red, rect, axes, single); + contribute(combine, stream, state.data(), scratch, dest, single, start, count, chunks); + } + start += count; + } +} + +struct ReduceDispatch { + template + void operator()(legate::TaskContext& ctx, CUfunction kernel) const { + if constexpr (supported && C != legate::Type::Code::BOOL) { + using T = legate::type_of_t; + auto op = static_cast(ctx.scalar(3).value()); + if (op == MapReduceOp::ADD) return run_reduction, D>(ctx, kernel); + if constexpr (C != legate::Type::Code::COMPLEX128) { + if (op == MapReduceOp::MUL) return run_reduction, D>(ctx, kernel); + } + if constexpr (C != legate::Type::Code::COMPLEX64 && C != legate::Type::Code::COMPLEX128) { + if (op == MapReduceOp::MIN) return run_reduction, D>(ctx, kernel); + if (op == MapReduceOp::MAX) return run_reduction, D>(ctx, kernel); + } + } + throw std::invalid_argument("mapreduce: unsupported reduction/type combination"); + } +}; + +struct FinishDispatch { + template + void operator()(legate::TaskContext& ctx, CUfunction kernel) const { + auto input = ctx.input(0).data(); + auto output = ctx.output(0).data(); + if (output.shape().empty()) return; + auto src = legate::type_dispatch(input.type().code(), PackArray{}, input); + auto dst = legate::type_dispatch(output.type().code(), PackArray{}, output); + std::vector state(padded_bytes_kernel_state, 0); + std::array args{state.data(), &src, &dst}; + if (ctx.scalar(1).size()) args[3] = const_cast(ctx.scalar(1).ptr()); + auto blocks = static_cast(std::min(MAX_PARTIALS, 1 + (dst.length - 1) / THREADS)); + launch(kernel, blocks, ctx.get_task_stream(), args.data()); + } +}; +} // namespace + +legate::Library get_mapreduce_library() { + return legate::Runtime::get_runtime()->find_library("cuNumeric_mapreduce"); +} + +void register_mapreduce_tasks() { + auto library = legate::Runtime::get_runtime()->create_library( + "cuNumeric_mapreduce", legate::ResourceConfig{}, std::make_unique()); + RunPTXMapReduceTask::register_variants(library); + RunPTXReduceFinishTask::register_variants(library); +} + +void RunPTXMapReduceTask::gpu_variant(legate::TaskContext context) { + auto kernel = lookup_ptx(context.scalar(0).value(), context.get_task_stream()); + auto dest = context.scalar(6).value() ? context.output(0) : context.reduction(0); + legate::double_dispatch(context.input(0).dim(), dest.type().code(), + ReduceDispatch{}, context, kernel); +} + +void RunPTXReduceFinishTask::gpu_variant(legate::TaskContext context) { + auto kernel = lookup_ptx(context.scalar(0).value(), context.get_task_stream()); + legate::dim_dispatch(context.output(0).dim(), FinishDispatch{}, context, kernel); +} +} // namespace ufi diff --git a/lib/cunumeric_jl_wrapper/src/memory.cpp b/lib/cunumeric_jl_wrapper/src/memory.cpp index 9efe0fb6e..c00b80e68 100644 --- a/lib/cunumeric_jl_wrapper/src/memory.cpp +++ b/lib/cunumeric_jl_wrapper/src/memory.cpp @@ -27,6 +27,7 @@ #include #include #include +#include #include "ndarray_c_api.h" @@ -40,6 +41,8 @@ static inline uint64_t query_machine_config_common( Realm::Processor::Kind proc_kind, Realm::Memory::Kind mem_kind) { Machine legion_machine{Machine::get_machine()}; uint64_t total_mem = 0; + std::set + seen; // Processors on one node share memory-query results. Machine::ProcessorQuery procs = Machine::ProcessorQuery(legion_machine).only_kind(proc_kind); @@ -57,7 +60,7 @@ static inline uint64_t query_machine_config_common( ++mit) { auto mem = *mit; assert(mem.kind() == mem_kind); - total_mem += mem.capacity(); + if (seen.insert(mem).second) total_mem += mem.capacity(); } } @@ -73,6 +76,8 @@ static inline uint64_t query_allocated_bytes_common( auto ctx = Legion::Runtime::get_context(); uint64_t current_bytes = 0; + std::set + seen; // Count each physical memory once, not per processor. Machine::ProcessorQuery procs = Machine::ProcessorQuery(legion_machine).only_kind(proc_kind); @@ -91,6 +96,7 @@ static inline uint64_t query_allocated_bytes_common( auto mem = *mit; assert(mem.kind() == mem_kind); + if (!seen.insert(mem).second) continue; size_t available = legion_runtime->query_available_memory(ctx, mem); size_t capacity = mem.capacity(); current_bytes += (capacity - available); diff --git a/lib/cunumeric_jl_wrapper/src/ndarray.cpp b/lib/cunumeric_jl_wrapper/src/ndarray.cpp index 9a925abc0..f5d02bd60 100644 --- a/lib/cunumeric_jl_wrapper/src/ndarray.cpp +++ b/lib/cunumeric_jl_wrapper/src/ndarray.cpp @@ -30,8 +30,13 @@ #include #include #include +#include +#include +#include #include +#include #include +#include #include #include "ndarray_c_api.h" @@ -57,6 +62,56 @@ struct CN_Store { legate::LogicalStore obj; }; +CN_NDArray* nda_empty_array(int32_t dim, const uint64_t* shape, CN_Type type) { + std::vector shp(shape, shape + dim); + auto* runtime = cupynumeric::CuPyNumericRuntime::get_runtime(); + return new CN_NDArray{runtime->create_array(shp, type.obj)}; +} + +bool nda_struct_layout_matches(int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets) { + if (fields <= 0) { + return false; + } + std::vector field_types; + field_types.reserve(fields); + for (int32_t i = 0; i < fields; ++i) { + field_types.push_back( + legate::primitive_type(static_cast(codes[i]))); + } + auto type = legate::struct_type(field_types, true); + auto layout = type.offsets(); + if (type.size() != size || layout.size() != static_cast(fields)) { + return false; + } + for (int32_t i = 0; i < fields; ++i) { + if (layout[i] != offsets[i]) { + return false; + } + } + return true; +} + +CN_NDArray* nda_empty_struct_array(int32_t dim, const uint64_t* shape, + int32_t fields, const int32_t* codes, + uint32_t size, const uint32_t* offsets) { + if (!nda_struct_layout_matches(fields, codes, size, offsets)) { + throw std::invalid_argument( + "Legate struct layout does not match Julia layout"); + } + std::vector field_types; + field_types.reserve(fields); + for (int32_t i = 0; i < fields; ++i) { + field_types.push_back( + legate::primitive_type(static_cast(codes[i]))); + } + auto type = legate::struct_type(field_types, true); + std::vector shp(shape, shape + dim); + auto store = + legate::Runtime::get_runtime()->create_store(legate::Shape(shp), type); + return new CN_NDArray{cupynumeric::as_array(store)}; +} + CN_NDArray* nda_zeros_array(int32_t dim, const uint64_t* shape, CN_Type type) { std::vector shp(shape, shape + dim); NDArray result = zeros(shp, type.obj); @@ -125,6 +180,24 @@ CN_NDArray* nda_unique(CN_NDArray* arr) { return new CN_NDArray{NDArray(std::move(result))}; } +CN_NDArray* nda_sort(CN_NDArray* arr, int32_t axis, bool stable) { + const char* kind = stable ? "stable" : "quicksort"; + NDArray result = + cupynumeric::sort(arr->obj, std::optional{axis}, kind); + return new CN_NDArray{NDArray(std::move(result))}; +} + +void nda_sort_inplace(CN_NDArray* arr, int32_t axis, bool stable) { + arr->obj.sort(arr->obj, false, std::optional{axis}, stable); +} + +CN_NDArray* nda_argsort(CN_NDArray* arr, int32_t axis, bool stable) { + const char* kind = stable ? "stable" : "quicksort"; + NDArray result = + cupynumeric::argsort(arr->obj, std::optional{axis}, kind); + return new CN_NDArray{NDArray(std::move(result))}; +} + CN_NDArray* nda_ravel(CN_NDArray* arr) { NDArray result = cupynumeric::ravel(arr->obj, "C"); return new CN_NDArray{NDArray(std::move(result))}; @@ -173,6 +246,59 @@ void nda_three_dot_arg(CN_NDArray* rhs1, CN_NDArray* rhs2, CN_NDArray* out) { out->obj.dot(rhs1->obj, rhs2->obj); } +CN_NDArray* nda_transpose_axes(CN_NDArray* arr, const int32_t* axes, + int32_t n) { + std::vector axis_vec(axes, axes + n); + NDArray result = cupynumeric::transpose(arr->obj, std::move(axis_vec)); + return new CN_NDArray{NDArray(std::move(result))}; +} + +CN_NDArray* nda_squeeze(CN_NDArray* arr, const int32_t* axes, int32_t n) { + if (n <= 0) { + NDArray result = cupynumeric::squeeze(arr->obj, std::nullopt); + return new CN_NDArray{NDArray(std::move(result))}; + } + std::vector axis_vec(axes, axes + n); + NDArray result = cupynumeric::squeeze( + arr->obj, + std::optional const>>{ + axis_vec}); + return new CN_NDArray{NDArray(std::move(result))}; +} + +CN_NDArray* nda_diagonal(CN_NDArray* arr, int32_t offset, int32_t axis1, + int32_t axis2) { + NDArray result = cupynumeric::diagonal(arr->obj, offset, axis1, axis2, true); + return new CN_NDArray{NDArray(std::move(result))}; +} + +void nda_contract(CN_NDArray* out, const char* lhs_modes, int32_t n_lhs, + CN_NDArray* rhs1, const char* rhs1_modes, int32_t n_rhs1, + CN_NDArray* rhs2, const char* rhs2_modes, int32_t n_rhs2, + const char* extent_keys, const int32_t* extents, + int32_t n_extents) { + std::vector lhs(lhs_modes, lhs_modes + n_lhs); + std::vector r1(rhs1_modes, rhs1_modes + n_rhs1); + std::vector r2(rhs2_modes, rhs2_modes + n_rhs2); + std::map mode2extent; + for (int32_t i = 0; i < n_extents; ++i) { + mode2extent.emplace(extent_keys[i], extents[i]); + } + out->obj.contract(lhs, rhs1->obj, r1, rhs2->obj, r2, mode2extent); +} + +// Julia passes the struct's bytes; the store's own type gives their layout. +void nda_fill_struct_array(CN_NDArray* arr, const void* value, uint64_t size) { + auto store = arr->obj.get_store(); + if (store.type().size() != size) { + throw std::invalid_argument( + "struct fill value size does not match the array type"); + } + if (store.volume() == 0) return; + legate::Runtime::get_runtime()->issue_fill( + store, legate::Scalar(store.type(), value, true)); +} + CN_NDArray* nda_copy(CN_NDArray* arr) { NDArray result = arr->obj.copy(); return new CN_NDArray{NDArray(std::move(result))}; @@ -182,6 +308,10 @@ void nda_assign(CN_NDArray* arr, CN_NDArray* other) { arr->obj.assign(other->obj); } +bool nda_overlaps(CN_NDArray* lhs, CN_NDArray* rhs) { + return lhs->obj.get_store().overlaps(rhs->obj.get_store()); +} + void nda_move(CN_NDArray* dst, CN_NDArray* src) { dst->obj.operator=(std::move(src->obj)); } @@ -237,18 +367,157 @@ void nda_unary_reduction(CN_NDArray* out, CuPyNumericUnaryRedCode op_code, out->obj.unary_reduction(op_code, input->obj); } +// COUNT_NONZERO's result is an integer count, not the input dtype. +// Passing res_dtype (not dtype) keeps the source type. dtype=int64 would +// cast the input before counting. +static std::optional unary_red_res_dtype( + CuPyNumericUnaryRedCode op_code) { + switch (op_code) { + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_COUNT_NONZERO: + return legate::int64(); + default: + return std::nullopt; + } +} + +static bool is_arg_reduction(CuPyNumericUnaryRedCode op_code) { + switch (op_code) { + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX: + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMIN: + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX: + case CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMIN: + return true; + default: + return false; + } +} + +} // extern "C" + +// Matches cupynumeric Argval { int64_t arg; T arg_value; }. +// Templates cannot live in the extern "C" block. +template +struct ArgvalCompat { + int64_t arg; + T arg_value; +}; + +template +static void fill_arg_identity(NDArray& acc, const legate::Type& argred_type, + bool is_argmax) { + ArgvalCompat id; + id.arg = std::numeric_limits::min(); + id.arg_value = is_argmax ? std::numeric_limits::lowest() + : std::numeric_limits::max(); + if (argred_type.size() != sizeof(id)) { + throw std::runtime_error("argred identity layout mismatch"); + } + acc.fill(Scalar(argred_type, &id, true)); +} + +static void fill_arg_identity(NDArray& acc, const legate::Type& src_type, + const legate::Type& argred_type, bool is_argmax) { + switch (src_type.code()) { + case legate::Type::Code::BOOL: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT8: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT16: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT32: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::INT64: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT8: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT16: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT32: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::UINT64: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::FLOAT32: + fill_arg_identity(acc, argred_type, is_argmax); + break; + case legate::Type::Code::FLOAT64: + fill_arg_identity(acc, argred_type, is_argmax); + break; + default: + throw std::runtime_error( + "argmax/argmin are not supported for this element type"); + } +} + +// Public unary_reduction fills identity via type_dispatch on the *output* +// type. For ARGMAX that output is a struct, which Legate rejects. Fill from +// the source dtype (like Python), then launch SCALAR_UNARY_RED and GETARG. +static CN_NDArray* nda_arg_reduction(CuPyNumericUnaryRedCode op_code, + CN_NDArray* input) { + const bool is_argmax = + op_code == CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX || + op_code == CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX; + auto* runtime = cupynumeric::CuPyNumericRuntime::get_runtime(); + const std::vector scalar_shape{}; + auto argred_type = runtime->get_argred_type(input->obj.type()); + // C++ get_argred_type does not attach redops; Python does this on first + // use. record_reduction_operator throws if called twice for the same type. + static std::unordered_set registered_argred; + if (registered_argred.insert(input->obj.type().code()).second) { + auto ids = cupynumeric_register_reduction_ops( + static_cast(input->obj.type().code())); + argred_type.record_reduction_operator( + legate::ReductionOpKind::MAX, + legate::GlobalRedopID{ids.argmax_redop_id}); + argred_type.record_reduction_operator( + legate::ReductionOpKind::MIN, + legate::GlobalRedopID{ids.argmin_redop_id}); + } + NDArray acc = runtime->create_array(scalar_shape, argred_type); + fill_arg_identity(acc, input->obj.type(), argred_type, is_argmax); + + auto task = + runtime->create_task(CuPyNumericOpCode::CUPYNUMERIC_SCALAR_UNARY_RED); + task.add_reduction(acc.get_store(), is_argmax ? legate::ReductionOpKind::MAX + : legate::ReductionOpKind::MIN); + task.add_input(input->obj.get_store()); + task.add_scalar_arg(Scalar(static_cast(op_code))); + task.add_scalar_arg(Scalar(input->obj.shape())); + task.add_scalar_arg(Scalar(false)); + runtime->submit(std::move(task)); + + NDArray idx = runtime->create_array(scalar_shape, legate::int64()); + idx.unary_op( + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_GETARG), + acc); + return new CN_NDArray{NDArray(std::move(idx))}; +} + +extern "C" { + CN_NDArray* nda_unary_reduction_axes(CuPyNumericUnaryRedCode op_code, CN_NDArray* input, const int32_t* axes, int32_t num_axes, bool keepdims) { + if (is_arg_reduction(op_code)) { + return nda_arg_reduction(op_code, input); + } std::vector axis_vec(axes, axes + num_axes); NDArray result = input->obj._perform_unary_reduction( static_cast(op_code), input->obj, axis_vec, - std::nullopt, // dtype - std::nullopt, // res_dtype - std::nullopt, // out - keepdims, {}, // args - std::nullopt, // initial - std::nullopt // where + std::nullopt, // dtype + unary_red_res_dtype(op_code), // res_dtype + std::nullopt, // out + keepdims, {}, // args + std::nullopt, // initial + std::nullopt // where ); return new CN_NDArray{NDArray(std::move(result))}; } @@ -291,4 +560,26 @@ CN_NDArray* nda_get_slice(CN_NDArray* arr, const CN_Slice* slices, CN_NDArray* nda_store_to_ndarray(CN_Store* st) { return new CN_NDArray{cupynumeric::as_array(st->obj)}; } + +// Mirrors cupynumeric deferred.searchsorted: fill + MIN/MAX reduction. +CN_NDArray* nda_searchsorted(CN_NDArray* a, CN_NDArray* v, bool left) { + auto* runtime = cupynumeric::CuPyNumericRuntime::get_runtime(); + NDArray out = runtime->create_array(v->obj.shape(), legate::int64()); + const int64_t n = static_cast(a->obj.size()); + out.fill(Scalar(left ? n : int64_t{0})); + + auto task = runtime->create_task(CuPyNumericOpCode::CUPYNUMERIC_SEARCHSORTED); + auto p_out = + task.add_reduction(out.get_store(), left ? legate::ReductionOpKind::MIN + : legate::ReductionOpKind::MAX); + task.add_input(a->obj.get_store()); + auto p_v = task.add_input(v->obj.get_store()); + task.add_constraint(legate::broadcast(p_v)); + task.add_constraint(legate::broadcast(p_out)); + task.add_constraint(legate::align(p_out, p_v)); + task.add_scalar_arg(Scalar(left)); + task.add_scalar_arg(Scalar(n)); + runtime->submit(std::move(task)); + return new CN_NDArray{NDArray(std::move(out))}; +} } // extern "C" diff --git a/lib/cunumeric_jl_wrapper/src/types.cpp b/lib/cunumeric_jl_wrapper/src/types.cpp index 867330ea1..f80fff258 100644 --- a/lib/cunumeric_jl_wrapper/src/types.cpp +++ b/lib/cunumeric_jl_wrapper/src/types.cpp @@ -19,149 +19,155 @@ */ #include "types.h" - #include "cupynumeric.h" +#include + void wrap_unary_ops(jlcxx::Module& mod) { - mod.add_bits("UnaryOpCode", - jlcxx::julia_type("CppEnum")); - mod.set_const("ABSOLUTE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ABSOLUTE); - mod.set_const("ANGLE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ANGLE); - mod.set_const("ARCCOS", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOS); - mod.set_const("ARCCOSH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOSH); - mod.set_const("ARCSIN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSIN); - mod.set_const("ARCSINH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSINH); - mod.set_const("ARCTAN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTAN); - mod.set_const("ARCTANH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTANH); - mod.set_const("CBRT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CBRT); - mod.set_const("CEIL", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CEIL); - mod.set_const("CLIP", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CLIP); - mod.set_const("CONJ", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CONJ); - mod.set_const("COPY", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COPY); - mod.set_const("COS", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COS); - mod.set_const("COSH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COSH); - mod.set_const("DEG2RAD", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_DEG2RAD); - mod.set_const("EXP", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP); - mod.set_const("EXP2", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP2); - mod.set_const("EXPM1", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXPM1); - mod.set_const("FLOOR", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FLOOR); - mod.set_const("FREXP", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FREXP); - mod.set_const("GETARG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_GETARG); - mod.set_const("IMAG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_IMAG); - mod.set_const("INVERT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_INVERT); - mod.set_const("ISFINITE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISFINITE); - mod.set_const("ISINF", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISINF); - mod.set_const("ISNAN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISNAN); - mod.set_const("LOG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG); - mod.set_const("LOG10", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG10); - mod.set_const("LOG1P", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG1P); - mod.set_const("LOG2", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG2); - mod.set_const("LOGICAL_NOT", - CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOGICAL_NOT); - mod.set_const("MODF", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_MODF); - mod.set_const("NEGATIVE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_NEGATIVE); - mod.set_const("POSITIVE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_POSITIVE); - mod.set_const("RAD2DEG", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RAD2DEG); - mod.set_const("REAL", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_REAL); - mod.set_const("RECIPROCAL", - CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RECIPROCAL); - mod.set_const("RINT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RINT); - mod.set_const("ROUND", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ROUND); - mod.set_const("SIGN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGN); - mod.set_const("SIGNBIT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGNBIT); - mod.set_const("SIN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIN); - mod.set_const("SINH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SINH); - mod.set_const("SQRT", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQRT); - mod.set_const("SQUARE", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQUARE); - mod.set_const("TAN", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TAN); - mod.set_const("TANH", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TANH); - mod.set_const("TRUNC", CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TRUNC); + + mod.add_enum("UnaryOpCode", + std::vector( + {"ABSOLUTE", "ANGLE", "ARCCOS", "ARCCOSH", "ARCSIN", "ARCSINH", "ARCTAN", "ARCTANH", + "CBRT", "CEIL", "CLIP", "CONJ", "COPY", "COS", "COSH", "DEG2RAD", + "EXP", "EXP2", "EXPM1", "FLOOR", "FREXP", "GETARG", "IMAG", "INVERT", + "ISFINITE", "ISINF", "ISNAN", "LOG", "LOG10", "LOG1P", "LOG2", + "LOGICAL_NOT", "MODF", "NEGATIVE", "POSITIVE", "RAD2DEG", "REAL", + "RECIPROCAL", "RINT", "ROUND", "SIGN", "SIGNBIT", "SIN", "SINH", + "SQRT", "SQUARE", "TAN", "TANH", "TRUNC"}), + std::vector({ + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ABSOLUTE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ANGLE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOS), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCCOSH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSIN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCSINH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTAN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ARCTANH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CBRT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CEIL), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CLIP), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_CONJ), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COPY), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COS), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_COSH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_DEG2RAD), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXP2), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_EXPM1), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FLOOR), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_FREXP), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_GETARG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_IMAG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_INVERT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISFINITE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISINF), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ISNAN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG10), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG1P), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOG2), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_LOGICAL_NOT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_MODF), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_NEGATIVE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_POSITIVE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RAD2DEG), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_REAL), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RECIPROCAL), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_RINT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_ROUND), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIGNBIT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SIN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SINH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQRT), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_SQUARE), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TAN), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TANH), + static_cast(CuPyNumericUnaryOpCode::CUPYNUMERIC_UOP_TRUNC), + + }) + ); } void wrap_unary_reds(jlcxx::Module& mod) { - mod.add_bits("UnaryRedCode", - jlcxx::julia_type("CppEnum")); - mod.set_const("ALL", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ALL); - mod.set_const("ANY", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ANY); - mod.set_const("ARGMAX", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX); - mod.set_const("ARGMIN", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMIN); - mod.set_const("CONTAINS", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_CONTAINS); - mod.set_const("COUNT_NONZERO", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_COUNT_NONZERO); - mod.set_const("MAX", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MAX); - mod.set_const("MIN", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MIN); - mod.set_const("NANARGMAX", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX); - mod.set_const("NANARGMIN", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMIN); - mod.set_const("NANMAX", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMAX); - mod.set_const("NANMIN", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMIN); - mod.set_const("NANPROD", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANPROD); - mod.set_const("NANSUM", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANSUM); - mod.set_const("PROD", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_PROD); - mod.set_const("SUM", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM); - mod.set_const("SUM_SQUARES", - CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM_SQUARES); - mod.set_const("VARIANCE", CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_VARIANCE); + + mod.add_enum("UnaryRedCode", + std::vector( + {"ALL", "ANY", "ARGMAX", "ARGMIN", "CONTAINS", "COUNT_NONZERO", "MAX", + "MIN", "NANARGMAX", "NANARGMIN", "NANMAX", "NANMIN", "NANPROD", + "NANSUM", "PROD", "SUM", "SUM_SQUARES", "VARIANCE"}), + std::vector({ + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ALL), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ANY), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_ARGMIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_CONTAINS), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_COUNT_NONZERO), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_MIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANARGMIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMAX), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANMIN), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANPROD), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_NANSUM), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_PROD), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_SUM_SQUARES), + static_cast(CuPyNumericUnaryRedCode::CUPYNUMERIC_RED_VARIANCE) + }) + ); } void wrap_binary_ops(jlcxx::Module& mod) { - mod.add_bits("BinaryOpCode", - jlcxx::julia_type("CppEnum")); - mod.set_const("ADD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ADD); - mod.set_const("ARCTAN2", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ARCTAN2); - mod.set_const("BITWISE_AND", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_AND); - mod.set_const("BITWISE_OR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_OR); - mod.set_const("BITWISE_XOR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_XOR); - mod.set_const("COPYSIGN", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_COPYSIGN); - mod.set_const("DIVIDE", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_DIVIDE); - mod.set_const("EQUAL", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_EQUAL); - mod.set_const("FLOAT_POWER", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOAT_POWER); - mod.set_const("FLOOR_DIVIDE", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOOR_DIVIDE); - mod.set_const("FMOD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FMOD); - mod.set_const("GCD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GCD); - mod.set_const("GREATER", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER); - mod.set_const("GREATER_EQUAL", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER_EQUAL); - mod.set_const("HYPOT", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_HYPOT); - mod.set_const("ISCLOSE", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ISCLOSE); - mod.set_const("LCM", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LCM); - mod.set_const("LDEXP", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LDEXP); - mod.set_const("LEFT_SHIFT", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LEFT_SHIFT); - mod.set_const("LESS", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS); - mod.set_const("LESS_EQUAL", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS_EQUAL); - mod.set_const("LOGADDEXP", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP); - mod.set_const("LOGADDEXP2", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP2); - mod.set_const("LOGICAL_AND", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_AND); - mod.set_const("LOGICAL_OR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_OR); - mod.set_const("LOGICAL_XOR", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_XOR); - mod.set_const("MAXIMUM", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MAXIMUM); - mod.set_const("MINIMUM", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MINIMUM); - mod.set_const("MOD", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MOD); - mod.set_const("MULTIPLY", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MULTIPLY); - mod.set_const("NEXTAFTER", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NEXTAFTER); - mod.set_const("NOT_EQUAL", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NOT_EQUAL); - mod.set_const("POWER", CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_POWER); - mod.set_const("RIGHT_SHIFT", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_RIGHT_SHIFT); - mod.set_const("SUBTRACT", - CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_SUBTRACT); + + mod.add_enum("BinaryOpCode", + std::vector({ + "ADD", "ARCTAN2", "BITWISE_AND", "BITWISE_OR", "BITWISE_XOR", + "COPYSIGN", "DIVIDE", "EQUAL", "FLOAT_POWER", "FLOOR_DIVIDE", + "FMOD", "GCD", "GREATER", "GREATER_EQUAL", "HYPOT", "ISCLOSE", "LCM", "LDEXP", "LEFT_SHIFT", + "LESS", "LESS_EQUAL", "LOGADDEXP", "LOGADDEXP2", "LOGICAL_AND", "LOGICAL_OR", "LOGICAL_XOR", + "MAXIMUM", "MINIMUM", "MOD", "MULTIPLY", "NEXTAFTER", + "NOT_EQUAL", "POWER", "RIGHT_SHIFT", "SUBTRACT" + }), + std::vector({ + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ADD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ARCTAN2), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_AND), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_OR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_BITWISE_XOR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_COPYSIGN), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_DIVIDE), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOAT_POWER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FLOOR_DIVIDE), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_FMOD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GCD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_GREATER_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_HYPOT), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_ISCLOSE), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LCM), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LDEXP), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LEFT_SHIFT), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LESS_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGADDEXP2), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_AND), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_OR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_LOGICAL_XOR), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MAXIMUM), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MINIMUM), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MOD), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_MULTIPLY), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NEXTAFTER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_NOT_EQUAL), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_POWER), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_RIGHT_SHIFT), + static_cast(CuPyNumericBinaryOpCode::CUPYNUMERIC_BINOP_SUBTRACT) + }) + ); } void wrap_linalg_ops(jlcxx::Module& mod) { @@ -169,10 +175,81 @@ void wrap_linalg_ops(jlcxx::Module& mod) { legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SOLVE}); mod.set_const("MP_SOLVE", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_SOLVE}); + mod.set_const("MP_POTRF", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_POTRF}); + mod.set_const("MP_QR", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_MP_QR}); + mod.set_const("POTRS", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_POTRS}); + mod.set_const("TRSM", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRSM}); + mod.set_const("SYRK", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SYRK}); + mod.set_const("GEMM", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEMM}); + mod.set_const( + "TRANSPOSE_COPY_2D", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRANSPOSE_COPY_2D}); + mod.set_const("TRILU", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_TRILU}); mod.set_const("SVD", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SVD}); mod.set_const("CQR", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_QR}); + mod.set_const("POTRF", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_POTRF}); mod.set_const("SYEV", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_SYEV}); mod.set_const("GEEV", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_GEEV}); } + +void wrap_fft_ops(jlcxx::Module& mod) { + mod.set_const("FFT", legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_FFT}); + mod.set_const("FFT_R2C", int32_t(CUPYNUMERIC_FFT_R2C)); + mod.set_const("FFT_C2R", int32_t(CUPYNUMERIC_FFT_C2R)); + mod.set_const("FFT_C2C", int32_t(CUPYNUMERIC_FFT_C2C)); + mod.set_const("FFT_D2Z", int32_t(CUPYNUMERIC_FFT_D2Z)); + mod.set_const("FFT_Z2D", int32_t(CUPYNUMERIC_FFT_Z2D)); + mod.set_const("FFT_Z2Z", int32_t(CUPYNUMERIC_FFT_Z2Z)); + mod.set_const("FFT_FORWARD", int32_t(CUPYNUMERIC_FFT_FORWARD)); + mod.set_const("FFT_INVERSE", int32_t(CUPYNUMERIC_FFT_INVERSE)); +} + +void wrap_bitgenerator_ops(jlcxx::Module& mod) { + mod.set_const( + "BITGENERATOR", + legate::LocalTaskID{CuPyNumericOpCode::CUPYNUMERIC_BITGENERATOR}); + + mod.set_const("BITGENOP_CREATE", int32_t(CUPYNUMERIC_BITGENOP_CREATE)); + mod.set_const("BITGENOP_DESTROY", int32_t(CUPYNUMERIC_BITGENOP_DESTROY)); + mod.set_const("BITGENOP_RAND_RAW", int32_t(CUPYNUMERIC_BITGENOP_RAND_RAW)); + mod.set_const("BITGENOP_DISTRIBUTION", + int32_t(CUPYNUMERIC_BITGENOP_DISTRIBUTION)); + + mod.set_const("BITGENTYPE_DEFAULT", uint32_t(CUPYNUMERIC_BITGENTYPE_DEFAULT)); + mod.set_const("BITGENTYPE_XORWOW", uint32_t(CUPYNUMERIC_BITGENTYPE_XORWOW)); + mod.set_const("BITGENTYPE_MRG32K3A", + uint32_t(CUPYNUMERIC_BITGENTYPE_MRG32K3A)); + mod.set_const("BITGENTYPE_MTGP32", uint32_t(CUPYNUMERIC_BITGENTYPE_MTGP32)); + mod.set_const("BITGENTYPE_MT19937", uint32_t(CUPYNUMERIC_BITGENTYPE_MT19937)); + mod.set_const("BITGENTYPE_PHILOX4_32_10", + uint32_t(CUPYNUMERIC_BITGENTYPE_PHILOX4_32_10)); + + mod.set_const("BITGENDIST_INTEGERS_16", + uint32_t(CUPYNUMERIC_BITGENDIST_INTEGERS_16)); + mod.set_const("BITGENDIST_INTEGERS_32", + uint32_t(CUPYNUMERIC_BITGENDIST_INTEGERS_32)); + mod.set_const("BITGENDIST_INTEGERS_64", + uint32_t(CUPYNUMERIC_BITGENDIST_INTEGERS_64)); + mod.set_const("BITGENDIST_UNIFORM_32", + uint32_t(CUPYNUMERIC_BITGENDIST_UNIFORM_32)); + mod.set_const("BITGENDIST_UNIFORM_64", + uint32_t(CUPYNUMERIC_BITGENDIST_UNIFORM_64)); + mod.set_const("BITGENDIST_NORMAL_32", + uint32_t(CUPYNUMERIC_BITGENDIST_NORMAL_32)); + mod.set_const("BITGENDIST_NORMAL_64", + uint32_t(CUPYNUMERIC_BITGENDIST_NORMAL_64)); + mod.set_const("BITGENDIST_EXPONENTIAL_32", + uint32_t(CUPYNUMERIC_BITGENDIST_EXPONENTIAL_32)); + mod.set_const("BITGENDIST_EXPONENTIAL_64", + uint32_t(CUPYNUMERIC_BITGENDIST_EXPONENTIAL_64)); +} diff --git a/lib/cunumeric_jl_wrapper/src/wrapper.cpp b/lib/cunumeric_jl_wrapper/src/wrapper.cpp index 562ad7af7..4ca6e15f5 100644 --- a/lib/cunumeric_jl_wrapper/src/wrapper.cpp +++ b/lib/cunumeric_jl_wrapper/src/wrapper.cpp @@ -18,10 +18,13 @@ * Nader Rahhal */ +#include #include #include +#include #include //needed for return type of toString methods #include +#include #include "accessors.h" #include "cupynumeric.h" @@ -33,6 +36,7 @@ #include "realm.h" #include "types.h" #include "ufi.h" +#include "mapreduce.h" struct WrapCppOptional { template @@ -53,12 +57,101 @@ void* nda_store_to_ndarray(legate::LogicalStore st) { return static_cast(new CN_NDArray{cupynumeric::as_array(st)}); } +// Legate.jl wraps ManualTask::add_input/add_output for a whole partition, but +// not the overloads taking a projection. GEEV needs one: its eigenvalue store +// has one fewer dimension than the launch domain. Julia picks the source +// dimensions; this only translates them into a SymbolicPoint. +static legate::SymbolicPoint make_projection(const std::vector& dims) { + std::vector exprs; + exprs.reserve(dims.size()); + for (auto d : dims) { + exprs.push_back(legate::dimension(static_cast(d))); + } + return legate::SymbolicPoint{std::move(exprs)}; +} + #if LEGATE_DEFINED(LEGATE_USE_CUDA) void register_tasks() { auto library = get_lib(); ufi::LoadPTXTask::register_variants(library); ufi::RunPTXTask::register_variants(library); ufi::RunPTXBroadcastTask::register_variants(library); + ufi::register_mapreduce_tasks(); +} + +static legate::Scalar mapreduce_payload(const void* ptr, size_t size) { + auto bytes = static_cast(ptr); + return legate::Scalar(std::vector(bytes, bytes + size)); +} + +static void submit_mapreduce(CN_NDArray* input, CN_NDArray* accumulator, + CN_NDArray* output, uint64_t axes, bool full, bool single, + ufi::MapReduceOp op, const std::string& kernel, + const std::string& contribute_kernel, + const std::string& finish_kernel, + const void* mapper, size_t mapper_size, + const void* finish, size_t finish_size) { + auto* rt = legate::Runtime::get_runtime(); + auto library = ufi::get_mapreduce_library(); + auto src = input->obj.get_store(); + // Physical descriptors and reduction accessors use at least one dimension. + if (src.dim() == 0) src = src.promote(0, 1); + if (src.dim() > 64) throw std::invalid_argument("mapreduce rank exceeds axis mask"); + // A dimensional singleton reduction is elementwise. Reuse the finish task + // to map and seed each value without constructing reduction privileges. + if (single && !full) { + auto dst = output->obj.get_store(); + if (dst.dim() == 0) dst = dst.promote(0, 1); + auto final = rt->create_task( + library, ufi::RunPTXReduceFinishTask::TASK_CONFIG.task_id()); + auto p_src = final.add_input(src); + auto p_dst = final.add_output(dst); + final.add_constraint(legate::align(p_src, p_dst)); + final.add_scalar_arg(legate::Scalar(finish_kernel)); + final.add_scalar_arg(mapreduce_payload(finish, finish_size)); + rt->submit(std::move(final)); + return; + } + auto acc = accumulator->obj.get_store(); + auto red = acc; + if (full) { + if (red.dim() == 0) red = red.promote(0, 1); + } else { + // Project from high to low, then promote from low to high: axis numbers + // refer to the original input throughout both transformations. + for (int d = red.dim() - 1; d >= 0; --d) + if ((axes >> d) & 1) red = red.project(d, 0); + for (int d = 0; d < src.dim(); ++d) + if ((axes >> d) & 1) red = red.promote(d, src.shape()[d]); + if (red.dim() == 0) red = red.promote(0, 1); + } + auto task = rt->create_task(library, ufi::RunPTXMapReduceTask::TASK_CONFIG.task_id()); + auto p_src = task.add_input(src); + // A singleton has no reduction arithmetic: even multiplying complex Inf by + // an identity can introduce NaNs. Give it an ordinary output privilege. + auto p_red = single ? task.add_output(red) + : task.add_reduction(red, static_cast(op)); + if (!full) task.add_constraint(legate::align(p_src, p_red)); + task.add_scalar_arg(legate::Scalar(kernel)); + task.add_scalar_arg(legate::Scalar(axes)); + task.add_scalar_arg(legate::Scalar(full)); + task.add_scalar_arg(legate::Scalar(static_cast(op))); + task.add_scalar_arg(mapreduce_payload(mapper, mapper_size)); + task.add_scalar_arg(legate::Scalar(contribute_kernel)); + task.add_scalar_arg(legate::Scalar(single)); + rt->submit(std::move(task)); + + if (finish_kernel.empty()) return; + auto dst = output->obj.get_store(); + if (acc.dim() == 0) acc = acc.promote(0, 1); + if (dst.dim() == 0) dst = dst.promote(0, 1); + auto final = rt->create_task(library, ufi::RunPTXReduceFinishTask::TASK_CONFIG.task_id()); + auto p_acc = final.add_input(acc); + auto p_dst = final.add_output(dst); + final.add_constraint(legate::align(p_acc, p_dst)); + final.add_scalar_arg(legate::Scalar(finish_kernel)); + final.add_scalar_arg(mapreduce_payload(finish, finish_size)); + rt->submit(std::move(final)); } #endif @@ -67,6 +160,23 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { wrap_binary_ops(mod); wrap_unary_reds(mod); wrap_linalg_ops(mod); + wrap_fft_ops(mod); + wrap_bitgenerator_ops(mod); + + mod.add_enum("MapReduceOp", + std::vector({ + "MAPREDUCE_ADD", + "MAPREDUCE_MUL", + "MAPREDUCE_MIN", + "MAPREDUCE_MAX" + }), + std::vector({ + static_cast(ufi::MapReduceOp::ADD), + static_cast(ufi::MapReduceOp::MUL), + static_cast(ufi::MapReduceOp::MIN), + static_cast(ufi::MapReduceOp::MAX) + }) + ); using jlcxx::ParameterList; using jlcxx::Parametric; @@ -96,6 +206,95 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { mod.method("get_lib", &get_lib); mod.method("nda_store_to_ndarray", &nda_store_to_ndarray); + // True when the loaded cuSolver provides cusolverDnXgeev. Without it + // cupynumeric has no GPU eigenvalue kernel for general matrices. + mod.method("cusolver_has_geev", &cupynumeric_cusolver_has_geev); + mod.method("cusolvermp_available", &cupynumeric_has_cusolvermp); + + // Match cuPyNumeric 26.06: the MP kernels use a single NCCL communicator. + mod.method("add_nccl_communicator", + [](legate::ManualTask& task) { task.add_communicator("nccl"); }); + mod.method("add_nccl_communicator", + [](legate::AutoTask& task) { task.add_communicator("nccl"); }); + + // Tiled Cholesky launches subrectangles in the original partition's color + // space. Bounds are inclusive and zero-based, as in Legion::Rect. + mod.method("create_linalg_task", [](legate::LocalTaskID id, int64_t row_lo, + int64_t col_lo, int64_t row_hi, + int64_t col_hi) { + auto domain = Legion::Domain{Legion::Rect<2>{ + Legion::Point<2>{row_lo, col_lo}, Legion::Point<2>{row_hi, col_hi}}}; + return legate::Runtime::get_runtime()->create_task(get_lib(), id, domain); + }); + mod.method("add_input_tile", + [](legate::ManualTask& task, + std::shared_ptr part, + uint64_t row, uint64_t col) { + std::vector color{row, col}; + task.add_input(part->get_child_store(color)); + }); + mod.method( + "add_input_column", + [](legate::ManualTask& task, + std::shared_ptr part, int32_t col) { + task.add_input(*part, + legate::SymbolicPoint{std::vector{ + legate::dimension(0), legate::constant(col)}}); + }); + + mod.method("add_input_proj", + [](legate::ManualTask& task, + std::shared_ptr part, + const std::vector& dims) { + task.add_input(*part, make_projection(dims)); + }); + mod.method("add_output_proj", + [](legate::ManualTask& task, + std::shared_ptr part, + const std::vector& dims) { + task.add_output(*part, make_projection(dims)); + }); + + // Marking a task as throwing also grows its leaf allocation pools by + // --max-exception-size (4096 bytes by default), which the cupynumeric + // decomposition tasks rely on: their mapper only declares enough zero-copy + // memory for a single status flag, while the batched GPU kernels allocate one + // per matrix. cupynumeric's own Python launchers always set this. + mod.method("task_throws_exception", [](legate::ManualTask& task, bool value) { + task.throws_exception(value); + }); + mod.method("task_throws_exception", [](legate::AutoTask& task, bool value) { + task.throws_exception(value); + }); + + // Legate.jl Scalar has no vector constructors. BITGENERATOR (and similar) + // tasks take fixed-array scalars; these helpers pack them from Julia + // pointers. + mod.method("add_vector_scalar_i64", + [](legate::AutoTask& task, const int64_t* p, int32_t n) { + std::vector v; + if (n > 0) { + v.assign(p, p + n); + } + task.add_scalar_arg(legate::Scalar(std::move(v))); + }); + mod.method("add_vector_scalar_f32", + [](legate::AutoTask& task, const float* p, int32_t n) { + std::vector v; + if (n > 0) { + v.assign(p, p + n); + } + task.add_scalar_arg(legate::Scalar(std::move(v))); + }); + mod.method("add_vector_scalar_f64", + [](legate::AutoTask& task, const double* p, int32_t n) { + std::vector v; + if (n > 0) { + v.assign(p, p + n); + } + task.add_scalar_arg(legate::Scalar(std::move(v))); + }); + auto ndarray_accessor = mod.add_type, TypeVar<2>>>("NDArrayAccessor"); ndarray_accessor @@ -110,6 +309,7 @@ JLCXX_MODULE define_julia_module(jlcxx::Module& mod) { #if LEGATE_DEFINED(LEGATE_USE_CUDA) mod.method("register_tasks", ®ister_tasks); + mod.method("submit_mapreduce", &submit_mapreduce); wrap_cuda_methods(mod); #endif } diff --git a/scripts/install_cxxwrap.sh b/scripts/install_cxxwrap.sh index c8e2892d2..705ab14ea 100755 --- a/scripts/install_cxxwrap.sh +++ b/scripts/install_cxxwrap.sh @@ -33,8 +33,8 @@ if [[ ! -d "$CUNUMERIC_ROOT_DIR" ]]; then exit 1 fi -JULIA='julia' -JULIA_PATH=$(which $JULIA) +JULIA=${JULIA:-julia} +JULIA_PATH=$(command -v "$JULIA") if [ -z "$JULIA_PATH" ]; then echo "Error: $JULIA is not installed or not in PATH." @@ -44,44 +44,42 @@ fi echo "Using $JULIA at: $JULIA_PATH" GIT_REPO="https://github.com/JuliaInterop/libcxxwrap-julia.git" -COMMIT_HASH="89e4699837bfa0929610c9e330889fb2df925b47" #(v14.2) +COMMIT_HASH="ee8a49b403ced7669c8fa56cec860f567f6510aa" #(v14.11) JULIA_CXXWRAP_SRC=$CUNUMERIC_ROOT_DIR/lib/libcxxwrap-julia if [ ! -d "$JULIA_CXXWRAP_SRC" ]; then - cd $CUNUMERIC_ROOT_DIR/lib - git clone $GIT_REPO + mkdir -p "$CUNUMERIC_ROOT_DIR/lib" + git clone "$GIT_REPO" "$JULIA_CXXWRAP_SRC" fi -cd $JULIA_CXXWRAP_SRC +cd "$JULIA_CXXWRAP_SRC" git fetch --tags git checkout $COMMIT_HASH # find julia dependency path -JULIA_DEP_PATH=$($JULIA -e 'println(DEPOT_PATH[1])') +JULIA_DEP_PATH=$("$JULIA_PATH" --startup-file=no -e 'print(DEPOT_PATH[1])') # https://github.com/JuliaInterop/libcxxwrap-julia/tree/v0.13.3?tab=readme-ov-file#configuring-and-building JULIA_CXXWRAP_DEV=$JULIA_DEP_PATH/dev/libcxxwrap_julia_jll JULIA_CXXWRAP=$JULIA_CXXWRAP_DEV/override -# Clean up whatever env is there right now and -# build default version of CxxWrap / libcxxwrap_julia -#* THIS COULD BREAK SOME USERS CODE IF THEY ALREADY OVERRIDE THIS PKG -cd $CUNUMERIC_ROOT_DIR -[ -f Manifest.toml ] && rm Manifest.toml -rm -rf $JULIA_CXXWRAP_DEV -julia -e 'using Pkg; Pkg.activate("."); Pkg.add("Legate")' -julia -e 'using Pkg; Pkg.activate("."); Pkg.precompile(["CxxWrap"])' - -# https://github.com/JuliaInterop/libcxxwrap-julia/tree/v0.13.3?tab=readme-ov-file#preparing-the-install-location -# this command will download https://github.com/JuliaBinaryWrappers/libcxxwrap_julia_jll.jl and install it in JULIA_DEP_PATH -julia -e 'using Pkg; Pkg.activate("."); Pkg.develop(PackageSpec(name="libcxxwrap_julia_jll")); import libcxxwrap_julia_jll; libcxxwrap_julia_jll.dev_jll()' - - -# JULIA_CXXWRAP_OVERRIDE=$JULIA_CXXWRAP/override/ -# Delete the default JLL installation of cxxwrap_julia -rm -rf $JULIA_CXXWRAP -mkdir $JULIA_CXXWRAP - -cmake -S $JULIA_CXXWRAP_SRC -B $JULIA_CXXWRAP -DJulia_EXECUTABLE=$JULIA_PATH -DCMAKE_BUILD_TYPE=Release -cd $JULIA_CXXWRAP -make -j "${JULIA_CPU_THREADS:-2}" +# Keep the resolved environment and developed JLL checkout. Only its generated +# override is disposable. Avoid importing/precompiling a possibly broken JLL +# before its replacement libraries have been built. +cd "$CUNUMERIC_ROOT_DIR" +JULIA_PKG_PRECOMPILE_AUTO=0 "$JULIA_PATH" --startup-file=no --project="$CUNUMERIC_ROOT_DIR" -e ' + using Pkg + checkout = joinpath(DEPOT_PATH[1], "dev", "libcxxwrap_julia_jll") + if isdir(checkout) + Pkg.develop(path=checkout) + else + Pkg.develop(PackageSpec(name="libcxxwrap_julia_jll"); shared=true) + end +' + +rm -rf "$JULIA_CXXWRAP" +mkdir -p "$JULIA_CXXWRAP" + +cmake -S "$JULIA_CXXWRAP_SRC" -B "$JULIA_CXXWRAP" \ + -DJulia_EXECUTABLE="$JULIA_PATH" -DCMAKE_BUILD_TYPE=Release +cmake --build "$JULIA_CXXWRAP" --parallel "${JULIA_CPU_THREADS:-2}" diff --git a/scripts/wrapper_changed.sh b/scripts/wrapper_changed.sh new file mode 100755 index 000000000..0b3edc67b --- /dev/null +++ b/scripts/wrapper_changed.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +# Exit 0 if the wrapper at REF (default HEAD) matches RELEASED_COMMIT (the +# source of the released wrapper JLL), 1 if it differs. + +set -euo pipefail + +readonly WRAPPER_PATH="lib/cunumeric_jl_wrapper" +readonly RELEASED_COMMIT_FILE="$WRAPPER_PATH/RELEASED_COMMIT" +ref="${1:-HEAD}" + +released="$(tr -d '[:space:]' < "$RELEASED_COMMIT_FILE")" +if ! git cat-file -e "${released}^{commit}" 2>/dev/null; then + git fetch --no-tags --depth=1 origin "$released" +fi + +git diff --quiet "$released" "$ref" -- "$WRAPPER_PATH" ":(exclude)$RELEASED_COMMIT_FILE" diff --git a/src/cnscalar.jl b/src/cnscalar.jl new file mode 100644 index 000000000..8ee80f6ab --- /dev/null +++ b/src/cnscalar.jl @@ -0,0 +1,303 @@ +export CNFloat, CNInt, CNUInt, CNBool, CNReal, CNComplex, CNScalar, DeviceScalar, + cnscalar, allowautofetch, @allowautofetch + +"A floating-point device scalar; wrapping its 0D storage does not synchronize." +struct CNFloat{T<:AbstractFloat,P} <: AbstractFloat + value::NDArray{T,0,P} +end + +"A signed integer device scalar backed by a 0D NDArray." +struct CNInt{T<:Signed,P} <: Signed + value::NDArray{T,0,P} +end + +"An unsigned integer device scalar backed by a 0D NDArray." +struct CNUInt{T<:Unsigned,P} <: Unsigned + value::NDArray{T,0,P} +end + +"A Boolean device scalar. Bool is concrete, so the wrapper subtypes Integer." +struct CNBool{T<:Bool,P} <: Integer + value::NDArray{T,0,P} +end + +"A complex device scalar backed by a 0D NDArray; wrapping does not synchronize." +struct CNComplex{T<:Complex,P} <: Number + value::NDArray{T,0,P} +end + +# Bounded wrapper branches allow both DeviceScalar{Float64} and +# DeviceScalar{ComplexF64}; an exact CNComplex{Float64} would violate its bound. +const CNReal{T} = Union{CNFloat{<:T},CNInt{<:T},CNUInt{<:T},CNBool{<:T}} +const CNScalar{T} = Union{CNReal{T},CNComplex{<:T}} +const DeviceScalar{T} = Union{NDArray{T,0},CNScalar{T}} +const _REAL_SCALAR_WRAPPERS = (CNFloat, CNInt, CNUInt, CNBool) +const _SCALAR_WRAPPERS = (_REAL_SCALAR_WRAPPERS..., CNComplex) +for (W, H) in ((CNFloat, AbstractFloat), (CNInt, Signed), (CNUInt, Unsigned), + (CNBool, Bool), (CNComplex, Complex)) + # Number's identity constructor otherwise conflicts with the generated + # field constructor when a wrapped scalar is passed back to its type. + @eval $W{T,P}(x::$W{T,P}) where {T<:$H,P} = x + @eval cnscalar(x::NDArray{T,0}) where {T<:$H} = $W(x) + @eval _scalar_type(::Type{T}) where {T<:$H} = $W{T} + @eval _scalar_eltype(::Type{<:$W{T}}) where {T} = T +end +cnscalar(x::CNScalar) = x +_scalar_result(x::NDArray{<:Number,0}) = cnscalar(x) +_scalar_result(x) = x + +_scale_storage(x::CNScalar) = x.value +# Host shortcuts must not inspect a device coefficient's value. +_host_iszero(::CNScalar) = false +_host_isone(::CNScalar) = false + +# Numeric data stays in backend storage; host control parameters are explicitly +# permission-checked for both wrapped and unwrapped device scalars. +_maybe_fetch(x) = x +function _maybe_fetch(x::DeviceScalar{<:Real}) + _assert_allowautofetch() + return only(_scale_storage(x)) +end + +_coefficient_type(x) = typeof(x) +_coefficient_type(x::DeviceScalar) = eltype(_scale_storage(x)) +_coefficient_as(::Type{T}, x::Number) where {T} = convert(T, x) +_coefficient_as(::Type{T}, x::DeviceScalar) where {T} = checked_promote_arr(_scale_storage(x), T) + +searchsortedfirst(a::NDArray{T,1}, x::CNScalar) where {T} = searchsortedfirst(a, x.value) +searchsortedlast(a::NDArray{T,1}, x::CNScalar) where {T} = searchsortedlast(a, x.value) + +function Base.fill!(a::NDArray, x::DeviceScalar) + a .= _scale_storage(x) + return a +end +function fill(x::DeviceScalar, dims::Dims) + a = cuNumeric.zeros(_coefficient_type(x), dims) + return fill!(a, x) +end +fill(x::DeviceScalar, dims::Int...) = fill(x, dims) +fill(x::DeviceScalar, dim::Int) = fill(x, (dim,)) + +""" + allowautofetch(f, allow=true) + allowautofetch(allow::Bool=true) + @allowautofetch expression + +Permit implicit host extraction of CNScalars for comparisons, predicates, and +conversion to host numeric types. Arithmetic stays on the backend. The do-block +and macro restore the calling task's previous permission, including on errors. +Permission is task-local and separate from scalar indexing and promotion. +This is not a fallback for arbitrary functions with unsupported argument types. +""" +allowautofetch(f::F, allow::Bool=true) where {F} = + task_local_storage(f, :cuNumericAllowAutoFetch, allow) +allowautofetch(allow::Bool=true) = (task_local_storage(:cuNumericAllowAutoFetch, allow); nothing) + +macro allowautofetch(ex) + quote + local previous = get(task_local_storage(), :cuNumericAllowAutoFetch, nothing) + task_local_storage(:cuNumericAllowAutoFetch, true) + @__tryfinally($(esc(ex)), + if isnothing(previous) + delete!(task_local_storage(), :cuNumericAllowAutoFetch) + else + task_local_storage(:cuNumericAllowAutoFetch, previous) + end) + end +end + +function _assert_allowautofetch() + get(task_local_storage(), :cuNumericAllowAutoFetch, false) && return nothing + throw(ArgumentError("Implicit CNScalar host extraction is disabled. Use " * + "allowautofetch() do ... end or @allowautofetch, or explicitly call fetch(x).")) +end + +""" + fetch(x::CNScalar) + fetch(x::NDArray) + +Retrieve a native Julia scalar, waiting for the result as needed. An NDArray +must contain exactly one element. Explicit fetching does not require +`allowautofetch` or `allowscalar` permission. +""" +Base.fetch(x::CNScalar) = only(x.value) +Base.only(x::CNScalar) = fetch(x) +# Preserve explicit scalar indexing of reduction results, including allowscalar. +# Number's default getindex would instead return the wrapper unchanged. +Base.getindex(x::CNScalar) = x.value[] +destroy!(x::CNScalar) = destroy!(x.value) +Base.copy(x::CNScalar) = cnscalar(copy(x.value)) +Base.broadcastable(x::CNScalar) = x.value +# Ref(device_scalar) still represents one backend value, not a host kernel arg. +Base.broadcastable(x::Base.RefValue{<:DeviceScalar}) = _scale_storage(x[]) +# Display is an intentional host extraction, just as for the backing 0D NDArray. +Base.show(io::IO, x::CNScalar) = show(io, x.value) +Base.show(io::IO, mime::MIME"text/plain", x::CNScalar) = show(io, mime, x.value) + +_scalar_operand(x::CNScalar) = x.value +_scalar_operand(x::Number) = x +_scalar_operand(x::NDArray{<:Any,0}) = x +_scalar_binary(f, x, y) = cnscalar(broadcast(f, _scalar_operand(x), _scalar_operand(y))) +function _scalar_compare(f, x, y) + _assert_allowautofetch() + result = broadcast(f, _scalar_operand(x), _scalar_operand(y)) + value = only(result) + destroy!(result) + return value +end +_scalar_host(x::CNScalar) = fetch(x) +_scalar_host(x::Number) = x +function _scalar_compare(f::Union{typeof(isless),typeof(isequal)}, x, y) + _assert_allowautofetch() + return f(_scalar_host(x), _scalar_host(y)) +end + +# Explicit intersections avoid ambiguities with Base's Real/Complex methods. +for op in (:+, :-, :*, :/, :^), W in _SCALAR_WRAPPERS + for H in (Real, Complex, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_binary($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_binary($op, x, y) + end + for V in _SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_binary($op, x, y) + end +end +for op in (:(==), :(!=), :isequal), W in _SCALAR_WRAPPERS + for H in (Real, Complex, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_compare($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_compare($op, x, y) + end + for V in _SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_compare($op, x, y) + end +end +for op in (:<, :<=, :>, :>=, :isless), W in _REAL_SCALAR_WRAPPERS + for H in (Real, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_compare($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_compare($op, x, y) + end + for V in _REAL_SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_compare($op, x, y) + end +end +for op in (:min, :max), W in _REAL_SCALAR_WRAPPERS + for H in (Real, AbstractFloat, Integer, Signed, Unsigned, Bool) + @eval Base.$op(x::$W, y::$H) = _scalar_binary($op, x, y) + @eval Base.$op(x::$H, y::$W) = _scalar_binary($op, x, y) + end + for V in _REAL_SCALAR_WRAPPERS + @eval Base.$op(x::$W, y::$V) = _scalar_binary($op, x, y) + end +end +Base.literal_pow(::typeof(^), x::CNScalar, ::Val{P}) where {P} = x ^ P + + +Base.:*(x::CNScalar, a::NDArray) = broadcast(*, x, a) +Base.:*(a::NDArray, x::CNScalar) = broadcast(*, a, x) +for op in (:+, :-, :*, :/, :^) + @eval begin + Base.$op(x::CNScalar, a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}) = _scalar_binary($op, x, a) + Base.$op(a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, x::CNScalar) = _scalar_binary($op, a, x) + end +end + +function _scalar_unary(f, x::CNScalar) + # Some backend unary kernels reject rank zero. A size-one view avoids host + # extraction and works on both CPU and GPU targets. + input = cuNumeric.reshape(x.value, 1) + output = broadcast(f, input) + result = cuNumeric.reshape(output, ()) + destroy!(input) + destroy!(output) + return cnscalar(result) +end +for op in (:-, :abs, :sqrt), W in _SCALAR_WRAPPERS + @eval Base.$op(x::$W) = _scalar_unary($op, x) +end +for op in (:conj, :real, :imag) + @eval Base.$op(x::CNComplex) = _scalar_unary($op, x) +end +for W in _SCALAR_WRAPPERS + @eval Base.:+(x::$W) = x + @eval Base.inv(x::$W) = one(eltype(x.value)) / x +end +for W in _REAL_SCALAR_WRAPPERS + @eval Base.real(x::$W) = x + @eval Base.conj(x::$W) = x + @eval Base.imag(x::$W) = zero(x) +end +Base.abs2(x::CNReal) = x * x +Base.abs2(x::CNComplex) = real(x * conj(x)) +Base.:!(x::CNReal{Bool}) = _scalar_unary(!, x) +for op in (:iszero, :isone, :isfinite, :isinf, :isnan), W in _SCALAR_WRAPPERS + @eval function Base.$op(x::$W) + _assert_allowautofetch() + return $op(fetch(x)) + end +end + +_scalar_convert(::Type{T}, x::CNScalar) where {T} = cnscalar(checked_promote_arr(x.value, T)) +_scalar_convert(::Type{T}, x::Number) where {T} = cnscalar(NDArray(convert(T, x))) +function _checked_scalar_convert(::Type{T}, x::CNScalar) where {T} + # These conversions can throw InexactError based on the value. A backend + # dtype cast would silently truncate or discard an imaginary component. + _assert_allowautofetch() + return cnscalar(NDArray(convert(T, fetch(x)))) +end +_scalar_convert(::Type{T}, x::CNComplex) where {T<:Real} = _checked_scalar_convert(T, x) +_scalar_convert(::Type{T}, x::CNFloat) where {T<:Integer} = + _checked_scalar_convert(T, x) +function _scalar_convert(::Type{T}, x::Union{CNInt,CNUInt,CNBool}) where {T<:Integer} + S = _scalar_eltype(typeof(x)) + if typemin(T) <= typemin(S) && typemax(T) >= typemax(S) + return cnscalar(checked_promote_arr(x.value, T)) + end + return _checked_scalar_convert(T, x) +end +for W in _SCALAR_WRAPPERS + @eval begin + $W{T}(x::Number) where {T} = _scalar_convert(T, x) + Base.convert(::Type{S}, x::Number) where {T,S<:$W{T}} = _scalar_convert(T, x) + Base.convert(::Type{S}, x::S) where {S<:$W} = x + Base.zero(::Type{S}) where {S<:$W} = cnscalar(NDArray(zero(_scalar_eltype(S)))) + Base.one(::Type{S}) where {S<:$W} = cnscalar(NDArray(one(_scalar_eltype(S)))) + end +end +Base.zero(x::CNScalar) = zero(typeof(x)) +Base.one(x::CNScalar) = one(typeof(x)) +Base.real(::Type{S}) where {S<:CNScalar} = _scalar_type(real(_scalar_eltype(S))) +Base.float(x::CNScalar) = _scalar_convert(float(_scalar_eltype(typeof(x))), x) +Base.float(x::CNFloat) = x +Base.float(::Type{S}) where {S<:CNScalar} = _scalar_type(float(_scalar_eltype(S))) + +for T in Base.uniontypes(SUPPORTED_ARRAY_TYPES), W in _SCALAR_WRAPPERS + @eval function Base.convert(::Type{$T}, x::$W) + _assert_allowautofetch() + return convert($T, fetch(x)) + end + @eval (::Type{$T})(x::$W) = convert($T, x) +end + +# Generic promotion keeps values device-backed; host conversion is a separate, +# permission-checked operation. Real and complex wrappers remain distinct. +for W in _SCALAR_WRAPPERS + @eval Base.promote_rule(::Type{S}, ::Type{T}) where {S<:$W,T<:Union{Real,Complex}} = + _scalar_type(promote_type(_scalar_eltype(S), T)) + for V in _SCALAR_WRAPPERS + @eval Base.promote_rule(::Type{S}, ::Type{T}) where {S<:$W,T<:$V} = + _scalar_type(promote_type(_scalar_eltype(S), _scalar_eltype(T))) + end +end +for T in Base.uniontypes(SUPPORTED_ARRAY_TYPES), W in _SCALAR_WRAPPERS + @eval Base.promote_rule(::Type{$T}, ::Type{S}) where {S<:$W} = + _scalar_type(promote_type($T, _scalar_eltype(S))) +end + +# Internal composition of reductions consumes storage, never a host scalar. +Base.mapreduce(f, op, x::CNScalar; kwargs...) = mapreduce(f, op, x.value; kwargs...) +_div_nelem(x::CNScalar, n::Integer) = x / n +function _sqrt_ndarray(x::CNScalar) + result = sqrt(x) + destroy!(x) + return result +end diff --git a/src/cuNumeric.jl b/src/cuNumeric.jl index bb27774a6..60df70eb2 100644 --- a/src/cuNumeric.jl +++ b/src/cuNumeric.jl @@ -23,13 +23,14 @@ module cuNumeric using Preferences using CNPreferences using LegatePreferences: LegatePreferences +# Load CUDACore before Legate/CxxWrap to avoid recompilation in its __init__. +using CUDACore: CUDACore +import CUDACore: CuArray using Legate using Libdl using CxxWrap using CUDATools: CUDATools -using CUDACore: CUDACore -import CUDACore: CuArray import KernelAbstractions: @kernel, @index import KernelAbstractions as KA @@ -38,24 +39,28 @@ using cunumeric_jl_wrapper_jll import Base: axes, convert, copy, copyto!, inv, isfinite, sqrt, -, +, *, ==, !=, isapprox, read, view, maximum, minimum, prod, sum, getindex, setindex!, - sum, prod + sum, prod, argmax, argmin using LinearAlgebra import LinearAlgebra: mul! +import AbstractFFTs: fft, ifft, bfft!, fft!, ifft! + using Random -import Random: rand! +import Random: rand!, randn!, randexp! + +using StaticArrays: SVector using StatsBase -import StatsBase: var, mean +import StatsBase: var, mean, std include(joinpath(@__DIR__, "../deps/version.jl")) include("utilities/preference.jl") -const HAS_CUDA = LegatePreferences.has_cuda_gpu() -if !HAS_CUDA - @warn "We couldn't find a CUDA-enabled GPU. If you have an NVIDIA GPU something might be wrong." -end +# Populated after Legate starts and resolves its automatic or explicit machine +# configuration. This reflects configured GPU targets, not merely visible hardware. +const HAS_CUDA = Ref(false) +@inline _has_gpu_target() = HAS_CUDA[] const DEFAULT_FLOAT = Float32 const DEFAULT_INT = Int32 @@ -72,6 +77,8 @@ const SUPPORTED_NUMERIC_TYPES = Union{ const SUPPORTED_SOLVE_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} const SUPPORTED_SVD_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} const SUPPORTED_QR_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const SUPPORTED_CHOLESKY_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const SUPPORTED_EIG_TYPES = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} const SUPPORTED_ARRAY_TYPES = Union{Bool,SUPPORTED_NUMERIC_TYPES} const SUPPORTED_TYPES = Union{SUPPORTED_ARRAY_TYPES,String} @@ -149,9 +156,28 @@ include("warnings.jl") # Compile-time so task scope instrumentation is fully elided when disabled. const TASK_SCOPE_NAMES = CNPreferences.TASK_SCOPE_NAMES +# Loaded at compile time; change through CNPreferences and restart Julia. +const MIN_SOLVE_MATRIX_SIZE = load_preference(CNPreferences, "MIN_SOLVE_MATRIX_SIZE", 2048) +const MIN_SOLVE_TILE_SIZE = load_preference(CNPreferences, "MIN_SOLVE_TILE_SIZE", 512) +const MIN_CHOLESKY_MATRIX_SIZE = load_preference(CNPreferences, "MIN_CHOLESKY_MATRIX_SIZE", 8192) +const MIN_CHOLESKY_TILE_SIZE = load_preference(CNPreferences, "MIN_CHOLESKY_TILE_SIZE", 2048) +const MIN_QR_MATRIX_SIZE = load_preference(CNPreferences, "MIN_QR_MATRIX_SIZE", 1048576) +const QR_TILE_SIZE = load_preference(CNPreferences, "QR_TILE_SIZE", 128) +const MAX_CHOLESKY_TILES_PER_PROC = load_preference(CNPreferences, "MAX_CHOLESKY_TILES_PER_PROC", 4) + +for key in ( + :MIN_SOLVE_MATRIX_SIZE, :MIN_SOLVE_TILE_SIZE, :MIN_CHOLESKY_MATRIX_SIZE, + :MIN_CHOLESKY_TILE_SIZE, :MIN_QR_MATRIX_SIZE, :QR_TILE_SIZE, :MAX_CHOLESKY_TILES_PER_PROC, +) + value = getfield(@__MODULE__, key) + value isa Int && value > 0 || throw(ArgumentError("$key must be a positive Int")) +end + # NDArray internal include("ndarray/detail/ndarray.jl") +include("ndarray/detail/distributed_linalg.jl") include("ndarray/detail/linalg.jl") +include("ndarray/detail/fft.jl") # Utilities include("cuda/strided_device_array.jl") @@ -166,14 +192,26 @@ const FUSE_BROADCAST_EXPRS = CNPreferences.FUSE_BROADCAST const FUSE_BROADCAST_MIN_OPS = CNPreferences.FUSE_BROADCAST_MIN_OPS # Functionality +include("cnscalar.jl") +include("ndarray/diagonal.jl") include("ndarray/promotion.jl") include("cuda/cuda_ptx_task.jl") include("ndarray/broadcast_fusion.jl") include("ndarray/broadcast.jl") include("ndarray/ndarray.jl") +include("ndarray/random/bitgenerator.jl") +include("ndarray/random/generator.jl") +include("ndarray/random/random.jl") include("ndarray/unary.jl") +include("ndarray/mapreduce.jl") +include("cuda/mapreduce.jl") include("ndarray/binary.jl") include("ndarray/linalg.jl") +include("ndarray/sort.jl") +include("ndarray/batched_linalg.jl") +include("ndarray/contract.jl") +include("ndarray/vector_linalg.jl") +include("ndarray/fft.jl") include("scoping/scoping.jl") # From https://github.com/JuliaGraphics/QML.jl/blob/dca239404135d85fe5d4afe34ed3dc5f61736c63/src/QML.jl#L147 @@ -194,8 +232,6 @@ function my_on_exit() return drain_pending_frees!() # flush before Legate tears down end -global cuNumeric_config_str::String = "" - ### These functions guard against a user trying ### to start multiple runtimes and also to allow ## package extensions which always try to re-load @@ -215,6 +251,12 @@ function _start_runtime() # AA = ArgcArgv([Base.julia_cmd()[1]]) cuNumeric.initialize_cunumeric(AA.argc, getargv(AA)) + num_gpus = Int(Legate.num_gpus()) + HAS_CUDA[] = num_gpus > 0 + _LINALG_RUNTIME[] = _LinalgRuntime( + cusolvermp_available(), num_gpus, Int(Legate.num_procs()) + ) + _init_deferred_free!() # record launch thread for deferred frees (memory.jl) # setup /src/memory.jl @@ -254,8 +296,6 @@ function __init__() _is_precompiling() && return nothing - _register_scoping_error_hint!() - # Cannot set LEGATE_CONFIG on CI machines used # to register packages. So we will just skip starting # legate/cunumeric when using registry CI machines. diff --git a/src/cuda/README.md b/src/cuda/README.md new file mode 100644 index 000000000..33136e060 --- /dev/null +++ b/src/cuda/README.md @@ -0,0 +1,36 @@ +# GPU mapped reductions + +The frontend in `../ndarray/mapreduce.jl` selects reduction policies, validates +shapes/types, and handles empty inputs. `mapreduce.jl` compiles and caches Julia +PTX kernels; capture values and array extents are runtime arguments. + +Each Legate GPU task: + +1. Packs the input tile's lower-bound pointer, extents, and element strides. + `_mr_offset` enumerates retained and reduced coordinates separately, so slices + need not be contiguous. +2. Maps values and reduces them in 256-thread blocks using shared memory. +3. Combines block partials into the task's reduction store, with one writer per + retained coordinate through an exclusive reduction accessor. + +**Legate combines contributions between tasks and GPUs.** The native glue in +`../../lib/cunumeric_jl_wrapper/src/mapreduce.cpp` packs descriptors and launches +PTX on the task stream. Scratch is task-local and bounded to 4,096 partials; +there is no input-sized mapped temporary. + +The reduction tasks use a separate Legate library with a mapper that reserves +64 KiB of framebuffer scratch per allocating task (4,096 × 16 bytes). The reduction +variant declares `has_allocations`; both variants declare that all device work +uses the task stream. Registering them with cuPyNumeric's mapper would omit this +scratch reservation and can abort even when device memory is available. + +Floating extrema use ordered unsigned keys to preserve NaNs and signed zeros. +Boolean product/extrema use 0/1 bytes. A finishing kernel decodes storage and +applies `init` once when needed; otherwise the accumulator is returned directly. +Singletons use output privileges to avoid identity arithmetic, while dimensional +sums/products retain Base's zero/one seed. + +Submission copies capture bytes into task-owned scalars. Temporary NDArray +handles are explicitly released after submission; native store handles use C++ +scope ownership. Warm calls neither extract host results nor insert execution +fences. Kernel registration uses the shared PTX loader on cache misses. diff --git a/src/cuda/cuda_ptx_task.jl b/src/cuda/cuda_ptx_task.jl index 87962b7fa..97502fda1 100644 --- a/src/cuda/cuda_ptx_task.jl +++ b/src/cuda/cuda_ptx_task.jl @@ -2,7 +2,11 @@ export @cuda_task, @launch, CUDATask struct CUDATask func::String - argtypes::NTuple{N,Type} where {N} #! THIS IS TYPE UNSTABLE + argtypes::Vector{DataType} + + function CUDATask(func, argtypes) + return new(convert(String, func), collect(DataType, argtypes)) + end end #! JUST PASS TYPES HERE INSTEAD OF CALLING typeof() @@ -16,57 +20,59 @@ function to_stdvec(::Type{T}, vec) where {T} return stdvec end -function add_padding(arr::NDArray, dims::Dims{N}; copy=false) where {N} - old_size = size(arr) - - @assert all(dims .>= old_size) "newdims must be ≥ current dims elementwise" - new = zeros(eltype(arr), dims) +@inline _launch_shape(arr::NDArray) = _launch_shape(arr, _padding(arr)) +@inline _launch_shape(arr::NDArray, ::Nothing) = size(arr) +@inline _launch_shape(::NDArray, padding::PaddedStorage) = padding.shape + +@inline _physical_array(arr::NDArray, ::Nothing) = arr +@inline _physical_array(::NDArray, padding::PaddedStorage) = padding.backing + +function _ensure_launch_padding!(arr::NDArray{T,N}, target_shape; copy=false) where {T,N} + padding = _padding(arr) + !isnothing(padding) && padding.shape == target_shape && return arr + isnothing(padding) && size(arr) == target_shape && return arr + @assert all(target_shape .>= size(arr)) "cannot pad $(size(arr)) to $target_shape" + + padded = zeros(T, target_shape) + slices = ntuple(d -> (0, size(arr, d)), N) + logical = nda_get_slice(padded, slice_array(slices...)) + aliases_parent = !isnothing(arr.parent) + copy && !aliases_parent && copyto!(logical, arr) + storage = PaddedStorage{T,N}( + padded, + aliases_parent ? logical : nothing, + target_shape, + ) - if copy # due to being an input. we don't need to copy outputs - slices = ntuple(d -> (0, Int(old_size[d])), length(old_size)) - s = nda_get_slice(new, slice_array(slices...)) - copyto!(s, arr) - destroy!(s) + if aliases_parent + old_padding = _padding(arr) + arr.padding = storage + !isnothing(old_padding) && _destroy_padded_storage!(old_padding) + else + destroy!(arr) + arr.ptr = logical.ptr + arr.nbytes = logical.nbytes + arr.padding = storage + logical.ptr = Ptr{Cvoid}(0) + logical.nbytes = 0 end - - nda_destroy_array(arr.ptr) - register_free!(arr.nbytes) - - # update pointer & update metadata - arr.ptr = new.ptr - arr.nbytes = new.nbytes - arr.padding = old_size # remember the prior (before the padding) - - # julia GC will call finalizer, but we manually cleaned it - new.ptr = Ptr{Cvoid}(0) - new.nbytes = 0 - return new.padding = nothing -end - -function add_padding(arr::NDArray, i::Int64; copy=false) - return add_padding(arr, (i,); copy=copy) + return arr end -function check_sz!(arr, maxshape; copy=false) - sz = cuNumeric.size(arr) - if maxshape != nothing - # currently require all ndarray inputs to be equal - alligned_equal_size = sz == maxshape - if !alligned_equal_size - cuNumeric.add_padding(arr, maxshape; copy=copy) - new_size = padded_shape(arr) - @warn "[Padding Added] $sz output is now $new_size" - end +function _sync_to_launch_padding!(arr::NDArray) + padding = _padding(arr) + if !isnothing(padding) && !isnothing(padding.staging) + copyto!(padding.staging, arr) end + return nothing end -function check_sz(arr, maxshape) - sz = cuNumeric.size(arr) - if maxshape != nothing - # currently require all ndarray inputs to be equal - alligned_equal_size = sz == maxshape - @assert alligned_equal_size +function _sync_from_launch_padding!(arr::NDArray) + padding = _padding(arr) + if !isnothing(padding) && !isnothing(padding.staging) + copyto!(arr, padding.staging) end + return nothing end # `get_store` returns a Julia-owned `LogicalArrayImplAllocated` that shares the @@ -74,42 +80,32 @@ end # array into the task; if we leave the temporary alive until GC, store refcounts # stay elevated and framebuffer reclaim stalls (fusion 1-GPU OOM under load). # Finalize the temporary immediately after the copy into the task. -function _add_task_array!(add_to, task, arr::NDArray) - st = cuNumeric.get_store(arr) - var = add_to(task, st) - finalize(st) - return var +function _add_task_array!(add_to, task, arr::NDArray; physical=false) + task_arr = physical ? _physical_array(arr, _padding(arr)) : arr + st = cuNumeric.get_store(task_arr) + try + return add_to(task, st) + finally + finalize(st) + end end function Launch(kernel::CUDATask, inputs::Tuple{Vararg{NDArray}}, outputs::Tuple{Vararg{NDArray}}, scalars::Tuple{Vararg{Any}}; - blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing, validate_shapes=true) - max_shape = if validate_shapes - # Generic PTX tasks retain the existing padding/shape behavior. - ndarrays = vcat(inputs..., outputs...) # returns (nbytes, position) - mx = findmax(arr -> arr.nbytes, ndarrays) # first elem nbytes - shape = size(ndarrays[mx[2]]) # second elem max position - @assert !isnothing(shape) - shape - else - # Fused linear broadcast verifies shapes match - nothing - end - + blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing) rt = Legate.get_runtime() lib = cuNumeric.get_lib() task = Legate.create_auto_task(rt, lib, taskid) + physical = taskid == cuNumeric.RUN_PTX input_vars = Vector{Legate.Variable}() for arr in inputs - validate_shapes && check_sz!(arr, max_shape; copy=true) - push!(input_vars, _add_task_array!(Legate.add_input, task, arr)) + push!(input_vars, _add_task_array!(Legate.add_input, task, arr; physical)) end output_vars = Vector{Legate.Variable}() for arr in outputs - validate_shapes && check_sz!(arr, max_shape; copy=false) - push!(output_vars, _add_task_array!(Legate.add_output, task, arr)) + push!(output_vars, _add_task_array!(Legate.add_output, task, arr; physical)) end # Reserved scalars: kernel_name (0), blocks (1,2,3), threads (4,5,6) @@ -136,15 +132,34 @@ function Launch(kernel::CUDATask, inputs::Tuple{Vararg{NDArray}}, end function launch(kernel::CUDATask, inputs, outputs, scalars; - blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing, validate_shapes=true) - return Launch(kernel, - isa(inputs, Tuple) ? inputs : (inputs,), - isa(outputs, Tuple) ? outputs : (outputs,), + blocks, threads, taskid=cuNumeric.RUN_PTX, ctx=nothing) + input_tuple = isa(inputs, Tuple) ? inputs : (inputs,) + output_tuple = isa(outputs, Tuple) ? outputs : (outputs,) + + # Custom tasks require equal physical shapes. Keep the padded backing so + # repeated launches do not allocate or copy again. + if taskid == cuNumeric.RUN_PTX + arrays = (input_tuple..., output_tuple...) + if !isempty(arrays) + rank = ndims(first(arrays)) + @assert all(ndims(arr) == rank for arr in arrays) "custom task arrays must have equal ranks" + max_shape = ntuple(d -> maximum(_launch_shape(arr)[d] for arr in arrays), rank) + foreach(arr -> _ensure_launch_padding!(arr, max_shape; copy=true), input_tuple) + foreach(arr -> _ensure_launch_padding!(arr, max_shape), output_tuple) + foreach(_sync_to_launch_padding!, input_tuple) + end + end + + result = Launch(kernel, + input_tuple, + output_tuple, isa(scalars, Tuple) ? scalars : (scalars,); blocks=isa(blocks, Tuple) ? blocks : (blocks,), threads=isa(threads, Tuple) ? threads : (threads,), - taskid=taskid, ctx=ctx, validate_shapes=validate_shapes, + taskid=taskid, ctx=ctx, ) + taskid == cuNumeric.RUN_PTX && foreach(_sync_from_launch_padding!, output_tuple) + return result end function ptx_task(ptx::String, kernel_name) @@ -153,7 +168,7 @@ function ptx_task(ptx::String, kernel_name) taskid = cuNumeric.LOAD_PTX # One point task per GPU so every GPU compiles the module. - ngpus = max(Int(Legate.num_gpus()), 1) + ngpus = max(_LINALG_RUNTIME[].gpus, 1) domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector((ngpus,)))) task = Legate.create_manual_task(rt, lib, taskid, domain) Legate.add_scalar(task, Legate.string_to_scalar(ptx)) diff --git a/src/cuda/mapreduce.jl b/src/cuda/mapreduce.jl new file mode 100644 index 000000000..771abfff0 --- /dev/null +++ b/src/cuda/mapreduce.jl @@ -0,0 +1,261 @@ +const _MR_THREADS = 256 +const _MR_PTX_CACHE = Dict{Any,String}() +const _MR_PTX_LOCK = ReentrantLock() + +struct MapReduceMap{F,OP,R,N} + f::F + op::OP + # Axes are runtime data; changing dims does not change the mapper/kernel type. + mask::NTuple{N,Bool} +end +MapReduceMap(f::F, op::OP, ::Type{R}, mask::NTuple{N,Bool}) where {F,OP,R,N} = + MapReduceMap{F,OP,R,N}(f, op, mask) + +# All 32 lanes execute each shuffle, including lanes without input. Only valid +# sources enter the arithmetic: padding with an identity changes signed zeros +# and can turn complex infinities into NaNs. +@inline function _mr_reduce_warp(op, value, lane, active) + offset = 16 + while offset > 0 + other = CUDACore.shfl_down_sync(0xffffffff, value, offset) + if lane + offset <= active + value = op(value, other) + end + offset >>= 1 + end + return value +end + +# The result is defined on thread 1. Both callers launch exactly _MR_THREADS +# threads and supply a contiguous prefix of valid thread-local accumulators. +@inline function _mr_reduce_block(op, value::S, active) where {S} + tid = Int(CUDACore.threadIdx().x) + lane = ((tid - 1) & 31) + 1 + warp = ((tid - 1) >> 5) + 1 + value = _mr_reduce_warp(op, value, lane, min(32, active - (warp - 1) * 32)) + shared = CUDACore.CuStaticSharedArray(S, _MR_THREADS ÷ 32) + lane == 1 && (@inbounds shared[warp] = value) + CUDACore.sync_threads() + if warp == 1 + warps = (active + 31) >> 5 + if lane <= warps + @inbounds value = shared[lane] + end + value = _mr_reduce_warp(op, value, lane, warps) + end + return value +end + +# In one dimension the selected index is already bounded by the only extent. +# Preserve the physical stride without computing a remainder for every element. +@inline _mr_offset(A::CuStridedDeviceArray{T,1}, other::Int, red::Int, mask::NTuple{1,Bool}) where {T} = + (mask[1] ? red : other) * A.strides[1] + +# Indices are relative to the PhysicalStore's lower bound, already reflected in +# the descriptor pointer. Unchecked unsigned division avoids device exceptions. +@inline function _mr_offset(A::CuStridedDeviceArray{T,N}, other::Int, red::Int, mask::NTuple{N,Bool}) where {T,N} + o, r = _bitcast_uint(other), _bitcast_uint(red) + offset = 0 + @inbounds for d in 1:N + extent = _bitcast_uint(A.dims[d]) + if mask[d] + offset += _bitcast_int(Core.Intrinsics.urem_int(r, extent)) * A.strides[d] + r = Core.Intrinsics.udiv_int(r, extent) + else + offset += _bitcast_int(Core.Intrinsics.urem_int(o, extent)) * A.strides[d] + o = Core.Intrinsics.udiv_int(o, extent) + end + end + return offset +end + +function _mr_partial_kernel(A, scratch::CuStridedDeviceArray{S,1}, mapper::MapReduceMap{F,OP,R,N}, start::Int, chunks::Int) where {S,F,OP,R,N} + tid = Int(CUDACore.threadIdx().x) + block = Int(CUDACore.blockIdx().x) - 1 + other = start + _bitcast_int(Core.Intrinsics.udiv_int(_bitcast_uint(block), _bitcast_uint(chunks))) + chunk = _bitcast_int(Core.Intrinsics.urem_int(_bitcast_uint(block), _bitcast_uint(chunks))) + nred = 1 + @inbounds for d in 1:N + mapper.mask[d] && (nred *= A.dims[d]) + end + value = _mr_identity(mapper.op, S) + i = chunk * _MR_THREADS + tid - 1 + first = true + while i < nred + offset = _mr_offset(A, other, i, mapper.mask) + x = unsafe_load(pointer(A), offset + 1, Val(_strided_align(A))) + mapped = _mr_encode(mapper.op, convert(R, mapper.f(x))) + value = first ? mapped : _mr_combine(mapper.op)(value, mapped) + first = false + i += chunks * _MR_THREADS + end + active = min(_MR_THREADS, nred - chunk * _MR_THREADS) + value = _mr_reduce_block(_mr_combine(mapper.op), value, active) + tid == 1 && (@inbounds scratch[block + 1] = value) + return nothing +end + +# One thread owns each retained coordinate. The C++ launcher obtains this +# pointer from an exclusive Legate reduction accessor, as cuPyNumeric's GEMV +# task does. Legate, not this kernel, combines contributions between tasks. +function _mr_contribute_kernel(scratch, dest, op, single::Bool, start::Int, count::Int, chunks::Int) + i = (Int(CUDACore.blockIdx().x) - 1) * Int(CUDACore.blockDim().x) + Int(CUDACore.threadIdx().x) + if i <= count + @inbounds value = scratch[(i - 1) * chunks + 1] + for c in 2:chunks + @inbounds value = _mr_combine(op)(value, scratch[(i - 1) * chunks + c]) + end + @inbounds dest[start + i] = single ? value : _mr_combine(op)(dest[start + i], value) + end + return nothing +end + +# A full reduction has one output. Cooperate across a block instead of making +# a single thread serially combine every partial. Invalid lanes never enter the +# tree: an extra identity operation can change signed zeros or complex infinities. +# Keep nested GPU calls statically dispatched on Julia 1.10 as well. +function _mr_contribute_full_kernel(scratch::CuStridedDeviceArray{S,1}, dest, op::OP, + single::Bool, start::Int, count::Int, chunks::Int) where {S,OP} + tid = Int(CUDACore.threadIdx().x) + active = min(chunks, _MR_THREADS) + value = _mr_identity(op, S) + if tid <= active + @inbounds value = scratch[tid] + for c in (tid + _MR_THREADS):_MR_THREADS:chunks + @inbounds value = _mr_combine(op)(value, scratch[c]) + end + end + value = _mr_reduce_block(_mr_combine(op), value, active) + if tid == 1 + @inbounds dest[start + 1] = single ? value : _mr_combine(op)(dest[start + 1], value) + end + return nothing +end + +struct MapReduceFinish{OP,R,I} + op::OP + init::I +end +MapReduceFinish(op::OP, ::Type{R}, init::I) where {OP,R,I} = MapReduceFinish{OP,R,I}(op, init) +@inline function (finish::MapReduceFinish{OP,R})(x) where {OP,R} + return _mr_finish(_mr_combine(finish.op), _mr_decode(finish.op, R, x), finish.init) +end + +struct MapReduceSingleton{F,OP,R,O,I} + f::F + op::OP + init::I +end +function MapReduceSingleton( + f::F, op::OP, ::Type{R}, ::Type{O}, init::I, +) where {F,OP,R,O,I} + return MapReduceSingleton{F,OP,R,O,I}(f, op, init) +end +@inline function (finish::MapReduceSingleton{F,OP,R,O})(x) where {F,OP,R,O} + mapped = convert(R, finish.f(x)) + value = _mr_finish(_mr_combine(finish.op), mapped, finish.init) + return convert(O, value) +end +function _mr_finish_kernel(src, dest, finish) + i = (Int(CUDACore.blockIdx().x) - 1) * Int(CUDACore.blockDim().x) + Int(CUDACore.threadIdx().x) + step = Int(CUDACore.gridDim().x) * Int(CUDACore.blockDim().x) + while i <= length(dest) + @inbounds dest[i] = finish(src[i]) + i += step + end + return nothing +end + +function _mr_kernel_name(kernel, types) + target = (CUDACore.capability(CUDACore.device()), _COMPATIBLE_PTX_VERSION[]) + key = (kernel, types, target) + return lock(_MR_PTX_LOCK) do + get!(_MR_PTX_CACHE, key) do + buf = IOBuffer() + _emit_compatible_ptx(buf, kernel, types) + ptx = String(take!(buf)) + original = extract_kernel_name(ptx) + name = original * "_mr_" * string(hash(ptx); base=16) + ptx_task(replace(ptx, original => name), name) + name + end + end +end + +_mr_dim_seed(op::_MR_OP, ::Type{R}, init, dims) where {R} = init +_mr_dim_seed(op::_MR_ADD, ::Type{R}, ::NoReductionInit, dims::_MR_DIMS) where {R} = zero(R) +_mr_dim_seed(op::_MR_MUL, ::Type{R}, ::NoReductionInit, dims::_MR_DIMS) where {R} = one(R) + +function _mr_finish_name(finish::F, ::Type{T}, ::Type{O}, ::Val{D}) where {F,T,O,D} + return _mr_kernel_name(_mr_finish_kernel, ( + CuStridedDeviceArray{T,D,CUDACore.AS.Global}, + CuStridedDeviceArray{O,D,CUDACore.AS.Global}, F, + )) +end + +function _mr_submit(A, accumulator, result, mask, full, single, redop, + name, contribute_name, finish_name, mapper, finish) + axis_bits = sum(d -> UInt64(mask[d]) << (d - 1), 1:length(mask)) + mapper_ref, finish_ref = Ref(mapper), Ref(finish) + @task_scope "mapreduce" begin + # Submission copies these bytes into task-owned scalars. Borrowing + # Refs avoids Julia-owned CxxWrap vector handles on every call. + GC.@preserve A accumulator result mapper_ref finish_ref begin + submit_mapreduce( + CxxWrap.CxxPtr{CN_NDArray}(A.ptr), + CxxWrap.CxxPtr{CN_NDArray}(accumulator.ptr), + CxxWrap.CxxPtr{CN_NDArray}(result.ptr), + axis_bits, full, single, redop, name, contribute_name, finish_name, + Base.unsafe_convert(Ptr{Cvoid}, mapper_ref), sizeof(mapper), + Base.unsafe_convert(Ptr{Cvoid}, finish_ref), sizeof(finish), + ) + end + end + return result +end + +function _mr_launch(f, op, A::NDArray{T,N}, ::Type{R}, ::Type{O}, mask, shape, init, dims, single::Bool) where {T,N,R,O} + S = _mr_storage(op, R) + D = max(N, 1) + OD = max(length(shape), 1) + seed = _mr_dim_seed(op, R, init, dims) + if single && !(dims isa Colon) + finish = MapReduceSingleton(f, op, R, O, seed) + finish_name = _mr_finish_name(finish, T, O, Val(OD)) + result = nda_empty_array(shape, O) + try + # The singleton submission never uses the accumulator or mapper. + return _mr_submit(A, result, result, mask, false, true, _mr_redop(op, S), + "", "", finish_name, nothing, finish) + catch + destroy!(result) + rethrow() + end + end + input_type = CuStridedDeviceArray{T,D,CUDACore.AS.Global} + scratch_type = CuStridedDeviceArray{S,1,CUDACore.AS.Global} + mapper = MapReduceMap(f, op, R, mask) + name = _mr_kernel_name(_mr_partial_kernel, (input_type, scratch_type, typeof(mapper), Int, Int)) + RD = dims isa Colon ? 1 : D + contribute_kernel = dims isa Colon ? _mr_contribute_full_kernel : _mr_contribute_kernel + contribute_name = _mr_kernel_name(contribute_kernel, ( + scratch_type, CuStridedDeviceArray{S,RD,CUDACore.AS.Global}, typeof(op), Bool, Int, Int, Int, + )) + finish = MapReduceFinish(op, R, seed) + needs_finish = S !== O || !(finish.init isa NoReductionInit) + finish_name = needs_finish ? _mr_finish_name(finish, S, O, Val(OD)) : "" + accumulator = single ? nda_empty_array(shape, S) : nda_full_array(shape, _mr_identity(op, S)) + result = nothing + try + result = needs_finish ? nda_empty_array(shape, O) : accumulator + _mr_submit(A, accumulator, result, mask, dims isa Colon, single, _mr_redop(op, S), + name, contribute_name, finish_name, mapper, finish) + # Returning here keeps the cleanup sentinel out of Julia 1.10 inference. + return result + catch + isnothing(result) || destroy!(result) + rethrow() + finally + result === accumulator || destroy!(accumulator) + end +end diff --git a/src/cuda/strided_device_array.jl b/src/cuda/strided_device_array.jl index 234044f4a..df33d258d 100644 --- a/src/cuda/strided_device_array.jl +++ b/src/cuda/strided_device_array.jl @@ -14,9 +14,9 @@ * limitations under the License. =# -# Device-side strided array packed by RunPTXBroadcastTask only. +# Device-side strided array shared by broadcast and mapped reduction tasks. # Layout must match C++ `CuStridedDeviceArray` in -# lib/cunumeric_jl_wrapper/src/cuda.cpp: +# lib/cunumeric_jl_wrapper/include/ptx.h: # ptr, maxsize, dims[N], strides[N] (element strides), length # # Dense RunPTXTask / @cuda_task still uses CUDA.jl CuDeviceArray (unchanged). diff --git a/src/memory.jl b/src/memory.jl index 02deb81f6..50705618e 100644 --- a/src/memory.jl +++ b/src/memory.jl @@ -102,7 +102,7 @@ hard_limit(; host=true) = _limit(hard_frac, host) function register_alloc!(nbytes::Integer) # assume device allocation if we have a GPU # the recalibration phase will fix any discrepancies - if HAS_CUDA + if _has_gpu_target() atomic_add!(current_device_bytes, nbytes) else atomic_add!(current_host_bytes, nbytes) @@ -116,7 +116,7 @@ function register_alloc!(nbytes::Integer) end function register_free!(nbytes::Integer) - if HAS_CUDA + if _has_gpu_target() atomic_sub!(current_device_bytes, nbytes) else atomic_sub!(current_host_bytes, nbytes) @@ -129,7 +129,7 @@ function recalibrate_allocator!() @assert recal_host_mem >= 0 atomic_xchg!(current_host_bytes, recal_host_mem) - if HAS_CUDA + if _has_gpu_target() recal_device_mem = ccall((:nda_query_allocated_device_memory, libnda), Int64, ()) @assert recal_device_mem >= 0 atomic_xchg!(current_device_bytes, recal_device_mem) diff --git a/src/ndarray/batched_linalg.jl b/src/ndarray/batched_linalg.jl new file mode 100644 index 000000000..90c929c8c --- /dev/null +++ b/src/ndarray/batched_linalg.jl @@ -0,0 +1,93 @@ +# Batched linear algebra: operations over a stack of matrices, dispatching on 3D +# arrays. These have no Base.LinearAlgebra counterpart, so they keep the +# `batched_` prefix and return plain tuples rather than `Factorization` objects. +# +# All of them take exactly one batch dimension, i.e. shape (b, m, m). See +# `MAX_BATCHED_DIM` for why. + +""" + cuNumeric.batched_solve(A, B) + +Solve a stack of linear systems `A * X = B`. + +`A` must have shape `(b, m, m)` and `B` shape `(b, m, n)`. The result has the +same shape as `B`. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a single system use [`cuNumeric.solve`](@ref). +""" +function batched_solve( + a::NDArray{A}, b::NDArray{B} +) where {A<:_SOLVE_ACCEPTED,B<:_SOLVE_ACCEPTED} + O = promote_type(_solve_eltype(A), _solve_eltype(B)) + return _solve_check_a_dims_batched( + checked_promote_arr(batched_solve, a, O), checked_promote_arr(batched_solve, b, O) + ) +end + +function batched_solve(a::NDArray, b::NDArray) + bad = eltype(a) <: _SOLVE_ACCEPTED ? eltype(b) : eltype(a) + return throw(ArgumentError("array type $bad is unsupported in batched_solve")) +end + +""" + cuNumeric.batched_cholesky(A) + +Cholesky factor of every matrix in the stack `A`, returned as a single array of +the same shape. Each `A[i, :, :]` is factored independently into a lower +triangular `L` with `A[i, :, :] ≈ L * L'`; the upper triangle is zeroed. + +`A` must have shape `(b, m, m)`. As in `LinearAlgebra.cholesky`, only the lower +triangle is read and Hermitian-ness is not checked. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. +""" +function batched_cholesky(a::NDArray{T,N}) where {T<:_CHOLESKY_ACCEPTED,N} + _assert_batched_dims(:batched_cholesky, N) + return _cholesky(checked_promote_arr(batched_cholesky, a, _cholesky_eltype(T))) +end + +function batched_cholesky(a::NDArray) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in batched_cholesky")) +end + +""" + cuNumeric.batched_eigen(A) -> (values, vectors) + +Eigenvalues and right eigenvectors of every matrix in the stack `A`. + +`A` must have shape `(b, m, m)`. `values` has shape `(b, m)` and `vectors` has +the shape of `A`, with `vectors[i, :, j]` the eigenvector for `values[i, j]`. + +Both results are always complex, even for real input. See `LinearAlgebra.eigen`. +""" +function batched_eigen(a::NDArray{T,N}) where {T<:_EIG_ACCEPTED,N} + _assert_batched_dims(:batched_eigen, N) + return _eig(checked_promote_arr(batched_eigen, a, _eig_eltype(T))) +end + +function batched_eigen(a::NDArray) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in batched_eigen")) +end + +""" + cuNumeric.batched_eigvals(A) + +Eigenvalues of every matrix in the stack `A`, as a complex array of shape +`(b, m)`. See [`cuNumeric.batched_eigen`](@ref). +""" +function batched_eigvals(a::NDArray{T,N}) where {T<:_EIG_ACCEPTED,N} + _assert_batched_dims(:batched_eigvals, N) + return _eigvals(checked_promote_arr(batched_eigvals, a, _eig_eltype(T))) +end + +function batched_eigvals(a::NDArray) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in batched_eigvals")) +end diff --git a/src/ndarray/binary.jl b/src/ndarray/binary.jl index 07745f483..2ad29d92d 100644 --- a/src/ndarray/binary.jl +++ b/src/ndarray/binary.jl @@ -1,14 +1,14 @@ # Still missing: -# # Base.copysign => cuNumeric.COPYSIGN, #* ANNOYING TO TEST -# #missing => cuNumeric.fmod, #same as mod in Julia? -# # Base.isapprox => cuNumeric.ISCLOSE, #* HANDLE rtol, atol kwargs!!! -# # Base.ldexp => cuNumeric.LDEXP, #* LHS FLOATS, RHS INTS -# #missing => cuNumeric.LOGADDEXP, -# #missing => cuNumeric.LOGADDEXP2, -# #missing => cuNumeric.NEXTAFTER, +# # Base.isapprox => cuNumeric.ISCLOSE, # rtol, atol kwargs +# # Base.ldexp => cuNumeric.LDEXP, # LHS floats, RHS ints +# # missing => cuNumeric.LOGADDEXP, +# # missing => cuNumeric.LOGADDEXP2, +# # missing => cuNumeric.NEXTAFTER, +# # Base.div / ÷ — FLOOR_DIVIDE matches Julia `fld` (toward -Inf), not +# # truncated `div`. Do not ship as `div`: `-7 ÷ 2` is -3 in Julia, -4 for fld. # Binary ops which are equivalent to Julia's broadcast syntax -global const binary_op_map = Dict{Function,BinaryOpCode}( +const binary_op_map = Dict{Function,BinaryOpCode}( Base.:+ => cuNumeric.ADD, Base.:* => cuNumeric.MULTIPLY, Base.:(-) => cuNumeric.SUBTRACT, @@ -24,22 +24,32 @@ global const binary_op_map = Dict{Function,BinaryOpCode}( Base.:(==) => cuNumeric.EQUAL, #* BE SURE TO DEFINE NON-BROADCASTED VERSION (BINARY_REDUCTION), Base.lcm => cuNumeric.LCM, Base.gcd => cuNumeric.GCD, - # Base.xor => cuNumeric.LOGICAL_XOR, #! DO LATER - # Base.:⊻ => cuNumeric.LOGICAL_XOR, #! DO LATER - # Base.div => cuNumeric.FLOOR_DIVIDE, #! THESE ARE IN-EXACT FOR INTS? - # Base.:(÷) => cuNumeric.FLOOR_DIVIDE, #! THESE ARE IN-EXACT FOR INTS? - # Base.:(>>) => cuNumeric.RIGHT_SHIFT, #! DO LATER - # Base.:(<<) => cuNumeric.LEFT_SHIFT, #! DO LATER - # Base.:(&&) => (cuNumeric.LOGICAL_AND, Bool, :same_as_input), #! CANNOT OVERLOAD WTF? (see Base.andand) - # Base.:(||) => (cuNumeric.LOGICAL_OR, Bool, :same_as_input), #! CANNOT OVERLOAD WTF? + Base.:(&) => cuNumeric.BITWISE_AND, # integers and Bool + Base.:(|) => cuNumeric.BITWISE_OR, + Base.:(⊻) => cuNumeric.BITWISE_XOR, + Base.:(<<) => cuNumeric.LEFT_SHIFT, # integers, not Bool + Base.:(>>) => cuNumeric.RIGHT_SHIFT, # integers, not Bool + Base.fld => cuNumeric.FLOOR_DIVIDE, # matches Julia fld, not div/÷ + Base.mod => cuNumeric.MOD, + Base.rem => cuNumeric.FMOD, + Base.:(%) => cuNumeric.FMOD, # Julia `%` is rem + Base.copysign => cuNumeric.COPYSIGN, # floats only + # Base.:(&&) => (cuNumeric.LOGICAL_AND, Bool, :same_as_input), # cannot overload (see Base.andand) + # Base.:(||) => (cuNumeric.LOGICAL_OR, Bool, :same_as_input), # cannot overload ) -global const floaty_binary_op_map = Dict{Function,BinaryOpCode}( +const floaty_binary_op_map = Dict{Function,BinaryOpCode}( Base.:/ => cuNumeric.DIVIDE, Base.hypot => cuNumeric.HYPOT, Base.atan => cuNumeric.ARCTAN2, ) +for julia_fn in (keys(binary_op_map)..., keys(floaty_binary_op_map)...) + @eval @inline _has_unfused_broadcast(::typeof($julia_fn), ::Val{2}) = true +end +@inline _has_unfused_broadcast(::typeof(+), ::Val{N}) where {N} = N >= 2 +@inline _has_unfused_broadcast(::typeof(*), ::Val{N}) where {N} = N >= 2 + ## SPECIAL CASES ## # Promote into out's eltype, then destroy any new temps (dispatch; no runtime !==). @inline function _nda_binary_op_promoted!( @@ -124,7 +134,7 @@ end function Base.:(-)(rhs1::NDArray{A,N}, rhs2::NDArray{B,N}) where {A,B,N} promote_shape(size(rhs1), size(rhs2)) T_OUT = __checked_promote_op(-, A, B) - out = cuNumeric.zeros(T_OUT, size(rhs1)) + out = NDArray{T_OUT}(undef, size(rhs1)) return _nda_binary_op_promoted!(out, cuNumeric.SUBTRACT, rhs1, rhs2) end @@ -132,12 +142,34 @@ end function Base.:(+)(rhs1::NDArray{A,N}, rhs2::NDArray{B,N}) where {A,B,N} promote_shape(size(rhs1), size(rhs2)) T_OUT = __checked_promote_op(+, A, B) - out = cuNumeric.zeros(T_OUT, size(rhs1)) + out = NDArray{T_OUT}(undef, size(rhs1)) return _nda_binary_op_promoted!(out, cuNumeric.ADD, rhs1, rhs2) end -Base.:(*)(val::V, arr::NDArray{A}) where {A,V} = _mul_scalar(__my_promote_type(A, V), val, arr) -Base.:(*)(arr::NDArray{A}, val::V) where {A,V} = val * arr +# Scalar-shaped arithmetic stays on the backend. Array-array + and - above +# already support rank zero; higher-rank * and / keep their linear algebra meaning. +# This deliberately departs from Julia's standard array API: a 0D Array is not +# a Number, and Base does not support this full set of scalar-style operations +# on it. NDArray supports them to keep reduction arithmetic asynchronous. +for op in (:*, :/, :^) + @eval function Base.$op( + a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, b::NDArray{<:SUPPORTED_ARRAY_TYPES,0} + ) + return broadcast($op, a, b) + end +end +for op in (:+, :-, :*, :/, :^) + @eval begin + Base.$op(a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, b::Number) = broadcast($op, a, b) + Base.$op(a::Number, b::NDArray{<:SUPPORTED_ARRAY_TYPES,0}) = broadcast($op, a, b) + end +end +Base.literal_pow(::typeof(^), a::NDArray{<:SUPPORTED_ARRAY_TYPES,0}, ::Val{P}) where {P} = a ^ P + +function Base.:(*)(val::V, arr::NDArray{A}) where {A,V<:Number} + return _mul_scalar(__my_promote_type(A, V), val, arr) +end +Base.:(*)(arr::NDArray{A}, val::V) where {A,V<:Number} = val * arr _mul_scalar(::Type{T}, val, arr::NDArray{T}) where {T} = nda_multiply_scalar(arr, T(val)) function _mul_scalar(::Type{U}, val, arr::NDArray) where {U} @@ -151,19 +183,20 @@ function Base.:(*)(rhs1::NDArray{A,2}, rhs2::NDArray{B,2}) where {A,B} size(rhs1, 2) == size(rhs2, 1) || throw(DimensionMismatch("Matrix dimensions incompatible: $(size(rhs1)) × $(size(rhs2))")) T = __my_promote_type(A, B) - out = cuNumeric.zeros(T, (size(rhs1, 1), size(rhs2, 2))) + dims = (size(rhs1, 1), size(rhs2, 2)) + out = size(rhs1, 2) == 0 ? cuNumeric.zeros(T, dims) : NDArray{T}(undef, dims) return _nda_three_dot_promoted!(rhs1, rhs2, out) end function Base.:(*)(rhs1::NDArray{Bool,2}, rhs2::NDArray{Bool,2}) - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Boolean arrays") ) end function Base.:(*)(rhs1::NDArray{<:Integer,2}, rhs2::NDArray{<:Integer,2}) #* this is a stupid..... - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Integer arrays") ) end @@ -221,14 +254,14 @@ end function LinearAlgebra.mul!(out::NDArray, rhs1::NDArray{Bool,2}, rhs2::NDArray{Bool,2}) #* Could just promote both inputs to Int32 - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Boolean arrays") ) end function LinearAlgebra.mul!(out::NDArray, rhs1::NDArray{<:Integer,2}, rhs2::NDArray{<:Integer,2}) #* this is a stupid..... - throw( + return throw( ArgumentError("cuNumeric.jl does not support matrix multiplication of two Integer arrays") ) end @@ -293,18 +326,6 @@ end return result end -# function Base.:(==)(lhs::NDArray{A}, rhs::NDArray{B}) where {A,B} -# error("Not implemented yet") -# #! REPLACE WITH ARRAY_EQUAL ONCE THAT IS WRAPPED -# #! or explicit call to nda_binary_reduction -# end - -# function Base.:(!=)(lhs::NDArray{A}, rhs::NDArray{B}) where {A,B} -# error("Not implemented yet") -# #! REPLACE WITH ARRAY_EQUAL ONCE THAT IS WRAPPED -# #! or explicit call to nda_binary_reduction -# end - # Specializations for 2 and -1 in unary.jl @inline function __broadcast( f::typeof(Base.literal_pow), out::NDArray, _, input::NDArray{T}, power::NDArray{T} @@ -319,6 +340,12 @@ function Base.map(f::Function, arr1::NDArray{A,N}, arr2::NDArray{B,N}) where {A, return f.(arr1, arr2) # Will try to call one of the functions generated above end -# function Base.map!(f::Function, dest::NDArray, arr1::NDArray, arr2::NDArray) -# return f -# end +for (julia_fn, _) in binary_op_map + @eval function Base.map!( + f::typeof($(julia_fn)), dest::NDArray{O,N}, arr1::NDArray{T,N}, arr2::NDArray{T,N} + ) where {O,T,N} + axes(dest) == axes(arr1) == axes(arr2) || + throw(DimensionMismatch("map! arrays must have matching axes")) + return __broadcast(f, dest, arr1, arr2) + end +end diff --git a/src/ndarray/broadcast.jl b/src/ndarray/broadcast.jl index 2d64c65ff..bb6812bf8 100644 --- a/src/ndarray/broadcast.jl +++ b/src/ndarray/broadcast.jl @@ -10,7 +10,7 @@ function map_cuda_type(::Type{cuNumeric.NDArrayStyle{N}}) where {N} end # Also can be HostMemory or UnifiedMemory function _nd_forbid_mix() - throw( + return throw( ArgumentError( "Broadcast between NDArray and other array types is not supported. " * "Convert explicitly to a single array type before broadcasting.", @@ -26,36 +26,73 @@ Base.BroadcastStyle(::DefaultArrayStyle{0}, a::NDArrayStyle) = a Base.BroadcastStyle(::NDArrayStyle, ::DefaultArrayStyle) = _nd_forbid_mix() Base.BroadcastStyle(::DefaultArrayStyle, ::NDArrayStyle) = _nd_forbid_mix() +# Like Base Diagonal vs Array: structured Diagonal style wins over dense NDArray +# so D.+A uses StructuredMatrixStyle{Diagonal} (densify to NDArray in diagonal.jl) +# instead of ArrayConflict → host Matrix + scalar indexing. +function Base.BroadcastStyle( + ::LinearAlgebra.StructuredMatrixStyle{<:Diagonal}, ::NDArrayStyle +) + return LinearAlgebra.StructuredMatrixStyle{Diagonal}() +end +function Base.BroadcastStyle( + ::NDArrayStyle, ::LinearAlgebra.StructuredMatrixStyle{<:Diagonal} +) + return LinearAlgebra.StructuredMatrixStyle{Diagonal}() +end + Base.broadcastable(A::NDArray) = A -#* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? -Base.similar(arr::NDArray, ::Type{T}, dims::Dims{N}) where {T,N} = cuNumeric.zeros(T, dims) +function _broadcast_allocate(::Type{T}, dims::Dims) where {T} + if _struct_storage_type(T) || + (isbitstype(T) && !isprimitivetype(T) && !(T <: SUPPORTED_TYPES)) + return nda_empty_array(dims, T) + end + return NDArray{T}(undef, dims) +end + +Base.similar(arr::NDArray, ::Type{T}, dims::Dims{N}) where {T,N} = _broadcast_allocate(T, dims) Base.similar(arr::NDArray, ::Type{T}, dims::Base.DimOrInd...) where {T} = similar(arr, T, dims) Base.similar(arr::NDArray{T,N}) where {T,N} = similar(arr, T, size(arr)) Base.similar(arr::NDArray{T}, dims::Tuple) where {T} = similar(arr, T, dims) Base.similar(arr::NDArray{T}, dims::Base.DimOrInd...) where {T} = similar(arr, T, dims) Base.similar(arr::NDArray, ::Type{T}) where {T} = similar(arr, T, size(arr)) -#* IS THERE A BETTER WAY TO ALLOCATE THE NEW ARRAY??? -Base.similar(::Type{NDArray{T}}, axes) where {T} = cuNumeric.zeros(T, Base.to_shape.(axes)) +# Prefer Dims over the axes catch-all: with StaticArrays loaded (GPU CI via CUDA), +# `similar(::Type{<:AbstractArray}, ::Tuple{})` is otherwise ambiguous between +# Base, StaticArrays, and our catch-all (0-d broadcast uses axes `()`). +Base.similar(::Type{NDArray{T}}, dims::Dims{N}) where {T,N} = _broadcast_allocate(T, dims) +function Base.similar( + ::Type{NDArray{T}}, + shape::Tuple{Union{Integer,Base.OneTo},Vararg{Union{Integer,Base.OneTo}}}, +) where {T} + return _broadcast_allocate(T, map(Int, Base.to_shape.(shape))) +end +Base.similar(::Type{NDArray{T}}, axes) where {T} = _broadcast_allocate(T, Base.to_shape.(axes)) function Base.similar(bc::Broadcasted{NDArrayStyle{N}}, ::Type{ElType}) where {N,ElType} return similar(NDArray{ElType}, axes(bc)) end function __broadcast(f::Function, _, args...) - #! WITH FUSION I THINK WE CAN SUPPORT THIS BY JUST CALLING MAP or MAP! return error( - """ - Tried to broadcast $(f). cuNumeric.jl does not support broadcasting user-defined functions yet. Please re-define \ - functions to match supported patterns. For example g(x) = x + 1 could be re-defined as \ - broadcast_g(x::NDArray) = x .+ 1. This can make the intention of code opaque to the reader, \ - but it is necessary until support is added.""", + "Broadcasting $(f) is not supported by cuNumeric's unfused broadcast path. " * + "Functions without a native broadcast implementation require GPU fusion, compatible array shapes, " * + "and a GPU-compilable function (fusion enabled: $(FUSE_BROADCAST_EXPRS)). " * + "Otherwise, rewrite the expression using supported broadcast operations.", ) end +# Low-storage Runge–Kutta methods use muladd. Express it as two supported +# broadcasts so the unfused path works and GPU fusion can still combine them. +@inline function __broadcast( + ::typeof(muladd), out::NDArray, a::NDArray, b::NDArray, c::NDArray +) + out .= a .* b .+ c + return out +end + # Get depth of Broadcast tree recursively # Need to call instantiate first -bcast_depth(bc::Base.Broadcast.Broadcasted) = maximum(bcast_depth, bc.args, init=0) + 1; +bcast_depth(bc::Base.Broadcast.Broadcasted) = maximum(bcast_depth, bc.args; init=0) + 1; bcast_depth(::Any) = 0 struct BrokenBroadcast{T} end @@ -63,18 +100,22 @@ Base.convert(::Type{BrokenBroadcast{T}}, x) where {T} = BrokenBroadcast{T}() Base.convert(::Type{BrokenBroadcast{T}}, x::BrokenBroadcast{T}) where {T} = x Base.eltype(::Type{BrokenBroadcast{T}}) where {T} = T +# Use cuNumeric promotion (`__recip_type` for inv, etc.), not Base.combine_eltypes +# — e.g. inv.(Int32) must allocate Float32, not Float64. +@inline function _broadcast_copy_eltype(bc::Broadcasted) + return __checked_promote_op(bc.f, Base.Broadcast.eltypes(bc.args)) +end + function Broadcast.copy(bc::Broadcasted{<:NDArrayStyle{0}}) - ElType = Broadcast.combine_eltypes(bc.f, bc.args) + ElType = _broadcast_copy_eltype(bc) if ElType == Union{} ElType = Nothing end - dest = copyto!(similar(bc, ElType), bc) - #! CHECK THIS DOESNT CAUSE ISSUES DUE TO BLOCKING NATURE - return @allowscalar dest[CartesianIndex()] + return copyto!(similar(bc, ElType), bc) end @inline function Broadcast.copy(bc::Broadcasted{<:NDArrayStyle}) - ElType = Broadcast.combine_eltypes(bc.f, bc.args) + ElType = _broadcast_copy_eltype(bc) if ElType == Union{} || !Base.allocatedinline(ElType) ElType = BrokenBroadcast{ElType} end @@ -91,15 +132,14 @@ __materialize(x::Base.RefValue{typeof(^)}) = x __materialize(x::Base.RefValue{Val{-1}}) = x # enables specialized reciprocal definition __materialize(x::Base.RefValue{Val{2}}) = x # enables specialized square definition __materialize(x::Base.RefValue{Val{V}}) where {V} = NDArray(V) # Use binary_op POWER for other literal powers +__materialize(x::Base.RefValue) = x # Catch unknown things... __materialize(x) = error("Unrecognized leaf in broadcast expression: $(x)") -# Scalar-only nested broadcasts (e.g. `s1 .* s2 .+ A`): the inner -# `Broadcasted(*, (s1, s2))` keeps DefaultArrayStyle{0}, not NDArrayStyle. -# Fold to a Number so the parent unravel sees a scalar leaf. +# Use Base for scalar-only broadcasts, including `literal_pow` wrappers. @inline function __materialize(bc::Broadcasted{<:DefaultArrayStyle{0}}) - return bc.f((__materialize.(bc.args))...) + return Base.materialize(bc) end function __materialize(bc::Broadcasted{<:NDArrayStyle}) @@ -107,6 +147,25 @@ function __materialize(bc::Broadcasted{<:NDArrayStyle}) return unravel_broadcast_tree(bc) end +# The C API is binary, so evaluate flattened `+` and `*` chains pairwise. +function _unravel_flattened_associative(f, args::Tuple, dest) + acc = first(args) + owns_acc = false + for (i, arg) in enumerate(Base.tail(args)) + bc = Base.broadcasted(f, acc, arg) + # Only the final binary operation may overwrite the destination. + next = if i == length(args) - 1 && !isnothing(dest) + unravel_broadcast_tree(Base.Broadcast.instantiate(bc), dest) + else + __materialize(bc) + end + owns_acc && acc isa NDArray && destroy!(acc) + acc = next + owns_acc = acc isa NDArray + end + return acc +end + # Destroy promote copies and non-leaf materialized NDArrays (nested results / Val{V}). @inline function _destroy_unfused_arg_temps!(orig, materialized, promoted) if promoted isa NDArray && promoted !== materialized @@ -118,8 +177,12 @@ end return nothing end -# Un-fused implementation of broadcast tree -function unravel_broadcast_tree(bc::Broadcasted) +# Un-fused implementation of broadcast tree. `dest`, when given, receives the +# top-level result directly if its eltype matches and no input partially overlaps it. +function unravel_broadcast_tree(bc::Broadcasted, dest=nothing) + if length(bc.args) > 2 && _is_flattened_associative(bc.f) + return _unravel_flattened_associative(bc.f, bc.args, dest) + end # Recursively materialize/unravel any nested broadcasts # until we reach a Broadcasted expression with only @@ -133,8 +196,7 @@ function unravel_broadcast_tree(bc::Broadcasted) T_IN = __my_promote_type(eltypes.parameters...) # type input arrays are promoted to in_args = unchecked_promote_arr.(materialized_args, T_IN) - # Allocate output array of proper size/type - out = similar(NDArray{T_OUT}, axes(bc)) + out = _unfused_output(dest, T_OUT, bc, in_args) # If the operation, "bc.f", is supported by cuNumeric, this # dispatches to a function calling the C-API. @@ -148,21 +210,36 @@ function unravel_broadcast_tree(bc::Broadcasted) return result end -# Slice destinations must assign into their parent store. +@inline _unfused_output(::Nothing, ::Type{T}, bc, in_args) where {T} = + similar(NDArray{T}, axes(bc)) + +@inline function _unfused_output(dest::NDArray{S}, ::Type{T}, bc, in_args) where {S,T} + # Elementwise ops may read and write the same array; only partial overlap + # (e.g. shifted views of one store) needs a temporary. + S === T && !any(x -> x isa NDArray && x !== dest && nda_overlaps(dest, x), in_args) && + return dest + return similar(NDArray{T}, axes(bc)) +end + +# Skips the temporary and store-back when the top-level op wrote into `dest`. +@inline function _unfused_into!(dest::NDArray, bc::Broadcasted) + result = unravel_broadcast_tree(bc, dest) + return result === dest ? dest : _copyto_unfused!(dest, result) +end + +# Preserve the destination store: other handles may already view it. @inline function _store_broadcast_result!( dest::NDArray{T}, temp_result::NDArray{T} ) where {T} - if _is_ndarray_slice(dest) - nda_assign(dest, temp_result) - destroy!(temp_result) - else - nda_move(dest, temp_result) - end + nda_assign(dest, temp_result) + destroy!(temp_result) return dest end @inline _copyto_unfused!(dest::NDArray{T}, temp_result::NDArray{T}) where {T} = - _store_broadcast_result!(dest, temp_result) + _store_broadcast_result!( + dest, temp_result + ) @inline function _copyto_unfused!(dest::NDArray{T}, temp_result::NDArray) where {T} promoted = checked_promote_arr(temp_result, T) @@ -181,18 +258,37 @@ end _broadcast_tree_length_args(Base.tail(args)) end -# Prefer fusion only when the tree has at least `FUSE_BROADCAST_MIN_OPS` ops. +# A single native operation uses the C API. Unknown functions need the GPU +# broadcast kernel even when the preference normally skips single-op fusion. +@inline _has_unfused_broadcast(f, ::Val) = false +@inline _has_unfused_broadcast(::typeof(muladd), ::Val{3}) = true +@inline _has_unfused_broadcast(::typeof(abs2), ::Val{1}) = true +@inline _has_unfused_broadcast(bc::Broadcasted) = + _has_unfused_broadcast(bc.f, Val(length(bc.args))) + +# Prefer fusion when the tree has enough ops, or a single op has no native path. # When that const is <= 1, every Broadcasted qualifies and the length check # compiles out (`@static`). @inline function _should_attempt_broadcast_fusion(dest::NDArray, bc::Broadcasted) @static if FUSE_BROADCAST_MIN_OPS <= 1 return can_fuse_linear_broadcast(dest, bc) else - return _broadcast_tree_length(bc) >= FUSE_BROADCAST_MIN_OPS && - can_fuse_linear_broadcast(dest, bc) + operation_count = _broadcast_tree_length(bc) + meets_fusion_minimum = operation_count >= FUSE_BROADCAST_MIN_OPS + single_operation_needs_fusion = + operation_count == 1 && !_has_unfused_broadcast(bc) + worth_fusing = meets_fusion_minimum || single_operation_needs_fusion + worth_fusing || return false + return can_fuse_linear_broadcast(dest, bc) end end +@inline function _identity_broadcast_source(bc::Broadcasted) + bc.f === identity && length(bc.args) == 1 || return nothing + source = only(bc.args) + return source isa NDArray ? source : nothing +end + @inline function _copyto!(dest::NDArray, bc::Broadcasted) axes(dest) == axes(bc) || Broadcast.throwdm(axes(dest), axes(bc)) isempty(dest) && return dest @@ -204,22 +300,56 @@ end ) end - # Fused writes `dest` in place (no post-fuse `nda_move`); promotion is - # checked pre-launch in `fuse_broadcast_tree!`. CPU vs GPU is compile-time - # via `@static if FUSE_BROADCAST_EXPRS && HAS_CUDA`. + # A same-type identity broadcast is an array assignment. Use the native + # path only for disjoint stores; overlapping slices need the broadcast + # temporary to preserve the original values. + source = _identity_broadcast_source(bc) + if source isa NDArray && eltype(dest) === eltype(source) && + axes(dest) == axes(source) && !nda_overlaps(dest, source) + return copyto!(dest, source) + end + + # Require an active GPU target so `--gpus 0` stays on the unfused path. # Fusion requires same-shaped NDArray leaves; otherwise fall back. - # Single-op exprs (length < `FUSE_BROADCAST_MIN_OPS`) stay unfused by default. - @static if FUSE_BROADCAST_EXPRS && HAS_CUDA - if _should_attempt_broadcast_fusion(dest, bc) + # Single native ops below `FUSE_BROADCAST_MIN_OPS` use the unfused C API. + @static if FUSE_BROADCAST_EXPRS + gpu_available = _has_gpu_target() + should_fuse = gpu_available && _should_attempt_broadcast_fusion(dest, bc) + if should_fuse return fuse_broadcast_tree!(dest, bc) else - return _copyto_unfused!(dest, unravel_broadcast_tree(bc)) + _assert_struct_broadcast_fused(dest, bc) + return _unfused_into!(dest, bc) end else - return _copyto_unfused!(dest, unravel_broadcast_tree(bc)) + _assert_struct_broadcast_fused(dest, bc) + return _unfused_into!(dest, bc) end end +# The unfused path runs cuPyNumeric operations, none of which read or produce records. +# TODO fuse struct results with size-1 extrusion and into 0-d destinations. +@inline function _assert_struct_broadcast_fused(dest::NDArray{T}, bc::Broadcasted) where {T} + role, S = _struct_storage_type(T) ? ("producing", T) : ("reading", _struct_leaf_type(bc)) + S === nothing && return nothing + throw( + ArgumentError( + "Broadcasts $(role) struct element type $(S) require GPU broadcast " * + "fusion with same-shaped NDArray inputs of rank at least 1 " * + "(fusion enabled: $(FUSE_BROADCAST_EXPRS), GPU available: $(_has_gpu_target()))", + ), + ) +end + +@inline _struct_leaf_type(_) = nothing +@inline _struct_leaf_type(::NDArray{T}) where {T} = _struct_storage_type(T) ? T : nothing +@inline _struct_leaf_type(bc::Broadcasted) = _struct_leaf_type_args(bc.args) +@inline _struct_leaf_type_args(::Tuple{}) = nothing +@inline function _struct_leaf_type_args(args::Tuple) + S = _struct_leaf_type(first(args)) + return S === nothing ? _struct_leaf_type_args(Base.tail(args)) : S +end + # Support .= @inline Base.copyto!(dest::NDArray, bc::Broadcasted{Nothing}) = _copyto!(dest, bc) @inline Base.copyto!(dest::NDArray, bc::Broadcasted{<:NDArrayStyle}) = _copyto!(dest, bc) diff --git a/src/ndarray/broadcast_fusion.jl b/src/ndarray/broadcast_fusion.jl index 6611edf11..164012f92 100644 --- a/src/ndarray/broadcast_fusion.jl +++ b/src/ndarray/broadcast_fusion.jl @@ -77,6 +77,16 @@ function _push_static_arg!(static_args, arg_plan, x) ), ) + # Static leaves are captured by the kernel closure, which the launcher does + # not pass to the device; only zero-size values (functions, `Val`) are safe. + # TODO pass isbits values such as structs as runtime scalars; the C++ + # launcher would need to align each scalar in the argument buffer. + sizeof(x) == 0 || throw( + ArgumentError( + "Broadcast fusion cannot pass $(repr(x)) of type $(typeof(x)) to the GPU " * + "kernel; pass numbers or NDArrays instead", + ), + ) push!(static_args, x) push!(arg_plan, StaticBroadcastArg{length(static_args)}()) return nothing @@ -127,16 +137,21 @@ end Int(CUDACore.threadIdx().x) end +# Widen before multiplying. C++ caps the grid; loops must still visit every element. +@inline _broadcast_grid_stride(axis) = + Int(getproperty(CUDACore.gridDim(), axis)) * Int(getproperty(CUDACore.blockDim(), axis)) + function make_linear_kernel(dest, bc::Base.Broadcast.Broadcasted, arg_plan, static_args) f = bc.f @kernel unsafe_indices = true function broadcast_kernel_linear_splat(dest, runtime_args...) I = _broadcast_linear_work_id() - if I <= length(dest) + while I <= length(dest) @inbounds args_modified = _materialize_broadcast_args( arg_plan, runtime_args, static_args, I ) @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + I += _broadcast_grid_stride(:x) end end @@ -159,12 +174,17 @@ function make_cartesian_kernel(dest, bc::Base.Broadcast.Broadcasted, arg_plan, s @kernel unsafe_indices = true function broadcast_kernel_cartesian_splat( dest, runtime_args... ) - I = _broadcast_cartesian_work_id() - if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) - @inbounds args_modified = _materialize_broadcast_args( - arg_plan, runtime_args, static_args, I - ) - @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + start = _broadcast_cartesian_work_id() + I = start + while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + @inbounds args_modified = _materialize_broadcast_args( + arg_plan, runtime_args, static_args, I + ) + @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + I += CartesianIndex(_broadcast_grid_stride(:y), 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:x)) end end @@ -192,12 +212,22 @@ function make_cartesian_kernel_3d( @kernel unsafe_indices = true function broadcast_kernel_cartesian_3d_splat( dest, runtime_args... ) - I = _broadcast_cartesian_work_id_3d() - if I[1] <= size(dest, 1) && I[2] <= size(dest, 2) && I[3] <= size(dest, 3) - @inbounds args_modified = _materialize_broadcast_args( - arg_plan, runtime_args, static_args, I - ) - @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf(f, args_modified...) + start = _broadcast_cartesian_work_id_3d() + I = start + while I[3] <= size(dest, 3) + while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + @inbounds args_modified = _materialize_broadcast_args( + arg_plan, runtime_args, static_args, I + ) + @inbounds dest[I] = Base.Broadcast._broadcast_getindex_evalf( + f, args_modified... + ) + I += CartesianIndex(_broadcast_grid_stride(:z), 0, 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:y), I[3]) + end + I = CartesianIndex(start[1], start[2], I[3] + _broadcast_grid_stride(:x)) end end @@ -270,8 +300,22 @@ Also refuses 0-d destinations: `RunPTXBroadcastTask` only supports dims in end # Same-shaped operands do not need Broadcast's dynamic index projection. +# +# Size-1 dimensions are special: Broadcast marks them `keeps=false` even when the +# leaf shape matches `dest` (e.g. length-1 vectors). Linear `I` is still valid in +# that case because the dimension only has index 1. +@inline function _extruded_ok_for_fusion(x::Base.Broadcast.Extruded) + keeps = x.keeps + for i in eachindex(keeps) + if !keeps[i] && size(x.x, i) != 1 + return false + end + end + return true +end + @inline function _unwrap_fusion_arg(x::Base.Broadcast.Extruded) - if all(x.keeps) + if _extruded_ok_for_fusion(x) return x.x end throw( @@ -309,7 +353,10 @@ end # checks stay in pre-flatten `_assert_fused_broadcast_promotion`. function _align_fused_runtime_args(runtime_args::Tuple) isempty(runtime_args) && return runtime_args - T_IN = __my_promote_type(map(eltype, runtime_args)...) + all(a -> !(a isa Number), runtime_args) && return runtime_args + numeric_types = filter(T -> T <: Number, map(eltype, runtime_args)) + isempty(numeric_types) && return runtime_args + T_IN = __my_promote_type(numeric_types...) return map(a -> unchecked_promote_scalar(a, T_IN), runtime_args) end @@ -431,7 +478,16 @@ function get_cuda_task( lock(_BCAST_PTX_CACHE_LOCK) do return get!(_BCAST_PTX_CACHE, key) do - ptx, threads, ctx = get_ptx(obj, DEST_T, ARG_TYPES...) + ptx, threads, ctx = try + get_ptx(obj, DEST_T, ARG_TYPES...) + catch err + err isa InterruptException && rethrow() + throw( + ErrorException( + "GPU broadcast function failed to fuse: $(sprint(showerror, err))" + ), + ) + end orig_name = extract_kernel_name(ptx) unique_name = orig_name * "_" * string(hash(ptx); base=16) @@ -605,8 +661,20 @@ end bc::Base.Broadcast.Broadcasted{S,Ax,F,Args} ) where {S,Ax,F,Args} eltypes = _fused_checked_eltypes(bc.args) - T_OUT = __checked_promote_op(bc.f, eltypes) - __my_promote_type(eltypes.parameters...) + T_OUT = if length(bc.args) > 2 && _is_flattened_associative(bc.f) + _checked_promote_associative(bc.f, eltypes.parameters...) + else + __checked_promote_op(bc.f, eltypes) + end + if bc.f isa StructConstructor + # Constructing a record keeps each field's declared type; its numeric + # inputs are not operands of a common arithmetic operation. + elseif bc.f === Base.literal_pow + __my_promote_type(eltypes.parameters...) + else + numeric_types = _numeric_broadcast_types(eltypes) + isempty(numeric_types) || __my_promote_type(numeric_types...) + end return T_OUT end @@ -639,6 +707,19 @@ end return Tuple{T1,rest.parameters...} end +# Like static leaves, the broadcast function is captured by the kernel closure, +# which the launcher does not pass to the device. +# TODO pass the closure's captured state to the kernel so closures can fuse. +@inline function _assert_kernel_function_has_no_data(f) + sizeof(f) == 0 || throw( + ArgumentError( + "Broadcast fusion cannot pass the captured variables of $(typeof(f)) to " * + "the GPU kernel; pass them as broadcast arguments instead", + ), + ) + return nothing +end + function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcast.Broadcasted} # Promotion checks use the pre-flatten tree (same shape as unfused unravel). _assert_fused_broadcast_promotion(dest, bc) @@ -653,6 +734,7 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas bc = Base.Broadcast.preprocess(dest, bc) bc = Base.Broadcast.instantiate(bc) bc = Base.Broadcast.flatten(bc) + _assert_kernel_function_has_no_data(bc.f) # Things like exponentiation generate arguments like Base.RefValue # which do not work with our pattern for making CUDA kernels as they are @@ -705,6 +787,9 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas input_ndarrays = tuple(unique_ndarrays...) + # Legion forbids overlapping input and output regions in one task. + output = any(nda -> nda_overlaps(dest, nda), unique_ndarrays) ? similar(dest) : dest + if BCAST_FUSION_DEBUG[] tree_str = _bcast_runtime_tree_str( bc_scope, ndarray_to_input_idx, actual_scalars @@ -721,23 +806,361 @@ function fuse_broadcast_tree!(dest::D, bc::B) where {D<:NDArray,B<:Base.Broadcas ) end - @task_scope _bcast_scope_name(bc_scope, ndarray_to_input_idx, actual_scalars) begin - # `blocks=1` is a placeholder; RunPTXBroadcastTask overwrites grid dims - # from the local PhysicalArray. `threads` is only the occupancy budget (tx). - # Scalars after ctx: num_kernel_args, arg_map... - launch( - fkm.cuda_task, - input_ndarrays, - (dest,), - (Int32(length(arg_map)), arg_map..., actual_scalars...); - blocks=1, - threads=fkm.threads, - taskid=cuNumeric.RUN_PTX_BROADCAST, - ctx=fkm.ctx, - validate_shapes=false, - ) + try + @task_scope _bcast_scope_name(bc_scope, ndarray_to_input_idx, actual_scalars) begin + # `blocks=1` is a placeholder; RunPTXBroadcastTask overwrites grid dims + # from the local PhysicalArray. `threads` is only the occupancy budget (tx). + # Scalars after ctx: num_kernel_args, arg_map... + launch( + fkm.cuda_task, + input_ndarrays, + (output,), + (Int32(length(arg_map)), arg_map..., actual_scalars...); + blocks=1, + threads=fkm.threads, + taskid=cuNumeric.RUN_PTX_BROADCAST, + ctx=fkm.ctx, + ) + end + output === dest || nda_assign(dest, output) + finally + output === dest || destroy!(output) end - # Fused kernel already wrote `dest` in place; promotion was checked pre-launch. return dest end + +# ============================================================================ +# Multi-output fused broadcast: materialize named intermediates in one launch. +# Segments (dependency order, root last) are flattened independently; a +# `MatRef{K}` leaf reads the K-th segment's per-element local. One kernel +# computes each segment into a local, stores it to that segment's output buffer, +# and chains locals into parents (segmented flatten + chained-local multi-store). +# ============================================================================ + +# Opaque scalar-like leaf that survives `Base.Broadcast.flatten` (never descended +# into, never wrapped in a Ref). +struct MatRef{K} end +MatRef(k::Int) = MatRef{k}() +Base.broadcastable(m::MatRef) = m +Base.Broadcast.BroadcastStyle(::Type{<:MatRef}) = Base.Broadcast.DefaultArrayStyle{0}() +Base.axes(::MatRef) = () +Base.ndims(::Type{<:MatRef}) = 0 + +# Third arg-plan variant (alongside Runtime/Static): read the K-th chained local. +struct LocalBroadcastArg{K} end + +Base.@propagate_inbounds @inline _materialize_ml_arg( + ::RuntimeBroadcastArg{J}, rt, sa, locals, I +) where {J} = _gpu_broadcast_getindex(getfield(rt, J), I) +Base.@propagate_inbounds @inline _materialize_ml_arg( + ::StaticBroadcastArg{J}, rt, sa, locals, I +) where {J} = getfield(sa, J) +Base.@propagate_inbounds @inline _materialize_ml_arg( + ::LocalBroadcastArg{K}, rt, sa, locals, I +) where {K} = getfield(locals, K) + +Base.@propagate_inbounds @inline _materialize_ml_args(::Tuple{}, rt, sa, locals, I) = () +Base.@propagate_inbounds @inline function _materialize_ml_args(plan::Tuple, rt, sa, locals, I) + return ( + @inbounds(_materialize_ml_arg(getfield(plan, 1), rt, sa, locals, I)), + @inbounds(_materialize_ml_args(Base.tail(plan), rt, sa, locals, I))..., + ) +end + +# Device-side: run each segment in order, store to its output, chain the local. +# Generate straight-line code because recursive tuple traversal eventually hits +# Julia's inference limit and leaves a dynamic call in GPU kernels on Julia 1.10. +Base.@propagate_inbounds @inline @generated function _run_segments( + segs::S, outs, rt, sa, locals::L, I +) where {S<:Tuple,L<:Tuple} + body = Expr(:block) + local_values = Any[:(getfield(locals, $k)) for k in 1:fieldcount(L)] + + for k in 1:fieldcount(S) + seg = gensym(:seg) + vals = gensym(:vals) + value = gensym(:value) + local_tuple = Expr(:tuple, local_values...) + push!( + body.args, + quote + $seg = getfield(segs, $k) + $vals = _materialize_ml_args( + getfield($seg, 2), rt, sa, $local_tuple, I + ) + $value = Base.Broadcast._broadcast_getindex_evalf( + getfield($seg, 1), $vals... + ) + @inbounds getfield(outs, $k)[I] = $value + end, + ) + push!(local_values, value) + end + + push!(body.args, :(nothing)) + return body +end + +# Dimension-dispatched (mirrors the single-output linear/cartesian kernels). +# `args` = (outputs[1:NOUT]..., runtime_args...); bounds from the first output. +function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val{2}) where {NOUT} + @kernel unsafe_indices = true function broadcast_kernel_multi_2d(args...) + start = _broadcast_cartesian_work_id() + I = start + dest = getfield(args, 1) + @inbounds while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + I += CartesianIndex(_broadcast_grid_stride(:y), 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:x)) + end + end + return broadcast_kernel_multi_2d +end + +function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val{3}) where {NOUT} + @kernel unsafe_indices = true function broadcast_kernel_multi_3d(args...) + start = _broadcast_cartesian_work_id_3d() + I = start + dest = getfield(args, 1) + @inbounds while I[3] <= size(dest, 3) + while I[2] <= size(dest, 2) + while I[1] <= size(dest, 1) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + I += CartesianIndex(_broadcast_grid_stride(:z), 0, 0) + end + I = CartesianIndex(start[1], I[2] + _broadcast_grid_stride(:y), I[3]) + end + I = CartesianIndex(start[1], start[2], I[3] + _broadcast_grid_stride(:x)) + end + end + return broadcast_kernel_multi_3d +end + +# 1-D and any other rank: linear indexing (matches the single-output default). +function make_multi_output_kernel(segs, ::Val{NOUT}, static_args, ::Val) where {NOUT} + @kernel unsafe_indices = true function broadcast_kernel_multi_linear(args...) + I = _broadcast_linear_work_id() + @inbounds while I <= length(getfield(args, 1)) + _run_segments(segs, args[1:NOUT], args[(NOUT + 1):end], static_args, (), I) + I += _broadcast_grid_stride(:x) + end + end + return broadcast_kernel_multi_linear +end + +# Flatten a segment and classify its leaves, deduping NDArrays into shared +# `runtime_args` and static leaves into shared `static_args`. +function _split_segment!(seg_bc, runtime_args, static_args, ndarray_idx) + flat = Base.Broadcast.flatten(seg_bc) + _assert_kernel_function_has_no_data(flat.f) + plan = Any[] + for leaf in flat.args + if leaf isa MatRef + push!(plan, LocalBroadcastArg{_matref_k(leaf)}()) + elseif leaf isa Base.RefValue + v = leaf[] + if v isa Number + push!(runtime_args, v) + push!(plan, RuntimeBroadcastArg{length(runtime_args)}()) + else + _push_static_arg!(static_args, plan, v) + end + elseif leaf isa NDArray || leaf isa Base.Broadcast.Extruded + nda = get_ndarray(leaf) + j = get!(() -> (push!(runtime_args, leaf); length(runtime_args)), + ndarray_idx, objectid(nda)) + push!(plan, RuntimeBroadcastArg{j}()) + elseif leaf isa Number + push!(runtime_args, leaf) + push!(plan, RuntimeBroadcastArg{length(runtime_args)}()) + elseif isbits(leaf) + _push_static_arg!(static_args, plan, leaf) + else + throw(ArgumentError("multi-output fusion: cannot lower leaf $(typeof(leaf))")) + end + end + return (flat.f, tuple(plan...)) +end + +_matref_k(::MatRef{K}) where {K} = K + +const _MULTI_PTX_CACHE = Dict{Any,Any}() +const _MULTI_PTX_CACHE_LOCK = ReentrantLock() + +# Compile + register the multi-output kernel -> (ctx, threads, CUDATask). Cached +# by (closure type, arg types) so a repeated fusion signature compiles once. +function get_multi_cuda_task(obj, out_arrs, runtime_args) + arg_types = (map_cuda_type.(typeof.(out_arrs))..., map_cuda_type.(typeof.(runtime_args))...) + key = (typeof(obj), arg_types) + lock(_MULTI_PTX_CACHE_LOCK) do + return get!(_MULTI_PTX_CACHE, key) do + ptx, threads, ctx = get_ptx(obj, arg_types...) + threads == 0 && return (ctx, 0, nothing) + orig = extract_kernel_name(ptx) + uname = orig * "_" * string(hash(ptx); base=16) + ptx = replace(ptx, orig => uname) + ptx_task(ptx, uname) + return (ctx, threads, CUDATask(uname, arg_types)) + end + end +end + +# First NDArray leaf across all segments; used as an allocation template. +function _first_ndarray(seg_bcs::Tuple) + for seg_bc in seg_bcs + for leaf in Base.Broadcast.flatten(seg_bc).args + leaf isa NDArray && return leaf + leaf isa Base.Broadcast.Extruded && return get_ndarray(leaf) + end + end + return throw(ArgumentError("multi-output fusion: no NDArray leaf to size buffers from")) +end + +# Result eltype of a segment, resolving `MatRef{k}` to `eltype(bufs[k])` (earlier +# segments already allocated). Lets each intermediate use its own eltype. +function _segment_eltype(flat, bufs) + ets = map(flat.args) do leaf + if leaf isa MatRef + eltype(bufs[_matref_k(leaf)]) + elseif leaf isa NDArray + eltype(leaf) + elseif leaf isa Base.Broadcast.Extruded + eltype(leaf.x) + else + typeof(leaf) + end + end + T = Base.promote_op(flat.f, ets...) + # Fall back to promoting the leaf eltypes when promote_op can't infer. + return isconcretetype(T) ? T : promote_type(ets...) +end + +# Render a segment's broadcast tree, showing `MatRef{k}` leaves as `seg{k}`. +function _bcast_multi_tree_str(bc) + return _bcast_tree_str(bc) do x + x isa MatRef && return "seg{$(_matref_k(x))}" + x isa NDArray && return "NDArray" + x isa Base.Broadcast.Extruded && return "NDArray" + x isa Number && return repr(x) + x isa Base.RefValue && return string("^", repr(x[])) + return string("<", typeof(x), ">") + end +end + +# Fused multi-output introspection (mirrors `_describe_fused_broadcast`). Enable +# with `cuNumeric.BCAST_FUSION_DEBUG[] = true`. +function _describe_fused_multi( + out_arrs, seg_bcs, input_ndarrays, actual_scalars, argmap, threads, ndrange +) + io = IOBuffer() + field(k, v) = println(io, " ", rpad(k, 8), v) + NOUT = length(out_arrs) + println(io, "\n", "="^40, " fused multi-output broadcast ($NOUT outputs)") + println(io, " segments (each materialized to its own output):") + for (i, seg) in enumerate(seg_bcs) + role = i == NOUT ? "root" : "seg{$i}" + println( + io, " ", rpad(role, 7), _ndarray_debug_summary(out_arrs[i]), + " <- ", _bcast_multi_tree_str(seg), + ) + end + field("inputs", "input{N} ($(length(input_ndarrays)) unique)") + for (i, nd) in enumerate(input_ndarrays) + println(io, " ", rpad(string(i - 1), 4), _ndarray_debug_summary(nd)) + end + isempty(actual_scalars) || field("scalars", join(repr.(actual_scalars), ", ")) + indexing = ndims(out_arrs[1]) in (2, 3) ? "cartesian" : "linear" + field( + "launch", + "host thread budget=$threads, indexing=$indexing, num_outputs=$NOUT, " * + "blocks=device(local tile), global_ndrange=$ndrange", + ) + field("arg_map", string(argmap)) + print(String(take!(io))) + return nothing +end + +# Launch one kernel writing each segment into preallocated `out_arrs[i]` +# (dependency order; `out_arrs[end]` is the root). +function _fused_multi_launch!(out_arrs::Tuple, seg_bcs::Tuple) + NOUT = length(seg_bcs) + runtime_args = Any[] + static_args = Any[] + ndarray_idx = Dict{UInt,Int}() + segs = Any[] + for seg_bc in seg_bcs + push!(segs, _split_segment!(seg_bc, runtime_args, static_args, ndarray_idx)) + end + segs = tuple(segs...) + static_args = tuple(static_args...) + + kernel = make_multi_output_kernel(segs, Val(NOUT), static_args, Val(ndims(out_arrs[1]))) + bck = kernel(CUDACore.CUDAKernels.CUDABackend()) + ctx, threads, task = get_multi_cuda_task(bck, out_arrs, tuple(runtime_args...)) + isnothing(task) && return out_arrs + + # arg_map: kernel args in order (outputs..., runtime_args...). + argmap = Int32[Int32(i) for i in 0:(NOUT - 1)] + input_ndarrays = NDArray[] + ndinput_idx = Dict{UInt,Int}() + actual_scalars = Any[] + for arg in runtime_args + if stores_cudevicearray(map_cuda_type(typeof(arg))) + nda = get_ndarray(arg) + j = get!(() -> (push!(input_ndarrays, nda); length(input_ndarrays) - 1), + ndinput_idx, objectid(nda)) + push!(argmap, Int32(NOUT + j)) + else + push!(argmap, Int32(-1 - length(actual_scalars))) + push!(actual_scalars, arg) + end + end + + if BCAST_FUSION_DEBUG[] + ndrange = ndims(out_arrs[1]) > 0 ? size(out_arrs[1]) : (1,) + _describe_fused_multi( + out_arrs, seg_bcs, input_ndarrays, actual_scalars, argmap, threads, ndrange + ) + end + + launch( + task, tuple(input_ndarrays...), out_arrs, + (Int32(length(argmap)), argmap..., actual_scalars...); + blocks=1, threads=threads, taskid=cuNumeric.RUN_PTX_BROADCAST, ctx=ctx, + ) + return out_arrs +end + +# Allocate a typed tuple so callers retain each segment's concrete NDArray type. +function _alloc_segment_buffers(template::NDArray, seg_bcs::Tuple, dims) + return _alloc_segment_buffers(template, seg_bcs, dims, ()) +end + +@inline _alloc_segment_buffers(template, ::Tuple{}, dims, bufs::Tuple) = bufs + +@inline function _alloc_segment_buffers(template, seg_bcs::Tuple, dims, bufs::Tuple) + flat = Base.Broadcast.flatten(first(seg_bcs)) + buf = similar(template, _segment_eltype(flat, bufs), dims) + return _alloc_segment_buffers(template, Base.tail(seg_bcs), dims, (bufs..., buf)) +end + +# `seg_bcs[1:end-1]` are materialized producers (dependency order); `seg_bcs[end]` +# writes `dest`. Producer buffers are allocated (returned so callers bind names). +function copyto_fused_multi!(dest::NDArray, seg_bcs::Tuple) + bufs = _alloc_segment_buffers(dest, seg_bcs[1:(end - 1)], size(dest)) + outs = (bufs..., dest) + _fused_multi_launch!(outs, seg_bcs) + return outs +end + +# Every segment gets a fresh buffer (all named results stay live). Returns the +# buffers in segment order so callers can bind each user name. +function copyto_fused_multi_alloc!(seg_bcs::Tuple) + tmpl = _first_ndarray(seg_bcs) + outs = _alloc_segment_buffers(tmpl, seg_bcs, size(tmpl)) + _fused_multi_launch!(outs, seg_bcs) + return outs +end diff --git a/src/ndarray/contract.jl b/src/ndarray/contract.jl new file mode 100644 index 000000000..8897cd335 --- /dev/null +++ b/src/ndarray/contract.jl @@ -0,0 +1,457 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +export contract!, contract, tensordot + +const _CONTRACT_NATIVE = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const _CONTRACT_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool} +const _CONTRACT_ACCEPTED = Union{_CONTRACT_NATIVE,_CONTRACT_PROMOTABLE} + +_contract_eltype(::Type{T}) where {T<:_CONTRACT_NATIVE} = T +_contract_eltype(::Type{<:_CONTRACT_PROMOTABLE}) = Float64 + +function _modes_as_chars(modes::AbstractString) + chars = Vector{UInt8}(undef, length(modes)) + for (i, c) in enumerate(modes) + isascii(c) || throw(ArgumentError("contract mode labels must be ASCII, got $(repr(c))")) + chars[i] = UInt8(c) + end + return chars +end + +function _modes_as_chars(modes::AbstractVector{UInt8}) + return collect(UInt8, modes) +end + +function _modes_as_chars(modes::AbstractVector{Char}) + chars = Vector{UInt8}(undef, length(modes)) + for (i, c) in enumerate(modes) + isascii(c) || throw(ArgumentError("contract mode labels must be ASCII, got $(repr(c))")) + chars[i] = UInt8(c) + end + return chars +end + +function _modes_as_chars(modes::AbstractVector{<:Integer}) + chars = Vector{UInt8}(undef, length(modes)) + for (i, m) in enumerate(modes) + v = Int(m) + (Int('a') - 1) + (0 <= v <= 255) || throw(ArgumentError("integer mode label $m is out of range")) + chars[i] = UInt8(v) + end + return chars +end + +function _modes_as_chars(modes::Tuple) + return _modes_as_chars(collect(modes)) +end + +function _require_modes(arr::NDArray, modes) + chars = _modes_as_chars(modes) + length(chars) == ndims(arr) || throw( + ArgumentError("expected $(ndims(arr)) mode labels, got $(length(chars))") + ) + if length(Base.unique(chars)) != length(chars) + throw(ArgumentError("duplicate mode labels are not allowed: $(modes)")) + end + return chars +end + +function _mode_extents(A::NDArray, Am, B::NDArray, Bm) + extents = Dict{UInt8,Int}() + for (m, s) in zip(Am, size(A)) + prev = get(extents, m, s) + prev == s || throw( + DimensionMismatch("mode $(Char(m)) has incompatible extents $prev and $s") + ) + extents[m] = s + end + for (m, s) in zip(Bm, size(B)) + prev = get(extents, m, s) + prev == s || throw( + DimensionMismatch("mode $(Char(m)) has incompatible extents $prev and $s") + ) + extents[m] = s + end + return extents +end + +function _check_mode_counts(Cm, Am, Bm) + counts = Dict{UInt8,Int}() + for m in Cm + counts[m] = get(counts, m, 0) + 1 + end + for m in Am + counts[m] = get(counts, m, 0) + 1 + end + for m in Bm + counts[m] = get(counts, m, 0) + 1 + end + for (m, c) in counts + (c == 2 || c == 3) || throw( + ArgumentError( + "mode $(Char(m)) appears $c times; each label must appear twice or three times across the output and inputs" + ), + ) + end + return nothing +end + +function _extent_arrays(extents) + n = length(extents) + keys_out = Vector{UInt8}(undef, n) + vals_out = Vector{Int32}(undef, n) + i = 0 + for (k, v) in extents + i += 1 + keys_out[i] = k + vals_out[i] = Int32(v) + end + return keys_out, vals_out +end + +# cupynumeric's MM shortcut asserts on untransposed NDArray shapes. Present a +# physical `ik,kj->ij` via Legate store transpose (no data copy) when needed. +function _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + if length(Cm) == 2 && length(Am) == 2 && length(Bm) == 2 + i, j = Cm[1], Cm[2] + k = (Am[1] == i || Am[1] == j) ? Am[2] : Am[1] + if k != i && k != j + left, Lm, right, Rm = (i == Am[1] || i == Am[2]) ? (A, Am, B, Bm) : (B, Bm, A, Am) + oL = Lm[1] != i + oR = Rm[2] != j + Lp = oL ? permutedims(left) : left + Rp = oR ? permutedims(right) : right + nda_contract(C, Cm, Lp, UInt8[i, k], Rp, UInt8[k, j], extent_keys, extent_vals) + oL && destroy!(Lp) + oR && destroy!(Rp) + return nothing + end + end + nda_contract(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + return nothing +end + +function _free_output_modes(A::NDArray, Am, B::NDArray, Bm) + counts = Dict{UInt8,Int}() + for m in Am + counts[m] = get(counts, m, 0) + 1 + end + for m in Bm + counts[m] = get(counts, m, 0) + 1 + end + Cm = UInt8[] + cshape = Int[] + for (m, s) in zip(Am, size(A)) + if counts[m] == 1 + push!(Cm, m) + push!(cshape, s) + end + end + for (m, s) in zip(Bm, size(B)) + if counts[m] == 1 + push!(Cm, m) + push!(cshape, s) + end + end + return Cm, Tuple(cshape) +end + +""" + contract!(C, Cmodes, A, Amodes, B, Bmodes; α=1, β=0) + +In-place pairwise tensor contraction `C = β * C + α * (A ⋆ B)`. + +`α` and `β` may be a Julia `Number` or a 0-d `NDArray`. + +`Amodes`, `Bmodes`, and `Cmodes` are mode labels for `A`, `B`, and `C`: an +ASCII `AbstractString`, a tuple/vector of `Char`, or a vector of integers +(`1` maps to `'a'`). Each label appears twice (a contracted or free index) or +three times (a batched / Hadamard index). Duplicate labels inside one array +are not allowed — extract a diagonal first with `cuNumeric.diagonal`. + +Supported element types are `Float32`, `Float64`, `ComplexF32`, and +`ComplexF64`. Integer and `Bool` inputs are converted to `Float64` under the +usual promotion rules. + +This is the pairwise primitive TensorOperations.jl can call later. It does not +parse einsum strings. Multi-GPU execution uses Legate tiling (not cuTensorMp). +""" +function contract!( + C::NDArray{TC}, + Cmodes, + A::NDArray{TA}, + Amodes, + B::NDArray{TB}, + Bmodes; + α=1, + β=0, +) where {TC<:_CONTRACT_NATIVE,TA<:_CONTRACT_ACCEPTED,TB<:_CONTRACT_ACCEPTED} + T = promote_type(_contract_eltype(TA), _contract_eltype(TB)) + T === TC || throw( + ArgumentError("contract! output has type $TC, but inputs promote to $T") + ) + + Ap = checked_promote_arr(contract!, A, T) + Bp = checked_promote_arr(contract!, B, T) + try + return _contract_same_type!(C, Cmodes, Ap, Amodes, Bp, Bmodes, _scale_storage(α), _scale_storage(β)) + finally + Ap !== A && destroy!(Ap) + Bp !== B && destroy!(Bp) + end +end + +function contract!(C::NDArray, Cmodes, A::NDArray, Amodes, B::NDArray, Bmodes; α=1, β=0) + bad = if eltype(C) <: _CONTRACT_NATIVE + (eltype(A) <: _CONTRACT_ACCEPTED ? eltype(B) : eltype(A)) + else + eltype(C) + end + return throw(ArgumentError("array type $bad is unsupported in contract!")) +end + +# Device scalar wrappers specialize this to expose storage without host extraction. +_scale_storage(x) = x +_host_iszero(x::Number) = iszero(x) +_host_iszero(::NDArray) = false +_host_isone(x::Number) = isone(x) +_host_isone(::NDArray) = false + +function _require_0d_scale(x::NDArray) + ndims(x) == 0 || throw( + ArgumentError("contract! scale factor must be a Number or a 0-d NDArray") + ) + return x +end + +function _contract_prepare( + C::NDArray{T}, Cmodes, A::NDArray{T}, Amodes, B::NDArray{T}, Bmodes +) where {T} + (C.ptr === A.ptr || C.ptr === B.ptr) && throw( + ArgumentError("contract! output must not alias either input") + ) + + Cm = _require_modes(C, Cmodes) + Am = _require_modes(A, Amodes) + Bm = _require_modes(B, Bmodes) + _check_mode_counts(Cm, Am, Bm) + extents = _mode_extents(A, Am, B, Bm) + for (m, s) in zip(Cm, size(C)) + expected = get(extents, m, -1) + expected == s || throw( + DimensionMismatch("output mode $(Char(m)) has size $s, expected $expected") + ) + end + extent_keys, extent_vals = _extent_arrays(extents) + return Cm, Am, Bm, extent_keys, extent_vals +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::Number, + β::Number, +) where {T} + Cm, Am, Bm, extent_keys, extent_vals = _contract_prepare(C, Cmodes, A, Amodes, B, Bmodes) + if isone(α) && iszero(β) + _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + return C + end + αT = convert(T, α) + if iszero(β) + _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= αT .* C + return C + end + βT = convert(T, β) + tmp = similar(C) + _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= βT .* C .+ αT .* tmp + destroy!(tmp) + return C +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::NDArray, + β::Number, +) where {T} + _require_0d_scale(α) + Cm, Am, Bm, extent_keys, extent_vals = _contract_prepare(C, Cmodes, A, Amodes, B, Bmodes) + α′ = eltype(α) === T ? α : as_type(α, T) + try + if iszero(β) + _nda_contract!(C, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= α′ .* C + return C + end + tmp = similar(C) + _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) + if isone(β) + C .= C .+ α′ .* tmp + else + βa = NDArray(convert(T, β)) + C .= βa .* C .+ α′ .* tmp + destroy!(βa) + end + destroy!(tmp) + return C + finally + α′ !== α && destroy!(α′) + end +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::Number, + β::NDArray, +) where {T} + _require_0d_scale(β) + αa = NDArray(convert(T, α)) + try + return _contract_same_type!(C, Cmodes, A, Amodes, B, Bmodes, αa, β) + finally + destroy!(αa) + end +end + +function _contract_same_type!( + C::NDArray{T}, + Cmodes, + A::NDArray{T}, + Amodes, + B::NDArray{T}, + Bmodes, + α::NDArray, + β::NDArray, +) where {T} + _require_0d_scale(α) + _require_0d_scale(β) + Cm, Am, Bm, extent_keys, extent_vals = _contract_prepare(C, Cmodes, A, Amodes, B, Bmodes) + α′ = eltype(α) === T ? α : as_type(α, T) + β′ = eltype(β) === T ? β : as_type(β, T) + try + tmp = similar(C) + _nda_contract!(tmp, Cm, A, Am, B, Bm, extent_keys, extent_vals) + C .= β′ .* C .+ α′ .* tmp + destroy!(tmp) + return C + finally + α′ !== α && destroy!(α′) + β′ !== β && destroy!(β′) + end +end + +""" + contract(A, Amodes, B, Bmodes; α=1) + +Allocate and return `α * (A ⋆ B)`. Output modes are the labels that appear +once, in the order they occur on `A` then `B` (classical Einstein). For a +batched or Hadamard product, allocate `C` yourself and call [`contract!`](@ref) +with explicit `Cmodes`. +""" +function contract( + A::NDArray{TA}, Amodes, B::NDArray{TB}, Bmodes; α=1 +) where {TA<:_CONTRACT_ACCEPTED,TB<:_CONTRACT_ACCEPTED} + T = promote_type(_contract_eltype(TA), _contract_eltype(TB)) + T <: _CONTRACT_NATIVE || + throw(ArgumentError("array type $T is unsupported in contract")) + + Am = _require_modes(A, Amodes) + Bm = _require_modes(B, Bmodes) + Cm, cshape = _free_output_modes(A, Am, B, Bm) + C = NDArray{T}(undef, cshape) + return contract!(C, Cm, A, Am, B, Bm; α=α, β=zero(T)) +end + +function contract(A::NDArray, Amodes, B::NDArray, Bmodes; α=1) + bad = eltype(A) <: _CONTRACT_ACCEPTED ? eltype(B) : eltype(A) + return throw(ArgumentError("array type $bad is unsupported in contract")) +end + +""" + tensordot(A, B, axes=2; α=1) + tensordot(A, B, (a_axes, b_axes); α=1) + +Contract `A` with `B` along the given 1-based axes. `axes::Integer` contracts +the last `axes` dimensions of `A` with the first `axes` of `B`. A tuple of axis +collections names the axes on each input. Remaining axes of `A` then `B` become +the output. +""" +function tensordot(A::NDArray, B::NDArray, axes::Integer=2; α=1) + n = Int(axes) + n < 0 && throw(ArgumentError("axes must be non-negative, got $n")) + na = ndims(A) + nb = ndims(B) + n > na && throw(ArgumentError("cannot contract $n axes of a $(na)-d array")) + n > nb && throw(ArgumentError("cannot contract $n axes of a $(nb)-d array")) + a_axes = ntuple(i -> na - n + i, n) + b_axes = ntuple(identity, n) + return tensordot(A, B, (a_axes, b_axes); α=α) +end + +function tensordot(A::NDArray, B::NDArray, axes::Tuple; α=1) + a_raw, b_raw = axes + a_axes = collect(Int, a_raw isa Integer ? (a_raw,) : a_raw) + b_axes = collect(Int, b_raw isa Integer ? (b_raw,) : b_raw) + length(a_axes) == length(b_axes) || + throw(ArgumentError("tensordot axis lists must have the same length")) + length(Base.unique(a_axes)) == length(a_axes) || + throw(ArgumentError("duplicate axes on first input: $a_axes")) + length(Base.unique(b_axes)) == length(b_axes) || + throw(ArgumentError("duplicate axes on second input: $b_axes")) + + na = ndims(A) + nb = ndims(B) + Am = Vector{UInt8}(undef, na) + Bm = Vector{UInt8}(undef, nb) + for i in 1:na + Am[i] = UInt8('a' + (i - 1)) + end + for i in 1:nb + Bm[i] = UInt8('A' + (i - 1)) + end + for (ai, bi) in zip(a_axes, b_axes) + (1 <= ai <= na) || throw(ArgumentError("axis $ai is out of range for $(na)-d array")) + (1 <= bi <= nb) || throw(ArgumentError("axis $bi is out of range for $(nb)-d array")) + size(A, ai) == size(B, bi) || throw( + DimensionMismatch( + "tensordot axes $ai and $bi have sizes $(size(A, ai)) and $(size(B, bi))" + ), + ) + Bm[bi] = Am[ai] + end + return contract(A, Am, B, Bm; α=α) +end diff --git a/src/ndarray/detail/distributed_linalg.jl b/src/ndarray/detail/distributed_linalg.jl new file mode 100644 index 000000000..abfd11fc8 --- /dev/null +++ b/src/ndarray/detail/distributed_linalg.jl @@ -0,0 +1,237 @@ +# Copyright 2024 NVIDIA Corporation +# SPDX-License-Identifier: Apache-2.0 +# Task construction follows cuPyNumeric 26.06's linalg/_solve.py, _qr.py, +# and _cholesky.py. +# Keep algorithm selection here; Legate owns placement and redistribution. +struct _LinalgRuntime + available::Bool + gpus::Int + procs::Int + mp_eligible::Bool +end + +function _LinalgRuntime(available::Bool, gpus::Int, procs::Int) + return _LinalgRuntime(available, gpus, procs, available && gpus > 1) +end + +# Populated once in _start_runtime(), including deferred initialization. +# These describe the configured machine for the lifetime of this runtime. +const _LINALG_RUNTIME = Ref(_LinalgRuntime(false, 0, 0)) + +struct _SingleProcLinalg end +struct _CuSolverMpLinalg end +struct _TiledCholesky end + +const _LINALG_CONJ_TRANSPOSE = Int32(2) + +_linalg_backend(op, a::NDArray) = _linalg_backend(op, size(a), _LINALG_RUNTIME[]) + +# Tuple length carries dimensionality in its type. Stacked systems never enter +# the MP selector; only their leading batch axes may be distributed. +_linalg_backend(::Val{:solve}, ::Tuple, ::_LinalgRuntime) = _SingleProcLinalg() + +function _linalg_backend(::Val{:solve}, shape::NTuple{2,Int}, rt::_LinalgRuntime) + use_mp = rt.mp_eligible && shape[1] >= MIN_SOLVE_MATRIX_SIZE + return use_mp ? _CuSolverMpLinalg() : _SingleProcLinalg() +end + +function _linalg_backend(::Val{:qr}, shape::NTuple{2,Int}, rt::_LinalgRuntime) + use_mp = rt.mp_eligible && prod(shape) >= MIN_QR_MATRIX_SIZE + return use_mp ? _CuSolverMpLinalg() : _SingleProcLinalg() +end + +_linalg_backend(::Val{:cholesky}, ::Tuple, ::_LinalgRuntime) = _SingleProcLinalg() + +function _linalg_backend(::Val{:cholesky}, shape::NTuple{2,Int}, rt::_LinalgRuntime) + rt.procs == 1 && return _SingleProcLinalg() + use_mp = rt.mp_eligible && shape[1] >= MIN_CHOLESKY_MATRIX_SIZE + return use_mp ? _CuSolverMpLinalg() : _TiledCholesky() +end + +function _linalg_scalars!(task, args...) + for arg in args + Legate.add_scalar(task, Legate.Scalar(arg)) + end + return nothing +end + +function _linalg_manual_task(id, lo::Tuple{Int,Int}, hi::Tuple{Int,Int}; throws=false) + task = create_linalg_task(id, Int64(lo[1]), Int64(lo[2]), Int64(hi[1]), Int64(hi[2])) + task_throws_exception(task, throws) + return task +end + +_submit_linalg_task(task) = Legate.submit_manual_task(Legate.get_runtime(), task) + +# Partitions own their store; submitted tasks retain their own references. +# Drop Julia's temporary owners promptly instead of waiting for GC. Recursion +# keeps earlier partitions protected if creating a later partition fails. +_with_linalg_partitions(f) = f() +function _with_linalg_partitions(f, spec::Tuple, specs::Tuple...) + store = nda_to_logical_store(first(spec)) + partition = try + Legate.partition_by_tiling(store, Base.tail(spec)...) + finally + finalize(store.handle) + end + try + return _with_linalg_partitions(specs...) do parts... + return f(partition, parts...) + end + finally + finalize(partition.handle) + end +end + +# Partition along rows, just as Python does. Every rank uses identical color +# spaces even when a reduced QR output has fewer rows than the input. +function _mp_row_partition(n::Int, gpus::Integer) + rows = cld(n, gpus) + return rows, (cld(n, rows), 1) +end + +_solve!(::_SingleProcLinalg, x, a, b) = solve_batched(a, b, x) + +function _solve!(::_CuSolverMpLinalg, x, a, b) + n, nrhs = size(a, 1), size(b, 2) + rows, colors = _mp_row_partition(n, _LINALG_RUNTIME[].gpus) + _with_linalg_partitions( + (a, (rows, n)), (b, (rows, nrhs)), (x, (rows, nrhs)) + ) do pa, pb, px + @task_scope "mp_solve" begin + task = _linalg_manual_task(MP_SOLVE, (0, 0), (colors[1] - 1, 0); throws=true) + Legate.add_input(task, pa) + Legate.add_input(task, pb) + Legate.add_output(task, px) + _linalg_scalars!(task, Int64(n), Int64(nrhs), Int64(MIN_SOLVE_TILE_SIZE)) + add_nccl_communicator(task) + _submit_linalg_task(task) + end + end + return x +end + +function _qr!(::_CuSolverMpLinalg, q, r, a) + m, n = size(a) + rows, colors = _mp_row_partition(m, _LINALG_RUNTIME[].gpus) + tiles = (rows, n) + _with_linalg_partitions((a, tiles), (q, tiles, colors), (r, tiles, colors)) do pa, pq, pr + @task_scope "mp_qr" begin + task = _linalg_manual_task(MP_QR, (0, 0), (colors[1] - 1, 0); throws=true) + Legate.add_input(task, pa) + Legate.add_output(task, pq) + Legate.add_output(task, pr) + _linalg_scalars!(task, Int64(m), Int64(n), Int64(QR_TILE_SIZE), Int64(QR_TILE_SIZE)) + add_nccl_communicator(task) + _submit_linalg_task(task) + end + end + return nothing +end + +_cholesky!(::_SingleProcLinalg, out, a) = potrf!(out, a; lower=true, zeroout=true) + +function _cholesky!(::_CuSolverMpLinalg, out, a) + @task_scope "mp_potrf" begin + rt = Legate.get_runtime() + task = Legate.create_auto_task(rt, get_lib(), MP_POTRF) + task_throws_exception(task, true) + ai = _add_task_array!(Legate.add_input, task, a) + oi = _add_task_array!(Legate.add_output, task, out) + Legate.add_constraint(task, Legate.align(oi, ai)) + _linalg_scalars!(task, Int64(size(a, 1)), Int64(MIN_CHOLESKY_TILE_SIZE)) + add_nccl_communicator(task) + Legate.submit_auto_task(rt, task) + _cholesky_tril!(out) + end + return out +end + +function _cholesky_tril!(out::NDArray) + rt = Legate.get_runtime() + task = Legate.create_auto_task(rt, get_lib(), TRILU) + _add_task_array!(Legate.add_output, task, out) + _add_task_array!(Legate.add_input, task, out) + # The third argument identifies Cholesky to the backend/mapper. + _linalg_scalars!(task, true, Int32(0), true) + Legate.submit_auto_task(rt, task) + return nothing +end + +function _cholesky_color_shape(n::Int, procs::Int) + (procs == 1 || n <= MIN_CHOLESKY_MATRIX_SIZE) && return (1, 1) + tiles = Int(procs) + while cld(n, tiles) > MIN_CHOLESKY_TILE_SIZE && + 2 * tiles <= procs * MAX_CHOLESKY_TILES_PER_PROC + tiles *= 2 + end + return (tiles, tiles) +end + +# Each task reads only tiles ready at this stage; Legate records the DAG from +# these inputs/outputs. No execution fence or host copy is needed between steps. +function _cholesky!(::_TiledCholesky, out, a) + n = size(a, 1) + initial = _cholesky_color_shape(n, _LINALG_RUNTIME[].procs) + tile = cld(n, initial[1]) + colors = cld(n, tile) + _with_linalg_partitions((a, (tile, tile)), (out, (tile, tile))) do pa, po + @task_scope "tiled_cholesky" begin + task = _linalg_manual_task(TRANSPOSE_COPY_2D, (0, 0), (colors - 1, colors - 1)) + Legate.add_output(task, po) + Legate.add_input(task, pa) + _submit_linalg_task(task) + for i in 0:(colors - 1) + _cholesky_potrf!(po, i) + _cholesky_trsm!(po, i, colors) + for k in (i + 1):(colors - 1) + _cholesky_syrk!(po, k, i) + _cholesky_gemm!(po, k, i, colors) + end + end + task = _linalg_manual_task(TRILU, (0, 0), (colors - 1, colors - 1)) + Legate.add_output(task, po) + Legate.add_input(task, po) + _linalg_scalars!(task, true, Int32(0), true) + _submit_linalg_task(task) + end + end + return out +end + +function _cholesky_potrf!(p, i) + task = _linalg_manual_task(POTRF, (i, i), (i, i); throws=true) + Legate.add_output(task, p) + Legate.add_input(task, p) + _linalg_scalars!(task, true, false) + return _submit_linalg_task(task) +end + +function _cholesky_trsm!(p, i, colors) + i + 1 >= colors && return nothing + task = _linalg_manual_task(TRSM, (i + 1, i), (colors - 1, i); throws=true) + Legate.add_output(task, p) + add_input_tile(task, p.handle, UInt64(i), UInt64(i)) + Legate.add_input(task, p) + # Right-side solve with the conjugate transpose of the lower factor. + _linalg_scalars!(task, false, true, _LINALG_CONJ_TRANSPOSE, false) + return _submit_linalg_task(task) +end + +function _cholesky_syrk!(p, k, i) + task = _linalg_manual_task(SYRK, (k, k), (k, k)) + Legate.add_output(task, p) + add_input_tile(task, p.handle, UInt64(k), UInt64(i)) + Legate.add_input(task, p) + return _submit_linalg_task(task) +end + +function _cholesky_gemm!(p, k, i, colors) + k + 1 >= colors && return nothing + task = _linalg_manual_task(GEMM, (k + 1, k), (colors - 1, k)) + Legate.add_output(task, p) + add_input_column(task, p.handle, Int32(i)) + add_input_tile(task, p.handle, UInt64(k), UInt64(i)) + Legate.add_input(task, p) + return _submit_linalg_task(task) +end diff --git a/src/ndarray/detail/fft.jl b/src/ndarray/detail/fft.jl new file mode 100644 index 000000000..25e0c4683 --- /dev/null +++ b/src/ndarray/detail/fft.jl @@ -0,0 +1,257 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +# cuFFT mixed-radix kernels only cover lengths whose prime factors are all +# <= 131. A larger factor forces Bluestein (slower, more scratch). +const CUFFT_MAX_EFFICIENT_PRIME = 131 +const _CUFFT_SMALL_PRIMES = ( + 2, + 3, + 5, + 7, + 11, + 13, + 17, + 19, + 23, + 29, + 31, + 37, + 41, + 43, + 47, + 53, + 59, + 61, + 67, + 71, + 73, + 79, + 83, + 89, + 97, + 101, + 103, + 107, + 109, + 113, + 127, + 131, +) + +function _has_large_prime_factor(n::Integer) + n < 2 && return false + m = Int(n) + for p in _CUFFT_SMALL_PRIMES + while m % p == 0 + m ÷= p + end + m == 1 && return false + end + return true +end + +function _unique_axes(axes0::NTuple{R,Int64}) where {R} + seen = zero(UInt64) + out = Vector{Int64}(undef, R) + n = 0 + for ax in axes0 + bit = one(UInt64) << ax + if seen & bit == 0 + seen |= bit + n += 1 + out[n] = ax + end + end + return resize!(out, n) +end + +function _axes_sorted(axes0::NTuple{R,Int64}) where {R} + for i in 2:R + axes0[i] < axes0[i - 1] && return false + end + return true +end + +function _operate_over_axes(axes0::NTuple{R,Int64}, ndim::Int) where {R} + unique_axes = _unique_axes(axes0) + return length(unique_axes) != R || R != ndim || !_axes_sorted(axes0) +end + +function _bluestein_mask( + axes0::NTuple{R,Int64}, in_size::NTuple{N,Int}, out_size::NTuple{N,Int} +) where {R,N} + mask = Int32(0) + slow = Tuple{Int,Int}[] + seen = zero(UInt64) + for ax in axes0 + bit = one(UInt64) << ax + seen & bit != 0 && continue + seen |= bit + jdim = Int(ax) + 1 + length_ax = max(in_size[jdim], out_size[jdim]) + if _has_large_prime_factor(length_ax) + mask |= Int32(1) << ax + push!(slow, (ax, length_ax)) + end + end + if !isempty(slow) + details = join(("axis $ax (length $len)" for (ax, len) in slow), ", ") + @warn "cuNumeric is computing an FFT over $details whose length has a prime factor > $CUFFT_MAX_EFFICIENT_PRIME, so cuFFT falls back to the Bluestein algorithm. You may notice significantly decreased performance and much higher GPU memory usage. Zero-padding the transformed axis to a length whose prime factors are all <= $CUFFT_MAX_EFFICIENT_PRIME (e.g. the next power of two) avoids this." + end + return mask +end + +function _assert_fft_gpu() + _has_gpu_target() && return nothing + return throw( + ErrorException( + "FFT requires a CUDA GPU; cupynumeric's FFT task has no CPU variant" + ), + ) +end + +_fft_kind(::Type{ComplexF32}) = Int32(cuNumeric.FFT_C2C) +_fft_kind(::Type{ComplexF64}) = Int32(cuNumeric.FFT_Z2Z) + +function _fft_dims(::NDArray{<:Any,N}) where {N} + N >= 1 || throw(ArgumentError("fft does not support 0-dimensional arrays")) + return ntuple(identity, Val(N)) +end + +function _fft_dims(A::NDArray, dim::Integer) + return _fft_dims(A, (Int(dim),)) +end + +function _fft_dims(A::NDArray{T,N}, dims) where {T,N} + N >= 1 || throw(ArgumentError("fft does not support 0-dimensional arrays")) + R = length(dims) + R >= 1 || throw(ArgumentError("fft dims must contain at least one dimension")) + region = ntuple(i -> Int(dims[i]), R) + seen = zero(UInt64) + for d in region + (1 <= d <= N) || throw(ArgumentError("fft dim $d is out of range for a $N-d array")) + bit = one(UInt64) << d + seen & bit != 0 && throw(ArgumentError("fft dims must be unique; got $region")) + seen |= bit + end + return region +end + +# Leading dimension is the batch; every remaining dim is transformed. +# Same CUPYNUMERIC_FFT task as `fft` — the batch axis is the one that may split. +function _fft_batch_dims(::NDArray{<:Any,N}) where {N} + N >= 2 || throw( + ArgumentError( + "batched_fft requires a leading batch dimension; got a $N-d array. Use fft for a single transform." + ), + ) + return ntuple(i -> i + 1, Val(N - 1)) +end + +function _ifft_scale(::Type{T}, sz::NTuple{N,Int}, dims::NTuple{R,Int}) where {T,N,R} + n = one(real(T)) + for d in dims + n *= real(T)(sz[d]) + end + return T(inv(n)) +end + +_fft_scope_name(direction::Int32) = + direction == Int32(cuNumeric.FFT_INVERSE) ? "ifft" : "fft" + +""" + fft_task!(out, inp, dims, direction; scale=false) + +Launch cupynumeric's `CUPYNUMERIC_FFT` auto task. `dims` are 1-based Julia +dimensions. `out` and `inp` must have the same shape and complex eltype; they +may be the same array (in-place C2C). When `scale` is true the inverse is +normalized in this same task scope (`out .*= 1/N`). +""" +function fft_task!( + out::NDArray{T,N}, + inp::NDArray{T,N}, + dims::NTuple{R,Int}, + direction::Int32; + scale::Bool=false, +) where {T<:SUPPORTED_COMPLEX_TYPES,N,R} + _assert_fft_gpu() + size(out) == size(inp) || + throw(DimensionMismatch("FFT output size $(size(out)) != input size $(size(inp))")) + + # cuPyNumeric maps an aliased input's store non-exactly, so once the task + # partitions across GPUs the output can land in a non-dense instance, which + # the GPU kernel rejects. Transform out of place and copy back instead. + if inp === out && _LINALG_RUNTIME[].gpus > 1 + tmp = similar(out) + try + fft_task!(tmp, inp, dims, direction; scale) + copyto!(out, tmp) + finally + destroy!(tmp) + end + return out + end + + axes0 = ntuple(i -> Int64(dims[i] - 1), Val(R)) + unique_axes = _unique_axes(axes0) + operate_over = _operate_over_axes(axes0, N) + # Warn on awkward lengths. cuFFT may take Bluestein internally; the + # 26.06 task has no bluestein_mask scalar, so we do not send one. + _bluestein_mask(axes0, size(inp), size(out)) + kind = _fft_kind(T) + + # Root the Julia wrappers through submission; pending tasks retain the stores. + # Internal FFT execution stays asynchronous and does not require extra copies. + GC.@preserve inp out begin + @task_scope _fft_scope_name(direction) begin + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + task = Legate.create_auto_task(rt, lib, cuNumeric.FFT) + cuNumeric.task_throws_exception(task, true) + + l_out = nda_to_logical_array(out) + l_in = inp === out ? l_out : nda_to_logical_array(inp) + + out_var = Legate.add_output(task, l_out) + in_var = Legate.add_input(task, l_in) + + # 26.06 fft_template.inl: kind, direction, operate_over_axes, then axes. + Legate.add_scalar(task, Legate.Scalar(kind)) + Legate.add_scalar(task, Legate.Scalar(direction)) + Legate.add_scalar(task, Legate.Scalar(operate_over)) + for ax in axes0 + Legate.add_scalar(task, Legate.Scalar(ax)) + end + + Legate.add_constraint(task, Legate.align(out_var, in_var)) + if N > length(unique_axes) + Legate.add_broadcast(task, l_in, CxxWrap.StdVector(UInt32.(unique_axes))) + else + Legate.add_broadcast(task, l_in) + end + + Legate.submit_auto_task(rt, task) + if scale + out .*= _ifft_scale(T, size(out), dims) + end + end + end + return out +end diff --git a/src/ndarray/detail/linalg.jl b/src/ndarray/detail/linalg.jl index 0362a84f5..aa65b9d34 100644 --- a/src/ndarray/detail/linalg.jl +++ b/src/ndarray/detail/linalg.jl @@ -1,21 +1,12 @@ -function choose_nd_color_shape(shape::NTuple{N,Int}) where {N} - color_shape = Base.ones(Int, N) - if N > 2 - color_shape[1] = Legate.num_procs() - done = false - while !done && color_shape[1] % 2 == 0 - weight_per_dim = [shape[i] / color_shape[i] for i in 1:(N - 2)] - max_weight, idx = findmax(weight_per_dim) - if weight_per_dim[idx] > 2 * weight_per_dim[1] - color_shape[1] ÷= 2 - color_shape[idx] *= 2 - else - done = true - end - end - end - return Tuple(color_shape) -end +# Only a single matrix or one batch axis is supported. Keep matrix axes whole. +choose_nd_color_shape(::NTuple{2,Int}) = (1, 1) +choose_nd_color_shape(::NTuple{3,Int}) = (_LINALG_RUNTIME[].procs, 1, 1) + +# One batch dimension is the ceiling for every batched op: +# - the POTRF task body is only instantiated for 2 <= DIM < 4 +# - Legate.jl's `domain_from_shape` builds launch domains for at most three +# dimensions and yields an empty domain past that +const MAX_BATCHED_DIM = 3 function prepare_manual_task_for_batched_matrices(full_shape::NTuple{N,Int}) where {N} initial_color_shape = choose_nd_color_shape(full_shape) @@ -32,25 +23,22 @@ function solve_batched(a::NDArray{T,N}, b::NDArray, x::NDArray) where {T,N} tilesize_a, color_shape = prepare_manual_task_for_batched_matrices(full_shape) tilesize_b = (tilesize_a[1:(end - 1)]..., nrhs) - store_a = nda_to_logical_store(a) - store_b = nda_to_logical_store(b) - store_x = nda_to_logical_store(x) - - tiled_a = Legate.partition_by_tiling(store_a, collect(tilesize_a)) - tiled_b = Legate.partition_by_tiling(store_b, collect(tilesize_b)) - tiled_x = Legate.partition_by_tiling(store_x, collect(tilesize_b)) - - @task_scope "solve" begin - rt = Legate.get_runtime() - domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) - lib = cuNumeric.get_lib() - task = Legate.create_manual_task(rt, lib, cuNumeric.SOLVE, domain) - - Legate.add_input(task, tiled_a) - Legate.add_input(task, tiled_b) - Legate.add_output(task, tiled_x) - - Legate.submit_manual_task(rt, task) + _with_linalg_partitions( + (a, tilesize_a), (b, tilesize_b), (x, tilesize_b) + ) do tiled_a, tiled_b, tiled_x + @task_scope "solve" begin + rt = Legate.get_runtime() + domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) + lib = cuNumeric.get_lib() + task = Legate.create_manual_task(rt, lib, cuNumeric.SOLVE, domain) + cuNumeric.task_throws_exception(task, true) + + Legate.add_input(task, tiled_a) + Legate.add_input(task, tiled_b) + Legate.add_output(task, tiled_x) + + Legate.submit_manual_task(rt, task) + end end end @@ -61,17 +49,47 @@ const _SOLVE_ACCEPTED = Union{SUPPORTED_SOLVE_TYPES,_SOLVE_PROMOTABLE} _solve_eltype(::Type{T}) where {T<:_SOLVE_PROMOTABLE} = Float64 _solve_eltype(::Type{T}) where {T<:SUPPORTED_SOLVE_TYPES} = T -# `a` must be at least 2D, `b` at least 1D. -function _solve_check_a_dims(a::NDArray{<:Any,0}, b::NDArray) - throw(ArgumentError("0-dimensional array given. Array must be at least two-dimensional")) +# `a` must be at least 2D, `b` at least 1D. `solve` takes only the 2D case; +# stacked systems go through `batched_solve`. +function _solve_check_a_dims_2d(a::NDArray{<:Any,0}, b::NDArray) + return throw(ArgumentError("0-dimensional array given. Array must be two-dimensional")) +end +function _solve_check_a_dims_2d(a::NDArray{<:Any,1}, b::NDArray) + return throw(ArgumentError("1-dimensional array given. Array must be two-dimensional")) +end +_solve_check_a_dims_2d(a::NDArray{<:Any,2}, b::NDArray) = _solve_check_b_dims(a, b) +function _solve_check_a_dims_2d(a::NDArray{<:Any,N}, b::NDArray) where {N} + return throw( + ArgumentError( + "$N-dimensional array given. Use `cuNumeric.batched_solve` for " * + "stacked systems of shape (...,m,m)", + ), + ) end -function _solve_check_a_dims(a::NDArray{<:Any,1}, b::NDArray) - throw(ArgumentError("1-dimensional array given. Array must be at least two-dimensional")) + +function _solve_check_a_dims_batched(a::NDArray{<:Any,N}, b::NDArray) where {N} + _assert_batched_dims(:batched_solve, N) + return _solve_check_b_dims(a, b) +end + +function _assert_batched_dims(f, N::Integer) + N < 3 && throw( + ArgumentError( + "$N-dimensional array given. `$f` requires shape (b,m,m); use the " * + "matching `LinearAlgebra` or `cuNumeric` function for a single matrix", + ), + ) + N > MAX_BATCHED_DIM && throw( + ArgumentError( + "$N-dimensional array given. `$f` supports at most one batch " * + "dimension, i.e. shape (b,m,m)", + ), + ) + return nothing end -_solve_check_a_dims(a::NDArray, b::NDArray) = _solve_check_b_dims(a, b) function _solve_check_b_dims(a::NDArray, b::NDArray{<:Any,0}) - throw(ArgumentError("0-dimensional array given. Array must be at least one-dimensional")) + return throw(ArgumentError("0-dimensional array given. Array must be at least one-dimensional")) end _solve_check_b_dims(a::NDArray, b::NDArray) = _solve(a, b) @@ -94,15 +112,102 @@ function _solve(a::NDArray{T,N}, b::NDArray{S,N}) where {T,S,N} " (size $(size(b)[end-1]) is different from $(size(a)[end]))", ), ) - prod(size(a)) == 0 || prod(size(b)) == 0 && return zeros(T, size(b)...) - x = zeros(T, size(b)...) - solve_batched(a, b, x) + size(a)[1:(end - 2)] == size(b)[1:(end - 2)] || + throw(ArgumentError("Batched matrices must have matching batch dimensions")) + x = NDArray{T}(undef, size(b)) + isempty(x) && return x + _solve!(_linalg_backend(Val(:solve), a), x, a, b) return x end # Mismatched batch dimensions function _solve(a::NDArray{T,N}, b::NDArray{S,M}) where {T,N,S,M} - throw(ArgumentError("Batched matrices require signature (...,m,m),(...,m,n)->(...,m,n)")) + return throw(ArgumentError("Batched matrices require signature (...,m,m),(...,m,n)->(...,m,n)")) +end + +# cholesky + +""" + potrf!(out, a; lower, zeroout) + +Run the cupynumeric `POTRF` task, writing the Cholesky factor of `a` into `out`. + +The task is batched over the leading dimension: for a `(b, m, m)` input it +factors each `m × m` block independently. `lower` selects which triangle holds +the factor, `zeroout` zeros the opposite triangle in-task. +""" +function potrf!(out::NDArray{T,N}, a::NDArray{T,N}; lower::Bool, zeroout::Bool) where {T,N} + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + + @task_scope "potrf" begin + task = Legate.create_auto_task(rt, lib, cuNumeric.POTRF) + cuNumeric.task_throws_exception(task, true) + + in_var = _add_task_array!(Legate.add_input, task, a) + out_var = _add_task_array!(Legate.add_output, task, out) + + Legate.add_scalar(task, Legate.Scalar(lower)) + Legate.add_scalar(task, Legate.Scalar(zeroout)) + + # Each matrix must live on one processor; only the batch axes may split. + Legate.add_constraint( + task, Legate.broadcast(in_var, CxxWrap.StdVector(UInt32[N - 2, N - 1])) + ) + Legate.add_constraint(task, Legate.align(out_var, in_var)) + + Legate.submit_auto_task(rt, task) + end + return out +end + +# eigen + +""" + geev!(a, ew, ev) + +Run the cupynumeric `GEEV` task, writing eigenvalues into `ew` and, unless `ev` +is `nothing`, right eigenvectors into `ev` as columns. + +`ew` has one fewer dimension than `a`, so the launch domain maps onto it through +a projection that drops the trailing axis. The task keys eigenvector computation +off the number of registered outputs, so `ev` must not be added when only +eigenvalues are wanted. +""" +function geev!(a::NDArray{T,N}, ew::NDArray, ev::Union{NDArray,Nothing}) where {T,N} + full_shape = size(a) + tilesize, color_shape = prepare_manual_task_for_batched_matrices(full_shape) + + tiled_a = Legate.partition_by_tiling(nda_to_logical_store(a), collect(tilesize)) + tiled_ew = Legate.partition_by_tiling( + nda_to_logical_store(ew), collect(tilesize[1:(end - 1)]) + ) + tiled_ev = if ev === nothing + nothing + else + Legate.partition_by_tiling(nda_to_logical_store(ev), collect(tilesize)) + end + + # 0-based source dimensions of the launch domain. + proj = CxxWrap.StdVector(Int32.(0:(N - 1))) + proj_ew = CxxWrap.StdVector(Int32.(0:(N - 2))) + + @task_scope "geev" begin + rt = Legate.get_runtime() + domain = Legate.domain_from_shape(Legate.Shape(Legate.to_cxx_vector(color_shape))) + lib = cuNumeric.get_lib() + task = Legate.create_manual_task(rt, lib, cuNumeric.GEEV, domain) + cuNumeric.task_throws_exception(task, true) + + cuNumeric.add_input_proj(task, tiled_a.handle, proj) + cuNumeric.add_output_proj(task, tiled_ew.handle, proj_ew) + if tiled_ev !== nothing + cuNumeric.add_output_proj(task, tiled_ev.handle, proj) + end + + Legate.submit_manual_task(rt, task) + end + return nothing end function svd_single(a::NDArray{T,N}, u::NDArray, s::NDArray, vh::NDArray) where {T,N} @@ -130,17 +235,22 @@ end function _svd(a::NDArray{T,2}, full_matrices::Bool) where {T} m, n = size(a) + # cupynumeric's SVD task is tall-skinny only (`assert(m >= n)` in the + # kernel). Catch it here so a wide matrix is a Julia error, not a + # process-killing C++ assert / abort. + m >= n || throw( + ArgumentError( + "svd only supports m >= n (got $(m)×$(n)); the backend does not factor wide matrices" + ), + ) k = min(m, n) S = real(T) - # cuSolver requires full square buffers regardless of full_matrices - u_buf = zeros(T, m, m) - s = zeros(S, k) - vh_buf = zeros(T, n, n) + + u_buf = NDArray{T}(undef, m, full_matrices ? m : k) + s = NDArray{S}(undef, k) + vh_buf = NDArray{T}(undef, full_matrices ? n : k, n) svd_single(a, u_buf, s, vh_buf) - # Backend factors are logically ordered; only thin strided views need materialization. - u = full_matrices ? u_buf : copy(u_buf[:, 1:k]) - vh = full_matrices ? vh_buf : copy(vh_buf[1:k, :]) - return u, s, vh + return u_buf, s, vh_buf end # svd runs on float/complex only — no integer backend @@ -149,54 +259,34 @@ const _SVD_ACCEPTED = Union{SUPPORTED_SVD_TYPES,_SVD_PROMOTABLE} _svd_eltype(::Type{T}) where {T<:_SVD_PROMOTABLE} = Float64 _svd_eltype(::Type{T}) where {T<:SUPPORTED_SVD_TYPES} = T -function _svd_check_dims(a::NDArray{<:Any,0}, full_matrices::Bool) - throw(ArgumentError("0-dimensional array given. Array must be at least two-dimensional")) -end - -function _svd_check_dims(a::NDArray{<:Any,1}, full_matrices::Bool) - throw(ArgumentError("1-dimensional array given. Array must be at least two-dimensional")) -end - -function _svd_check_dims(a::NDArray{<:Any,2}, full_matrices::Bool) - return _svd(a, full_matrices) -end - -function _svd_check_dims(a::NDArray, full_matrices::Bool) - throw(ArgumentError("cuNumeric does not yet support stacked 2d arrays")) -end - # qr -function qr_single(a::NDArray{T,N}, q::NDArray, r::NDArray) where {T,N} +function _qr!(::_SingleProcLinalg, q, r, a) rt = Legate.get_runtime() lib = cuNumeric.get_lib() task = Legate.create_auto_task(rt, lib, cuNumeric.CQR) + cuNumeric.task_throws_exception(task, true) - l_a = nda_to_logical_array(a) - l_q = nda_to_logical_array(q) - l_r = nda_to_logical_array(r) - - Legate.add_input(task, l_a) - Legate.add_output(task, l_q) - Legate.add_output(task, l_r) - - Legate.add_broadcast(task, l_a) - Legate.add_broadcast(task, l_q) - Legate.add_broadcast(task, l_r) + ai = _add_task_array!(Legate.add_input, task, a) + qi = _add_task_array!(Legate.add_output, task, q) + ri = _add_task_array!(Legate.add_output, task, r) + for variable in (ai, qi, ri) + Legate.add_constraint(task, Legate.broadcast(variable)) + end - return Legate.submit_auto_task(rt, task) + Legate.submit_auto_task(rt, task) + return nothing end function _qr(a::NDArray{T,2}) where {T} m, n = size(a) k = min(m, n) - # cuSolver requires full square buffers regardless of output shape - q_buf = zeros(T, m, m) - r_buf = zeros(T, n, n) - qr_single(a, q_buf, r_buf) - # Host conversion assumes contiguous storage, so materialize the economy slices. - q = copy(q_buf[:, 1:k]) - r = copy(r_buf[1:k, :]) + # CQR writes dense column-major economy factors with leading dimensions + # m for Q and k for R. Square buffers give R the wrong stride when m < n. + q = NDArray{T}(undef, m, k) + r = NDArray{T}(undef, k, n) + k == 0 && return q, r + _qr!(_linalg_backend(Val(:qr), a), q, r, a) return q, r end @@ -205,18 +295,87 @@ const _QR_ACCEPTED = Union{SUPPORTED_QR_TYPES,_QR_PROMOTABLE} _qr_eltype(::Type{T}) where {T<:_QR_PROMOTABLE} = Float64 _qr_eltype(::Type{T}) where {T<:SUPPORTED_QR_TYPES} = T -function _qr_check_dims(a::NDArray{<:Any,0}) - throw(ArgumentError("0-dimensional array given. Array must be at least two-dimensional")) +# cholesky/eigen guards. Both run in floating point only, so int/bool inputs +# promote to Float64 the same way solve/svd/qr do. + +const _CHOLESKY_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool} +const _CHOLESKY_ACCEPTED = Union{SUPPORTED_CHOLESKY_TYPES,_CHOLESKY_PROMOTABLE} +_cholesky_eltype(::Type{T}) where {T<:_CHOLESKY_PROMOTABLE} = Float64 +_cholesky_eltype(::Type{T}) where {T<:SUPPORTED_CHOLESKY_TYPES} = T + +const _EIG_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool} +const _EIG_ACCEPTED = Union{SUPPORTED_EIG_TYPES,_EIG_PROMOTABLE} +_eig_eltype(::Type{T}) where {T<:_EIG_PROMOTABLE} = Float64 +_eig_eltype(::Type{T}) where {T<:SUPPORTED_EIG_TYPES} = T + +# GEEV always produces complex eigenvalues and eigenvectors, even for real input. +_eig_complex_eltype(::Type{Float32}) = ComplexF32 +_eig_complex_eltype(::Type{ComplexF32}) = ComplexF32 +_eig_complex_eltype(::Type{Float64}) = ComplexF64 +_eig_complex_eltype(::Type{ComplexF64}) = ComplexF64 + +function _check_square_matrices(f, a::NDArray{<:Any,N}) where {N} + N < 2 && throw( + ArgumentError( + "$N-dimensional array given. Array must be at least two-dimensional" + ), + ) + sz = size(a) + sz[end - 1] != sz[end] && + throw(ArgumentError("Last 2 dimensions of the array must be square in $f")) + sz[end] == 0 && throw(ArgumentError("Input shape dimension 0 not allowed in $f")) + return nothing end -function _qr_check_dims(a::NDArray{<:Any,1}) - throw(ArgumentError("1-dimensional array given. Array must be at least two-dimensional")) +""" + _cholesky(a) + +Lower Cholesky factor of each trailing square block of `a`, with the upper +triangle zeroed. Only the lower triangle of the input is read; the input is +assumed Hermitian without being checked, matching cupynumeric. +""" +function _cholesky(a::NDArray{T,N}) where {T,N} + _check_square_matrices(:cholesky, a) + out = NDArray{T}(undef, size(a)) + _cholesky!(_linalg_backend(Val(:cholesky), a), out, a) + return out end -function _qr_check_dims(a::NDArray{<:Any,2}) - return _qr(a) +""" + _eig(a) + +Eigenvalues and right eigenvectors of each trailing square block of `a`. Both +are always complex. +""" +function _eig(a::NDArray{T,N}) where {T,N} + ew = _alloc_eigenvalues(a) + ev = NDArray{_eig_complex_eltype(T)}(undef, size(a)) + geev!(a, ew, ev) + return ew, ev end -function _qr_check_dims(a::NDArray) - throw(ArgumentError("cuNumeric does not yet support stacked 2d arrays")) +""" + _eigvals(a) + +Eigenvalues only. Registering just the one output is what tells the backend task +to skip the eigenvector computation. +""" +function _eigvals(a::NDArray) + ew = _alloc_eigenvalues(a) + geev!(a, ew, nothing) + return ew +end + +function _alloc_eigenvalues(a::NDArray{T,N}) where {T,N} + _check_square_matrices(:eigen, a) + _assert_geev_available() + return NDArray{_eig_complex_eltype(T)}(undef, size(a)[1:(end - 1)]) +end + +function _assert_geev_available() + (_has_gpu_target() && !cuNumeric.cusolver_has_geev()) && error( + "eigen requires cusolverDnXgeev, which the installed cuSolver does not " * + "provide. Upgrade CUDA (12.6.2 or newer) or run without GPUs.", + ) + return nothing end diff --git a/src/ndarray/detail/ndarray.jl b/src/ndarray/detail/ndarray.jl index 5d4d621c0..a449d9ea7 100644 --- a/src/ndarray/detail/ndarray.jl +++ b/src/ndarray/detail/ndarray.jl @@ -41,23 +41,46 @@ end get_n_dim(ptr::NDArray_t) = Int(ccall((:nda_array_dim, libnda), Int32, (NDArray_t,), ptr)) -abstract type AbstractNDArray{T<:SUPPORTED_TYPES,N} end +abstract type AbstractNDArray{T,N} <: AbstractArray{T,N} end + +@inline _struct_storage_type(::Type{T}) where {T} = + isbitstype(T) && !(T <: SUPPORTED_TYPES) && !isprimitivetype(T) && + fieldcount(T) > 0 && + all(F -> F <: SUPPORTED_ARRAY_TYPES, fieldtypes(T)) + +# cuPyNumeric has no record-typed operations: struct stores are packed, +# unpacked, copied and compared by the fused GPU broadcast kernel. +# TODO CPU variants for struct pack/unpack so host transfer works without a GPU. +@inline _struct_kernel_available() = FUSE_BROADCAST_EXPRS && _has_gpu_target() + +@inline function _assert_struct_kernel(op, ::Type{T}) where {T} + _struct_kernel_available() || throw( + ArgumentError( + "$(op) of NDArrays with struct element type $(T) requires GPU broadcast " * + "fusion (fusion enabled: $(FUSE_BROADCAST_EXPRS), GPU available: $(_has_gpu_target()))", + ), + ) + return nothing +end + +# Runtime padding uses an abstract field to break the recursive storage definition. +abstract type AbstractPaddedStorage{T,N} end @doc""" The NDArray type represents a multi-dimensional array in cuNumeric. It is a wrapper around a Legate array and provides various methods for array manipulation and operations. Finalizer calls `nda_destroy_array` to clean up the underlying Legate array when the NDArray is garbage collected. """ -mutable struct NDArray{T,N,PADDED,P} <: AbstractNDArray{T,N} +mutable struct NDArray{T,N,P} <: AbstractNDArray{T,N} ptr::NDArray_t nbytes::Int64 - padding::Union{Nothing,NTuple{N,Int}} + padding::Union{Nothing,AbstractPaddedStorage{T,N}} parent::P function NDArray(ptr::NDArray_t, ::Type{T}, ::Val{N}) where {T,N} nbytes = cuNumeric.nda_nbytes(ptr) cuNumeric.register_alloc!(nbytes) - handle = new{T,N,false,Nothing}(ptr, nbytes, nothing, nothing) + handle = new{T,N,Nothing}(ptr, nbytes, nothing, nothing) finalizer(_finalize_ndarray!, handle) return handle end @@ -66,26 +89,53 @@ mutable struct NDArray{T,N,PADDED,P} <: AbstractNDArray{T,N} function NDArray(ptr::NDArray_t, ::Type{T}, ::Val{N}, parent::P) where {T,N,P} nbytes = cuNumeric.nda_nbytes(ptr) cuNumeric.register_alloc!(nbytes) - handle = new{T,N,false,P}(ptr, nbytes, nothing, parent) + handle = new{T,N,P}(ptr, nbytes, nothing, parent) finalizer(_finalize_ndarray!, handle) return handle end end +struct PaddedStorage{T,N} <: AbstractPaddedStorage{T,N} + backing::NDArray{T,N,Nothing} + staging::Union{Nothing,NDArray{T,N,NDArray{T,N,Nothing}}} + shape::NTuple{N,Int} +end + +# Narrow the abstract field to its concrete storage type. +@inline _padding(arr::NDArray{T,N}) where {T,N} = + arr.padding::Union{Nothing,PaddedStorage{T,N}} + +function _finalize_padded_storage!(storage::PaddedStorage) + !isnothing(storage.staging) && finalize(storage.staging) + finalize(storage.backing) + return nothing +end + +function _destroy_padded_storage!(storage::PaddedStorage) + !isnothing(storage.staging) && destroy!(storage.staging) + destroy!(storage.backing) + return nothing +end + # May run off the launch thread, so defer the Legate free to drain_pending_frees!. # Accounting is atomic and safe to do here immediately. function _finalize_ndarray!(arr::NDArray) ptr = arr.ptr - ptr == C_NULL && return nothing arr.ptr = Ptr{Cvoid}(0) nbytes = arr.nbytes arr.nbytes = 0 - nbytes > 0 && register_free!(nbytes) - _enqueue_free!(ptr) + padding = _padding(arr) + arr.padding = nothing + + if ptr != C_NULL + nbytes > 0 && register_free!(nbytes) + _enqueue_free!(ptr) + end + !isnothing(padding) && _finalize_padded_storage!(padding) return nothing end -@inline _is_ndarray_slice(arr::NDArray) = arr.parent isa NDArray +@inline _is_ndarray_slice(arr::NDArray) = arr.parent isa NDArray || !isnothing(_padding(arr)) """ destroy!(arr::NDArray) @@ -102,6 +152,9 @@ function destroy!(arr::NDArray) arr.nbytes = 0 nbytes > 0 && register_free!(nbytes) end + padding = _padding(arr) + arr.padding = nothing + !isnothing(padding) && _destroy_padded_storage!(padding) return arr end @@ -130,6 +183,44 @@ _scope_op(kind, op_code) = string(kind, "#", Int32(op_code)) NDArray(value::T) where {T<:SUPPORTED_TYPES} = nda_full_array((), value) # construction +# Internal outputs only: callers must overwrite every element before any read. +function nda_empty_array(dims::Dims{N}, ::Type{T}) where {T,N} + shape = collect(UInt64, dims) + if _struct_storage_type(T) + fields = fieldtypes(T) + codes = Int32[Int32(Legate.code(Legate.to_legate_type(F))) for F in fields] + offsets = UInt32[UInt32(fieldoffset(T, i)) for i in eachindex(fields)] + layout_matches = ccall((:nda_struct_layout_matches, libnda), Bool, + (Int32, Ptr{Int32}, UInt32, Ptr{UInt32}), + Int32(length(fields)), codes, UInt32(sizeof(T)), offsets) + layout_matches || throw( + ArgumentError( + "Legate cannot store $T: its field offsets or size differ from Julia's layout" + ), + ) + ptr = @task_scope "empty_struct" begin + ccall((:nda_empty_struct_array, libnda), NDArray_t, + (Int32, Ptr{UInt64}, Int32, Ptr{Int32}, UInt32, Ptr{UInt32}), + Int32(N), shape, Int32(length(fields)), codes, UInt32(sizeof(T)), offsets) + end + return NDArray(ptr, T, Val(N)) + end + if isbitstype(T) && !isprimitivetype(T) && !(T <: SUPPORTED_TYPES) + throw( + ArgumentError( + "Unsupported isbits element type $T: struct fields must be supported scalar types" + ), + ) + end + legate_type = Legate.to_legate_type(T) + ptr = @task_scope "empty" begin + ccall((:nda_empty_array, libnda), + NDArray_t, (Int32, Ptr{UInt64}, Legate.LegateTypeAllocated), + Int32(N), shape, legate_type) + end + return NDArray(ptr, T, Val(N)) +end + function nda_zeros_array(dims::Dims{N}, ::Type{T}) where {T,N} shape = collect(UInt64, dims) legate_type = Legate.to_legate_type(T) @@ -155,22 +246,79 @@ function nda_full_array(dims::Dims{N}, value::T) where {T,N} return NDArray(ptr, T, Val(N)) end -function nda_random(arr::NDArray, gen_code) - @task_scope "rand!" begin - ccall((:nda_random, libnda), - Cvoid, (NDArray_t, Int32), - arr.ptr, Int32(gen_code)) +# Legacy Float64-only CUPYNUMERIC_RAND wrappers; unused after BitGenerator. +# function nda_random(arr::NDArray, gen_code) +# @task_scope "rand!" begin +# ccall((:nda_random, libnda), +# Cvoid, (NDArray_t, Int32), +# arr.ptr, Int32(gen_code)) +# end +# end +# +# function nda_random_array(dims::Dims{N}) where {N} +# shape = collect(UInt64, dims) +# ptr = @task_scope "rand" begin +# ccall((:nda_random_array, libnda), +# NDArray_t, (Int32, Ptr{UInt64}), +# Int32(N), shape) +# end +# return NDArray(ptr, Float64, Val(N)) #* T is always Float64 cause of cupynumeric +# end + +# Pack an SVector as a Legate fixed-array scalar. Empty vectors pass a null ptr. +function _add_vector_scalar!(add!, task, ::SVector{0,T}) where {T} + add!(task, Ptr{T}(C_NULL), Int32(0)) + return nothing +end +function _add_vector_scalar!(add!, task, v::SVector{N,T}) where {N,T} + ref = Ref(v) + GC.@preserve ref begin + add!(task, Ptr{T}(Base.unsafe_convert(Ptr{SVector{N,T}}, ref)), Int32(N)) end + return nothing end -function nda_random_array(dims::Dims{N}) where {N} - shape = collect(UInt64, dims) - ptr = @task_scope "rand" begin - ccall((:nda_random_array, libnda), - NDArray_t, (Int32, Ptr{UInt64}), - Int32(N), shape) +# Match cupynumeric/_thunk/deferred.py::bitgenerator_distribution via Julia +# Legate tasking. Vector scalars still go through tiny C++ helpers because +# Legate.jl Scalar has no std::vector constructors. +function nda_bitgenerator_distribution!( + arr::NDArray, + handle::Int32, + generator_type::UInt32, + seed::UInt64, + flags::UInt32, + distribution::UInt32, + strides::SVector{N,Int64}, + intparams::SVector{NI,Int64}, + floatparams::SVector{NF,Float32}, + doubleparams::SVector{ND,Float64}, +) where {N,NI,NF,ND} + isempty(arr) && return arr + + @task_scope "bitgenerator" begin + rt = Legate.get_runtime() + lib = cuNumeric.get_lib() + task = Legate.create_auto_task(rt, lib, cuNumeric.BITGENERATOR) + + st = cuNumeric.get_store(arr) + Legate.add_output(task, st) + finalize(st) + + Legate.add_scalar(task, Legate.Scalar(Int32(cuNumeric.BITGENOP_DISTRIBUTION))) + Legate.add_scalar(task, Legate.Scalar(handle)) + Legate.add_scalar(task, Legate.Scalar(generator_type)) + Legate.add_scalar(task, Legate.Scalar(seed)) + Legate.add_scalar(task, Legate.Scalar(flags)) + Legate.add_scalar(task, Legate.Scalar(distribution)) + + _add_vector_scalar!(cuNumeric.add_vector_scalar_i64, task, strides) + _add_vector_scalar!(cuNumeric.add_vector_scalar_i64, task, intparams) + _add_vector_scalar!(cuNumeric.add_vector_scalar_f32, task, floatparams) + _add_vector_scalar!(cuNumeric.add_vector_scalar_f64, task, doubleparams) + + Legate.submit_auto_task(rt, task) end - return NDArray(ptr, Float64, Val(N)) #* T is always Float64 cause of cupynumeric + return arr end function nda_get_slice(arr::NDArray{T,N}, slices::Vector{Slice}) where {T,N} @@ -184,11 +332,16 @@ function nda_get_slice(arr::NDArray{T,N}, slices::Vector{Slice}) where {T,N} return NDArray(ptr, T, Val(N), arr) end +@inline nda_overlaps(lhs::NDArray, rhs::NDArray) = + ccall( + (:nda_overlaps, libnda), Cuchar, (NDArray_t, NDArray_t), lhs.ptr, rhs.ptr + ) != 0 + # queries nda_array_dim(arr::NDArray) = ccall((:nda_array_dim, libnda), Int32, (NDArray_t,), arr.ptr) nda_array_size(arr::NDArray) = ccall((:nda_array_size, libnda), - Int32, (NDArray_t,), arr.ptr) + UInt64, (NDArray_t,), arr.ptr) # C API returns uint64_t, not an axis/count int32_t. function nda_array_type_code(arr::NDArray) return ccall((:nda_array_type_code, libnda), Int32, (NDArray_t,), arr.ptr) @@ -236,7 +389,47 @@ function nda_fill_array(arr::NDArray{T}, value::T) where {T} return nothing end +# Struct stores are filled from the value's bytes; Legate has their layout. +# TODO fill!, fill, and struct setindex! call Legate's issue_fill directly. Move +# them onto the fused broadcast kernel, as struct copies are, once runtime +# scalar arguments keep their own types instead of promoting to a common one. +function nda_fill_struct_array(arr::NDArray{T}, value::T) where {T} + val = Ref(value) + GC.@preserve val begin + @task_scope "fill!" begin + ccall((:nda_fill_struct_array, libnda), + Cvoid, (NDArray_t, Ptr{Cvoid}, UInt64), + arr.ptr, Base.unsafe_convert(Ptr{T}, val), UInt64(sizeof(T))) + end + end + return nothing +end + +# cuPyNumeric has no record kernels, so struct copies run the fused broadcast +# kernel, which already packs struct stores, slices, and views. +function _nda_assign_struct(arr::NDArray{T}, other::NDArray{T}) where {T} + size(arr) == size(other) || throw( + DimensionMismatch("cannot copy an array of size $(size(other)) into size $(size(arr))") + ) + isempty(arr) && return nothing + _assert_struct_kernel("Copying", T) + if ndims(arr) == 0 + # The fused kernel needs a rank; a 0-d reshape views the same element. + dest, src = nda_reshape_array(arr, (1,)), nda_reshape_array(other, (1,)) + try + _nda_assign_struct(dest, src) + finally + destroy!(dest) + destroy!(src) + end + else + arr .= StructIdentity().(other) + end + return nothing +end + function nda_assign(arr::NDArray{T}, other::NDArray{T}) where {T} + _struct_storage_type(T) && return _nda_assign_struct(arr, other) @task_scope "copyto!" begin ccall((:nda_assign, libnda), Cvoid, (NDArray_t, NDArray_t), @@ -245,6 +438,11 @@ function nda_assign(arr::NDArray{T}, other::NDArray{T}) where {T} end function nda_copy(arr::NDArray{T,N}) where {T,N} + if _struct_storage_type(T) + out = nda_empty_array(size(arr), T) + _nda_assign_struct(out, arr) + return out + end ptr = @task_scope "copy" begin ccall((:nda_copy, libnda), NDArray_t, (NDArray_t,), @@ -280,6 +478,17 @@ function nda_binary_op!(out::NDArray, op_code::BinaryOpCode, rhs1::NDArray, rhs2 return out end +function nda_binary_reduction!( + out::NDArray, op_code::BinaryOpCode, rhs1::NDArray, rhs2::NDArray +) + @task_scope _scope_op("binary_red", op_code) begin + ccall((:nda_binary_reduction, libnda), + Cvoid, (NDArray_t, BinaryOpCode, NDArray_t, NDArray_t), + out.ptr, op_code, rhs1.ptr, rhs2.ptr) + end + return out +end + function nda_unary_op!(out::NDArray, op_code::UnaryOpCode, input::NDArray) @task_scope _scope_op("unary", op_code) begin ccall((:nda_unary_op, libnda), @@ -340,6 +549,42 @@ function nda_unique(arr::NDArray{T}) where {T} return NDArray(ptr, T, Val(1)) end +function nda_sort(arr::NDArray{T,N}, axis::Int32, stable::Bool) where {T,N} + ptr = @task_scope "sort" begin + ccall((:nda_sort, libnda), + NDArray_t, (NDArray_t, Int32, Bool), + arr.ptr, axis, stable) + end + return NDArray(ptr, T, Val(N)) +end + +function nda_sort_inplace(arr::NDArray, axis::Int32, stable::Bool) + @task_scope "sort!" begin + ccall((:nda_sort_inplace, libnda), + Cvoid, (NDArray_t, Int32, Bool), + arr.ptr, axis, stable) + end + return arr +end + +function nda_argsort(arr::NDArray{<:Any,N}, axis::Int32, stable::Bool) where {N} + ptr = @task_scope "argsort" begin + ccall((:nda_argsort, libnda), + NDArray_t, (NDArray_t, Int32, Bool), + arr.ptr, axis, stable) + end + return NDArray(ptr, Int64, Val(N)) +end + +function nda_searchsorted(a::NDArray, v::NDArray{<:Any,N}, left::Bool) where {N} + ptr = @task_scope "searchsorted" begin + ccall((:nda_searchsorted, libnda), + NDArray_t, (NDArray_t, NDArray_t, Bool), + a.ptr, v.ptr, left) + end + return NDArray(ptr, Int64, Val(N)) +end + function nda_ravel(arr::NDArray) ptr = @task_scope "ravel" begin ccall((:nda_ravel, libnda), @@ -418,7 +663,7 @@ function nda_trace( (NDArray_t, Int32, Int32, Int32, Legate.LegateTypeAllocated), arr.ptr, offset, a1, a2, legate_type) end - return NDArray(ptr, T, Val(1)) + return NDArray(ptr, T, Val(0)) end # transpose reverses the axes: element type and rank are preserved @@ -431,6 +676,62 @@ function nda_transpose(arr::NDArray{T,N}) where {T,N} return NDArray(ptr, T, Val(N)) end +# Arbitrary axis permutation; rank is preserved. `axes` are 0-based. +function nda_transpose_axes(arr::NDArray{T,N}, axes::Vector{Int32}) where {T,N} + axes_c = collect(Int32, axes) + ptr = @task_scope "permutedims" begin + ccall((:nda_transpose_axes, libnda), + NDArray_t, (NDArray_t, Ptr{Int32}, Int32), + arr.ptr, axes_c, Int32(length(axes_c))) + end + return NDArray(ptr, T, Val(N)) +end + +# Rank-changing: drop size-1 axes. Empty `axes` drops every size-1 axis. +function nda_squeeze(arr::NDArray, axes::Vector{Int32}) + axes_c = collect(Int32, axes) + ptr = @task_scope "squeeze" begin + ccall((:nda_squeeze, libnda), + NDArray_t, (NDArray_t, Ptr{Int32}, Int32), + arr.ptr, axes_c, Int32(length(axes_c))) + end + return NDArray(ptr) +end + +function nda_diagonal(arr::NDArray, offset::Int32, axis1::Int32, axis2::Int32) + ptr = @task_scope "diagonal" begin + ccall((:nda_diagonal, libnda), + NDArray_t, (NDArray_t, Int32, Int32, Int32), + arr.ptr, offset, axis1, axis2) + end + return NDArray(ptr) +end + +function nda_contract( + out::NDArray, + lhs_modes::Vector{UInt8}, + rhs1::NDArray, + rhs1_modes::Vector{UInt8}, + rhs2::NDArray, + rhs2_modes::Vector{UInt8}, + extent_keys::Vector{UInt8}, + extents::Vector{Int32}, +) + @task_scope "contract" begin + ccall((:nda_contract, libnda), + Cvoid, + ( + NDArray_t, Ptr{UInt8}, Int32, NDArray_t, Ptr{UInt8}, Int32, + NDArray_t, Ptr{UInt8}, Int32, Ptr{UInt8}, Ptr{Int32}, Int32, + ), + out.ptr, lhs_modes, Int32(length(lhs_modes)), + rhs1.ptr, rhs1_modes, Int32(length(rhs1_modes)), + rhs2.ptr, rhs2_modes, Int32(length(rhs2_modes)), + extent_keys, extents, Int32(length(extents))) + end + return out +end + function nda_attach_external(arr::Array{T,N}; shape::Dims{N}=size(arr)) where {T,N} st = Legate.attach_external_row_major(arr; shape) # Use the CxxWrap method for type-safe interaction @@ -453,7 +754,14 @@ function get_ptr(arr::NDArray{T,N}) where {T,N} # store with the NDArray; finalize after use (same pin class as `_add_task_array!`). st_handle = get_store(arr) # LogicalArrayImplAllocated (returned by value) la = Legate.LogicalArray{T,N}(st_handle, size(arr)) - ptr = Legate.get_ptr(la) + # Legate.get_ptr(::LogicalArray) leaves PhysicalArray/PhysicalStore handles + # to GC. Their destructors can unmap regions, which must run on the runtime + # thread, just like the logical handle below. Keep and release them here. + physical = Legate.get_physical_array(la) + data = Legate.data(physical) + ptr = Legate.get_ptr(data) + finalize(data) + finalize(physical) finalize(st_handle) return ptr end @@ -534,11 +842,12 @@ end Return the size of the given `NDArray`. """ -shape(arr::NDArray{<:Any,N,true}) where {N} = arr.padding - -function shape(arr::NDArray{<:Any,N,false}) where {N} - shp = cuNumeric.nda_array_shape(arr) - return ntuple(i -> Int(shp[i]), Val(N)) +function shape(arr::NDArray{<:Any,N}) where {N} + # Rank is known from the type; avoid a rank query and temporary shape Vector. + shp = Ref{NTuple{N,UInt64}}() + ccall((:nda_array_shape, libnda), + Cvoid, (NDArray_t, Ref{NTuple{N,UInt64}}), arr.ptr, shp) + return map(Int, shp[]) end @doc""" diff --git a/src/ndarray/diagonal.jl b/src/ndarray/diagonal.jl new file mode 100644 index 000000000..19099d946 --- /dev/null +++ b/src/ndarray/diagonal.jl @@ -0,0 +1,664 @@ +###### diag / _eye / trace ###### + +export diagonal + +@doc""" + cuNumeric.diag(arr::NDArray; k=0) + +Extract the k-th diagonal from a 2D `NDArray`. +""" +function diag(arr::NDArray; k::Int=0) + return nda_diag(arr, Int32(k)) +end + +LinearAlgebra.diag(arr::NDArray{<:Any,2}, k::Integer=0) = nda_diag(arr, Int32(k)) + +""" + cuNumeric.diagonal(arr::NDArray; offset=0, dims=(1, 2)) + +Extract the diagonal of `arr` along two 1-based axes `dims`. The result has rank +`ndims(arr) - 1`, with the diagonal stored as the last axis. `offset` selects a +superdiagonal (`> 0`) or subdiagonal (`< 0`), matching `LinearAlgebra.diag`. + +For a 2D matrix this is the same 1D diagonal as [`diag`](@ref). For `N > 2` the +non-diagonal axes are kept in order, followed by the extracted diagonal. +""" +function diagonal(arr::NDArray; offset::Integer=0, dims::NTuple{2,Integer}=(1, 2)) + nd = ndims(arr) + nd < 2 && throw(ArgumentError("diagonal requires ndims >= 2")) + d1 = Int(dims[1]) + d2 = Int(dims[2]) + (1 <= d1 <= nd && 1 <= d2 <= nd) || throw( + ArgumentError("dims $dims are out of range for $(nd)-d array") + ) + d1 == d2 && throw(ArgumentError("diagonal dims must be distinct, got $dims")) + return nda_diagonal(arr, Int32(offset), Int32(d1 - 1), Int32(d2 - 1)) +end + +# Internal dense identity used by UniformScaling / Diagonal densify helpers. +# Prefer `LinearAlgebra.I` / `NDArray{T}(I, n, n)` / `one(A)` in user code. +function _eye(::Type{T}, rows::Int) where {T} + return nda_eye(Int32(rows), T) +end +_eye(rows::Int) = _eye(DEFAULT_FLOAT, rows) + +@doc""" + cuNumeric.trace(arr::NDArray; offset=0, a1=0, a2=1) + +Compute the trace (sum of a diagonal) of the `NDArray`. +Returns a 0-dimensional `NDArray`. The accumulator type follows promotions of +other reductions like `sum`. +""" +function trace(arr::NDArray{T,2}; offset::Int=0, a1::Int=0, a2::Int=1) where {T} + LinearAlgebra.checksquare(arr) + T_OUT = Base.promote_op(Base.sum, Vector{T}) + return cnscalar(nda_trace(arr, Int32(offset), Int32(a1), Int32(a2), T_OUT)) +end + +function LinearAlgebra.tr(arr::NDArray{<:Any,2}) + return cuNumeric.trace(arr) +end + +###### Diagonal constructors ###### + +const DiagonalNDArray{T} = Diagonal{T,<:NDArray{T,1}} + +function LinearAlgebra.Diagonal(arr::NDArray{T,1}) where {T} + return Diagonal{T,typeof(arr)}(arr) +end + +function LinearAlgebra.Diagonal(arr::NDArray{T,2}) where {T} + return Diagonal(diag(arr)) +end + +# Note: Matrix{T} === Array{T,2}, so do not also define Array{T,2}(...). +# Use Base.zeros — bare `zeros` resolves to cuNumeric.zeros inside this module. +function Base.Matrix{T}(D::DiagonalNDArray) where {T} + dv = Array(D.diag) + n = length(dv) + B = Base.zeros(T, n, n) + @inbounds for i in 1:n + B[i, i] = dv[i] + end + return B +end +Base.Matrix(D::DiagonalNDArray{T}) where {T} = Matrix{T}(D) + +function Base.show(io::IO, D::DiagonalNDArray) + return show(io, Diagonal(Array(D.diag))) +end + +function Base.show(io::IO, ::MIME"text/plain", D::DiagonalNDArray) + # Keep the real Diagonal{T,<:NDArray} in the summary; only densify the + # diagonal vector for Base's ⋅-style body formatting. + summary(io, D) + isempty(D) && return nothing + println(io, ":") + Base.print_array(io, Diagonal(Array(D.diag))) + return nothing +end + +###### Diagonal operators ###### + +@inline _diag_vec(D::DiagonalNDArray) = D.diag +@inline _row_scale(d::NDArray{<:Any,1}) = reshape(d, (length(d), 1)) +@inline _col_scale(d::NDArray{<:Any,1}) = reshape(d, (1, length(d))) + +function Base.:*(D::DiagonalNDArray, A::NDArray{<:Any,2}) + size(A, 1) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return _row_scale(_diag_vec(D)) .* A +end + +function Base.:*(A::NDArray{<:Any,2}, D::DiagonalNDArray) + size(A, 2) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return A .* _col_scale(_diag_vec(D)) +end + +function Base.:*(D::DiagonalNDArray, v::NDArray{<:Any,1}) + length(v) == size(D, 1) || throw( + DimensionMismatch("vector length $(length(v)) does not match diagonal $(size(D,1))") + ) + return _diag_vec(D) .* v +end + +function LinearAlgebra.lmul!(D::DiagonalNDArray, B::NDArray) + return mul!(B, D, B) +end + +function LinearAlgebra.rmul!(A::NDArray, D::DiagonalNDArray) + return mul!(A, A, D) +end + +function LinearAlgebra.mul!( + C::NDArray, D::DiagonalNDArray, A::Union{NDArray{<:Any,1},NDArray{<:Any,2}} +) + size(C) == size(A) || + throw(DimensionMismatch("diagonal product destination must have size $(size(A))")) + size(A, 1) == size(D, 1) || + throw(DimensionMismatch("diagonal product dimensions do not match")) + d = ndims(A) == 1 ? _diag_vec(D) : _row_scale(_diag_vec(D)) + C .= d .* A + d === _diag_vec(D) || destroy!(d) + return C +end + +function LinearAlgebra.mul!(C::NDArray, A::NDArray{<:Any,2}, D::DiagonalNDArray) + size(C) == size(A) || + throw(DimensionMismatch("diagonal product destination must have size $(size(A))")) + size(A, 2) == size(D, 1) || + throw(DimensionMismatch("diagonal product dimensions do not match")) + d = _col_scale(_diag_vec(D)) + C .= A .* d + destroy!(d) + return C +end + +function Base.:\(D::DiagonalNDArray, B::NDArray{<:Any,1}) + length(B) == size(D, 1) || throw( + DimensionMismatch("vector length $(length(B)) does not match diagonal $(size(D,1))") + ) + return B ./ _diag_vec(D) +end + +function Base.:\(D::DiagonalNDArray, B::NDArray{<:Any,2}) + size(B, 1) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(B,1))×$(size(B,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + d = _row_scale(_diag_vec(D)) + result = B ./ d + destroy!(d) + return result +end + +function Base.:/(A::NDArray{<:Any,2}, D::DiagonalNDArray) + size(A, 2) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + d = _col_scale(_diag_vec(D)) + result = A ./ d + destroy!(d) + return result +end + +function LinearAlgebra.ldiv!(D::DiagonalNDArray, B::NDArray{<:Any,1}) + length(B) == size(D, 1) || + throw(DimensionMismatch("vector length $(length(B)) does not match diagonal $(size(D,1))")) + B ./= _diag_vec(D) + return B +end + +function LinearAlgebra.ldiv!(D::DiagonalNDArray, B::NDArray{<:Any,2}) + size(B, 1) == size(D, 1) || + throw(DimensionMismatch("diagonal division dimensions do not match")) + d = _row_scale(_diag_vec(D)) + B ./= d + destroy!(d) + return B +end + +function LinearAlgebra.rdiv!(A::NDArray{<:Any,2}, D::DiagonalNDArray) + size(A, 2) == size(D, 1) || + throw(DimensionMismatch("diagonal division dimensions do not match")) + d = _col_scale(_diag_vec(D)) + A ./= d + destroy!(d) + return A +end + +function Base.inv(D::DiagonalNDArray{T}) where {T} + # Base Julia checks and throws a SingularException. We cannot do + # that without unwrapping the NDArray to a Julia scalar. + return Diagonal(inv.(_diag_vec(D))) +end + +LinearAlgebra.det(D::DiagonalNDArray) = prod(_diag_vec(D)) +LinearAlgebra.tr(D::DiagonalNDArray{<:Number}) = sum(_diag_vec(D)) +Base.sum(D::DiagonalNDArray) = sum(_diag_vec(D)) + +# Include structural zeros before reducing so nonfinite entries propagate, +# without overflowing a product of finite diagonal entries first. +function Base.prod(D::DiagonalNDArray{T}) where {T<:Number} + n = size(D, 1) + n == 0 && return cnscalar(NDArray(one(T))) + n == 1 && return sum(_diag_vec(D)) + values = zero(T) .* _diag_vec(D) + # All finite terms are zero; sum also supports complex backend types. + result = sum(values) + destroy!(values) + return result +end + +function Base.maximum(D::DiagonalNDArray{T}) where {T<:Number} + maxdiag = maximum(_diag_vec(D)) + size(D, 1) > 1 && return max(zero(T), maxdiag) + return maxdiag +end + +function Base.minimum(D::DiagonalNDArray{T}) where {T<:Number} + mindiag = minimum(_diag_vec(D)) + size(D, 1) > 1 && return min(zero(T), mindiag) + return mindiag +end + +Base.iszero(D::DiagonalNDArray) = iszero(_diag_vec(D)) +function Base.isone(D::DiagonalNDArray{T}) where {T} + return all(_diag_vec(D) .== one(T)) +end + +# Base walks `iszero(D.diag)` by scalar iteration; keep on-device via `iszero(D)`. +function LinearAlgebra.istriu(D::DiagonalNDArray, k::Integer=0) + return k <= 0 ? cnscalar(NDArray(true)) : iszero(D) +end +function LinearAlgebra.istril(D::DiagonalNDArray, k::Integer=0) + return k >= 0 ? cnscalar(NDArray(true)) : iszero(D) +end + +# Real Diagonal is always Hermitian/symmetric in Base; Complex Hermitian needs isreal(diag). +LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Real}) = cnscalar(NDArray(true)) +function LinearAlgebra.ishermitian(D::DiagonalNDArray{<:Complex}) + return all(imag(_diag_vec(D)) .== zero(real(eltype(D)))) +end +LinearAlgebra.issymmetric(D::DiagonalNDArray{<:Number}) = cnscalar(NDArray(true)) + +# Base `isposdef(D) = all(isposdef, D.diag)` scalar-iterates. +function LinearAlgebra.isposdef(D::DiagonalNDArray{T}) where {T<:Real} + isempty(D) && return cnscalar(NDArray(true)) + return all(_diag_vec(D) .> zero(T)) +end +function LinearAlgebra.isposdef(D::DiagonalNDArray{T}) where {T<:Complex} + # isposdef(z) = isreal(z) && real(z) > 0 — keep on-device, no host densify. + d = _diag_vec(D) + return all((imag(d) .== zero(real(T))) .& (real(d) .> zero(real(T)))) +end + +###### Eigen / related ###### + +# Base: eigvals(D::Diagonal{<:Number}) = copy(D.diag). Keep NDArray (package style). +function LinearAlgebra.eigvals(D::DiagonalNDArray{<:Number}; permute::Bool=true, scale::Bool=true) + return copy(_diag_vec(D)) +end + +# Unsorted eigen: values are a copy of the diagonal (NDArray); vectors are NDArray I. +# Keyword `sortby` is not accepted on this override. +function LinearAlgebra.eigen( + D::DiagonalNDArray; + permute::Bool=true, + scale::Bool=true, +) + Td = Base.promote_op(/, eltype(D), eltype(D)) + return Eigen(copy(_diag_vec(D)), _eye(Td, size(D, 1))) +end + +function LinearAlgebra.eigvecs( + D::DiagonalNDArray; + permute::Bool=true, + scale::Bool=true, +) + return eigen(D; permute=permute, scale=scale).vectors +end + +# Real logdet is sum(log.(diag)) on-device. Complex logdet / other Base LinearAlgebra +# ops without overrides fall through and may scalar-index `.diag`. +LinearAlgebra.logdet(D::DiagonalNDArray{<:Real}) = sum(log.(_diag_vec(D))) + +# Operator / entrywise norms from the diagonal only (no host densify). +for op in (:norm, :opnorm, :cond) + @eval LinearAlgebra.$op(D::DiagonalNDArray, p::NDArray{<:Real,0}) = $op(D, _maybe_fetch(p)) +end + +function LinearAlgebra.opnorm(D::DiagonalNDArray, p::Real=2) + p = _maybe_fetch(p) + if !(p == 1 || p == 2 || p == Inf) + throw(ArgumentError(lazy"invalid p-norm p=$p. Valid: 1, 2, Inf")) + end + isempty(D) && return cnscalar(NDArray(float(real(zero(eltype(D)))))) + return maximum(abs.(_diag_vec(D))) +end + +function LinearAlgebra.norm(D::DiagonalNDArray, p::Real=2) + p = _maybe_fetch(p) + R = float(real(eltype(D))) + isempty(D) && return cnscalar(NDArray(zero(R))) + d = abs.(_diag_vec(D)) + # Canonicalize host orders, including signed zero, to share specializations. + result = _diagonal_norm(d, Val(Float64(p) + 0.0), R) + destroy!(d) + return result +end + +function _diagonal_norm(d, ::Val{0.0}, ::Type{R}) where {R} + mask = d .!= zero(eltype(d)) + values = as_type(mask, R) + result = sum(values) + destroy!(values) + destroy!(mask) + return result +end + +_diagonal_norm(d, ::Val{1.0}, ::Type) = sum(d) +_diagonal_norm(d, ::Val{2.0}, ::Type) = sqrt(sum(d .^ 2)) +_diagonal_norm(d, ::Val{Inf}, ::Type) = maximum(d) + +function _diagonal_norm(d, ::Val{-Inf}, ::Type{R}) where {R} + # With structural zeros, every negative norm is zero (or NaN). Reuse the + # -1 case to propagate NaNs, which the backend minimum would discard. + return length(d) > 1 ? _diagonal_norm(d, Val(-1.0), R) : minimum(d) +end + +function _diagonal_norm(d, ::Val{P}, ::Type{R}) where {P,R} + total = sum(d .^ R(P)) + # Each structural zero contributes Inf for a finite negative p. + # Adding that contribution retains NaNs from diagonal entries. + length(d) > 1 && P < 0 && (total = total + R(Inf)) + return total ^ inv(R(P)) +end + +function LinearAlgebra.cond(D::DiagonalNDArray, p::Real=2) + p = _maybe_fetch(p) + if !(p == 1 || p == 2 || p == Inf) + throw(ArgumentError(lazy"invalid p-norm p=$p. Valid: 1, 2, Inf")) + end + isempty(D) && return cnscalar(NDArray(float(one(real(eltype(D)))))) + dabs = abs.(_diag_vec(D)) + result = maximum(dabs) / minimum(dabs) + destroy!(dabs) + return result +end + +function Base.:+(A::NDArray{T,2}, D::DiagonalNDArray) where {T} + size(A, 1) == size(A, 2) == size(D, 1) || throw( + DimensionMismatch( + "matrix is $(size(A,1))×$(size(A,2)), but diagonal is $(size(D,1))×$(size(D,2))" + ), + ) + return A + (_row_scale(_diag_vec(D)) .* _eye(eltype(D), size(D, 1))) +end +Base.:+(D::DiagonalNDArray, A::NDArray{<:Any,2}) = A + D + +Base.:-(A::NDArray{<:Any,2}, D::DiagonalNDArray) = A + (-D) +Base.:-(D::DiagonalNDArray, A::NDArray{<:Any,2}) = D + (-A) + +###### UniformScaling (LinearAlgebra.I) ###### + +@inline function _uniformscaling_eye(::Type{R}, n::Integer, λ) where {R} + E = _eye(R, Int(n)) + return isone(λ) ? E : nda_multiply_scalar(E, R(λ)) +end + +function _uniformscaling_eye(::Type{R}, n::Integer, λ::NDArray{<:Any,0}) where {R} + E = _eye(R, Int(n)) + E .*= _coefficient_as(R, λ) + return E +end + +_uniformscale_mul(::Type{T}, λ::Number, a::NDArray) where {T} = _mul_scalar(T, λ, a) +_uniformscale_mul(::Type{T}, λ::NDArray{<:Any,0}, a::NDArray) where {T} = + a .* _coefficient_as(T, λ) + +function NDArray{T}(J::LinearAlgebra.UniformScaling, dims::Dims{2}) where {T} + A = NDArray{T}(undef, dims) + copyto!(A, J) + return A +end +function NDArray{T}(J::LinearAlgebra.UniformScaling, m::Integer, n::Integer) where {T} + return NDArray{T}(J, Dims((Int(m), Int(n)))) +end +NDArray(J::LinearAlgebra.UniformScaling{T}, dims::Dims{2}) where {T} = NDArray{_coefficient_type(J.λ)}(J, dims) +function NDArray(J::LinearAlgebra.UniformScaling{T}, m::Integer, n::Integer) where {T} + return NDArray(J, Dims((Int(m), Int(n)))) +end + +function Base.copyto!(A::NDArray{T,2}, J::LinearAlgebra.UniformScaling) where {T} + m, n = size(A) + if _host_iszero(J.λ) + return fill!(A, zero(T)) + elseif m == n + return copyto!(A, _uniformscaling_eye(T, m, _scale_storage(J.λ))) + else + fill!(A, zero(T)) + k = min(m, n) + A[1:k, 1:k] = _uniformscaling_eye(T, k, _scale_storage(J.λ)) + return A + end +end + +function Base.:+(A::NDArray{T,2}, J::LinearAlgebra.UniformScaling) where {T} + LinearAlgebra.checksquare(A) + R = Base.promote_op(+, T, _coefficient_type(J.λ)) + return A + _uniformscaling_eye(R, size(A, 1), _scale_storage(J.λ)) +end +Base.:+(J::LinearAlgebra.UniformScaling, A::NDArray{<:Any,2}) = A + J + +Base.:-(A::NDArray{<:Any,2}, J::LinearAlgebra.UniformScaling) = A + (-J) +function Base.:-(J::LinearAlgebra.UniformScaling, A::NDArray{<:Any,2}) + return (-A) + J +end + +# Scale by λ without promoting the array (A * I must not Bool→Float32 promote). +function Base.:*(A::NDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + return _uniformscale_mul(T, _scale_storage(J.λ), A) +end +function Base.:*(J::LinearAlgebra.UniformScaling, A::NDArray{T}) where {T} + return _uniformscale_mul(T, _scale_storage(J.λ), A) +end + +function Base.one(A::NDArray{T,2}) where {T} + LinearAlgebra.checksquare(A) + return _eye(T, size(A, 1)) +end +function Base.oneunit(A::NDArray{T,2}) where {T} + LinearAlgebra.checksquare(A) + return _eye(T, size(A, 1)) +end + +###### Diagonal ↔ UniformScaling ###### + +# Keep Diagonal structure: D + λI == Diagonal(d .+ λ), not a dense matrix. +function Base.:+(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + R = Base.promote_op(+, T, _coefficient_type(J.λ)) + return Diagonal(_diag_vec(D) .+ _coefficient_as(R, J.λ)) +end +Base.:+(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray) = D + J + +Base.:-(D::DiagonalNDArray, J::LinearAlgebra.UniformScaling) = D + (-J) +function Base.:-(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray{T}) where {T} + R = Base.promote_op(-, _coefficient_type(J.λ), T) + return Diagonal(_coefficient_as(R, J.λ) .- _diag_vec(D)) +end + +function Base.:*(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + return Diagonal(_uniformscale_mul(T, _scale_storage(J.λ), _diag_vec(D))) +end +Base.:*(J::LinearAlgebra.UniformScaling, D::DiagonalNDArray) = D * J + +function Base.copyto!(D::DiagonalNDArray{T}, J::LinearAlgebra.UniformScaling) where {T} + fill!(_diag_vec(D), _coefficient_as(T, J.λ)) + return D +end + +###### Diagonal broadcast ###### + +# LinearAlgebra's StructuredMatrixStyle{Diagonal} path (structuredbroadcast.jl) +# fills `dest.diag[i]` via `Broadcast._broadcast_getindex(bc, (i,i))`, which +# scalar-indexes the NDArray. CUDA/GPUArrays hit the same Base path and do not +# special-case Diagonal broadcast either. +# +# For structure-preserving / zero-preserving broadcasts we lower to a 1D +# broadcast on `.diag` (NDArrayStyle). Fusion then applies to that vector +# broadcast as usual; there is no separate Diagonal-matrix fusion. +# +# Densifying out-of-place broadcasts (e.g. `D .+ 1`, `D .+ A`) allocate a dense +# NDArray and expand Diagonal leaves to `diag .* I` before the usual NDArray +# `copyto!` path — matching Base, which densifies to Matrix. In-place writes +# into Diagonal that would fill off-diagonals still throw ArgumentError. + +@inline _has_diagonal_ndarray(@nospecialize(_)) = false +@inline _has_diagonal_ndarray(::DiagonalNDArray) = true +@inline function _has_diagonal_ndarray(bc::Broadcast.Broadcasted) + return _has_diagonal_ndarray_args(bc.args) +end +@inline _has_diagonal_ndarray_args(::Tuple{}) = false +@inline function _has_diagonal_ndarray_args(args::Tuple) + return _has_diagonal_ndarray(getfield(args, 1)) || + _has_diagonal_ndarray_args(Base.tail(args)) +end + +# Dense NDArray leaves (not wrapped in Diagonal). Used to avoid scalar-indexing +# when building the in-place densifying off-diagonal ArgumentError. +@inline _has_plain_ndarray(@nospecialize(_)) = false +@inline _has_plain_ndarray(::NDArray) = true +@inline function _has_plain_ndarray(bc::Broadcast.Broadcasted) + return _has_plain_ndarray_args(bc.args) +end +@inline _has_plain_ndarray_args(::Tuple{}) = false +@inline function _has_plain_ndarray_args(args::Tuple) + return _has_plain_ndarray(getfield(args, 1)) || + _has_plain_ndarray_args(Base.tail(args)) +end + +@inline _first_diagonal_ndarray_diag(D::DiagonalNDArray) = D.diag +@inline function _first_diagonal_ndarray_diag(bc::Broadcast.Broadcasted) + return _first_diagonal_ndarray_diag_args(bc.args) +end +@inline function _first_diagonal_ndarray_diag_args(args::Tuple) + a = getfield(args, 1) + return if _has_diagonal_ndarray(a) + _first_diagonal_ndarray_diag(a) + else + _first_diagonal_ndarray_diag_args(Base.tail(args)) + end +end + +# Replace Diagonal leaves with their `.diag` vectors; keep scalars / Refs / etc. +@inline _diag_bc_arg(D::Diagonal) = D.diag +@inline function _diag_bc_arg(bc::Broadcast.Broadcasted) + return Broadcast.broadcasted(bc.f, map(_diag_bc_arg, bc.args)...) +end +@inline _diag_bc_arg(x) = x + +@inline function _diagonal_broadcast_preserves_structure(bc::Broadcast.Broadcasted) + # `+` / `-` on Diagonal+Diagonal are zero-preserving (Base's fzeropreserving), + # not isstructurepreserving. Prefer either so D.+D lowers to `.diag` broadcast. + return LinearAlgebra.isstructurepreserving(bc) || LinearAlgebra.fzeropreserving(bc) +end + +# Materialize Diagonal{NDArray} as a dense matrix (same pattern as A ± D). +@inline function _densify_diagonal_ndarray(D::DiagonalNDArray) + d = _diag_vec(D) + n = length(d) + # nda_eye(1) currently yields an unreadable store; build 1×1 from the diag. + n <= 1 && return reshape(copy(d), (n, n)) + return _row_scale(d) .* _eye(eltype(D), n) +end + +# For densifying Structured → dense NDArray copyto!: expand Diagonal leaves. +@inline _expand_diagonal_ndarray_bc_arg(D::DiagonalNDArray) = _densify_diagonal_ndarray(D) +@inline function _expand_diagonal_ndarray_bc_arg(bc::Broadcast.Broadcasted) + return Broadcast.broadcasted(bc.f, map(_expand_diagonal_ndarray_bc_arg, bc.args)...) +end +@inline _expand_diagonal_ndarray_bc_arg(x) = x + +# Off-diagonal Diagonal getindex uses `diagzero` (no `.diag` read), so evaluating +# an off-diagonal broadcast index is safe without `@allowscalar` when every +# non-Diagonal leaf is a scalar. Dense NDArray leaves must not be indexed. +# Used for in-place densifying rejection only (out-of-place densifies). +function _throw_densifying_diagonal_broadcast(bc::Broadcast.Broadcasted) + axs = axes(bc) + if length(axs) >= 2 && length(axs[1]) >= 2 && length(axs[2]) >= 2 + if !_has_plain_ndarray(bc) + v = @inbounds Broadcast._broadcast_getindex(bc, CartesianIndex(2, 1)) + throw( + ArgumentError( + "cannot set off-diagonal entry (2, 1) to a nonzero value ($v)" + ), + ) + end + throw( + ArgumentError( + "cannot set off-diagonal entry (2, 1) to a nonzero value; " * + "broadcast over Diagonal with NDArray diagonal would densify", + ), + ) + end + return throw( + ArgumentError( + "broadcast over Diagonal with NDArray diagonal is not structure-preserving " * + "and would densify; in-place densifying broadcast is not supported", + ), + ) +end + +function Base.similar( + bc::Broadcast.Broadcasted{LinearAlgebra.StructuredMatrixStyle{Diagonal}}, + ::Type{ElType}, +) where {ElType} + inds = axes(bc) + n = length(inds[1]) + if _has_diagonal_ndarray(bc) + if _diagonal_broadcast_preserves_structure(bc) + d = _first_diagonal_ndarray_diag(bc) + return Diagonal(similar(d, ElType, (n,))) + end + # Match Base: densify out-of-place to a dense array (NDArray, not Matrix). + return similar(NDArray{ElType}, inds) + elseif _diagonal_broadcast_preserves_structure(bc) + return LinearAlgebra.structured_broadcast_alloc(bc, Diagonal, ElType, n) + else + return similar( + convert(Broadcast.Broadcasted{Broadcast.DefaultArrayStyle{ndims(bc)}}, bc), + ElType, + ) + end +end + +@inline function _copyto_diagonal_ndarray!( + dest::DiagonalNDArray, bc::Broadcast.Broadcasted +) + axes(bc) == axes(dest) || Broadcast.throwdm(axes(bc), axes(dest)) + # Lower to NDArrayStyle vector broadcast so `_copyto!` / fusion apply. + copyto!(_diag_vec(dest), Broadcast.instantiate(_diag_bc_arg(bc))) + return dest +end + +# Out-of-place densify: destination is dense NDArray from `similar` above. +function Base.copyto!( + dest::NDArray, + bc::Broadcast.Broadcasted{LinearAlgebra.StructuredMatrixStyle{Diagonal}}, +) + axes(dest) == axes(bc) || Broadcast.throwdm(axes(dest), axes(bc)) + isempty(dest) && return dest + expanded = Broadcast.instantiate(_expand_diagonal_ndarray_bc_arg(bc)) + return _copyto!(dest, expanded) +end + +function Base.copyto!( + dest::DiagonalNDArray, + bc::Broadcast.Broadcasted{<:LinearAlgebra.StructuredMatrixStyle}, +) + if !LinearAlgebra.isvalidstructbc(dest, bc) + # 1×1: Base's generic path only writes the diagonal; no off-diagonals to reject. + size(dest, 1) <= 1 || return _throw_densifying_diagonal_broadcast(bc) + end + return _copyto_diagonal_ndarray!(dest, bc) +end + +# Safety net if a densifying Structured broadcast is converted to Nothing +# (Base's `isvalidstructbc` fallback) before reaching the method above. +function Base.copyto!(dest::DiagonalNDArray, bc::Broadcast.Broadcasted{Nothing}) + if !_diagonal_broadcast_preserves_structure(bc) + size(dest, 1) <= 1 || return _throw_densifying_diagonal_broadcast(bc) + end + return _copyto_diagonal_ndarray!(dest, bc) +end diff --git a/src/ndarray/fft.jl b/src/ndarray/fft.jl new file mode 100644 index 000000000..68cab8a97 --- /dev/null +++ b/src/ndarray/fft.jl @@ -0,0 +1,214 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +export fft, ifft, bfft!, fft!, ifft!, batched_fft, batched_ifft, batched_fft!, batched_ifft! + +const _FFT_PROMOTABLE = Union{SUPPORTED_INT_TYPES,Bool,SUPPORTED_FLOAT_TYPES} +const _FFT_ACCEPTED = Union{SUPPORTED_COMPLEX_TYPES,_FFT_PROMOTABLE} + +_fft_eltype(::Type{ComplexF32}) = ComplexF32 +_fft_eltype(::Type{ComplexF64}) = ComplexF64 +_fft_eltype(::Type{Float32}) = ComplexF32 +_fft_eltype(::Type{Float64}) = ComplexF64 +_fft_eltype(::Type{T}) where {T<:_FFT_PROMOTABLE} = ComplexF64 + +function _fft_complex(A::NDArray{T}) where {T<:_FFT_ACCEPTED} + return unchecked_promote_arr(A, _fft_eltype(T)) +end + +function _fft_out_of_place( + A::NDArray{<:_FFT_ACCEPTED}, dims::NTuple{R,Int}, direction::Int32, scale::Bool +) where {R} + inp = _fft_complex(A) + out = cuNumeric.zeros(eltype(inp), size(inp)) + fft_task!(out, inp, dims, direction; scale) + # Promotion allocated a new complex buffer; the task already holds the store. + inp !== A && destroy!(inp) + return out +end + +""" + fft(A::NDArray, [dims]) + +Unnormalized complex discrete Fourier transform of `A`, matching +`AbstractFFTs.fft`. By default every dimension is transformed. `dims` selects a +subset (Julia 1-based dimensions). + +Real and integer inputs are converted to complex (`Float32` → `ComplexF32`, +everything else → `ComplexF64`). The result has the same shape as `A`. + +This is GPU-only: cupynumeric's FFT task has no CPU variant. Multi-GPU +execution is limited to batching over dimensions that are not transformed. + +There is no reusable cuFFT plan; each call launches the `CUPYNUMERIC_FFT` task. +""" +function fft(A::NDArray{<:_FFT_ACCEPTED}) + return _fft_out_of_place(A, _fft_dims(A), Int32(cuNumeric.FFT_FORWARD), false) +end +function fft(A::NDArray{<:_FFT_ACCEPTED}, dims) + return _fft_out_of_place(A, _fft_dims(A, dims), Int32(cuNumeric.FFT_FORWARD), false) +end + +function fft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in fft")) +end +function fft(A::NDArray, ::Any) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in fft")) +end + +""" + ifft(A::NDArray, [dims]) + +Normalized inverse discrete Fourier transform of `A`, matching +`AbstractFFTs.ifft`. Equivalent to the unnormalized inverse scaled by `1/N`, +where `N` is the product of the transformed lengths. + +See [`fft`](@ref). +""" +function ifft(A::NDArray{<:_FFT_ACCEPTED}) + return _fft_out_of_place(A, _fft_dims(A), Int32(cuNumeric.FFT_INVERSE), true) +end +function ifft(A::NDArray{<:_FFT_ACCEPTED}, dims) + return _fft_out_of_place(A, _fft_dims(A, dims), Int32(cuNumeric.FFT_INVERSE), true) +end + +function ifft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in ifft")) +end +function ifft(A::NDArray, ::Any) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in ifft")) +end + +""" + fft!(A::NDArray, [dims]) + +In-place [`fft`](@ref). `A` must already be `ComplexF32` or `ComplexF64`. +""" +function fft!(A::NDArray{T,N}) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(A, A, _fft_dims(A), Int32(cuNumeric.FFT_FORWARD)) +end +function fft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(A, A, _fft_dims(A, dims), Int32(cuNumeric.FFT_FORWARD)) +end + +function fft!(A::NDArray) + return throw(ArgumentError("fft! requires a complex NDArray; got $(eltype(A))")) +end +function fft!(A::NDArray, ::Any) + return throw(ArgumentError("fft! requires a complex NDArray; got $(eltype(A))")) +end + +""" + ifft!(A::NDArray, [dims]) + +In-place [`ifft`](@ref). `A` must already be `ComplexF32` or `ComplexF64`. +""" +function ifft!(A::NDArray{T,N}) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return ifft!(A, _fft_dims(A)) +end +function ifft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} + region = _fft_dims(A, dims) + return fft_task!(A, A, region, Int32(cuNumeric.FFT_INVERSE); scale=true) +end + +function bfft!(A::NDArray{T,N}) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return bfft!(A, _fft_dims(A)) +end +function bfft!(A::NDArray{T,N}, dims) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(A, A, _fft_dims(A, dims), Int32(cuNumeric.FFT_INVERSE)) +end + +""" + bfft!(dest::NDArray, src::NDArray) + +Write the unnormalized inverse FFT of `src` into `dest` without changing `src`. +Both arrays must have the same shape and complex eltype. +""" +function bfft!( + dest::NDArray{T,N}, src::NDArray{T,N} +) where {T<:SUPPORTED_COMPLEX_TYPES,N} + return fft_task!(dest, src, _fft_dims(src), Int32(cuNumeric.FFT_INVERSE)) +end + +function bfft!(A::NDArray) + return throw(ArgumentError("bfft! requires a complex NDArray; got $(eltype(A))")) +end +function bfft!(A::NDArray, ::Any) + return throw(ArgumentError("bfft! requires a complex NDArray; got $(eltype(A))")) +end + +function ifft!(A::NDArray) + return throw(ArgumentError("ifft! requires a complex NDArray; got $(eltype(A))")) +end +function ifft!(A::NDArray, ::Any) + return throw(ArgumentError("ifft! requires a complex NDArray; got $(eltype(A))")) +end + +""" + cuNumeric.batched_fft(A) + +FFT every trailing dimension of `A`, treating `size(A, 1)` as a batch. + +`A` must be at least 2-d. A `(b, n)` stack is `b` independent 1-d transforms; +`(b, n, m)` is `b` independent 2-d transforms. This is the same +`CUPYNUMERIC_FFT` task as [`fft`](@ref); the leading axis is the one that may +be partitioned across GPUs. +""" +function batched_fft(A::NDArray{<:_FFT_ACCEPTED}) + return fft(A, _fft_batch_dims(A)) +end +function batched_fft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in batched_fft")) +end + +""" + cuNumeric.batched_ifft(A) + +Normalized inverse of [`batched_fft`](@ref). +""" +function batched_ifft(A::NDArray{<:_FFT_ACCEPTED}) + return ifft(A, _fft_batch_dims(A)) +end +function batched_ifft(A::NDArray) + return throw(ArgumentError("array type $(eltype(A)) is unsupported in batched_ifft")) +end + +""" + cuNumeric.batched_fft!(A) + +In-place [`batched_fft`](@ref). `A` must already be complex. +""" +function batched_fft!(A::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return fft!(A, _fft_batch_dims(A)) +end +function batched_fft!(A::NDArray) + return throw(ArgumentError("batched_fft! requires a complex NDArray; got $(eltype(A))")) +end + +""" + cuNumeric.batched_ifft!(A) + +In-place [`batched_ifft`](@ref). `A` must already be complex. +""" +function batched_ifft!(A::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return ifft!(A, _fft_batch_dims(A)) +end +function batched_ifft!(A::NDArray) + return throw(ArgumentError("batched_ifft! requires a complex NDArray; got $(eltype(A))")) +end diff --git a/src/ndarray/linalg.jl b/src/ndarray/linalg.jl index 075820f85..5f41eaba9 100644 --- a/src/ndarray/linalg.jl +++ b/src/ndarray/linalg.jl @@ -1,48 +1,208 @@ +export NDArrayQR + # Type/dim guards dispatch on one argument at a time, then forward to `_solve`. """ cuNumeric.solve(A, b) -Solve linear system(s) `A * x = b`. +Solve the linear system `A * x = b`. + +`A` must be a square `(m, m)` matrix. `b` must have shape `(m,)` or `(m, n)`. +The result has the same shape as `b`. -`A` must have shape `(..., m, m)`. `b` must have shape `(..., m)` or `(..., m, n)`. -The result has the same shape as `b`. Batch dimensions are supported; the -implementation always uses the batched Legate `SOLVE` path. +Large systems automatically use cuSolverMp when it is available and Legate has +multiple active GPUs. Algorithm selection follows cuPyNumeric 26.06. Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. -Integer or `Bool` inputs promote to `Float64` only when promotion is allowed. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a stack of systems of shape `(b, m, m)` use +[`cuNumeric.batched_solve`](@ref). + +See also `\\`. """ -function solve(a::NDArray{<:_SOLVE_ACCEPTED}, b::NDArray{<:_SOLVE_ACCEPTED}) - A, B = eltype(a), eltype(b) +function solve(a::NDArray{A}, b::NDArray{B}) where {A<:_SOLVE_ACCEPTED,B<:_SOLVE_ACCEPTED} O = promote_type(_solve_eltype(A), _solve_eltype(B)) - # int/bool -> float is an implicit promotion, disallowed unless `allowpromotion` - A <: _SOLVE_PROMOTABLE && assertpromotion(solve, A, O) - B <: _SOLVE_PROMOTABLE && assertpromotion(solve, B, O) - return _solve_check_a_dims(unchecked_promote_arr(a, O), unchecked_promote_arr(b, O)) + return _solve_check_a_dims_2d( + checked_promote_arr(solve, a, O), checked_promote_arr(solve, b, O) + ) end function solve(a::NDArray, b::NDArray) bad = eltype(a) <: _SOLVE_ACCEPTED ? eltype(b) : eltype(a) - throw(ArgumentError("array type $bad is unsupported in solve")) + return throw(ArgumentError("array type $bad is unsupported in solve")) +end + +""" + A \\ b + +Solve `A * x = b` for a square 2D `NDArray` `A`. Equivalent to +[`cuNumeric.solve`](@ref). +""" +Base.:\(a::NDArray{<:Any,2}, b::NDArray{<:Any,1}) = solve(a, b) +Base.:\(a::NDArray{<:Any,2}, b::NDArray{<:Any,2}) = solve(a, b) + +""" + LinearAlgebra.cholesky(A::NDArray{T,2}) -> Cholesky + +Cholesky factorization of the Hermitian positive-definite matrix `A`, returned as +a `LinearAlgebra.Cholesky` object holding the lower factor `L` with `A ≈ L * L'`. + +Only the lower triangle of `A` is read and, unlike Base, it is *not* checked for +being Hermitian. A non-positive-definite input raises an `ErrorException` from +the task rather than `LinearAlgebra.PosDefException`, so the `check` keyword is +not supported. + +Large matrices use cuSolverMp when available with multiple active GPUs. +Other multi-processor configurations use cuPyNumeric's tiled Cholesky algorithm. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a stack of matrices of shape `(b, m, m)` use +[`cuNumeric.batched_cholesky`](@ref). +""" +function LinearAlgebra.cholesky(a::NDArray{T,2}) where {T<:_CHOLESKY_ACCEPTED} + factors = _cholesky(checked_promote_arr(cholesky, a, _cholesky_eltype(T))) + return LinearAlgebra.Cholesky(factors, 'L', 0) +end + +function LinearAlgebra.cholesky(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in cholesky")) +end + +""" + LinearAlgebra.eigen(A::NDArray{T,2}) -> Eigen + +Eigenvalues and right eigenvectors of the square matrix `A`, returned as a +`LinearAlgebra.Eigen` object. Column `j` of `F.vectors` is the eigenvector for +`F.values[j]`. + +Values and vectors are **always complex**, even when `A` is real with real +eigenvalues. This follows the underlying LAPACK `geev` path and differs from +Base, which returns real factors for such input. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. + +For a stack of matrices of shape `(b, m, m)` use +[`cuNumeric.batched_eigen`](@ref). +""" +function LinearAlgebra.eigen(a::NDArray{T,2}) where {T<:_EIG_ACCEPTED} + return LinearAlgebra.Eigen(_eig(checked_promote_arr(eigen, a, _eig_eltype(T)))...) +end + +function LinearAlgebra.eigen(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in eigen")) +end + +""" + LinearAlgebra.eigvals(A::NDArray{T,2}) + +Eigenvalues of the square matrix `A` as a complex `NDArray`. See +`LinearAlgebra.eigen` for the supported element types and for why the +result is always complex. +""" +function LinearAlgebra.eigvals(a::NDArray{T,2}) where {T<:_EIG_ACCEPTED} + return _eigvals(checked_promote_arr(eigvals, a, _eig_eltype(T))) +end + +function LinearAlgebra.eigvals(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in eigvals")) end -function svd(a::NDArray{<:_SVD_ACCEPTED}, full_matrices::Bool=true) - A = eltype(a) - O = _svd_eltype(A) - A <: _SVD_PROMOTABLE && assertpromotion(svd, A, O) - return _svd_check_dims(unchecked_promote_arr(a, O), full_matrices) +""" + LinearAlgebra.eigvecs(A::NDArray{T,2}) + +Right eigenvectors of the square matrix `A`, as columns of a complex `NDArray`. +See `LinearAlgebra.eigen` for the supported element types. +""" +LinearAlgebra.eigvecs(a::NDArray{<:Any,2}) = LinearAlgebra.eigen(a).vectors + +""" + LinearAlgebra.svd(A::NDArray{T,2}; full=false) -> SVD + +Singular value decomposition of `A`, returned as a `LinearAlgebra.SVD` object +with `A ≈ F.U * Diagonal(F.S) * F.Vt`. + +With `m, n = size(A)` and `k = min(m, n)`, `full=false` gives an `m × k` `F.U` +and a `k × n` `F.Vt`, and `full=true` gives `m × m` and `n × n`. `F.S` has length +`k` and is real-valued for both real and complex input. + +Destructuring an `SVD` yields `(U, S, V)` — the adjoint of `F.Vt`, not `F.Vt` +itself. `F.V` is a lazy `Adjoint` wrapper, so operating on it falls back to +scalar indexing until `adjoint(::NDArray)` is implemented; prefer `F.Vt`. + +The backend only factors tall or square matrices (`m >= n`). A wide input +throws `ArgumentError` rather than hitting the C++ `m >= n` assert. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. +""" +function LinearAlgebra.svd(a::NDArray{T,2}; full::Bool=false) where {T<:_SVD_ACCEPTED} + return LinearAlgebra.SVD(_svd(checked_promote_arr(svd, a, _svd_eltype(T)), full)...) end -function svd(a::NDArray, full_matrices::Bool=true) - throw(ArgumentError("array type $(eltype(a)) is unsupported in svd")) +function LinearAlgebra.svd(a::NDArray{<:Any,2}; full::Bool=false) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in svd")) end -function qr(a::NDArray{<:_QR_ACCEPTED}) - A = eltype(a) - O = _qr_eltype(A) - A <: _QR_PROMOTABLE && assertpromotion(qr, A, O) - return _qr_check_dims(unchecked_promote_arr(a, O)) +""" + NDArrayQR <: LinearAlgebra.Factorization + +QR factorization of an `NDArray`, holding the explicit factors `Q` and `R` with +`A ≈ Q * R`. + +The backend returns materialized factors rather than the packed Householder +representation Base uses, so this is a distinct type from `LinearAlgebra.QR` and +`QRCompactWY`. It destructures as `(Q, R)` and exposes `F.Q` and `F.R`. +""" +struct NDArrayQR{T,M<:NDArray{T,2}} <: LinearAlgebra.Factorization{T} + Q::M + R::M +end + +Base.iterate(F::NDArrayQR) = (F.Q, Val(:R)) +Base.iterate(F::NDArrayQR, ::Val{:R}) = (F.R, Val(:done)) +Base.iterate(::NDArrayQR, ::Val{:done}) = nothing +Base.size(F::NDArrayQR) = (size(F.Q, 1), size(F.R, 2)) +Base.size(F::NDArrayQR, d::Integer) = d == 1 ? size(F.Q, 1) : size(F.R, d) + +function Base.show(io::IO, ::MIME"text/plain", F::NDArrayQR) + summary(io, F) + println(io) + println(io, "Q factor: ", summary(F.Q)) + print(io, "R factor: ", summary(F.R)) + return nothing +end + +""" + LinearAlgebra.qr(A::NDArray{T,2}) -> NDArrayQR + +Reduced QR factorization of `A`, with `A ≈ F.Q * F.R`. For an `m × n` input and +`k = min(m, n)`, `F.Q` is `m × k` and `F.R` is `k × n`. + +Large matrices automatically use cuSolverMp when available with multiple active +GPUs, including tall and wide inputs. + +See [`NDArrayQR`](@ref) for why this is not a `LinearAlgebra.QRCompactWY`. + +Accepted element types are `Float32`, `Float64`, `ComplexF32`, and `ComplexF64`. +Integer and `Bool` inputs are converted to `Float64`. As everywhere else in +the package, that conversion needs `@allowpromotion` only when it widens the +element type, so `Int64` and `UInt64` pass through silently. +""" +function LinearAlgebra.qr(a::NDArray{T,2}) where {T<:_QR_ACCEPTED} + return NDArrayQR(_qr(checked_promote_arr(qr, a, _qr_eltype(T)))...) end -function qr(a::NDArray) - throw(ArgumentError("array type $(eltype(a)) is unsupported in qr")) +function LinearAlgebra.qr(a::NDArray{<:Any,2}) + return throw(ArgumentError("array type $(eltype(a)) is unsupported in qr")) end diff --git a/src/ndarray/mapreduce.jl b/src/ndarray/mapreduce.jl new file mode 100644 index 000000000..409e11f09 --- /dev/null +++ b/src/ndarray/mapreduce.jl @@ -0,0 +1,157 @@ +# Mapping and reduction policies are separate: sum/prod widen small integers, +# whereas mapreduce with +/* uses the ordinary scalar operators. +struct NoReductionInit end +const _MR_ADD = Union{typeof(+),typeof(Base.add_sum)} +const _MR_MUL = Union{typeof(*),typeof(Base.mul_prod)} +const _MR_EXTREMA = Union{typeof(min),typeof(max)} +const _MR_OP = Union{_MR_ADD,_MR_MUL,_MR_EXTREMA} +const _MR_DIMS = Union{Integer,Tuple} + +_mr_operator(op::_MR_OP) = op +_mr_operator(op) = throw(ArgumentError("mapreduce supports only +, *, min, and max")) +_mr_combine(::_MR_ADD) = (+) +_mr_combine(::_MR_MUL) = (*) +_mr_combine(op::_MR_EXTREMA) = op +_mr_redop(::_MR_ADD, ::Type) = MAPREDUCE_ADD +_mr_redop(::_MR_MUL, ::Type) = MAPREDUCE_MUL +_mr_redop(::typeof(min), ::Type) = MAPREDUCE_MIN +_mr_redop(::typeof(max), ::Type) = MAPREDUCE_MAX + +_mr_storage(::_MR_OP, ::Type{T}) where {T} = T +# Legate's bitwise reducers do not provide Bool specializations. Product and +# extrema on 0/1 bytes preserve the Boolean operations without custom reducers. +_mr_storage(::Union{_MR_MUL,_MR_EXTREMA}, ::Type{Bool}) = UInt8 +_mr_storage(::_MR_EXTREMA, ::Type{Float32}) = UInt32 +_mr_storage(::_MR_EXTREMA, ::Type{Float64}) = UInt64 +_mr_encode(::_MR_OP, x) = x +_mr_encode(::Union{_MR_MUL,_MR_EXTREMA}, x::Bool) = UInt8(x) +_mr_nan_key(::typeof(min), ::Type{U}) where {U} = zero(U) +_mr_nan_key(::typeof(max), ::Type{U}) where {U} = typemax(U) +@inline function _mr_encode(op::_MR_EXTREMA, x::T) where {T<:Union{Float32,Float64}} + U = _mr_storage(op, T) + bits = reinterpret(U, x) + sign = one(U) << (8sizeof(U) - 1) + return isnan(x) ? _mr_nan_key(op, U) : (bits & sign == 0 ? bits ⊻ sign : ~bits) +end +_mr_decode(::_MR_OP, ::Type{T}, x) where {T} = x +_mr_decode(::Union{_MR_MUL,_MR_EXTREMA}, ::Type{Bool}, x) = !iszero(x) +@inline function _mr_decode(op::_MR_EXTREMA, ::Type{T}, x::U) where {T<:Union{Float32,Float64},U<:Unsigned} + x == _mr_nan_key(op, U) && return T(NaN) + sign = one(U) << (8sizeof(U) - 1) + return reinterpret(T, x & sign == 0 ? ~x : x ⊻ sign) +end + +_mr_identity(::_MR_ADD, ::Type{T}) where {T} = zero(T) +# -0 is neutral even when all full-reduction inputs are negative zero. +_mr_identity(::_MR_ADD, ::Type{T}) where {T<:AbstractFloat} = -zero(T) +_mr_identity(::_MR_ADD, ::Type{Complex{T}}) where {T} = Complex{T}(-zero(T), -zero(T)) +_mr_identity(::_MR_MUL, ::Type{T}) where {T} = one(T) +_mr_identity(::typeof(min), ::Type{T}) where {T} = typemax(T) +_mr_identity(::typeof(max), ::Type{T}) where {T} = typemin(T) + +function _mr_dims(shape::Dims{N}, ::Colon) where {N} + return ntuple(_ -> true, max(N, 1)), () +end +function _mr_dims(shape::Dims{N}, dims) where {N} + region = dims isa Integer ? (dims,) : dims + region isa Tuple || throw(ArgumentError("dims must be :, an integer, or a tuple of integers")) + Base.reduced_indices(map(Base.OneTo, shape), region) # Base's validation, including redundant axes + mask = ntuple(d -> d in region, max(N, 1)) + return mask, ntuple(d -> mask[d] ? 1 : shape[d], N) +end + +_mr_scalar_capture(::Type{T}) where {T<:SUPPORTED_ARRAY_TYPES} = true +_mr_scalar_capture(::Type{T}) where {T} = isbitstype(T) && all(_mr_scalar_capture, fieldtypes(T)) && !isprimitivetype(T) +struct MapReduceConvert{T} end +(::MapReduceConvert{T})(x) where {T} = T(x) +_mr_callable(f) = f +_mr_callable(::Type{T}) where {T<:SUPPORTED_ARRAY_TYPES} = MapReduceConvert{T}() +function _mr_mapped_type(f::F, ::Type{T}) where {F,T} + _mr_scalar_capture(F) || throw(ArgumentError("mapreduce requires an isbits callable with scalar captures; captured arrays and pointers are unsupported")) + M = Base.promote_op(f, T) + isconcretetype(M) && M <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError("mapreduce mapping must infer a supported scalar result type; inferred $M")) + return M +end + +_mr_supported_accumulator(op, ::Type{R}) where {R} = R +_mr_supported_accumulator(::_MR_MUL, ::Type{ComplexF64}) = + throw(ArgumentError("ComplexF64 product accumulators are unsupported: Legate has no corresponding built-in reducer")) + +function _mr_accumulator(op::_MR_OP, ::Type{M}) where {M} + op isa _MR_EXTREMA && M <: Complex && throw(ArgumentError("min/max mapreduce does not support complex mapped values")) + R = Base.promote_op(Base.reduce_first, typeof(op), M) + isconcretetype(R) && R <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError("unsupported mapreduce accumulator type $R")) + return _mr_supported_accumulator(op, R) +end +_mr_accumulator(op, M, ::NoReductionInit, dims::_MR_DIMS) = _mr_accumulator(op, M) +_mr_accumulator(op, M, init, ::Colon) = _mr_accumulator(op, M) +function _mr_accumulator(op, M, init::I, dims::_MR_DIMS) where {I} + # Narrowing after each update is not an associative distributed reduction. + R = _mr_accumulator(op, M) + op isa _MR_EXTREMA && I <: Complex && throw(ArgumentError("min/max mapreduce does not support complex init")) + Base.promote_op(_mr_combine(op), I, R) === I || throw(ArgumentError("dimensional mapreduce requires init's type to hold the accumulator without narrowing; use init::$R")) + return _mr_supported_accumulator(op, I) +end + +_mr_finish(op, x, ::NoReductionInit) = x +_mr_finish(op, x, init) = op(init, x) +_mr_output_type(op, R, ::NoReductionInit, dims::_MR_DIMS) = R +_mr_output_type(op, R, init, ::Colon) = Base.promote_op(_mr_combine(op), typeof(init), R) +_mr_output_type(op, R, init, dims::_MR_DIMS) = typeof(init) +_mr_output_type(op, R, ::NoReductionInit, ::Colon) = R + +_mr_empty(f, op, T, M, ::NoReductionInit, ::Colon) = Base.mapreduce_empty(f, op, T) +_mr_empty(f, op, T, M, init, dims) = init +_mr_empty(f, op::_MR_ADD, T, M, ::NoReductionInit, dims::_MR_DIMS) = zero(_mr_accumulator(op, M)) +_mr_empty(f, op::_MR_MUL, T, M, ::NoReductionInit, dims::_MR_DIMS) = one(_mr_accumulator(op, M)) +# Base rejects empty reduced axes before scalar reduction dispatch (which can +# throw MethodError on Julia 1.10). abs/abs2 maxima have a separate zero seed. +_mr_empty(f, op::_MR_EXTREMA, T, M, ::NoReductionInit, dims::_MR_DIMS) = + throw(ArgumentError("reducing over an empty collection is not allowed")) +_mr_empty(f::Union{typeof(abs),typeof(abs2)}, op::typeof(max), T, M, ::NoReductionInit, dims::_MR_DIMS) = + Base.mapreduce_empty(f, op, T) + +""" + mapreduce(f, op, A::NDArray; dims=:, init) + +Fuse a scalar mapping with a distributed GPU reduction. Supported operators are +`+`, `*`, `min`, and `max`. Full reductions return an `CNScalar`; explicit +dimensions retain singleton axes. `sum(f, A)` and `prod(f, A)` use Base's integer +widening rules, subject to `allowpromotion`. + +Requires an active GPU target and a type-stable GPU-compilable callable with only +isbits scalar captures. One input array is supported; complex results support +only addition/product; ComplexF64 product accumulators are unsupported by Legate. +Narrowing dimensional `init` types are unsupported. +Floating-point results may differ in rounding with partitioning; extrema preserve +NaNs and signed zeros, but not NaN payloads. See the mapped-reductions documentation. +""" +function Base.mapreduce(f::F, op::OP, A::NDArray{T}; dims=:, init=NoReductionInit()) where {F,OP,T} + op = _mr_operator(op) + input_shape = size(A) + mask, shape = _mr_dims(input_shape, dims) + _has_gpu_target() || throw(ArgumentError("mapped reductions currently require a Legate GPU target")) + mapper = _mr_callable(f) + M = _mr_mapped_type(mapper, T) + init isa NoReductionInit || (isbitstype(typeof(init)) && init isa SUPPORTED_ARRAY_TYPES) || throw(ArgumentError("init must be a supported scalar number")) + R = _mr_accumulator(op, M, init, dims) + O = _mr_output_type(op, R, init, dims) + isconcretetype(O) && O <: SUPPORTED_ARRAY_TYPES || throw(ArgumentError("unsupported mapreduce output type $O")) + is_wider_type(M, T) && assertpromotion(f, T, M) + is_wider_type(R, M) && assertpromotion(op, M, R) + is_wider_type(O, R) && assertpromotion(op, R, O) + nreduce = prod(d -> mask[d] ? input_shape[d] : 1, 1:ndims(A); init=1) + if nreduce == 0 + value = _mr_empty(f, op, T, M, init, dims) + return _scalar_result(nda_full_array(shape, value)) + end + any(iszero, input_shape) && return _scalar_result(nda_zeros_array(shape, O)) + return _scalar_result(_mr_launch(mapper, op, A, R, O, mask, shape, init, dims, nreduce == 1)) +end + +Base.mapreduce(f, op, A::NDArray, B::AbstractArray, rest::AbstractArray...; kwargs...) = + throw(ArgumentError("mapreduce currently supports one input NDArray")) + +for (name, op) in ((:sum, Base.add_sum), (:prod, Base.mul_prod), (:minimum, min), (:maximum, max)) + @eval Base.$name(f, A::NDArray; kwargs...) = mapreduce(f, $op, A; kwargs...) +end diff --git a/src/ndarray/ndarray.jl b/src/ndarray/ndarray.jl index 7dd0b2779..9cfaf2d55 100644 --- a/src/ndarray/ndarray.jl +++ b/src/ndarray/ndarray.jl @@ -18,7 +18,9 @@ * Nader Rahhal =# -export unwrap +export squeeze + +# See TODO.md (Base / LinearAlgebra sections) for AbstractArray and LA gaps. @doc""" cuNumeric.transpose(arr::NDArray) @@ -29,39 +31,79 @@ function transpose(arr::NDArray) return nda_transpose(arr) end -@doc""" - cuNumeric.eye([T=Float32,] rows::Int) +""" + Base.permutedims(arr::NDArray, perm) -Create a 2D identity `NDArray` of size `rows × rows` with element type `T`. -The default type is Float32 if not specified. +Permute the dimensions of `arr` according to `perm`, a 1-based permutation of +`1:ndims(arr)`. Rank is preserved. See also [`transpose`](@ref). """ -function eye(::Type{T}, rows::Int) where {T} - return nda_eye(Int32(rows), T) -end -function eye(rows::Int) - return eye(DEFAULT_FLOAT, rows) +function Base.permutedims(arr::NDArray{T,N}, perm) where {T,N} + length(perm) == N || throw( + ArgumentError("permutation length $(length(perm)) does not match ndims = $N") + ) + p = ntuple(i -> Int(perm[i]), Val(N)) + used = Base.falses(N) + axes = Vector{Int32}(undef, N) + for i in 1:N + d = p[i] + (1 <= d <= N) || throw(ArgumentError("permutation index $d is out of range for ndims = $N")) + used[d] && throw(ArgumentError("permutation $perm is not a permutation of 1:$N")) + used[d] = true + axes[i] = Int32(d - 1) + end + return nda_transpose_axes(arr, axes) end -@doc""" - cuNumeric.trace(arr::NDArray; offset=0, a1=0, a2=1) +Base.permutedims(arr::NDArray{<:Any,2}) = permutedims(arr, (2, 1)) -Compute the trace (sum of a diagonal) of the `NDArray`. -The accumulator type follows promotions of other reductions like 'sum'. """ -function trace(arr::NDArray{T}; offset::Int=0, a1::Int=0, a2::Int=1) where {T} - T_OUT = Base.promote_op(Base.sum, Vector{T}) - return nda_trace(arr, Int32(offset), Int32(a1), Int32(a2), T_OUT) -end + squeeze(arr::NDArray) + squeeze(arr::NDArray, dims) -@doc""" - cuNumeric.diag(arr::NDArray; k=0) +Drop size-1 dimensions of `arr`. With no `dims`, every size-1 axis is removed. +With `dims` (a 1-based integer or collection), only those axes are removed and +each must have size 1. -Extract the k-th diagonal from a 2D `NDArray`. +See also `Base.dropdims`. """ -function diag(arr::NDArray; k::Int=0) - return nda_diag(arr, Int32(k)) +function squeeze(arr::NDArray) + any(==(1), size(arr)) || return arr + return nda_squeeze(arr, Int32[]) +end + +function squeeze(arr::NDArray, dims) + axes = _squeeze_axes(arr, dims) + isempty(axes) && return arr + return nda_squeeze(arr, axes) +end + +function Base.dropdims(arr::NDArray; dims) + return squeeze(arr, dims) +end + +function _squeeze_axes(arr::NDArray, dims) + nd = ndims(arr) + axes = Int32[] + seen = Set{Int}() + for d in _as_dims(dims) + ax = Int(d) + (1 <= ax <= nd) || + throw(ArgumentError("dimension $ax is out of range for $(nd)-d array")) + ax in seen && throw(ArgumentError("duplicate dimension $ax")) + push!(seen, ax) + size(arr, ax) == 1 || throw( + DimensionMismatch( + "cannot drop dimension $ax of size $(size(arr, ax)); expected size 1" + ), + ) + push!(axes, Int32(ax - 1)) + end + return axes end +_as_dims(d::Integer) = (Int(d),) +_as_dims(dims) = Tuple(Int(d) for d in dims) + @doc""" cuNumeric.ravel(arr::NDArray) @@ -71,15 +113,6 @@ function ravel(arr::NDArray) return nda_ravel(arr) end -@doc""" - cuNumeric.unique(arr::NDArray) - -Return a new `NDArray` containing the unique elements of the input `arr`. -""" -function unique(arr::NDArray) - return nda_unique(arr) -end - @doc""" Base.copy(arr::NDArray) @@ -111,7 +144,39 @@ copyto!(a, b); a[1,1] ``` """ -Base.copyto!(arr::NDArray{T,N}, other::NDArray{T,N}) where {T,N} = nda_assign(arr, other) +@inline function Base.copyto!(arr::NDArray{T,N}, other::NDArray{T,N}) where {T,N} + nda_assign(arr, other) + return arr +end + +function Base.copyto!(dest::NDArray{T,N}, src::Array{T,N}) where {T,N} + attached = _nda_from_julia_array(src) + try + GC.@preserve src attached begin + copyto!(dest, attached) + issue_execution_fence(; block=true) + end + finally + destroy!(attached) + end + return dest +end + +# Borrow a host vector only until the copy completes; avoid the constructor's +# separate owned copy and fence. Struct packing keeps its existing path. +function Base.copyto!(dest::NDArray{T,1}, src::Vector{T}) where {T<:SUPPORTED_ARRAY_TYPES} + size(dest) == size(src) || throw(DimensionMismatch("source and destination sizes differ")) + isempty(src) && return dest + GC.@preserve src begin + attached = nda_attach_external(src) + GC.@preserve attached dest begin + copyto!(dest, attached) + issue_execution_fence(; block=true) + end + destroy!(attached) + end + return dest +end @doc""" as_type(arr::NDArray, t::Type{T}) where {T} @@ -143,53 +208,143 @@ end # get_ptr is a blocking call that grabs the physical store # we have not tested across multiple processes or devices yet -function (::Type{<:Array{A}})(arr::NDArray{B,0}) where {A,B} - out = Array{A}(undef) +# NDArray-specific overrides of Core's AbstractArray constructors (NDArray <: +# AbstractArray): exact `Array{T}` / `Array{T,N}` / `Array` signatures so we win +# over `Array{T,N}(::AbstractArray)` (which would scalar-index). Bulk path uses +# `_copy_to_julia_array`; 1-d has specialized same-type and converting paths. +struct StructConstructor{T} end +@inline (::StructConstructor{T})(fields...) where {T} = T(fields...) +@inline (::StructConstructor{T})(fields...) where {T<:NamedTuple} = T(fields) + +struct StructField{I} end +@inline (::StructField{I})(value) where {I} = getfield(value, I) + +struct StructIdentity end +@inline (::StructIdentity)(value) = value + +function (::Type{Array{T}})(arr::NDArray{S,0}) where {T,S} + out = Array{T,0}(undef) allowscalar() do - return out[] = convert(A, arr[]) + return out[] = convert(T, arr[]) end return out end -function (::Type{<:Array{A}})(arr::NDArray{B,1}) where {A,B} - return make_array(A, Ptr{A}(get_ptr(arr)), size(arr)) +function (::Type{Array{T}})(arr::NDArray{T,1}) where {T} + # Legate.get_ptr requests a write accessor. A singleton can be backed by + # a read-only future, so copy into attached host storage before mapping it. + return _copy_to_julia_array(arr) +end + +function (::Type{Array{T}})(arr::NDArray{S,1}) where {T,S} + return copyto!(Vector{T}(undef, length(arr)), _copy_to_julia_array(arr)) end # Copy logically into Julia's column-major storage. # Legate may map an NDArray in C or Fortran order. -function _copy_to_julia_array(arr::NDArray{T,N}) where {T,N} +function _copy_to_julia_array(arr::NDArray{T,0}) where {T} + _struct_storage_type(T) || return _copy_to_julia_array_impl(arr) + # Struct fields are projected by the fused kernel, which needs a rank. + vector = nda_reshape_array(arr, (1,)) + try + return Base.reshape(_copy_to_julia_array(vector), ()) + finally + destroy!(vector) + end +end + +_copy_to_julia_array(arr::NDArray) = _copy_to_julia_array_impl(arr) + +function _copy_to_julia_array_impl(arr::NDArray{T,N}) where {T,N} + if _struct_storage_type(T) + out = Array{T}(undef, size(arr)) + isempty(out) && return out + _assert_struct_kernel("Host transfer", T) + fields = ntuple(fieldcount(T)) do i + projected = StructField{i}().(arr) + try + Array(projected) + finally + destroy!(projected) + end + end + for i in eachindex(out) + out[i] = StructConstructor{T}()(ntuple(j -> fields[j][i], fieldcount(T))...) + end + return out + end out = Array{T}(undef, size(arr)) + isempty(out) && return out store = Legate.attach_external_col_major(out) ptr = cuNumeric.nda_store_to_ndarray(store.handle) finalize(store.handle) attached = NDArray(ptr, T, Val(N), out) - copyto!(attached, arr) - get_ptr(attached) # Block until the copy into `out` completes. + try + copyto!(attached, arr) + get_ptr(attached) # Block until the copy into `out` completes. + finally + destroy!(attached) + end return out end -function (::Type{<:Array{A}})(arr::NDArray{B}) where {A,B} +function (::Type{Array{T}})(arr::NDArray{S,N}) where {T,S,N} out = _copy_to_julia_array(arr) - return A === B ? out : copyto!(Array{A}(undef, size(arr)), out) + return T === S ? out : copyto!(Array{T}(undef, size(arr)), out) end -function (::Type{<:Array})(arr::NDArray{B}) where {B} - return Array{B}(arr) -end +(::Type{Array{T,N}})(arr::NDArray{S,N}) where {T,S,N} = Array{T}(arr) + +(::Type{Array})(arr::NDArray{T,N}) where {T,N} = Array{T}(arr) # conversion from Base Julia array to NDArray # Julia Arrays are column-major; Legate stores are row-major. For N>=2 we # materialize a C-ordered buffer via permutedims, attach it with the original # shape, and keep that buffer as `parent` for lifetime. +function _nda_from_julia_struct_array(arr::Array{T,N}) where {T,N} + isempty(arr) && return nda_empty_array(size(arr), T) + _assert_struct_kernel("Construction from a host Array", T) + # `map` keeps Bool fields as Array{Bool}; broadcasting would build a BitArray. + fields = ntuple(i -> NDArray(map(StructField{i}(), arr)), fieldcount(T)) + try + return StructConstructor{T}().(fields...) + finally + foreach(destroy!, fields) + end +end + +# A 0-d store cannot run the fused kernel, but can be filled with its element. +function _nda_from_julia_struct_array(arr::Array{T,0}) where {T} + out = nda_empty_array((), T) + nda_fill_struct_array(out, arr[]) + return out +end + function _nda_from_julia_array(arr::Array{T,0}) where {T} + _struct_storage_type(T) && return _nda_from_julia_struct_array(arr) return cuNumeric.nda_attach_external(arr) end function _nda_from_julia_array(arr::Array{T,1}) where {T} - return cuNumeric.nda_attach_external(arr) + _struct_storage_type(T) && return _nda_from_julia_struct_array(arr) + # Prototype: the attachment borrows Julia memory only for this copy. + # Preserve the source through completion, not just task submission. + GC.@preserve arr begin + attached = cuNumeric.nda_attach_external(arr) + try + out = copy(attached) + GC.@preserve attached out begin + issue_execution_fence(; block=true) + end + return out + finally + destroy!(attached) + end + end end function _nda_from_julia_array(arr::Array{T,N}) where {T,N} + _struct_storage_type(T) && return _nda_from_julia_struct_array(arr) tmp = collect(permutedims(arr, reverse(ntuple(identity, Val(N))))) return cuNumeric.nda_attach_external(tmp; shape=size(arr)) end @@ -243,7 +398,7 @@ dim(::NDArray{T,N}) where {T,N} = N::Int Base.ndims(::NDArray{T,N}) where {T,N} = N::Int @doc""" Base.size(arr::NDArray) - Base.size(arr::NDArray, dim::Int) + Base.size(arr::NDArray, dim::Integer) Return the size of the given `NDArray`. @@ -258,12 +413,13 @@ size(arr, 2) ``` """ Base.size(arr::NDArray{<:Any,N}) where {N} = cuNumeric.shape(arr) -Base.size(arr::NDArray, dim::Int) = Base.size(arr)[dim] +Base.size(arr::NDArray, dim::Integer) = dim <= ndims(arr) ? size(arr)[dim] : 1 Base.isempty(arr::NDArray) = any(==(0), size(arr)) +Base.length(arr::NDArray) = prod(size(arr)) @doc""" - Base.firstindex(arr::NDArray, dim::Int) - Base.lastindex(arr::NDArray, dim::Int) + Base.firstindex(arr::NDArray, dim::Integer) + Base.lastindex(arr::NDArray, dim::Integer) Base.lastindex(arr::NDArray) Provide the first and last valid indices along a given dimension `dim` for `NDArray`. @@ -276,44 +432,40 @@ lastindex(arr, 2) lastindex(arr) ``` """ -Base.firstindex(arr::NDArray, dim::Int) = 1 -Base.lastindex(arr::NDArray, dim::Int) = Base.size(arr, dim) -Base.lastindex(arr::NDArray) = Base.size(arr, 1) +Base.firstindex(arr::NDArray, dim::Integer) = 1 +Base.lastindex(arr::NDArray, dim::Integer) = size(arr, dim) +Base.lastindex(arr::NDArray) = length(arr) +Base.IndexStyle(::Type{<:NDArray}) = IndexCartesian() Base.axes(arr::NDArray) = Base.OneTo.(size(arr)) Base.view(arr::NDArray, inds...) = arr[inds...] # NDArray slices are views by default. -Base.IndexStyle(::NDArray) = IndexCartesian() - -function Base.show(io::IO, arr::NDArray{T,0}) where {T} - allowscalar() do - return print(io, "NDArray{$(T),0}(", repr(arr[]), ")") - end +# All-colon getindex copies, but view/dotview must retain the original store. +# A separate handle is required because @accelerate frees temporary views. +function Base.view(arr::NDArray{T,N}, ::Vararg{Colon,N}) where {T,N} + return nda_get_slice(arr, slice_array((nothing, nothing))) end +Base.view(arr::NDArray{T,0}) where {T} = nda_reshape_array(arr, ()) -function Base.show(io::IO, ::MIME"text/plain", arr::NDArray{T,0}) where {T} - println(io, "0-dimensional NDArray{$(T),0}") - allowscalar() do - return print(io, arr[]) - end +function Base.show(io::IO, arr::NDArray{T,0}) where {T} + print(io, summary(arr), "(") + @allowscalar show(io, arr[]) + return print(io, ")") end -function Base.show(io::IO, arr::NDArray{T,N}) where {T,N} - return print(io, "NDArray{$(T),$(N)} with size ", size(arr)) +# Used by print(arr), println(arr), and nested displays +function Base.show(io::IO, arr::NDArray) + return show(io, Array(arr)) end -function Base.show(io::IO, ::MIME"text/plain", arr::NDArray{T,N}) where {T,N} - println(io, "NDArray{$(T),$(N)} with size ", size(arr)) - return Base.print_array(io, Array(arr)) -end +# Used for full REPL display +function Base.show(io::IO, ::MIME"text/plain", arr::NDArray) + summary(io, arr) -function Base.print(arr::NDArray{T}) where {T} - return Base.show(stdout, arr) -end + isempty(arr) && return nothing -function Base.println(arr::NDArray{T}) where {T} - Base.show(stdout, arr) - return print("\n") + println(io, ":") + return Base.print_array(io, Array(arr)) end #### ARRAY INDEXING AND SLICES #### @@ -332,8 +484,11 @@ end Overloads `Base.getindex` and `Base.setindex!` to support multidimensional indexing and slicing on `cuNumeric.NDArray`s. -Slicing supports combinations of `Int`, `UnitRange`, and `Colon()` for selecting ranges of rows and columns. -The use of all colons (`arr[:]`, `arr[:, :]`, etc.) returns a new Julia `Array` containing a copy of the data. +Slicing supports combinations of `Integer`, `UnitRange`, and `Colon()` for selecting ranges of rows and columns. +Using one colon per dimension (`v[:]`, `A[:, :]`, etc.) returns an `NDArray` +copy. `view` and dotted assignment share the original storage instead. +Linear range/colon indexing of multidimensional arrays is unsupported; use one +index per dimension. Assignment also supports: - Writing NDArray slices to NDArray regions @@ -349,10 +504,13 @@ Array(A) ``` """ ##### REGULAR ARRAY INDEXING #### -function Base.getindex(arr::NDArray{T,N}, idxs::Vararg{Int,N}) where {T<:SUPPORTED_NUMERIC_TYPES,N} +@inline function Base.getindex( + arr::NDArray{T,N}, idxs::Vararg{Integer,N} +) where {T<:SUPPORTED_NUMERIC_TYPES,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("getindex") acc = NDArrayAccessor{T,N}() - return read(acc, arr.ptr, to_cpp_index(idxs)) + return read(acc, arr.ptr, to_cpp_index(Int.(idxs))) end function Base.getindex(arr::NDArray{T,0}) where {T<:SUPPORTED_NUMERIC_TYPES} @@ -362,10 +520,11 @@ function Base.getindex(arr::NDArray{T,0}) where {T<:SUPPORTED_NUMERIC_TYPES} return read(acc, arr.ptr, zero_index) end -function Base.getindex(arr::NDArray{Bool,N}, idxs::Vararg{Int,N}) where {N} +@inline function Base.getindex(arr::NDArray{Bool,N}, idxs::Vararg{Integer,N}) where {N} + @boundscheck checkbounds(arr, idxs...) assertscalar("getindex") acc = NDArrayAccessor{CxxWrap.CxxBool,N}() - return read(acc, arr.ptr, to_cpp_index(idxs)) + return read(acc, arr.ptr, to_cpp_index(Int.(idxs))) end function Base.getindex(arr::NDArray{Bool,0}) @@ -376,17 +535,26 @@ function Base.getindex(arr::NDArray{Bool,0}) end #! TODO SUPPORT CONVERSION OF VALUES -function Base.setindex!(arr::NDArray{T,N}, value::T, idxs::Vararg{Int,N}) where {T,N} +@inline function Base.setindex!( + arr::NDArray{T,N}, value::T, idxs::Vararg{Integer,N} +) where {T,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("setindex!") return _setindex!(Val{N}(), arr, value, idxs...) end -function Base.setindex!(arr::NDArray{Complex{T},N}, value::T, idxs::Vararg{Int,N}) where {T,N} +@inline function Base.setindex!( + arr::NDArray{Complex{T},N}, value::T, idxs::Vararg{Integer,N} +) where {T,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("setindex!") return _setindex!(Val{N}(), arr, Complex{T}(value), idxs...) end -function Base.setindex!(arr::NDArray{T,N}, value, idxs::Vararg{Int,N}) where {T,N} +@inline function Base.setindex!( + arr::NDArray{T,N}, value, idxs::Vararg{Integer,N} +) where {T,N} + @boundscheck checkbounds(arr, idxs...) assertscalar("setindex!") return _setindex!(Val{N}(), arr, convert(T, value), idxs...) end @@ -402,122 +570,302 @@ function _setindex!(::Val{0}, arr::NDArray{Bool,0}, value::Bool) end function _setindex!( - ::Val{N}, arr::NDArray{T,N}, value::T, idxs::Vararg{Int,N} + ::Val{N}, arr::NDArray{T,N}, value::T, idxs::Vararg{Integer,N} ) where {T<:SUPPORTED_NUMERIC_TYPES,N} acc = NDArrayAccessor{T,N}() - return write(acc, arr.ptr, to_cpp_index(idxs), value) + return write(acc, arr.ptr, to_cpp_index(Int.(idxs)), value) end -function _setindex!(::Val{N}, arr::NDArray{Bool,N}, value::Bool, idxs::Vararg{Int,N}) where {N} +function _setindex!( + ::Val{N}, arr::NDArray{Bool,N}, value::Bool, idxs::Vararg{Integer,N} +) where {N} acc = NDArrayAccessor{CxxWrap.CxxBool,N}() - return write(acc, arr.ptr, to_cpp_index(idxs), value) + return write(acc, arr.ptr, to_cpp_index(Int.(idxs)), value) +end + +# Struct elements have no typed accessor; move one element through a slice. +@inline _struct_element_slice(arr::NDArray{T,N}, idxs::Vararg{Integer,N}) where {T,N} = + nda_get_slice(arr, slice_array(map(_zero_based_index, idxs)...)) + +function Base.getindex(arr::NDArray{T,0}) where {T} + _struct_storage_type(T) || throw(Base.CanonicalIndexError("getindex", typeof(arr))) + assertscalar("getindex") + return _copy_to_julia_array(arr)[] +end + +@inline function Base.getindex(arr::NDArray{T,N}, idxs::Vararg{Integer,N}) where {T,N} + _struct_storage_type(T) || throw(Base.CanonicalIndexError("getindex", typeof(arr))) + @boundscheck checkbounds(arr, idxs...) + assertscalar("getindex") + element = _struct_element_slice(arr, idxs...) + try + return only(_copy_to_julia_array(element)) + finally + destroy!(element) + end +end + +function _setindex!(::Val{N}, arr::NDArray{T,N}, value::T, idxs::Vararg{Integer,N}) where {T,N} + _struct_storage_type(T) || throw(Base.CanonicalIndexError("setindex!", typeof(arr))) + N == 0 && return nda_fill_struct_array(arr, value) + element = _struct_element_slice(arr, idxs...) + try + return nda_fill_struct_array(element, value) + finally + destroy!(element) + end end #### START OF SLICING #### -# LHS slices from `nda_get_slice` are invisible to `@analyze_lifetimes`; destroy +# LHS slices from `nda_get_slice` are invisible to `@accelerate`; destroy # the view handle after submitting the assign so they cannot pile up under Julia # GC (which sees each NDArray as ~pointer-sized). -function _setindex_slice!(lhs::NDArray, rhs::NDArray, slices) +function _setindex_slice!(lhs::NDArray{T}, rhs::NDArray, slices) where {T} s = nda_get_slice(lhs, slices) - copyto!(s, rhs) - destroy!(s) + try + # cuPyNumeric's assign broadcasts rhs to the slice shape; check first so a + # mismatch raises DimensionMismatch as in Base instead of writing silently. + Base.setindex_shape_check(rhs, size(s)...) + src = checked_promote_arr(rhs, T) + shaped = size(src) == size(s) ? src : reshape(src, size(s)) + copyto!(s, shaped) + shaped === src || destroy!(shaped) + src === rhs || destroy!(src) + finally + destroy!(s) + end return nothing end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Colon, j::Int64) - return _setindex_slice!(lhs, rhs, slice_array((0, Base.size(lhs, 1)), (j-1, j))) +@inline _zero_based_index(i::Integer) = (Int(i) - 1, Int(i)) +@inline _zero_based_range(i::AbstractUnitRange{<:Integer}) = (Int(first(i)) - 1, Int(last(i))) + +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, ::Colon, j::Integer +) where {T} + @boundscheck checkbounds(lhs, :, j) + return _setindex_slice!( + lhs, rhs, slice_array((0, size(lhs, 1)), _zero_based_index(j)) + ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Int64, j::Colon) - return _setindex_slice!(lhs, rhs, slice_array((i-1, i))) +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, i::Integer, ::Colon +) where {T} + @boundscheck checkbounds(lhs, i, :) + return _setindex_slice!(lhs, rhs, slice_array(_zero_based_index(i))) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::UnitRange, j::Colon) +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, i::AbstractUnitRange{<:Integer}, ::Colon +) where {T} + @boundscheck checkbounds(lhs, i, :) return _setindex_slice!( - lhs, rhs, slice_array((first(i) - 1, last(i)), (0, Base.size(lhs, 2))) + lhs, rhs, slice_array(_zero_based_range(i), (0, size(lhs, 2))) ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Colon, j::UnitRange) +@inline function Base.setindex!( + lhs::NDArray{T,2}, rhs::NDArray, ::Colon, j::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(lhs, :, j) return _setindex_slice!( - lhs, rhs, slice_array((0, Base.size(lhs, 1)), (first(j) - 1, last(j))) + lhs, rhs, slice_array((0, size(lhs, 1)), _zero_based_range(j)) ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::UnitRange, j::Int64) - return _setindex_slice!(lhs, rhs, slice_array((first(i) - 1, last(i)), (j-1, j))) -end - -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::Int64, j::UnitRange) - return _setindex_slice!(lhs, rhs, slice_array((i-1, i), (first(j) - 1, last(j)))) +@inline function Base.setindex!( + lhs::NDArray{T,2}, + rhs::NDArray, + i::AbstractUnitRange{<:Integer}, + j::Integer, +) where {T} + @boundscheck checkbounds(lhs, i, j) + return _setindex_slice!( + lhs, rhs, slice_array(_zero_based_range(i), _zero_based_index(j)) + ) end -function Base.setindex!(lhs::NDArray, rhs::NDArray, i::UnitRange, j::UnitRange) +@inline function Base.setindex!( + lhs::NDArray{T,2}, + rhs::NDArray, + i::Integer, + j::AbstractUnitRange{<:Integer}, +) where {T} + @boundscheck checkbounds(lhs, i, j) return _setindex_slice!( - lhs, rhs, slice_array((first(i) - 1, last(i)), (first(j) - 1, last(j))) + lhs, rhs, slice_array(_zero_based_index(i), _zero_based_range(j)) ) end -function Base.getindex(arr::NDArray, i::Colon, j::Int64) - return nda_get_slice(arr, slice_array((0, Base.size(arr, 1)), (j-1, j))) +@inline function Base.setindex!( + lhs::NDArray{T,2}, + rhs::NDArray, + i::AbstractUnitRange{<:Integer}, + j::AbstractUnitRange{<:Integer}, +) where {T} + @boundscheck checkbounds(lhs, i, j) + return _setindex_slice!( + lhs, rhs, slice_array(_zero_based_range(i), _zero_based_range(j)) + ) end -function Base.getindex(arr::NDArray, i::Int64, j::Colon) - return nda_get_slice(arr, slice_array((i-1, i))) +@inline function Base.setindex!( + lhs::NDArray{T,1}, rhs::NDArray, i::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(lhs, i) + return _setindex_slice!(lhs, rhs, slice_array(_zero_based_range(i))) end -function Base.getindex(arr::NDArray, i::UnitRange, j::Colon) +@inline function Base.getindex(arr::NDArray{T,2}, ::Colon, j::Integer) where {T} + @boundscheck checkbounds(arr, :, j) return nda_get_slice( - arr, slice_array((first(i) - 1, last(i)), (0, Base.size(arr, 2))) + arr, slice_array((0, size(arr, 1)), _zero_based_index(j)) ) end -function Base.getindex(arr::NDArray, i::Colon, j::UnitRange) +@inline function Base.getindex(arr::NDArray{T,2}, i::Integer, ::Colon) where {T} + @boundscheck checkbounds(arr, i, :) + return nda_get_slice(arr, slice_array(_zero_based_index(i))) +end + +@inline function Base.getindex( + arr::NDArray{T,2}, i::AbstractUnitRange{<:Integer}, ::Colon +) where {T} + @boundscheck checkbounds(arr, i, :) return nda_get_slice( - arr, slice_array((0, Base.size(arr, 1)), (first(j) - 1, last(j))) + arr, slice_array(_zero_based_range(i), (0, size(arr, 2))) ) end -function Base.getindex(arr::NDArray, i::UnitRange, j::Int64) - return nda_get_slice(arr, slice_array((first(i) - 1, last(i)), (j-1, j))) +@inline function Base.getindex( + arr::NDArray{T,2}, ::Colon, j::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(arr, :, j) + return nda_get_slice( + arr, slice_array((0, size(arr, 1)), _zero_based_range(j)) + ) end -function Base.getindex(arr::NDArray, i::Int64, j::UnitRange) - return nda_get_slice(arr, slice_array((i-1, i), (first(j) - 1, last(j)))) +@inline function Base.getindex( + arr::NDArray{T,2}, i::AbstractUnitRange{<:Integer}, j::Integer +) where {T} + @boundscheck checkbounds(arr, i, j) + return nda_get_slice( + arr, slice_array(_zero_based_range(i), _zero_based_index(j)) + ) end -function Base.getindex(arr::NDArray, i::UnitRange, j::UnitRange) +@inline function Base.getindex( + arr::NDArray{T,2}, i::Integer, j::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(arr, i, j) return nda_get_slice( - arr, slice_array((first(i) - 1, last(i)), (first(j) - 1, last(j))) + arr, slice_array(_zero_based_index(i), _zero_based_range(j)) ) end -function Base.getindex(arr::NDArray, i::UnitRange) +@inline function Base.getindex( + arr::NDArray{T,2}, + i::AbstractUnitRange{<:Integer}, + j::AbstractUnitRange{<:Integer}, +) where {T} + @boundscheck checkbounds(arr, i, j) return nda_get_slice( - arr, slice_array((first(i) - 1, last(i))) + arr, slice_array(_zero_based_range(i), _zero_based_range(j)) ) end -Base.getindex(arr::NDArray{T}, c::Vararg{Colon,N}) where {T,N} = Base.copy(arr) -function Base.setindex!(arr::NDArray{T}, rhs::NDArray{T}, c::Vararg{Colon,N}) where {T,N} +@inline function Base.getindex( + arr::NDArray{T,1}, i::AbstractUnitRange{<:Integer} +) where {T} + @boundscheck checkbounds(arr, i) + return nda_get_slice(arr, slice_array(_zero_based_range(i))) +end + +function Base.getindex(arr::NDArray, i::AbstractUnitRange{<:Integer}) + throw(ArgumentError( + "linear range indexing of multidimensional NDArrays is unsupported; " * + "use one index per dimension" + )) +end + +function Base.setindex!(arr::NDArray, rhs::NDArray, i::AbstractUnitRange{<:Integer}) + throw(ArgumentError( + "linear range assignment to multidimensional NDArrays is unsupported; " * + "use one index per dimension" + )) +end + +@inline function Base.getindex( + arr::NDArray{T,N}, c::Vararg{Colon,N} +) where {T,N} + return Base.copy(arr) +end + +function Base.getindex(arr::NDArray, ::Vararg{Colon}) + throw(ArgumentError( + "linear colon indexing of multidimensional NDArrays is unsupported; " * + "use one colon per dimension" + )) +end + +@inline function Base.setindex!( + arr::NDArray{T,N}, rhs::NDArray{T}, c::Vararg{Colon,N} +) where {T,N} return Base.copyto!(arr, rhs) end -function Base.setindex!(arr::NDArray{T,2}, val::T, i::Colon, j::Int64) where {T} - s = nda_get_slice(arr, slice_array((0, Base.size(arr, 1)), (j-1, j))) +function Base.setindex!(arr::NDArray{T}, rhs::NDArray{T}, ::Vararg{Colon}) where {T} + throw(ArgumentError( + "linear colon assignment to multidimensional NDArrays is unsupported; " * + "use one colon per dimension" + )) +end + +@inline function Base.setindex!( + arr::NDArray{T,2}, val::T, ::Colon, j::Integer +) where {T} + @boundscheck checkbounds(arr, :, j) + s = nda_get_slice( + arr, slice_array((0, size(arr, 1)), _zero_based_index(j)) + ) nda_fill_array(s, val) return destroy!(s) end -function Base.setindex!(arr::NDArray{T,2}, val::T, i::Int64, j::Colon) where {T} - s = nda_get_slice(arr, slice_array((i-1, i))) +@inline function Base.setindex!( + arr::NDArray{T,2}, val::T, i::Integer, ::Colon +) where {T} + @boundscheck checkbounds(arr, i, :) + s = nda_get_slice(arr, slice_array(_zero_based_index(i))) nda_fill_array(s, val) return destroy!(s) end -Base.fill!(arr::NDArray{T}, val::T) where {T} = nda_fill_array(arr, val) +@inline function Base.fill!(arr::NDArray{T}, val::SUPPORTED_ARRAY_TYPES) where {T} + nda_fill_array(arr, convert(T, val)) + return arr +end + +function Base.fill!(arr::NDArray{T}, val) where {T} + _struct_storage_type(T) || return invoke(fill!, Tuple{AbstractArray,Any}, arr, val) + nda_fill_struct_array(arr, convert(T, val)) + return arr +end #### INITIALIZATION OF NDARRAYS #### +@doc""" + NDArray{T}(undef, dims::Int...) + NDArray{T}(undef, dims::Dims) + +Allocate an uninitialized `NDArray{T}`. Every element must be assigned before it is read. +""" +NDArray{T}(::UndefInitializer, dims::Dims{N}) where {T<:SUPPORTED_ARRAY_TYPES,N} = + nda_empty_array(dims, T) +NDArray{T}(::UndefInitializer, dims::Int...) where {T<:SUPPORTED_ARRAY_TYPES} = + NDArray{T}(undef, dims) + @doc""" cuNumeric.fill(val::T, dims::Dims) cuNumeric.fill(val::T, dims::Int...) @@ -534,11 +882,16 @@ function fill(val::T, dims::Dims) where {T<:SUPPORTED_TYPES} return nda_full_array(dims, val) end -function fill(val::T, dims::Int...) where {T<:SUPPORTED_TYPES} +function fill(val::T, dims::Dims) where {T} + _struct_storage_type(T) || throw(MethodError(fill, (val, dims))) + return fill!(nda_empty_array(dims, T), val) +end + +function fill(val::T, dims::Int...) where {T} return fill(val, dims) end -function fill(val::T, dim::Int) where {T<:SUPPORTED_TYPES} +function fill(val::T, dim::Int) where {T} return fill(val, (dim,)) end @@ -654,47 +1007,6 @@ function ones() return ones(DEFAULT_FLOAT) end -@doc""" - cuNumeric.rand!(arr::NDArray{Float64}) - -Fill `arr` in-place with uniform random `Float64` values. -""" -Random.rand!(arr::NDArray{Float64}) = cuNumeric.nda_random(arr, 0) -function Random.rand!(arr::NDArray{T}) where {T} - return error("rand! only supports NDArray{Float64} for now. Cast with cuNumeric.as_type.") -end - -# Backend only generates Float64. Same-type path needs no cast; other floats -# convert then eagerly drop the Float64 source so it cannot leak until GC. -@doc""" - cuNumeric.rand([T=Float32,] dims::Int...) - cuNumeric.rand([T=Float32,] dims::Tuple) - -Create a new `NDArray` filled with uniform random values. - -The backend currently supports only `Float64` draws. Other floating types are -converted automatically. - -# Examples -```@repl -cuNumeric.rand(2, 2) -cuNumeric.rand((4, 1)) -A = cuNumeric.zeros(Float64, 2, 2); cuNumeric.rand!(A) -``` -""" -rand(::Type{Float64}, dims::Dims) = cuNumeric.nda_random_array(dims) - -function rand(::Type{T}, dims::Dims) where {T<:AbstractFloat} - arrfp64 = cuNumeric.nda_random_array(dims) - arr = cuNumeric.as_type(arrfp64, T) - destroy!(arrfp64) - return arr -end - -rand(::Type{T}, dims::Int...) where {T<:AbstractFloat} = cuNumeric.rand(T, dims) -rand(dims::Dims) = cuNumeric.rand(DEFAULT_FLOAT, dims) -rand(dims::Int...) = cuNumeric.rand(DEFAULT_FLOAT, dims) - #### OPERATIONS #### @doc""" reshape(arr::NDArray, dims::Dims{N}; copy::Val{C}=Val(false)) where {N,C} @@ -718,7 +1030,8 @@ reshape(arr, (3, 4); copy=Val(true)) # `copy` is a type parameter via Val{C}, so the default path constant-folds # and stays type-stable (needed by solve's 1D-rhs reshape). -function reshape(arr::NDArray, i::Dims{N}; copy::Val{C}=Val(false)) where {N,C} +function reshape(arr::NDArray{T}, i::Dims{N}; copy::Val{C}=Val(false)) where {T,N,C} + _struct_storage_type(T) && return _reshape_struct(arr, i) reshaped = nda_reshape_array(arr, i) if C copied = Base.copy(reshaped) @@ -728,26 +1041,56 @@ function reshape(arr::NDArray, i::Dims{N}; copy::Val{C}=Val(false)) where {N,C} return reshaped end +# cuPyNumeric copies non-view reshapes with a typed kernel that rejects records. +# Reshape each numeric field and rebuild, so struct reshapes always copy. +# TODO return a view when the reshape needs no copy, as numeric reshapes do. +function _reshape_struct(arr::NDArray{T}, dims::Dims) where {T} + prod(dims) == length(arr) || throw( + DimensionMismatch( + "new dimensions $(dims) must be consistent with array length $(length(arr))" + ), + ) + isempty(arr) && return nda_empty_array(dims, T) + _assert_struct_kernel("Reshaping", T) + fields = ntuple(fieldcount(T)) do i + projected = StructField{i}().(arr) + try + reshape(projected, dims; copy=Val(true)) + finally + destroy!(projected) + end + end + try + return StructConstructor{T}().(fields...) + finally + foreach(destroy!, fields) + end +end + function reshape(arr::NDArray, i::Int...; copy::Val{C}=Val(false)) where {C} return reshape(arr, i; copy=Val{C}()) end # Ignore the scalar indexing here... -unwrap(x::NDArray{<:Any,0}) = @allowscalar x[] -unwrap(x::NDArray{<:Any,1}) = @allowscalar x[][1] # assumes 1 element - -@doc""" - ==(arr1::NDArray, arr2::NDArray) +Base.only(x::NDArray{T,0}) where {T} = @allowscalar x[] -Check if two NDArrays are equal element-wise. +function Base.only(x::NDArray{T,N}) where {T,N} + length(x) == 1 || + throw(ArgumentError("collection must contain exactly 1 element")) -Returns `true` if both arrays have the same shape and all corresponding elements are equal. -Currently supports arrays up to 3 dimensions. For higher dimensions, returns `false` with a warning. + return @allowscalar x[firstindex(x)] +end -!!! warning +Base.fetch(x::NDArray) = only(x) - This function uses scalar indexing and should not be used in production code. This is meant for testing. +@doc""" + ==(arr1::NDArray, arr2::NDArray) + !=(arr1::NDArray, arr2::NDArray) +Element-wise equality reduced to a 0-d `NDArray{Bool}` (not a Julia `Bool`). +Same shape and values yields true; mismatched shape or rank yields false. +Mixed dtypes follow Julia `==` (promote, then compare). Use `fetch` or `A[]` +(with `allowscalar`) for a host value. Broadcast `.==` / `.!=` stay elementwise. # Examples ```@repl @@ -758,12 +1101,69 @@ c = cuNumeric.zeros(2, 2) a == c ``` """ -function Base.:(==)(arr1::NDArray{T,N}, arr2::NDArray{T,N}) where {T,N} - return nda_array_equal(arr1, arr2) #DOESNT RETURN SCALAR +function Base.:(==)(a::NDArray, b::NDArray) + size(a) == size(b) || return cnscalar(NDArray(false)) + if _struct_storage_type(eltype(a)) || _struct_storage_type(eltype(b)) + return _struct_array_equal(a, b) + end + return cnscalar(_array_equal_impl(a, b)) +end + +struct StructEqual end +@inline (::StructEqual)(x, y) = x == y + +# cuPyNumeric cannot compare records. Evaluate the element type's own `==` +# in the fused kernel so user-defined equality and NaN semantics match Base. +function _struct_array_equal(a::NDArray, b::NDArray) + isempty(a) && return cnscalar(NDArray(true)) + _assert_struct_kernel("Comparison", _struct_storage_type(eltype(a)) ? eltype(a) : eltype(b)) + if ndims(a) == 0 + return cnscalar(NDArray(_copy_to_julia_array(a) == _copy_to_julia_array(b))) + end + equal = StructEqual().(a, b) + try + return all(equal) + finally + destroy!(equal) + end +end + +function Base.:(!=)(a::NDArray, b::NDArray) + return !(a == b) +end + +function _array_equal_impl(a::NDArray{T}, b::NDArray{T}) where {T} + out = cuNumeric.zeros(Bool) + return nda_binary_reduction!(out, cuNumeric.EQUAL, a, b) +end + +function _array_equal_impl(a::NDArray{A}, b::NDArray{B}) where {A,B} + T = promote_type(A, B) + return _array_equal_promoted(a, b, T) end -function Base.:(!=)(arr1::NDArray{T,N}, arr2::NDArray{T,N}) where {T,N} - return !(arr1 == arr2) +function _array_equal_promoted(a::NDArray{T}, b::NDArray{T}, ::Type{T}) where {T} + return _array_equal_impl(a, b) +end +function _array_equal_promoted(a::NDArray, b::NDArray{T}, ::Type{T}) where {T} + p1 = unchecked_promote_arr(a, T) + result = _array_equal_impl(p1, b) + destroy!(p1) + return result +end +function _array_equal_promoted(a::NDArray{T}, b::NDArray, ::Type{T}) where {T} + p2 = unchecked_promote_arr(b, T) + result = _array_equal_impl(a, p2) + destroy!(p2) + return result +end +function _array_equal_promoted(a::NDArray, b::NDArray, ::Type{T}) where {T} + p1 = unchecked_promote_arr(a, T) + p2 = unchecked_promote_arr(b, T) + result = _array_equal_impl(p1, p2) + destroy!(p1) + destroy!(p2) + return result end @doc""" @@ -826,15 +1226,56 @@ isapprox(julia_arr, arr2) """ function Base.isapprox(julia_array::AbstractArray{T}, arr::NDArray{T}; atol=0, rtol=0) where {T} #! REPLCE THIS WITH BIN_OP isapprox - return compare(julia_array, arr, atol, rtol) + return compare(julia_array, arr, _maybe_fetch(atol), _maybe_fetch(rtol)) end function Base.isapprox(arr::NDArray{T}, julia_array::AbstractArray{T}; atol=0, rtol=0) where {T} - return compare(julia_array, arr, atol, rtol) + return compare(julia_array, arr, _maybe_fetch(atol), _maybe_fetch(rtol)) end function Base.isapprox(arr::NDArray{T}, arr2::NDArray{T}; atol=0, rtol=0) where {T} - return compare(arr, arr2, atol, rtol) + return compare(arr, arr2, _maybe_fetch(atol), _maybe_fetch(rtol)) +end + +# HDF5 signature. A leftover empty/truncated file from a crashed write has no +# header, and opening it in a Legate HDF5 task aborts the runtime. +const _HDF5_MAGIC = UInt8[0x89, 0x48, 0x44, 0x46, 0x0d, 0x0a, 0x1a, 0x0a] + +function _hdf5_check_dataset(dataset::AbstractString) + return isempty(dataset) && throw(ArgumentError("HDF5 dataset name must be non-empty")) +end + +function _is_hdf5_file(path::AbstractString) + isfile(path) || return false + filesize(path) < length(_HDF5_MAGIC) && return false + return open(path, "r") do io + return read(io, length(_HDF5_MAGIC)) == _HDF5_MAGIC + end +end + +# A truncated leftover `.h5` is not a valid file and makes HDF5CombineVDS abort. +# Leave a real HDF5 file in place so Legate can truncate it. Do not touch +# `*_legate_vds`: that directory holds the payload of a VDS write, and we +# cannot tell a stale sidecar from one a later read still needs. +function _prepare_h5write(path::AbstractString, dataset::AbstractString) + _hdf5_check_dataset(dataset) + isdir(path) && throw(ArgumentError("h5write path must be a file, got directory $path")) + parent = dirname(path) + if !isempty(parent) && parent != "." && !isdir(parent) + throw(ArgumentError("h5write parent directory does not exist: $parent")) + end + if ispath(path) && !_is_hdf5_file(path) + rm(path; force=true) + end + return nothing +end + +function _prepare_h5read(path::AbstractString, dataset::AbstractString) + _hdf5_check_dataset(dataset) + isdir(path) && throw(ArgumentError("h5read path must be a file, got directory $path")) + isfile(path) || throw(ArgumentError("HDF5 file does not exist: $path")) + _is_hdf5_file(path) || throw(ArgumentError("not an HDF5 file: $path")) + return nothing end """ @@ -842,12 +1283,17 @@ end Write an `NDArray` directly to an HDF5 dataset without a host copy or dimension flip. +A leftover empty or truncated `.h5` from a crashed write is removed first. +A valid HDF5 file is left in place so Legate can overwrite it. The +`*_legate_vds` sidecar is not touched: it holds the payload of a VDS write. + # Arguments - `path`: Path to the HDF5 file. - `dataset`: Name of the dataset to write. - `arr`: The array to write. """ function h5write(path::String, dataset::String, arr::NDArray{T,N}) where {T,N} + _prepare_h5write(path, dataset) st_handle = get_store(arr) # NDArrays are row-major, so this writes straight through (no dim flip, no warning). la = Legate.LogicalArray{T,N}(st_handle, size(arr)) @@ -867,6 +1313,7 @@ Read a dataset from an HDF5 file into an `NDArray`. - `layout`: On-disk memory order, either `:row` (default) or `:col`. """ function h5read(path::String, dataset::String; kwargs...) + _prepare_h5read(path, dataset) la = Legate.h5read(path, dataset; kwargs...) T = eltype(la) N = Int(Legate.dim(la)) diff --git a/src/ndarray/promotion.jl b/src/ndarray/promotion.jl index b13437fbb..1c1ac8ad6 100644 --- a/src/ndarray/promotion.jl +++ b/src/ndarray/promotion.jl @@ -5,7 +5,15 @@ is_wider_type(::Type{A}, ::Type{B}) where {A,B} = sizeof(A) > sizeof(B) checked_promote_arr(arr::NDArray{T}, ::Type{T}) where {T} = arr function checked_promote_arr(arr::NDArray{T}, ::Type{S}) where {T,S} - is_wider_type(S, T) && assertpromotion(promote_type, T, S) + return checked_promote_arr(promote_type, arr, S) +end + +# `op` only names the caller in the error message, so that e.g. cholesky reports +# itself rather than `promote_type`. +checked_promote_arr(op, arr::NDArray{T}, ::Type{T}) where {T} = arr + +function checked_promote_arr(op, arr::NDArray{T}, ::Type{S}) where {T,S} + is_wider_type(S, T) && assertpromotion(op, T, S) return as_type(arr, S) end @@ -22,9 +30,64 @@ unchecked_promote_scalar(x, ::Type) = x unchecked_promote_arr(::Base.RefValue{typeof(^)}, ::Type{T}) where {T} = typeof(Base.:(^)) unchecked_promote_arr(::Base.RefValue{Val{V}}, ::Type{T}) where {T,V} = Val{V} +@inline _is_flattened_associative(f) = f === (+) || f === (*) + __checked_promote_op(op, ::Type{Tuple{A}}) where {A} = __checked_promote_op(op, A) __checked_promote_op(op, ::Type{Tuple{A,B}}) where {A,B} = __checked_promote_op(op, A, B) +# Ref-wrapped functions and other static broadcast arguments participate in +# result-type inference, but they are not numeric inputs to promote. +@inline function _numeric_broadcast_types(::Type{Args}) where {Args<:Tuple} + return filter(T -> T <: Number, tuple(Args.parameters...)) +end + +@inline function _smallest_numeric_broadcast_type(::Type{Args}) where {Args<:Tuple} + types = _numeric_broadcast_types(Args) + return isempty(types) ? nothing : foldl(smaller_type, types) +end + +@inline function _broadcast_result_type(op, ::Type{T}) where {T} + (T <: SUPPORTED_ARRAY_TYPES || _struct_storage_type(T)) || throw( + ArgumentError( + "Broadcast function $(op) cannot produce an NDArray: unsupported result type $(T)" + ), + ) + return T +end + +# Julia flattens dotted `+` and `*` chains into n-ary Broadcasted nodes. Fold +# their input types pairwise, matching both the binary C API and fused path. +@inline function __checked_promote_op( + op::Union{typeof(+),typeof(*)}, ::Type{Args} +) where {Args<:Tuple{Any,Any,Any,Vararg{Any}}} + return _checked_promote_associative(op, Args.parameters...) +end + +# Resolve the 4+-argument overlap with the general custom-function method. +@inline function __checked_promote_op( + op::Union{typeof(+),typeof(*)}, ::Type{Args} +) where {Args<:Tuple{Any,Any,Any,Any,Vararg{Any}}} + return _checked_promote_associative(op, Args.parameters...) +end + +@inline function __checked_promote_op( + op, ::Type{Tuple{A,B,C}} +) where {A,B,C} + T = _broadcast_result_type(op, Base.promote_op(op, A, B, C)) + S = _smallest_numeric_broadcast_type(Tuple{A,B,C}) + (S === nothing || !(T <: Number)) || (is_wider_type(T, S) && assertpromotion(op, S, T)) + return T +end + +@inline function __checked_promote_op( + op, ::Type{Args} +) where {Args<:Tuple{Any,Any,Any,Any,Vararg{Any}}} + T = _broadcast_result_type(op, Base.promote_op(op, Args.parameters...)) + S = _smallest_numeric_broadcast_type(Args) + (S === nothing || !(T <: Number)) || (is_wider_type(T, S) && assertpromotion(op, S, T)) + return T +end + # Path for literal powers @inline function __checked_promote_op( f::typeof(Base.literal_pow), a::Type{Tuple{_,ARR_TYPE,Val{POWER}}} @@ -51,24 +114,34 @@ __recip_type(::Type{Int64}) = Float64 __recip_type(::Type{Bool}) = DEFAULT_FLOAT @inline function __checked_promote_op(op, ::Type{A}) where {A} - T = Base.promote_op(op, A) - is_wider_type(T, A) && assertpromotion(op, A, T) + T = _broadcast_result_type(op, Base.promote_op(op, A)) + T <: SUPPORTED_ARRAY_TYPES && is_wider_type(T, A) && assertpromotion(op, A, T) return T end @inline function __checked_promote_op(op, ::Type{A}, ::Type{A}) where {A} - T = Base.promote_op(op, A, A) - is_wider_type(T, A) && assertpromotion(op, A, T) + T = _broadcast_result_type(op, Base.promote_op(op, A, A)) + T <: Number && is_wider_type(T, A) && assertpromotion(op, A, T) return T end @inline function __checked_promote_op(op, ::Type{A}, ::Type{B}) where {A,B} - T = Base.promote_op(op, A, B) - S = smaller_type(A, B) - is_wider_type(T, S) && assertpromotion(op, S, T) + T = _broadcast_result_type(op, Base.promote_op(op, A, B)) + S = _smallest_numeric_broadcast_type(Tuple{A,B}) + (S === nothing || !(T <: Number)) || (is_wider_type(T, S) && assertpromotion(op, S, T)) return T end +@inline _checked_promote_associative(op, ::Type{A}, ::Type{B}) where {A,B} = + __checked_promote_op(op, A, B) + +@inline function _checked_promote_associative( + op, ::Type{A}, ::Type{B}, ::Type{C}, rest::Type... +) where {A,B,C} + T = __checked_promote_op(op, A, B) + return _checked_promote_associative(op, T, C, rest...) +end + # For literal powers which are often Int64, do not check for promotion to double # The result of promote_op with a literal integer power is always the base type # Base.promote_op(^, Float32, Int64) == Float32 diff --git a/src/ndarray/random/bitgenerator.jl b/src/ndarray/random/bitgenerator.jl new file mode 100644 index 000000000..a72240ac0 --- /dev/null +++ b/src/ndarray/random/bitgenerator.jl @@ -0,0 +1,154 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +# Mirrors cupynumeric.random._bitgenerator.BitGenerator and the three +# engines Python actually subclasses: XORWOW, MRG32k3a, PHILOX4_32_10. +# CREATE is lazy: the first DISTRIBUTION task materializes the generator. + +""" + BitGenerator + +Abstract supertype for cuRAND engines used by [`Generator`](@ref cuNumeric.Generator). Concrete +types are [`XORWOW`](@ref cuNumeric.XORWOW) (default), [`MRG32k3a`](@ref cuNumeric.MRG32k3a), and +[`PHILOX4_32_10`](@ref cuNumeric.PHILOX4_32_10). +""" +abstract type BitGenerator end + +const _bitgen_id_lock = ReentrantLock() +const _next_bitgen_id = Ref{Int32}(0) +const _bitgen_zombies = Int32[] + +function _next_bitgenerator_id() + return lock(_bitgen_id_lock) do + _next_bitgen_id[] += Int32(1) + return _next_bitgen_id[] + end +end + +function _record_bitgenerator_zombie(handle::Int32) + handle == 0 && return nothing + return lock(_bitgen_id_lock) do + push!(_bitgen_zombies, handle) + return nothing + end +end + +# C-order strides (NDArray stores are row-major). Julia col-major is only +# handled in Array conversions. DISTRIBUTION writes a dense ptr and ignores this. +function _c_order_strides(shape::NTuple{N,Int}) where {N} + return SVector{N,Int64}( + ntuple(Val(N)) do i + p = Int64(1) + @inbounds for d in (i + 1):N + p *= Int64(shape[d]) + end + return p + end, + ) +end + +_c_order_strides(::Tuple{}) = SVector{0,Int64}() + +const _EMPTY_INT64 = SVector{0,Int64}() +const _EMPTY_FLOAT32 = SVector{0,Float32}() +const _EMPTY_FLOAT64 = SVector{0,Float64}() + +function _make_bitgenerator(::Type{B}, seed, flags) where {B<:BitGenerator} + handle = _next_bitgenerator_id() + seed64 = seed === nothing ? UInt64(time_ns()) : UInt64(seed) + bg = B(handle, seed64, UInt32(flags)) + finalizer(_finalize_bitgenerator!, bg) + return bg +end + +function _finalize_bitgenerator!(bg::BitGenerator) + handle = bg.handle + bg.handle = Int32(0) + _record_bitgenerator_zombie(handle) + return nothing +end + +""" + XORWOW(seed=nothing; flags=0) + +Default cuRAND BitGenerator (xorwow). `seed=nothing` draws from `time_ns()`. +""" +mutable struct XORWOW <: BitGenerator + handle::Int32 + seed::UInt64 + flags::UInt32 +end +function XORWOW(seed::Union{Integer,Nothing}=nothing; flags::Integer=0) + return _make_bitgenerator(XORWOW, seed, flags) +end + +""" + MRG32k3a(seed=nothing; flags=0) + +cuRAND MRG32k3a BitGenerator. +""" +mutable struct MRG32k3a <: BitGenerator + handle::Int32 + seed::UInt64 + flags::UInt32 +end +function MRG32k3a(seed::Union{Integer,Nothing}=nothing; flags::Integer=0) + return _make_bitgenerator(MRG32k3a, seed, flags) +end + +""" + PHILOX4_32_10(seed=nothing; flags=0) + +cuRAND Philox4_32_10 BitGenerator. +""" +mutable struct PHILOX4_32_10 <: BitGenerator + handle::Int32 + seed::UInt64 + flags::UInt32 +end +function PHILOX4_32_10(seed::Union{Integer,Nothing}=nothing; flags::Integer=0) + return _make_bitgenerator(PHILOX4_32_10, seed, flags) +end + +generator_type(::XORWOW) = cuNumeric.BITGENTYPE_XORWOW +generator_type(::MRG32k3a) = cuNumeric.BITGENTYPE_MRG32K3A +generator_type(::PHILOX4_32_10) = cuNumeric.BITGENTYPE_PHILOX4_32_10 + +function _bitgenerator_distribution!( + arr::NDArray, + bg::BitGenerator, + distribution, + intparams::SVector{NI,Int64}, + floatparams::SVector{NF,Float32}, + doubleparams::SVector{ND,Float64}, +) where {NI,NF,ND} + return nda_bitgenerator_distribution!( + arr, + bg.handle, + UInt32(generator_type(bg)), + bg.seed, + bg.flags, + UInt32(distribution), + _c_order_strides(size(arr)), + intparams, + floatparams, + doubleparams, + ) +end diff --git a/src/ndarray/random/generator.jl b/src/ndarray/random/generator.jl new file mode 100644 index 000000000..3b91081b5 --- /dev/null +++ b/src/ndarray/random/generator.jl @@ -0,0 +1,270 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +""" + Generator(bit_generator) + +cuPyNumeric-style RNG wrapping a [`BitGenerator`](@ref cuNumeric.BitGenerator). Module-level +[`rand`](@ref cuNumeric.rand), [`randn`](@ref cuNumeric.randn), [`randexp`](@ref cuNumeric.randexp) use a process-global +XORWOW generator; pass a different engine to `default_rng` for Philox or MRG32k3a. +""" +struct Generator{B<:BitGenerator} + bit_generator::B +end + +const _RNG_INT_TYPES = Union{Int16,Int32,Int64} +const _static_generator = Ref{Union{Nothing,Generator{XORWOW}}}(nothing) + +""" + default_rng() + default_rng(seed) + default_rng(::Type{<:BitGenerator}, seed=nothing; flags=0) + default_rng(bit_generator) + default_rng(generator) + +Return a [`Generator`](@ref cuNumeric.Generator). The no-argument and integer-seed methods use +[`XORWOW`](@ref cuNumeric.XORWOW). This does **not** replace the process-global generator used +by module-level `rand` / `randn`. + +```julia +g = cuNumeric.default_rng(cuNumeric.PHILOX4_32_10, 1234) +cuNumeric.random(g, Float32, (8, 8)) +``` +""" +function default_rng() + return Generator(XORWOW()) +end + +function default_rng(seed::Integer) + return Generator(XORWOW(seed)) +end + +function default_rng( + ::Type{B}, seed::Union{Integer,Nothing}=nothing; flags::Integer=0 +) where {B<:BitGenerator} + return Generator(B(seed; flags)) +end + +function default_rng(bg::BitGenerator) + return Generator(bg) +end + +function default_rng(g::Generator) + return g +end + +function get_static_generator() + gen = _static_generator[] + if gen === nothing + gen = default_rng() + _static_generator[] = gen + end + return gen::Generator{XORWOW} +end + +function random!(g::Generator, arr::NDArray{Float32}) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_UNIFORM_32, + _EMPTY_INT64, SVector{2,Float32}(0, 1), _EMPTY_FLOAT64, + ) + return arr +end + +function random!(g::Generator, arr::NDArray{Float64}) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_UNIFORM_64, + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{2,Float64}(0, 1), + ) + return arr +end + +# No native Bool distribution. Draw {0,1} as Int16 and compare; that is the +# usual path to NDArray{Bool}. Do not as_type to Bool (CxxWrap.CxxBool). +function random!(g::Generator, arr::NDArray{Bool}) + copyto!(arr, random(g, Bool, size(arr))) + return arr +end + +# as_type first: `float .* Complex` is a widening promotion and is disallowed. +# Does not take ownership of `re` / `imag_part`. +function _pack_complex!( + out::NDArray{Complex{T}}, re::NDArray{T}, imag_part::NDArray{T} +) where {T<:SUPPORTED_FLOAT_TYPES} + CT = Complex{T} + re_c = as_type(re, CT) + im_c = as_type(imag_part, CT) + out .= re_c .+ im_c .* CT(0, 1) + destroy!(re_c) + destroy!(im_c) + return out +end + +# No native complex BitGenerator dist. Independent real/imag uniforms, like +# Julia `rand(Complex{T})` (unit square, not the unit disk). +function random!(g::Generator, arr::NDArray{Complex{T}}) where {T<:SUPPORTED_FLOAT_TYPES} + re = random(g, T, size(arr)) + imag_part = random(g, T, size(arr)) + _pack_complex!(arr, re, imag_part) + destroy!(re) + destroy!(imag_part) + return arr +end + +function random!(g::Generator, arr::NDArray{T}) where {T} + return error( + "random! supports Float32, Float64, ComplexF32, ComplexF64, Bool, Int16, Int32, and Int64 NDArray storage" + ) +end + +function random(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + arr = NDArray{T}(undef, dims) + random!(g, arr) + return arr +end + +function random(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + arr = NDArray{T}(undef, dims) + random!(g, arr) + return arr +end + +function random(g::Generator, ::Type{Bool}, dims::Dims) + return integers(g, Int16, dims; low=0, high=2) .!= Int16(0) +end + +function _randn!(g::Generator, arr::NDArray{Float32}; loc::Union{Real,DeviceScalar{<:Real}}=0, scale::Union{Real,DeviceScalar{<:Real}}=1) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_NORMAL_32, + _EMPTY_INT64, SVector{2,Float32}(Float32(_maybe_fetch(loc)), Float32(_maybe_fetch(scale))), _EMPTY_FLOAT64, + ) + return arr +end + +function _randn!(g::Generator, arr::NDArray{Float64}; loc::Union{Real,DeviceScalar{<:Real}}=0, scale::Union{Real,DeviceScalar{<:Real}}=1) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_NORMAL_64, + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{2,Float64}(Float64(_maybe_fetch(loc)), Float64(_maybe_fetch(scale))), + ) + return arr +end + +function randn!(g::Generator, arr::NDArray{Float32}) + return _randn!(g, arr) +end + +function randn!(g::Generator, arr::NDArray{Float64}) + return _randn!(g, arr) +end + +# Independent N(0, 1/2) real/imag so E[|z|²] = 1, matching Julia `randn(Complex{T})`. +# No loc/scale kwargs: shift or scale in user code (`μ .+ σ .* Z`). +function randn!(g::Generator, arr::NDArray{Complex{T}}) where {T<:SUPPORTED_FLOAT_TYPES} + s = 1 / sqrt(T(2)) + re = NDArray{T}(undef, size(arr)) + imag_part = NDArray{T}(undef, size(arr)) + _randn!(g, re; scale=s) + _randn!(g, imag_part; scale=s) + _pack_complex!(arr, re, imag_part) + destroy!(re) + destroy!(imag_part) + return arr +end + +function randn!(g::Generator, arr::NDArray{T}) where {T} + return error( + "randn! only supports Float32, Float64, ComplexF32, and ComplexF64 NDArray storage" + ) +end + +function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + arr = NDArray{T}(undef, dims) + randn!(g, arr) + return arr +end + +function randn(g::Generator, ::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + arr = NDArray{T}(undef, dims) + randn!(g, arr) + return arr +end + +function randexp!(g::Generator, arr::NDArray{Float32}; scale::Union{Real,DeviceScalar{<:Real}}=1) + s = Float32(_maybe_fetch(scale)) + s > 0 || throw(ArgumentError("scale must be positive, got $scale")) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_EXPONENTIAL_32, + _EMPTY_INT64, SVector{1,Float32}(s), _EMPTY_FLOAT64, + ) + return arr +end + +function randexp!(g::Generator, arr::NDArray{Float64}; scale::Union{Real,DeviceScalar{<:Real}}=1) + s = Float64(_maybe_fetch(scale)) + s > 0 || throw(ArgumentError("scale must be positive, got $scale")) + _bitgenerator_distribution!( + arr, g.bit_generator, cuNumeric.BITGENDIST_EXPONENTIAL_64, + _EMPTY_INT64, _EMPTY_FLOAT32, SVector{1,Float64}(s), + ) + return arr +end + +function randexp!(g::Generator, arr::NDArray{T}; scale::Union{Real,DeviceScalar{<:Real}}=1) where {T} + return error("randexp! only supports Float32 and Float64 NDArray storage") +end + +function randexp( + g::Generator, ::Type{T}, dims::Dims; scale::Union{Real,DeviceScalar{<:Real}}=1 +) where {T<:SUPPORTED_FLOAT_TYPES} + arr = NDArray{T}(undef, dims) + randexp!(g, arr; scale=scale) + return arr +end + +_integer_distribution(::Type{Int16}) = cuNumeric.BITGENDIST_INTEGERS_16 +_integer_distribution(::Type{Int32}) = cuNumeric.BITGENDIST_INTEGERS_32 +_integer_distribution(::Type{Int64}) = cuNumeric.BITGENDIST_INTEGERS_64 + +function _integer_distribution(::Type{T}) where {T} + return throw(ArgumentError("integer random only supports Int16, Int32, and Int64. Got $T.")) +end + +# Kernel draws [low, high); Julia `a:b` is mapped to low=a, high=b+1. +function integers!( + g::Generator, arr::NDArray{T}; low::Integer, high::Integer +) where {T<:_RNG_INT_TYPES} + Int64(high) <= Int64(low) && throw(ArgumentError("low >= high")) + _bitgenerator_distribution!( + arr, g.bit_generator, _integer_distribution(T), + SVector{2,Int64}(Int64(low), Int64(high)), _EMPTY_FLOAT32, _EMPTY_FLOAT64, + ) + return arr +end + +function integers!(g::Generator, arr::NDArray{T}; low::Integer, high::Integer) where {T} + return throw(ArgumentError("integer random only supports Int16, Int32, and Int64")) +end + +function integers( + g::Generator, ::Type{T}, dims::Dims; low::Integer, high::Integer +) where {T<:_RNG_INT_TYPES} + arr = NDArray{T}(undef, dims) + integers!(g, arr; low=low, high=high) + return arr +end diff --git a/src/ndarray/random/random.jl b/src/ndarray/random/random.jl new file mode 100644 index 000000000..6608ef9cf --- /dev/null +++ b/src/ndarray/random/random.jl @@ -0,0 +1,218 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +# Module-level convenience API, mirroring cupynumeric.random._random.py. +# All draws go through get_static_generator(). + +@doc""" + Random.rand!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + Random.rand!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + Random.rand!(arr::NDArray{<:Union{Int16,Int32,Int64}}) + Random.rand!(arr::NDArray{Bool}) + +Fill `arr` in-place with uniform random values. + +Floating arrays are filled from `[0, 1)`. Complex arrays draw independent +real/imag uniforms in `[0, 1)` (the unit square). Integer arrays use the full +range of the element type, matching Julia `rand(T)`. `Bool` arrays are fair +coin flips. +""" +function Random.rand!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + return random!(get_static_generator(), arr) +end + +function Random.rand!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return random!(get_static_generator(), arr) +end + +function Random.rand!(arr::NDArray{T}) where {T<:_RNG_INT_TYPES} + return integers!(get_static_generator(), arr; low=typemin(T), high=typemax(T)) +end + +function Random.rand!(arr::NDArray{Bool}) + return random!(get_static_generator(), arr) +end + +function Random.rand!(arr::NDArray{T}) where {T} + return error( + "rand! supports Float32, Float64, ComplexF32, ComplexF64, Bool, Int16, Int32, and Int64 NDArray storage" + ) +end + +@doc""" + cuNumeric.rand([T=Float32,] dims::Int...) + cuNumeric.rand([T=Float32,] dims::Tuple) + cuNumeric.rand(r::AbstractUnitRange, dims...) + +Create a new `NDArray` filled with uniform random values. + +Floating types (`Float32` / `Float64`) are drawn natively in `[0, 1)`. +`ComplexF32` / `ComplexF64` draw independent real/imag uniforms (unit square). +`Bool` is a fair coin flip. Integer types `Int16`, `Int32`, and `Int64` use the +full range of `T`. A unit range (`1:10`) draws inclusive integers, matching +Julia `rand(1:10, dims...)`. + +Uses a process-global [`XORWOW`](@ref cuNumeric.XORWOW) generator. For a different engine or +seed, see [`default_rng`](@ref cuNumeric.default_rng) and [`Generator`](@ref cuNumeric.Generator). + +# Examples +```@repl +cuNumeric.rand(2, 2) +cuNumeric.rand((4, 1)) +cuNumeric.rand(0:9, 4, 4) +cuNumeric.rand(Bool, 8) +A = cuNumeric.zeros(Float32, 2, 2); cuNumeric.rand!(A) +``` +""" +function rand(::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + return random(get_static_generator(), T, dims) +end + +function rand(::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + return random(get_static_generator(), T, dims) +end + +function rand(::Type{T}, dims::Dims) where {T<:_RNG_INT_TYPES} + return integers(get_static_generator(), T, dims; low=typemin(T), high=typemax(T)) +end + +rand(::Type{Bool}, dims::Dims) = random(get_static_generator(), Bool, dims) + +function rand( + ::Type{T}, dims::Int... +) where {T<:Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES,_RNG_INT_TYPES,Bool}} + return cuNumeric.rand(T, dims) +end +rand(dims::Dims) = cuNumeric.rand(DEFAULT_FLOAT, dims) +rand(dims::Int...) = cuNumeric.rand(DEFAULT_FLOAT, dims) + +function _unitrange_bounds(r::AbstractUnitRange{<:Integer}) + lo = Int64(first(r)) + hi = Int64(last(r)) + hi < lo && throw(ArgumentError("empty range $r")) + return lo, hi + Int64(1) # kernel is [low, high) +end + +function rand(r::AbstractUnitRange{T}, dims::Dims) where {T<:_RNG_INT_TYPES} + lo, hi = _unitrange_bounds(r) + return integers(get_static_generator(), T, dims; low=lo, high=hi) +end + +rand(r::AbstractUnitRange{<:_RNG_INT_TYPES}, dims::Int...) = cuNumeric.rand(r, dims) + +function rand(r::AbstractUnitRange{Bool}, dims::Dims) + first(r) == last(r) && return fill(first(r), dims) + first(r) == false && last(r) == true && return cuNumeric.rand(Bool, dims) + return throw(ArgumentError("empty range $r")) +end +rand(r::AbstractUnitRange{Bool}, dims::Int...) = cuNumeric.rand(r, dims) + +@doc""" + Random.randn!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + Random.randn!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + +Fill `arr` in-place with standard normal samples. Real arrays have mean 0 and +variance 1. Complex arrays match Julia `randn(Complex{T})`: independent real +and imag parts with variance `1/2`, so `E[|z|²] = 1`. +""" +function Random.randn!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + return randn!(get_static_generator(), arr) +end + +function Random.randn!(arr::NDArray{<:SUPPORTED_COMPLEX_TYPES}) + return randn!(get_static_generator(), arr) +end + +function Random.randn!(arr::NDArray{T}) where {T} + return error( + "randn! only supports Float32, Float64, ComplexF32, and ComplexF64 NDArray storage" + ) +end + +@doc""" + cuNumeric.randn([T=Float32,] dims::Int...) + cuNumeric.randn([T=Float32,] dims::Tuple) + +Create a new `NDArray` filled with standard normal samples. Complex types +match Julia `randn(Complex{T})` (`E[|z|²] = 1`). + +Uses a process-global [`XORWOW`](@ref cuNumeric.XORWOW) generator. For a different +engine, pass a [`Generator`](@ref cuNumeric.Generator) to `randn!`. Shift or scale +in user code (`μ .+ σ .* Z`); `loc`/`scale` are not part of the public API. + +# Examples +```@repl +cuNumeric.randn(2, 2) +cuNumeric.randn(Float64, 1000) +``` +""" +function randn(::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + return randn(get_static_generator(), T, dims) +end + +function randn(::Type{T}, dims::Dims) where {T<:SUPPORTED_COMPLEX_TYPES} + return randn(get_static_generator(), T, dims) +end + +function randn( + ::Type{T}, dims::Int... +) where {T<:Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES}} + return cuNumeric.randn(T, dims) +end +randn(dims::Dims) = cuNumeric.randn(DEFAULT_FLOAT, dims) +randn(dims::Int...) = cuNumeric.randn(DEFAULT_FLOAT, dims) + +@doc""" + Random.randexp!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + +Fill `arr` in-place with exponential samples of scale 1 (mean 1), matching +Julia `randexp`. +""" +function Random.randexp!(arr::NDArray{<:SUPPORTED_FLOAT_TYPES}) + return randexp!(get_static_generator(), arr) +end + +function Random.randexp!(arr::NDArray{T}) where {T} + return error("randexp! only supports Float32 and Float64 NDArray storage") +end + +@doc""" + cuNumeric.randexp([T=Float32,] dims::Int...) + cuNumeric.randexp([T=Float32,] dims::Tuple) + +Create a new `NDArray` filled with exponential samples of scale 1 (mean 1), +matching Julia `randexp`. + +Uses a process-global [`XORWOW`](@ref cuNumeric.XORWOW) generator. For a +different engine or scale, use `randexp!(generator, arr; scale)`. + +# Examples +```@repl +cuNumeric.randexp(2, 2) +cuNumeric.randexp(Float64, 1000) +``` +""" +function randexp(::Type{T}, dims::Dims) where {T<:SUPPORTED_FLOAT_TYPES} + return randexp(get_static_generator(), T, dims) +end + +randexp(::Type{T}, dims::Int...) where {T<:SUPPORTED_FLOAT_TYPES} = cuNumeric.randexp(T, dims) +randexp(dims::Dims) = cuNumeric.randexp(DEFAULT_FLOAT, dims) +randexp(dims::Int...) = cuNumeric.randexp(DEFAULT_FLOAT, dims) diff --git a/src/ndarray/sort.jl b/src/ndarray/sort.jl new file mode 100644 index 000000000..39c3f992c --- /dev/null +++ b/src/ndarray/sort.jl @@ -0,0 +1,136 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +# Julia 1-based dims -> 0-based cupynumeric axis. +function _checked_sort_axis(N::Int, dims::Integer) + 1 <= dims <= N || throw(ArgumentError("dims=$dims is invalid for a $N-d array")) + return Int32(dims - 1) +end + +@doc""" + cuNumeric.sort(v::NDArray{T,1}; stable::Bool=false) + cuNumeric.sort(A::NDArray; dims::Integer, stable::Bool=false) + +Return a copy of `A` sorted in ascending order. + +This is **not** `Base.sort`. Call it as `cuNumeric.sort`. `alg`, `lt`, `by`, +`rev`, and `order` are not accepted. + +For rank greater than 1, `dims` is required (Julia 1-based). `stable=true` +sets the cupynumeric stable-sort flag (`kind="stable"`); the default is +unstable (`kind="quicksort"`). These names do not run Julia's QuickSort or +MergeSort. Complex values are ordered lexicographically by `(real, imag)`. +""" +function sort(arr::NDArray{T,1}; stable::Bool=false) where {T} + return nda_sort(arr, Int32(-1), stable) +end + +function sort(arr::NDArray{T,N}; dims::Integer, stable::Bool=false) where {T,N} + return nda_sort(arr, _checked_sort_axis(N, dims), stable) +end + +@doc""" + cuNumeric.sort!(v::NDArray{T,1}; stable::Bool=false) + cuNumeric.sort!(A::NDArray; dims::Integer, stable::Bool=false) + +Sort `A` in place. Same kwargs as [`cuNumeric.sort`](@ref). Not `Base.sort!`. +""" +function sort!(arr::NDArray{T,1}; stable::Bool=false) where {T} + return nda_sort_inplace(arr, Int32(-1), stable) +end + +function sort!(arr::NDArray{T,N}; dims::Integer, stable::Bool=false) where {T,N} + return nda_sort_inplace(arr, _checked_sort_axis(N, dims), stable) +end + +function _searchsorted_impl(a::NDArray{TA,1}, v::NDArray{TV,N}, left::Bool) where {TA,TV,N} + (TA <: Complex || TV <: Complex) && + throw(ArgumentError("searchsorted is not supported for complex arrays")) + U = promote_type(TA, TV) + a2 = unchecked_promote_arr(a, U) + v2 = unchecked_promote_arr(v, U) + raw = nda_searchsorted(a2, v2, left) + a2 !== a && destroy!(a2) + v2 !== v && destroy!(v2) + return left ? _indices_to_one_based(raw) : raw +end + +@doc""" + cuNumeric.searchsortedfirst(a::NDArray{T,1}, x) + cuNumeric.searchsortedlast(a::NDArray{T,1}, x) + +Insertion indices into a 1-d sorted `a`. `x` may be a `Number` or an +`NDArray` of needles. + +Scalar queries return a 0-d `NDArray{Int64}` (not a Julia `Int`); use +`fetch` for an `Int`. Array queries return an `NDArray{Int64}` with the +shape of the needles. Indices are 1-based. + +Not `Base.searchsortedfirst` / `searchsortedlast`. `a` must already be sorted +ascending. `lt` / `by` / `rev` are not accepted. Complex arrays are not +supported. +""" +function searchsortedfirst(a::NDArray{T,1}, v::NDArray) where {T} + return _searchsorted_impl(a, v, true) +end + +function searchsortedfirst(a::NDArray{T,1}, x::Number) where {T} + needle = NDArray(x) + result = searchsortedfirst(a, needle) + destroy!(needle) + return result +end + +function searchsortedlast(a::NDArray{T,1}, v::NDArray) where {T} + return _searchsorted_impl(a, v, false) +end + +function searchsortedlast(a::NDArray{T,1}, x::Number) where {T} + needle = NDArray(x) + result = searchsortedlast(a, needle) + destroy!(needle) + return result +end + +@doc""" + cuNumeric.searchsorted(a::NDArray{T,1}, x::Number) + +`searchsortedfirst(a, x):searchsortedlast(a, x)` as a `UnitRange`, matching +Base's scalar search. Materializes two 0-d index arrays via `fetch`. +Not `Base.searchsorted`. +""" +function searchsorted(a::NDArray{T,1}, x::Union{Number,DeviceScalar}) where {T} + lo = searchsortedfirst(a, x) + hi = searchsortedlast(a, x) + return fetch(lo):fetch(hi) +end + +@doc""" + cuNumeric.unique(A::NDArray) -> NDArray{T,1} + +Sorted unique elements of `A`, flattened to 1-d. + +This is **not** `Base.unique`, which keeps first-occurrence order and does +not sort. `dims`, `return_index`, `return_inverse`, and `return_counts` +are not accepted; cupynumeric does not implement them. +""" +function unique(arr::NDArray) + return nda_unique(arr) +end diff --git a/src/ndarray/unary.jl b/src/ndarray/unary.jl index 0c0df9cac..0091e9c96 100644 --- a/src/ndarray/unary.jl +++ b/src/ndarray/unary.jl @@ -1,4 +1,4 @@ -global const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( +const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( Base.acos => cuNumeric.ARCCOS, Base.acosh => cuNumeric.ARCCOSH, Base.asin => cuNumeric.ARCSIN, @@ -24,51 +24,88 @@ global const floaty_unary_ops_no_args = Dict{Function,UnaryOpCode}( Base.tanh => cuNumeric.TANH, ) -global const unary_op_map_no_args = Dict{Function,UnaryOpCode}( +const unary_op_map_no_args = Dict{Function,UnaryOpCode}( Base.abs => cuNumeric.ABSOLUTE, - # Base.conj => cuNumeric.CONJ, #! NEED TO SUPPORT COMPLEX TYPES FIRST + # Base.conj => cuNumeric.CONJ, # handled as a special case below Base.:(-) => cuNumeric.NEGATIVE, - # Base.frexp => cuNumeric.FREXP, #* annoying returns tuple - # missing => cuNumeric.GETARG, #not in numpy? - # Base.imag => cuNumeric.IMAG, #! NEED TO SUPPORT COMPLEX TYPES FIRST - # missing => cuNumerit.INVERT, # no bitwise not in julia? - # Base.isfinite => cuNumeric.ISFINITE, #* dont feel like looking into Inf rn - # Base.isinf => cuNumeric.ISINF, #* dont feel like looking into Inf rn - # Base.isnan => cuNumeric.ISNAN, #* dont feel like looking into Inf rn - # Base.modf => cuNumeric.MODF, #* annoying returns tuple - #missing => cuNumeric.POSITIVE, #What is this even for + # Base.frexp => cuNumeric.FREXP, # returns a tuple + # missing => cuNumeric.GETARG, + # Base.imag => cuNumeric.IMAG, # handled as a special case below + Base.:(~) => cuNumeric.INVERT, # integers only; kernel rejects Bool + Base.isfinite => cuNumeric.ISFINITE, + Base.isinf => cuNumeric.ISINF, + Base.isnan => cuNumeric.ISNAN, + # Base.modf => cuNumeric.MODF, # returns a tuple + # missing => cuNumeric.POSITIVE, Base.sign => cuNumeric.SIGN, - # Base.signbit => cuNumeric.SIGNBIT, #! Doesnt support Bool, I do not feel like dealing with this right now... + Base.signbit => cuNumeric.SIGNBIT, # floats only; kernel rejects Bool/int + Base.ceil => cuNumeric.CEIL, # floats only + Base.floor => cuNumeric.FLOOR, # floats only + Base.trunc => cuNumeric.TRUNC, # floats only + # 1-arg `round.(A)` only (see __broadcast below). ROUND needs extra_args. + Base.round => cuNumeric.RINT, ) +for julia_fn in (keys(floaty_unary_ops_no_args)..., keys(unary_op_map_no_args)..., + identity, real, imag, conj, inv, !) + @eval @inline _has_unfused_broadcast(::typeof($julia_fn), ::Val{1}) = true +end +# Positional rounding modes are rejected by the native path, too. +@inline _has_unfused_broadcast(::typeof(round), ::Val) = true +@inline _has_unfused_broadcast(::typeof(Base.literal_pow), ::Val{3}) = true + ### SPECIAL CASES ### +# `dest .= src` lowers to `identity.(src)`. Treat identity like the native +# unary operation it is so ordinary Julia broadcast assignment works for +# NDArrays, including writable slices. +@inline function __broadcast(::typeof(identity), out::NDArray, input::NDArray) + return nda_unary_op!(out, cuNumeric.COPY, input) +end + +# Real abs2 is a single native square. +@inline function __broadcast(::typeof(abs2), out::NDArray{T}, input::NDArray{T}) where {T<:Real} + return nda_unary_op!(out, cuNumeric.SQUARE, input) +end + +# Complex abs2 has no matching native opcode. Take the magnitude into a real +# temporary and square it, keeping the unfused path available without a GPU. +@inline function __broadcast( + ::typeof(abs2), out::NDArray{T}, input::NDArray{Complex{T}} +) where {T<:SUPPORTED_FLOAT_TYPES} + magnitude = similar(out) + nda_unary_op!(magnitude, cuNumeric.ABSOLUTE, input) + nda_unary_op!(out, cuNumeric.SQUARE, magnitude) + destroy!(magnitude) + return out +end + # Needed to support != Base.:(!)(input::NDArray{Bool,0}) = nda_unary_op!(similar(input), cuNumeric.LOGICAL_NOT, input) Base.:(!)(input::NDArray{Bool,1}) = nda_unary_op!(similar(input), cuNumeric.LOGICAL_NOT, input) # Non-broadcasted version of negation function Base.:(-)(input::NDArray{T}) where {T} - out = cuNumeric.zeros(T, size(input)) + out = NDArray{T}(undef, size(input)) return nda_unary_op!(out, cuNumeric.NEGATIVE, input) end function Base.real(input::NDArray{T}) where {T<:Complex} T_OUT = Base.promote_op(real, T) - out = cuNumeric.zeros(T_OUT, size(input)) + out = NDArray{T_OUT}(undef, size(input)) return nda_unary_op!(out, cuNumeric.REAL, input) end Base.real(input::NDArray{<:Real}) = input function Base.imag(input::NDArray{T}) where {T<:Complex} T_OUT = Base.promote_op(imag, T) - out = cuNumeric.zeros(T_OUT, size(input)) + out = NDArray{T_OUT}(undef, size(input)) return nda_unary_op!(out, cuNumeric.IMAG, input) end Base.imag(input::NDArray{T}) where {T<:Real} = cuNumeric.zeros(T, size(input)) function Base.conj(input::NDArray{T}) where {T<:Complex} - out = cuNumeric.zeros(T, size(input)) + out = NDArray{T}(undef, size(input)) return nda_unary_op!(out, cuNumeric.CONJ, input) end Base.conj(input::NDArray{<:Real}) = input @@ -87,7 +124,7 @@ end # Fallbacks for Real types @inline function __broadcast(f::typeof(Base.real), out::NDArray, input::NDArray{<:Real}) # real(real_array) is just the array - return nda_unary_op!(out, cuNumeric.IDENTITY, input) + return nda_unary_op!(out, cuNumeric.COPY, input) end @inline function __broadcast(f::typeof(Base.imag), out::NDArray, input::NDArray{<:Real}) # imag(real_array) is all zeros @@ -95,7 +132,7 @@ end end @inline function __broadcast(f::typeof(Base.conj), out::NDArray, input::NDArray{<:Real}) # conj(real_array) is just the array - return nda_unary_op!(out, cuNumeric.IDENTITY, input) + return nda_unary_op!(out, cuNumeric.COPY, input) end function Base.:(-)(input::NDArray{Bool}) @@ -105,6 +142,17 @@ function Base.:(-)(input::NDArray{Bool}) return out end +# Broadcast `.-` on Bool: Julia `-true === -1`, so promote then NEGATIVE. +@inline function __broadcast( + ::typeof(Base.:(-)), out::NDArray{O}, input::NDArray{Bool} +) where {O<:Integer} + assertpromotion(".-", Bool, O) + promoted = unchecked_promote_arr(input, O) + result = nda_unary_op!(out, cuNumeric.NEGATIVE, promoted) + destroy!(promoted) + return result +end + function Base.sqrt(input::NDArray{T,2}) where {T} return error("cuNumeric.jl does not support matrix square root.") end @@ -118,7 +166,7 @@ end @inline function __broadcast( ::typeof(Base.literal_pow), out::NDArray{O}, _, input::NDArray{O}, ::Type{Val{-1}} ) where {O} - nda_move(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED return out end @@ -126,19 +174,19 @@ end ::typeof(Base.literal_pow), out::NDArray{O}, _, input::NDArray, ::Type{Val{-1}} ) where {O} promoted = checked_promote_arr(input, O) # always a new array when eltype ≠ O - nda_move(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED destroy!(promoted) return out end @inline function __broadcast(::typeof(Base.inv), out::NDArray{O}, input::NDArray{O}) where {O} - nda_move(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ input) #! REPLACE WITH RECIP ONCE FIXED return out end @inline function __broadcast(::typeof(Base.inv), out::NDArray{O}, input::NDArray) where {O} promoted = checked_promote_arr(input, O) # always a new array when eltype ≠ O - nda_move(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED + _store_broadcast_result!(out, O(1) ./ promoted) #! REPLACE WITH RECIP ONCE FIXED destroy!(promoted) return out end @@ -169,6 +217,32 @@ for (julia_fn, op_code) in unary_op_map_no_args end end +# INVERT rejects Bool; on Bool, Julia `~` is the same as `!`. +@inline function __broadcast(::typeof(Base.:(~)), out::NDArray{Bool}, input::NDArray{Bool}) + return nda_unary_op!(out, cuNumeric.LOGICAL_NOT, input) +end + +@noinline function _unsupported_round_broadcast() + return throw( + ArgumentError( + "cuNumeric.jl only supports round.(A) (default RoundNearest / IEEE rint). " * + "digits, sigdigits, and RoundingMode are not supported.", + ), + ) +end + +# Reject keyword forms before fusion captures Julia's keyword wrapper as a GPU callable. +@inline function Base.Broadcast.broadcasted_kwsyntax( + ::typeof(Base.round), ::NDArray; kwargs... +) + return _unsupported_round_broadcast() +end + +# Only the 1-arg `round.(A)` method above is supported (IEEE rint / RoundNearest). +@inline function __broadcast(::typeof(Base.round), ::NDArray, ::NDArray, extra...) + return _unsupported_round_broadcast() +end + # Some functions always return floats even when given integers # in the case where the output is determined to be float, but # the input is integer, we first promote the input to float. @@ -192,29 +266,9 @@ for (julia_fn, op_code) in floaty_unary_ops_no_args end end -# global const unary_op_map_with_args = Dict{Function, Int}( -# Base.angle => Int(cuNumeric.ANGLE), -# Base.ceil => Int(cuNumeric.CEIL), #* HAS EXTRA ARGS -# Base.clamp => Int(cuNumeric.CLIP), #* HAS EXTRA ARGS -# Base.floor => cuNumeric.FLOOR, #! Doesnt support Bool, I do not feel like dealing with this right now... -# Base.trunc => Int(cuNumeric.TRUNC) #* HAS EXTRA ARGS -# missing => Int(cuNumeric.RINT), #figure out which version of round -# missing => Int(cuNumeric.ROUND), #figure out which version of round -# ) - -# for (base_func, op_code) in unary_op_map_with_args -# @eval begin -# @doc """ -# $($(Symbol(base_func))) : A unary operation acting on an NDArray -# """ -# function $(Symbol(base_func))(input::NDArray, args...) -# out = cuNumeric.zeros(eltype(input), size(input)) # not sure this is ok for performance -# extra_args = cuNumeric.StdVector{cuNumeric.LegateScalar}([LegateScalar(a) for a in args]) -# unary_op(out, $(op_code), input, extra_args) -# return out -# end -# end -# end +# CLIP / clamp needs extra_args (lo, hi). nda_unary_op! currently ccall's +# without extra scalars, so clamp is not wired. Do not revive the old +# StdVector{LegateScalar} path unless that C API exists. @doc""" Supported Unary Reduction Operations @@ -228,11 +282,18 @@ The following unary reduction operations are supported and can be applied direct • `minimum` • `prod` • `sum` + • `mean` + • `var` / `std` (sample / `corrected=true`; real types only) + • `argmax` / `argmin` (1-d only) -These operations follow standard Julia semantics. +Full reductions return an **`CNScalar`** backed by a 0D NDArray. Use `fetch` or +`only` when you need a host value. Reduction over specific dimensions is supported via the `dims` keyword argument, -following the same semantics as Julia's base reduction functions. +following the same keepdims semantics as Julia's base reduction functions. +Multi-axis `dims=(1,2)` is implemented as sequential single-axis reductions +(the C++ kernel accepts only one axis at a time). `argmax`/`argmin` are 1-d +only (Base's N-d / `dims=` path returns `CartesianIndex`). Examples -------- @@ -242,6 +303,7 @@ A = cuNumeric.ones(5) maximum(A) sum(A) +mean(A) # Reduce over a specific dimension B = cuNumeric.ones(3, 4) @@ -252,11 +314,9 @@ sum(B, dims=2) # 3×1 result sum(B, dims=(1,2)) # 1×1 result ``` """ -global const unary_reduction_map = Dict{Function,UnaryRedCode}( - # Base.argmax => cuNumeric.ARGMAX, #* WILL BE OFF BY 1 - # Base.argmin => cuNumeric.ARGMIN, #* WILL BE OFF BY 1 +const unary_reduction_map = Dict{Function,UnaryRedCode}( + # ARGMAX/ARGMIN: 1-d Base.argmax/argmin below, not this map. #missing => cuNumeric.CONTAINS, # strings or also integral types - #missing => cuNumeric.COUNT_NONZERO, # Base.count(!Base.iszero, arr) Base.maximum => cuNumeric.MAX, Base.minimum => cuNumeric.MIN, #missing => cuNumeric.NANARGMAX, @@ -266,12 +326,10 @@ global const unary_reduction_map = Dict{Function,UnaryRedCode}( #missing => cuNumeric.NANPROD, Base.prod => cuNumeric.PROD, Base.sum => cuNumeric.SUM, - #missing => cuNumeric.SUM_SQUARES, - # StatsBase.var => cuNumeric.VARIANCE #! dies horribly?? wth + # VARIANCE opcode is unused: compose sample var from mean / sum instead. ) -#! IT WOULD BE NICE IF THESE JUST RETURNED SCALARS WHEN APPROPRIATE -# #*TODO HOW TO GET THESE ACTING ON CERTAIN DIMS +# Public full reductions wrap backend 0D NDArrays as CNScalars. function _unary_reduction_apply(out, op_code, input::NDArray{T}, ::Type{T}) where {T} return nda_unary_reduction(out, op_code, input) @@ -295,7 +353,15 @@ function _unary_reduction_axes_apply(op_code, input::NDArray, ::Type{U}, axes) w return result end +@inline function _assert_numeric_reduction(base_func, ::Type{T}) where {T} + _struct_storage_type(T) && throw( + ArgumentError("$(base_func) does not support NDArrays of struct element type $(T)") + ) + return nothing +end + function _unary_reduction_impl(base_func, op_code, input::NDArray{T}, ::Colon) where {T} + _assert_numeric_reduction(base_func, T) T_OUT = Base.promote_op(base_func, Vector{T}) is_wider_type(T_OUT, T) && assertpromotion(base_func, T, T_OUT) out = cuNumeric.zeros(T_OUT) @@ -303,30 +369,35 @@ function _unary_reduction_impl(base_func, op_code, input::NDArray{T}, ::Colon) w end function _unary_reduction_impl(base_func, op_code, input::NDArray{T,N}, dims::Integer) where {T,N} + _assert_numeric_reduction(base_func, T) T_OUT = Base.promote_op(base_func, Vector{T}) is_wider_type(T_OUT, T) && assertpromotion(base_func, T, T_OUT) axes = Int32[dims - 1] return _unary_reduction_axes_apply(op_code, input, T_OUT, axes) end +# cupynumeric throws if axes.size() > 1. Compose keepdims single-axis reductions +# so `sum(A; dims=(1,2))` matches Julia's 1×1 (etc.) shape. function _unary_reduction_impl(base_func, op_code, input::NDArray{T,N}, dims::Tuple) where {T,N} - if length(dims) > 1 - error( - "$(base_func): reducing over multiple dimensions is not yet supported. Got dims=$dims" - ) + n = length(dims) + n == 0 && return copy(input) + n == 1 && return _unary_reduction_impl(base_func, op_code, input, dims[1]) + result = input + owned = false + for d in dims + next = _unary_reduction_impl(base_func, op_code, result, d) + owned && destroy!(result) + result = next + owned = true end - # single element tuple - T_OUT = Base.promote_op(base_func, Vector{T}) - is_wider_type(T_OUT, T) && assertpromotion(base_func, T, T_OUT) - axes = Int32[dims[1] - 1] - return _unary_reduction_axes_apply(op_code, input, T_OUT, axes) + return result end # Generate code for all unary reductions. for (base_func, op_code) in unary_reduction_map @eval begin function $(Symbol(base_func))(input::NDArray{T,N}; dims=Colon()) where {T,N} - return _unary_reduction_impl($base_func, $(op_code), input, dims) + return _scalar_result(_unary_reduction_impl($base_func, $(op_code), input, dims)) end end end @@ -336,29 +407,178 @@ function _bool_reduction_impl(op_code, input::NDArray{Bool}, ::Colon) return nda_unary_reduction(out, op_code, input) end +function _bool_reduction_impl(op_code, input::NDArray{Bool}, dim::Integer) + return nda_unary_reduction_axes(op_code, input, Int32[dim - 1], true) +end + +function _bool_reduction_impl(op_code, input::NDArray{Bool}, dims::Tuple) + n = length(dims) + n == 0 && return copy(input) + n == 1 && return _bool_reduction_impl(op_code, input, dims[1]) + result = input + owned = false + for d in dims + next = _bool_reduction_impl(op_code, result, d) + owned && destroy!(result) + result = next + owned = true + end + return result +end + function _bool_reduction_impl(op_code, input::NDArray{Bool}, dims) - axes = collect(Int32, (d - 1 for d in (dims isa Integer ? (dims,) : dims))) - return nda_unary_reduction_axes(op_code, input, axes, true) + return _bool_reduction_impl(op_code, input, Tuple(dims)) end function Base.all(input::NDArray{Bool}; dims=Colon()) - return _bool_reduction_impl(cuNumeric.ALL, input, dims) + return _scalar_result(_bool_reduction_impl(cuNumeric.ALL, input, dims)) end function Base.any(input::NDArray{Bool}; dims=Colon()) - return _bool_reduction_impl(cuNumeric.ANY, input, dims) + return _scalar_result(_bool_reduction_impl(cuNumeric.ANY, input, dims)) +end + +# Compare on-device against `zero(T)` / `_eye(T, n)` (identity filled with `one(T)`). +# Returns a 0D `NDArray{Bool}` — not a Julia `Bool`. +function Base.iszero(A::NDArray{T}) where {T} + return all(A .== zero(T)) +end +function Base.isone(A::NDArray{T,2}) where {T} + m, n = size(A) + m != n && return cnscalar(NDArray(false)) # LinearAlgebra.isone: only square matrices + return all(A .== _eye(T, m)) end # Boolean multiplication is logical conjunction. cuPyNumeric's PROD reduction # uses a numeric fill identity, which Legate rejects for a Boolean target. function Base.prod(input::NDArray{Bool}; dims=Colon()) - return _unary_reduction_impl(Base.prod, cuNumeric.ALL, input, dims) + return _scalar_result(_unary_reduction_impl(Base.prod, cuNumeric.ALL, input, dims)) +end + +# Number of elements a reduction with `dims` collapses. Used by mean/var/std. +_reduction_nelem(arr::NDArray, ::Colon) = Int(prod(size(arr))) +_reduction_nelem(arr::NDArray, dim::Integer) = Int(size(arr, dim)) +function _reduction_nelem(arr::NDArray, dims::Tuple) + n = 1 + for d in dims + n *= Int(size(arr, d)) + end + return n +end + +# Divide an NDArray by a count without going through 0-d broadcast, which +# unwraps to a Julia scalar in Broadcast.copy. +function _div_nelem(arr::NDArray{T}, n::Integer) where {T} + FT = float(T) + return (FT(1) / FT(n)) * arr +end + +""" + mean(A::NDArray; dims=:) + +Arithmetic mean of `A`. Full reduction returns an `CNScalar`, not a host +scalar. With `dims`, the reduced axes are kept as size 1, matching Base. +""" +function mean(arr::NDArray; dims=Colon()) + s = sum(arr; dims=dims) + result = _div_nelem(s, _reduction_nelem(arr, dims)) + destroy!(s) + return result +end + +""" + var(A::NDArray; corrected=true, mean=nothing, dims=:) + std(A::NDArray; corrected=true, mean=nothing, dims=:) + +Sample variance and standard deviation (`corrected=true`, divisor `n-1`), +matching Julia / StatsBase. Real types only. Returns an `CNScalar` or dimension-preserving +`NDArray`, not a Julia scalar. +""" +function var(arr::NDArray{T}; corrected::Bool=true, mean=nothing, dims=Colon()) where {T<:Real} + μ = isnothing(mean) ? cuNumeric.mean(arr; dims=dims) : mean + centered = arr .- μ + isnothing(mean) && μ isa Union{NDArray,CNScalar} && destroy!(μ) + sq = centered .^ 2 + destroy!(centered) + s = sum(sq; dims=dims) + destroy!(sq) + n = _reduction_nelem(arr, dims) + denom = corrected ? n - 1 : n + result = _div_nelem(s, denom) + destroy!(s) + return result +end + +function std(arr::NDArray{T}; corrected::Bool=true, mean=nothing, dims=Colon()) where {T<:Real} + return _sqrt_ndarray(var(arr; corrected=corrected, mean=mean, dims=dims)) end -#! ONLY ADD ONCE REDUCTIONS RETURN A SCALAR -# function StatsBase.mean(arr::NDArray{T}) where T -# return sum(arr) ./ prod(size(arr)) +# SQRT kernel rejects 0-d (shape [] vs [1]). Wrap the host sqrt back into a 0-d array. +function _sqrt_ndarray(v::NDArray{T,0}) where {T} + s = T(sqrt(fetch(v))) + destroy!(v) + return NDArray(s) +end + +function _sqrt_ndarray(v::NDArray{T}) where {T} + out = similar(v) + nda_unary_op!(out, cuNumeric.SQRT, v) + destroy!(v) + return out +end + +# function _count_nonzero(input::NDArray, dims) +# nz = input .!= zero(eltype(input)) +# result = sum(nz; dims=dims) +# destroy!(nz) +# return result # end +# +# """ +# count(A::NDArray{Bool}; dims=:) +# count(!iszero, A::NDArray; dims=:) +# +# Count `true` values in a `Bool` array, or nonzeros in a numeric array. +# Returns an `CNScalar` or dimension-preserving `NDArray` of integers, not a Julia `Int`. +# """ +# function count(arr::NDArray{Bool}; dims=Colon()) +# return _count_nonzero(arr, dims) +# end +# +# function count(::ComposedFunction{typeof(!),typeof(iszero)}, arr::NDArray; dims=Colon()) +# return _count_nonzero(arr, dims) +# end + +# Kernel ARGMAX/ARGMIN are 0-based. Julia indices are 1-based. +function _indices_to_one_based(raw::NDArray{Int64}) + result = nda_add_scalar(raw, Int64(1)) + destroy!(raw) + return result +end + +""" + argmax(A::NDArray{<:Any,1}) + +1-based index of the first extremum, as a 0-d `NDArray{Int64}` (not a Julia +`Int`). 1-d only. Complex arrays are not supported. +""" +function argmax(arr::NDArray{T,1}) where {T} + T <: Complex && throw(ArgumentError("argmax/argmin are not supported for complex arrays")) + raw = nda_unary_reduction_axes(cuNumeric.ARGMAX, arr, Int32[], false) + return _indices_to_one_based(raw) +end + +""" + argmin(A::NDArray{<:Any,1}) + +1-based index of the first extremum, as a 0-d `NDArray{Int64}` (not a Julia +`Int`). 1-d only. Complex arrays are not supported. +""" +function argmin(arr::NDArray{T,1}) where {T} + T <: Complex && throw(ArgumentError("argmax/argmin are not supported for complex arrays")) + raw = nda_unary_reduction_axes(cuNumeric.ARGMIN, arr, Int32[], false) + return _indices_to_one_based(raw) +end # function Base.reduce(f::Function, arr::NDArray) # return f(arr) diff --git a/src/ndarray/vector_linalg.jl b/src/ndarray/vector_linalg.jl new file mode 100644 index 000000000..5981b9dec --- /dev/null +++ b/src/ndarray/vector_linalg.jl @@ -0,0 +1,191 @@ +# LinearAlgebra interfaces retain asynchronous NDArray reduction results. +const _LA_FLOAT = Union{SUPPORTED_FLOAT_TYPES,SUPPORTED_COMPLEX_TYPES} +const _LA_INTEGER = Union{SUPPORTED_INT_TYPES,Bool} + +_matmul_eltype(::Type{T}) where {T<:_LA_FLOAT} = T +function _matmul_eltype(::Type{T}) where {T<:_LA_INTEGER} + throw(ArgumentError( + "NDArray matrix multiplication does not support integer-integer operands (including Bool); " * + "the inputs promote to $T. Convert an operand to a floating-point type explicitly.", + )) +end + +""" + mul!(y::NDArray, A::NDArray, x::NDArray) + +Store the matrix-vector product `A * x` in `y`. +""" +function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + return mul!(y, A, x, true, false) +end + +_mul_output_scale(x::Number, ::Type{T}) where {T} = convert(T, x) +_mul_output_scale(x::NDArray, ::Type{T}) where {T} = x + +function _linalg_mul!(C::NDArray{T}, cm, A::NDArray{TA}, am, B::NDArray{TB}, bm, α, β) where {T,TA,TB} + required = _matmul_eltype(promote_type(TA, TB)) + promote_type(required, T) === T || throw(ArgumentError("mul! output type $T cannot hold promoted input type $required")) + Ap = checked_promote_arr(mul!, A, T) + Bp = checked_promote_arr(mul!, B, T) + α = _scale_storage(α) + β = _scale_storage(β) + if _host_iszero(α) || isempty(A) || isempty(B) || isempty(C) + _contract_prepare(C, cm, Ap, am, Bp, bm) + if !isempty(C) + if _host_iszero(β) + fill!(C, zero(T)) + else + C .*= _mul_output_scale(β, T) + end + end + else + _contract_same_type!(C, cm, Ap, am, Bp, bm, α, β) + end + Ap !== A && destroy!(Ap) + Bp !== B && destroy!(Bp) + return C +end + +""" + mul!(y::NDArray, A::NDArray, x::NDArray, α, β) + +Store `α * A * x + β * y` in `y`. The destination must not alias an input. +`α` and `β` accept host numbers, 0D `NDArray{T,0}` values, or `CNScalar` wrappers. +""" +function LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, α::Union{Number,DeviceScalar}, β::Union{Number,DeviceScalar}) + return _linalg_mul!(y, "i", A, "ij", x, "j", α, β) +end + +""" + mul!(C::NDArray, A::NDArray, B::NDArray, α, β) + +Store `α * A * B + β * C` in `C`. The destination must not alias an input. +`α` and `β` accept host numbers, 0D `NDArray{T,0}` values, or `CNScalar` wrappers. +""" +function LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, B::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, α::Union{Number,DeviceScalar}, β::Union{Number,DeviceScalar}) + return _linalg_mul!(C, "ij", A, "ik", B, "kj", α, β) +end + +# Resolve intersections with LinearAlgebra's host Number signatures. +LinearAlgebra.mul!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, α::Number, β::Number) = + _linalg_mul!(y, "i", A, "ij", x, "j", α, β) +LinearAlgebra.mul!(C::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, A::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, B::NDArray{<:SUPPORTED_ARRAY_TYPES,2}, α::Number, β::Number) = + _linalg_mul!(C, "ij", A, "ik", B, "kj", α, β) + +function Base.:*(A::NDArray{TA,2}, x::NDArray{TX,1}) where {TA<:SUPPORTED_ARRAY_TYPES,TX<:SUPPORTED_ARRAY_TYPES} + T = _matmul_eltype(promote_type(TA, TX)) + size(A, 2) == length(x) || throw(DimensionMismatch("matrix-vector dimensions do not match")) + y = NDArray{T}(undef, size(A, 1)) + return mul!(y, A, x) +end + +_dot_eltype(::Type{T}) where {T} = T +_dot_eltype(::Type{Bool}) = Int +_dot_same_type(x::NDArray{T,1}, y::NDArray{T,1}) where {T<:Real} = nda_dot(x, y) +function _dot_same_type(x::NDArray{T,1}, y::NDArray{T,1}) where {T<:Complex} + cx = conj.(x) + result = nda_dot(cx, y) + destroy!(cx) + return result +end + +""" + dot(x::NDArray, y::NDArray) + +Return the vector inner product as a `CNScalar`. Complex inputs conjugate `x`. +""" +function LinearAlgebra.dot(x::NDArray{TX,1}, y::NDArray{TY,1}) where {TX<:SUPPORTED_ARRAY_TYPES,TY<:SUPPORTED_ARRAY_TYPES} + length(x) == length(y) || throw(DimensionMismatch("dot vector lengths do not match")) + T = _dot_eltype(promote_type(TX, TY)) + xp = checked_promote_arr(dot, x, T) + yp = checked_promote_arr(dot, y, T) + result = isempty(x) ? cuNumeric.zeros(T, ()) : _dot_same_type(xp, yp) + xp !== x && destroy!(xp) + yp !== y && destroy!(yp) + return cnscalar(result) +end + +_norm_nonzero(v) = ifelse(iszero(v), zero(real(v)), one(real(v))) + +LinearAlgebra.norm(x::NDArray{<:SUPPORTED_ARRAY_TYPES}, p::NDArray{<:Real,0}) = + norm(x, _maybe_fetch(p)) + +""" + norm(x::NDArray, p::Real=2) + +Return the entrywise `p`-norm as a real `CNScalar`, not a matrix operator norm. +Dense-array norms currently require a GPU. Unscaled accumulation can overflow +or underflow. +""" +function LinearAlgebra.norm(x::NDArray{T}, p::Real=2) where {T<:_LA_FLOAT} + p = _maybe_fetch(p) + R = real(T) + isempty(x) && return cnscalar(cuNumeric.zeros(R, ())) + p == 0 && return sum(_norm_nonzero, x) + p == 1 && return sum(abs, x) + p == Inf && return maximum(abs, x) + p == -Inf && return minimum(abs, x) + isnan(p) && return cnscalar(NDArray(R(NaN))) + exponent = R(p) + total = p == 2 ? sum(abs2, x) : sum(v -> abs(v)^exponent, x) + # Optimization opportunity: a specialized reduction could fuse the root into + # its final combine kernel, after all partitions' contributions are combined. + # dims=() uses the elementwise singleton path for this 0D result: one root + # kernel, without unwrapping or launching another full reduction. + result = p == 2 ? mapreduce(sqrt, +, total; dims=()) : + mapreduce(v -> v^inv(exponent), +, total; dims=()) + destroy!(total) + return result +end + +function LinearAlgebra.norm(x::NDArray{T}, p::Real=2) where {T<:_LA_INTEGER} + xp = checked_promote_arr(norm, x, float(T)) + result = norm(xp, p) + destroy!(xp) + return result +end + +""" + axpy!(α, x::NDArray, y::NDArray) + +Update and return `y` with `y = α * x + y`. +""" +function LinearAlgebra.axpy!(α::Union{Number,DeviceScalar}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + length(x) == length(y) || throw(DimensionMismatch("axpy! vector lengths do not match")) + _host_iszero(α) && return y + y .= α .* x .+ y + return y +end + +""" + axpby!(α, x::NDArray, β, y::NDArray) + +Update and return `y` with `y = α * x + β * y`. +""" +function LinearAlgebra.axpby!(α::Union{Number,DeviceScalar}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, β::Union{Number,DeviceScalar}, y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + length(x) == length(y) || throw(DimensionMismatch("axpby! vector lengths do not match")) + _host_iszero(α) && _host_isone(β) && return y + y .= α .* x .+ β .* y + return y +end + +function LinearAlgebra.rmul!(x::NDArray{<:SUPPORTED_ARRAY_TYPES}, α::Number) + x .*= α + return x +end + +function LinearAlgebra.lmul!(α::Number, x::NDArray{<:SUPPORTED_ARRAY_TYPES}) + x .= α .* x + return x +end + +# Keep the Number signatures above to resolve Base's AbstractArray/Number +# intersections. Raw device scalars share their implementation via the wrapper. +LinearAlgebra.rmul!(x::NDArray{<:SUPPORTED_ARRAY_TYPES}, α::NDArray{<:Any,0}) = rmul!(x, cnscalar(α)) +LinearAlgebra.lmul!(α::NDArray{<:Any,0}, x::NDArray{<:SUPPORTED_ARRAY_TYPES}) = lmul!(cnscalar(α), x) + +function LinearAlgebra.ldiv!(y::NDArray{<:SUPPORTED_ARRAY_TYPES,1}, D::DiagonalNDArray{<:SUPPORTED_ARRAY_TYPES}, x::NDArray{<:SUPPORTED_ARRAY_TYPES,1}) + length(x) == length(y) == size(D, 1) || throw(DimensionMismatch("diagonal solve dimensions do not match")) + y .= x ./ _diag_vec(D) + return y +end diff --git a/src/scoping/accelerate.jl b/src/scoping/accelerate.jl new file mode 100644 index 000000000..2ddd6c47a --- /dev/null +++ b/src/scoping/accelerate.jl @@ -0,0 +1,264 @@ +using MacroTools: MacroTools + +# Rejected everywhere: control flow makes last-use freeing unsound (a temp freed +# after its textual last use could be revived on another path). +const _CONTROL_FLOW_HEADS = (:if, :elseif, :for, :while, :try, :do, :break, :continue, :&&, :||) + +_is_function_lhs(::Any) = false +function _is_function_lhs(lhs::Expr) + lhs.head === :call && return true + lhs.head === :where && return _is_function_lhs(first(lhs.args)) + return false +end + +function _reject_nonstraightline(body) + MacroTools.postwalk(body) do node + node isa Expr || return node + if node.head === :function || node.head === :-> || + (node.head === :(=) && _is_function_lhs(first(node.args))) + error("@accelerate: nested/anonymous function definitions are not supported") + end + if node.head in _CONTROL_FLOW_HEADS + error( + "@accelerate: control flow (`$(node.head)`) is not supported; " * + "only straight-line code can be accelerated", + ) + end + return node + end + return nothing +end + +# Function args are caller-owned: protected roots, never freed or fused away. +function _argument_symbols(def) + names = Set{Symbol}() + for arg in Iterators.flatten((get(def, :args, Any[]), get(def, :kwargs, Any[]))) + name, _, _, _ = MacroTools.splitarg(arg) + name isa Symbol && push!(names, name) + end + return names +end + +# Function form: validate, expand `@.`, normalize trailing `return`, run the +# lifetime/fusion passes protecting `protected_roots` (the args). +function _accelerate_rewrite(body, caller::Module, protected_roots::Set{Symbol}) + _reject_nonstraightline(body) + body = _normalize_return(_expand_dot_macros(body, caller)) + on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing + return process_ndarray_scope(body; on_rewrite, protected_roots) +end + +# `begin`/expr form: 1:1 Julia scope (no `let`) — named bindings stay live. +# On GPU, same-shape chains fuse into one multi-output launch; otherwise only +# anonymous temporaries (slices) are freed. +function _accelerate_block_soft(block, caller::Module) + _reject_nonstraightline(block) + nb = _normalize_return(_expand_dot_macros(block, caller)) + on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing + fallback = process_ndarray_scope( + nb; on_rewrite, protected_roots=_assigned_symbols(nb) + ) + @static if FUSE_BROADCAST_EXPRS + fused = _try_fuse_block_multi(nb) + if !isnothing(fused) + return quote + if cuNumeric._has_gpu_target() + $fused + else + $fallback + end + end + end + end + # Protect named bindings; free only anonymous temps. + return fallback +end + +# Flatten a `let` node's bindings + body into one statement block. +function _let_body(letexpr::Expr) + stmts = Any[] + for part in letexpr.args + if part isa Expr && part.head === :block + append!(stmts, part.args) + elseif part isa Expr && part.head === :(=) + push!(stmts, part) + elseif !isnothing(part) + push!(stmts, part) + end + end + return Expr(:block, stmts...) +end + +# `let` form: hard scope. Full analysis — combine single-use producers, free +# every non-returned temp — re-wrapped in a `let` so only the result escapes. +function _accelerate_block_hard(letexpr, caller::Module) + body = _let_body(letexpr) + _reject_nonstraightline(body) + nb = _normalize_return(_expand_dot_macros(body, caller)) + on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing + rewritten = process_ndarray_scope(nb; on_rewrite, protected_roots=Set{Symbol}()) + bindings = union(_assigned_symbols(nb), _assigned_symbols(rewritten)) + return _lexical_scope(rewritten, bindings) +end + +# Drop the leading dot: `.+` -> `+`, `.^` -> `^`. +_undot(op::Symbol) = Symbol(chop(string(op); head=1, tail=0)) + +# Dotted RHS -> lazy `Base.broadcasted(...)`: chain vars become `MatRef{k}`, +# slice leaves are hoisted into `hoisted` (temp => slice) to free post-launch. +# `nothing` when not lowerable (caller falls back). +function _to_broadcasted(expr, idx::AbstractDict{Symbol,Int}, hoisted::Vector) + if expr isa Symbol + haskey(idx, expr) && return :(cuNumeric.MatRef($(idx[expr]))) + return expr + end + expr isa Expr || return expr + if expr.head === :ref + # Slice temp: bail if it indexes a chain var, else hoist to free later. + any(s -> haskey(idx, s), walk_symbols(expr)) && return nothing + tmp = gensym(:slice) + push!(hoisted, tmp => expr) + return tmp + end + if expr.head === :call && expr.args[1] isa Symbol && _is_broadcast_op(expr.args[1]) + cargs = map(a -> _to_broadcasted(a, idx, hoisted), expr.args[2:end]) + any(isnothing, cargs) && return nothing + return Expr(:call, :(Base.broadcasted), _undot(expr.args[1]), cargs...) + end + if expr.head === :. && length(expr.args) == 2 && + expr.args[2] isa Expr && expr.args[2].head === :tuple + cargs = map(a -> _to_broadcasted(a, idx, hoisted), expr.args[2].args) + any(isnothing, cargs) && return nothing + return Expr(:call, :(Base.broadcasted), expr.args[1], cargs...) + end + # Non-dotted scalar leaf; unsafe if it reads a chain var as a scalar. + any(s -> haskey(idx, s), walk_symbols(expr)) && return nothing + return expr +end + +function _is_top_broadcast(rhs) + return ( + rhs isa Expr && rhs.head === :call && rhs.args[1] isa Symbol && + _is_broadcast_op(rhs.args[1]) + ) || + ( + rhs isa Expr && rhs.head === :. && length(rhs.args) == 2 && + rhs.args[2] isa Expr && rhs.args[2].head === :tuple + ) +end + +# SSA chain of `sym = ` (+ optional trailing return) -> one +# multi-output launch materializing each result. `nothing` -> caller falls back. +function _try_fuse_block_multi(block) + stmts = _scope_statements(block) + isnothing(stmts) && return nothing + stmts = filter(s -> !(s isa LineNumberNode), stmts) + isempty(stmts) && return nothing + + assigns = stmts + ret = nothing + if isnothing(_assignment(last(stmts))) + ret = last(stmts) + assigns = stmts[1:(end - 1)] + end + length(assigns) >= 2 || return nothing + + syms = Symbol[] + idx = Dict{Symbol,Int}() + seg_exprs = Any[] + hoisted = Pair{Symbol,Any}[] # slice temp => slice expr + for stmt in assigns + a = _assignment(stmt) + isnothing(a) && return nothing + a.lhs isa Symbol || return nothing # no indexed-assign in this path + a.lhs in syms && return nothing # SSA: no reassignment + _is_top_broadcast(a.rhs) || return nothing # must be a real broadcast + seg = _to_broadcasted(a.rhs, idx, hoisted) + isnothing(seg) && return nothing + push!(seg_exprs, seg) + push!(syms, a.lhs) + idx[a.lhs] = length(syms) + end + outs = gensym(:outs) + slice_binds = [:($t = $e) for (t, e) in hoisted] + slice_frees = [:(cuNumeric.maybe_insert_delete($t)) for (t, _) in hoisted] + binds = [:($(syms[i]) = $outs[$i]) for i in eachindex(syms)] + value = isnothing(ret) ? last(syms) : ret + return quote + $(slice_binds...) # materialize slice views + $outs = cuNumeric.copyto_fused_multi_alloc!(($(seg_exprs...),)) + $(slice_frees...) # free them after the launch + $(binds...) + $value + end +end + +# AST `@accelerate` emits (pre-`esc`); shared with `@show_lifetimes`. Dispatch: +# function def / `let` (hard scope) / `begin`-expr (soft, 1:1 Julia scope). +function _accelerate_expand(input, caller::Module) + if MacroTools.isdef(input) + def = MacroTools.splitdef(input) + def[:body] = _accelerate_rewrite(def[:body], caller, _argument_symbols(def)) + return MacroTools.combinedef(def) + elseif input isa Expr && input.head === :let + return _accelerate_block_hard(input, caller) + end + return _accelerate_block_soft(input, caller) +end + +@doc""" + @accelerate function f(args...) ... end + @accelerate begin ... end + @accelerate let ... end + @accelerate expr + +Optimize straight-line array code by coordinating CUDA broadcast fusion within +expressions, fusion across broadcast statements, and scope-aware cleanup of +materialized temporaries. Control flow and nested/anonymous functions are +rejected. + +Single-use producers may fuse across intervening read-only calculations when +their inputs are unchanged. Unknown calls remain barriers. Function-form fusion +across writes to other array arguments checks storage overlap at runtime and +retains materialized intermediates when those arguments overlap. + +Four forms determine which values must remain valid: + + * **function** (preferred): arguments and returned values are protected; + non-returned locals may fuse into consumers or be freed after their last use. + * **`begin`**: creates no new Julia scope, so every named binding stays live; + eligible GPU chains may use one multi-output kernel that materializes them. + * **`let`**: creates a local scope; only the result escapes, so other locals may + fuse away or be freed after their last use. + * **expression**: materializes and returns one expression; eligible operations + fuse within it and transient temporaries are released. + +```julia +@accelerate function step(u, v) # c may fuse away; the result is returned + c = u .* v + return c .^ 2 +end +a, b = @accelerate begin # a and b both stay live, one GPU launch + a = x .* y + b = a .+ 1 + (a, b) +end +result = @accelerate (x .+ y .* z) +``` +""" +macro accelerate(input) + return esc(_accelerate_expand(input, __module__)) +end + +@doc""" + @show_lifetimes function f(args...) ... end + @show_lifetimes begin ... end + @show_lifetimes let ... end + +Print the exact expansion [`@accelerate`](@ref) produces for the same input +(all forms), without running it; inserted frees are highlighted. Pure AST work. +""" +macro show_lifetimes(input) + expansion = _accelerate_expand(input, __module__) + return :(print_lifetime_analysis($(QuoteNode(expansion)))) +end diff --git a/src/scoping/broadcast_lifetimes.jl b/src/scoping/broadcast_lifetimes.jl index 47bb6b301..49ae1c63e 100644 --- a/src/scoping/broadcast_lifetimes.jl +++ b/src/scoping/broadcast_lifetimes.jl @@ -74,20 +74,21 @@ function rewrite_broadcast_lifetimes(scope) return :($lhs = $new_rhs), temps end - # A `.=` RHS is a broadcast tree: only its slices are hoisted. + # A dotted-assignment RHS is a broadcast tree: only its slices are hoisted. broadcast_assignment = _broadcast_assignment(expr) if !isnothing(broadcast_assignment) (; lhs, rhs) = broadcast_assignment + op = expr.head # NDArray slices are writable views. Hoist the destination slice so # the fused broadcast writes through it, then destroy its handle. lhs_reference = _reference(lhs) if isnothing(lhs_reference) new_lhs, lhs_temps = rewrite_materialized(lhs) else - new_lhs, lhs_temps = fresh_tmp(lhs) + new_lhs, lhs_temps = fresh_tmp(:(Base.@view $lhs)) end new_rhs, rhs_temps = rewrite_lazy_broadcast(rhs, Dict{Any,Symbol}()) - return Expr(:(.=), new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) + return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) end reference = _reference(expr) @@ -115,9 +116,30 @@ function rewrite_broadcast_lifetimes(scope) return _prepend_statements(rewritten, temps), assigned_vars end -function process_broadcast_lifetime_scope(scope; on_rewrite=nothing) - # Returned producers must stay materialized, so exempt them from fusion. - protected = _returned_symbols(scope) - scope = InterBroadcastFusion.rewrite_scope(scope; on_rewrite, protected) - return _process_lifetime_scope(scope, rewrite_broadcast_lifetimes) +# Scalars/immutable scalar parameter records cannot share mutable array storage. +_fusion_disjoint(a, b) = isbitstype(typeof(a)) || isbitstype(typeof(b)) +_fusion_disjoint(a::AbstractArray, b::AbstractArray) = !Base.mightalias(a, b) +_fusion_disjoint(a::NDArray, b::AbstractArray) = false +_fusion_disjoint(a::AbstractArray, b::NDArray) = false +_fusion_disjoint(a::NDArray, b::NDArray) = !nda_overlaps(a, b) + +function process_broadcast_lifetime_scope( + scope; on_rewrite=nothing, protected_roots=Set{Symbol}() +) + # Returned producers and caller-owned roots stay materialized: exempt from fusion. + protected = union(_returned_symbols(scope), protected_roots) + checks = Tuple{Symbol,Symbol}[] + guard_roots = setdiff(protected_roots, _assigned_symbols(scope)) + rewritten = InterBroadcastFusion.rewrite_scope(scope; + on_rewrite, protected, guard_roots, alias_checks=checks) + fast = _process_lifetime_scope(rewritten, rewrite_broadcast_lifetimes; protected_roots) + isempty(checks) && return fast + + # Analyze each straight-line branch separately, then wrap the complete + # lifetime-managed bodies. Overlapping inputs retain materialized producers. + fallback = InterBroadcastFusion.rewrite_scope(scope; protected) + slow = _process_lifetime_scope(fallback, rewrite_broadcast_lifetimes; protected_roots) + conditions = [:(cuNumeric._fusion_disjoint($a, $b)) for (a, b) in checks] + condition = reduce((a, b) -> Expr(:&&, a, b), conditions) + return Expr(:if, condition, fast, slow) end diff --git a/src/scoping/inter_broadcast_fusion.jl b/src/scoping/inter_broadcast_fusion.jl index d69bff4bd..453574ee6 100644 --- a/src/scoping/inter_broadcast_fusion.jl +++ b/src/scoping/inter_broadcast_fusion.jl @@ -15,39 +15,153 @@ using ..ScopingUtils # # The pass is syntax-only and has no NDArray or cuNumeric dependencies. -function _substitute_symbols(expr, replacements::Dict{Symbol,Any}) - assignment = _assignment(expr) - isnothing(assignment) && return _replace_symbols(expr, replacements) - assignment.lhs isa Symbol || return _replace_symbols(expr, replacements) - rhs = _replace_symbols(assignment.rhs, replacements) - return :($(assignment.lhs) = $rhs) +# These records exist only during macro expansion, not during array execution. + +# Array/scalar inputs read by an expression, plus names whose rebinding would +# change its meaning if evaluation is delayed until a later statement. +struct ReadDependencies + inputs::Set{Symbol} + bindings::Set{Symbol} # Also includes callable names that could be rebound. +end + +# A statement such as `tmp = A .+ B` that produces a single-use temporary. +# Records its expression and the statement positions of its definition and use. +struct BroadcastProducer + name::Symbol + definition::Int + consumer::Int + expression::Any +end + +# Temporaries to replace with their expressions at the use site, together with +# any runtime disjointness checks needed to make that delayed evaluation safe. +struct FusionPlan + producers::Dict{Int,BroadcastProducer} + alias_checks::Vector{Tuple{Symbol,Symbol}} +end +FusionPlan() = FusionPlan(Dict{Int,BroadcastProducer}(), Tuple{Symbol,Symbol}[]) + +# Substitutions accumulated while applying a plan, with original statement +# positions and before/after expressions used to report the rewrites. +struct RewriteState + replacements::Dict{Symbol,Any} + sources::Dict{Symbol,Vector{Int}} + events::Vector{NamedTuple} +end +RewriteState() = RewriteState(Dict{Symbol,Any}(), Dict{Symbol,Vector{Int}}(), NamedTuple[]) + +# Only move arithmetic/indexing expressions across other statements. Unknown +# calls may mutate their arguments (even when their names do not end in `!`). +# This name-based allowlist assumes ordinary numerical methods; it does not +# prove that a shadowed function or an overloaded method is free of side effects. +const _READONLY_CALLS = Set((:+, :-, :*, :/, :^, :%, :fld, :cld, :mod, :rem, + :(:), :abs, :abs2, :sqrt, :exp, :log, :sin, :cos, :tan, :inv, :min, :max, + :ifelse, :iszero, :isfinite, :isnan, :identity, :size, :axes, :length, + :firstindex, :lastindex, :eltype, :one, :zero, :<, :>, :<=, :>=, :(==), :(!=), + :Bool, :Int, :Int8, :Int16, :Int32, :Int64, :UInt, :UInt8, :UInt16, :UInt32, + :UInt64, :Float16, :Float32, :Float64, :ComplexF32, :ComplexF64)) + +_read_dependencies!(deps, ::Any) = false +_read_dependencies!(deps, ::Number) = true +_read_dependencies!(deps, ::QuoteNode) = true + +function _read_dependencies!(deps, expr::Symbol) + expr in (:end, :(:), :nothing, :true, :false) || push!(deps, expr) + return true +end + +function _read_dependencies!(deps, expr::Expr) + reference = _reference(expr) + if !isnothing(reference) + return reference.array isa Symbol && + _read_dependencies!(deps, reference.array) && + all(x -> _read_dependencies!(deps, x), reference.indices) + end + # Parameter fields such as args.dt; writes to properties remain barriers. + if expr.head === :. && length(expr.args) == 2 && expr.args[2] isa QuoteNode + return _read_dependencies!(deps, expr.args[1]) + end + call = _call(expr) + isnothing(call) && (call = _dotcall(expr)) + isnothing(call) && return false + f = call.f + _is_broadcast_op(f) && (f = Symbol(chop(string(f); head=1, tail=0))) + return f in _READONLY_CALLS && all(x -> _read_dependencies!(deps, x), call.args) +end + +_is_readonly(expr) = _read_dependencies!(Set{Symbol}(), expr) + +function _dependencies(expr) + inputs = Set{Symbol}() + _read_dependencies!(inputs, expr) || return nothing + return ReadDependencies(inputs, Set(walk_symbols(expr))) end -function _indexed_assignment_base(stmt) +function _statement_assignment(stmt) assignment = _assignment(stmt) - isnothing(assignment) && return nothing - reference = _reference(assignment.lhs) + return isnothing(assignment) ? _broadcast_assignment(stmt) : assignment +end + +# A simple destination whose indexing has no unknown effects. Property writes +# and calls that compute destinations are deliberately excluded. +function _write_target(lhs) + lhs isa Symbol && return lhs + reference = _reference(lhs) isnothing(reference) && return nothing reference.array isa Symbol || return nothing + all(_is_readonly, reference.indices) || return nothing return reference.array end -function _safe_to_delay_broadcast( - stmts, def_idx::Int, use_idx::Int, dependencies::Set{Symbol}, lazy_defs::Set{Int} -) - for i in (def_idx + 1):(use_idx - 1) - stmt = stmts[i] - i in lazy_defs && continue +function _known_assignment(stmt) + assignment = _statement_assignment(stmt) + isnothing(assignment) && return false + return !isnothing(_write_target(assignment.lhs)) && _is_readonly(assignment.rhs) +end + +function _entry_guard_valid(stmts, write_index) + # An earlier unknown call could replace storage after the entry check. + return all(i -> _known_assignment(stmts[i]), 1:(write_index - 1)) +end + +function _write_checks(stmts, index, assignment, deps::ReadDependencies, guard_roots) + _is_readonly(assignment.rhs) || return nothing + target = _write_target(assignment.lhs) + isnothing(target) && return nothing + target in guard_roots || return nothing + _entry_guard_valid(stmts, index) || return nothing + # Entry checks can name stable arguments, not locals created later. + issubset(deps.inputs, guard_roots) || return nothing + target in deps.inputs && return nothing + return [(target, input) for input in Base.sort!(collect(deps.inputs); by=string)] +end - # An indexed write to an unrelated array does not invalidate the lazy - # producer. Any other intervening statement is conservatively a barrier. - mutated = _indexed_assignment_base(stmt) - if !isnothing(mutated) && !(mutated in dependencies) +function _delay_checks(stmts, producer::BroadcastProducer, rhs, guard_roots) + deps = _dependencies(rhs) + isnothing(deps) && return nothing + checks = Tuple{Symbol,Symbol}[] + for i in (producer.definition + 1):(producer.consumer - 1) + assignment = _assignment(stmts[i]) + if !isnothing(assignment) && assignment.lhs isa Symbol + assignment.lhs in deps.bindings && return nothing + _is_readonly(assignment.rhs) || return nothing continue end - return false + isnothing(assignment) && (assignment = _broadcast_assignment(stmts[i])) + isnothing(assignment) && return nothing + write_checks = _write_checks(stmts, i, assignment, deps, guard_roots) + isnothing(write_checks) && return nothing + append!(checks, write_checks) end - return true + return checks +end + +function _substitute_symbols(expr, replacements::Dict{Symbol,Any}) + assignment = _assignment(expr) + isnothing(assignment) && return _replace_symbols(expr, replacements) + assignment.lhs isa Symbol || return _replace_symbols(expr, replacements) + rhs = _replace_symbols(assignment.rhs, replacements) + return :($(assignment.lhs) = $rhs) end function _single_use_index(stmts, symbol::Symbol, def_idx::Int) @@ -70,7 +184,7 @@ function _source_indices(expr, replacement_sources) append!(indices, get(replacement_sources, symbol, Int[])) end unique!(indices) - sort!(indices) + Base.sort!(indices) return indices end @@ -89,68 +203,85 @@ function _fuse_into_destination(stmt) return Expr(:(.=), assignment.lhs, assignment.rhs) end -function _rewrite_scope(scope, protected) - stmts = _scope_statements(scope) - isnothing(stmts) && return scope, NamedTuple[] - - definitions = Dict{Symbol,Tuple{Int,Any}}() - lazy_defs = Set{Int}() - for (i, stmt) in enumerate(stmts) - assignment = _assignment(stmt) - if !isnothing(assignment) && assignment.lhs isa Symbol && - _is_broadcast_syntax(assignment.rhs) - definitions[assignment.lhs] = (i, assignment.rhs) - push!(lazy_defs, i) - end - end +function _broadcast_consumer(stmt) + assignment = _statement_assignment(stmt) + rhs = isnothing(assignment) ? stmt : assignment.rhs + _is_broadcast_syntax(rhs) && _is_readonly(rhs) || return false + return isnothing(assignment) || !isnothing(_write_target(assignment.lhs)) +end - inlineable = Dict{Symbol,Tuple{Int,Any}}() - for (sym, (def_idx, rhs)) in definitions - # Never inline a returned producer; it must escape as a real NDArray. - sym in protected && continue - use_idx = _single_use_index(stmts, sym, def_idx) - isnothing(use_idx) && continue - dependencies = Set(walk_symbols(rhs)) - if !_safe_to_delay_broadcast(stmts, def_idx, use_idx, dependencies, lazy_defs) - continue - end - inlineable[sym] = (def_idx, rhs) +function _producer(stmts, index, protected) + assignment = _assignment(stmts[index]) + isnothing(assignment) && return nothing + name, rhs = assignment.lhs, assignment.rhs + name isa Symbol && _is_broadcast_syntax(rhs) || return nothing + name in protected && return nothing + consumer = _single_use_index(stmts, name, index) + isnothing(consumer) && return nothing + _broadcast_consumer(stmts[consumer]) || return nothing + return BroadcastProducer(name, index, consumer, rhs) +end + +function _plan_fusion(stmts, protected, guard_roots) + plan = FusionPlan() + expanded = Dict{Symbol,Any}() + for index in eachindex(stmts) + producer = _producer(stmts, index, protected) + isnothing(producer) && continue + # Include the original inputs of already-elided producers. Otherwise a + # later rebind can become invisible through a chain like t -> s -> out. + rhs = _replace_symbols(producer.expression, expanded) + checks = _delay_checks(stmts, producer, rhs, guard_roots) + isnothing(checks) && continue + plan.producers[index] = producer + expanded[producer.name] = rhs + append!(plan.alias_checks, checks) end + unique!(plan.alias_checks) + return plan +end - replacements = Dict{Symbol,Any}() - replacement_sources = Dict{Symbol,Vector{Int}}() - removed = Set(first(info) for info in values(inlineable)) - def_symbols = Dict(info[1] => sym for (sym, info) in inlineable) - fusion_events = NamedTuple[] - rewritten = Any[] +function _record_producer!(state::RewriteState, producer::BroadcastProducer) + indices = _source_indices(producer.expression, state.sources) + push!(indices, producer.definition) + state.sources[producer.name] = indices + state.replacements[producer.name] = + _substitute_symbols(producer.expression, state.replacements) + return nothing +end - for (i, original_stmt) in enumerate(stmts) - if i in removed - sym = def_symbols[i] - assignment = _assignment(original_stmt) - source_indices = _source_indices(assignment.rhs, replacement_sources) - push!(source_indices, i) - replacement_sources[sym] = source_indices - replacements[sym] = _substitute_symbols(inlineable[sym][2], replacements) - continue - end +function _rewrite_consumer!(state::RewriteState, stmts, index) + original = stmts[index] + indices = _source_indices(original, state.sources) + stmt = _substitute_symbols(original, state.replacements) + isempty(indices) && return stmt + + stmt = _fuse_into_destination(stmt) + before = Expr(:block, (stmts[i] for i in indices)..., original) + push!(state.events, (; before, fused=stmt)) + return stmt +end - source_indices = _source_indices(original_stmt, replacement_sources) - stmt = _substitute_symbols(original_stmt, replacements) - - if !isempty(source_indices) - stmt = _fuse_into_destination(stmt) - before = Expr( - :block, - (stmts[source_idx] for source_idx in source_indices)..., - original_stmt, - ) - push!(fusion_events, (; before, fused=stmt)) +function _apply_plan(scope, stmts, plan::FusionPlan) + state = RewriteState() + rewritten = Any[] + for index in eachindex(stmts) + producer = get(plan.producers, index, nothing) + if isnothing(producer) + push!(rewritten, _rewrite_consumer!(state, stmts, index)) + else + _record_producer!(state, producer) end - push!(rewritten, stmt) end + return Expr(scope.head, rewritten...), state.events +end - return Expr(scope.head, rewritten...), fusion_events +function _rewrite_scope(scope, protected, guard_roots) + stmts = _scope_statements(scope) + isnothing(stmts) && return scope, NamedTuple[], Tuple{Symbol,Symbol}[] + plan = _plan_fusion(stmts, protected, guard_roots) + rewritten, events = _apply_plan(scope, stmts, plan) + return rewritten, events, plan.alias_checks end """ @@ -161,9 +292,18 @@ rewritten scope. Symbols in `protected` — typically whatever the scope returns are never fused so they stay materialized. When provided, `on_rewrite` is called with a named tuple containing the `before` and `fused` expressions for each rewrite. + +Nonadjacent producers may cross read-only bindings that do not rebind their +inputs. With `guard_roots` and an `alias_checks` collector, writes to stable +arguments may also be crossed: the caller must guard the returned rewrite with +disjointness checks for those `(destination, input)` pairs and provide a fallback. """ -function rewrite_scope(scope; on_rewrite=nothing, protected=Set{Symbol}()) - rewritten, fusion_events = _rewrite_scope(scope, protected) +function rewrite_scope(scope; on_rewrite=nothing, protected=Set{Symbol}(), + guard_roots=Set{Symbol}(), alias_checks=nothing) + # Callers without a guard collector receive only statically safe rewrites. + roots = isnothing(alias_checks) ? Set{Symbol}() : guard_roots + rewritten, fusion_events, checks = _rewrite_scope(scope, protected, roots) + isnothing(alias_checks) || append!(alias_checks, checks) if !isnothing(on_rewrite) for event in fusion_events on_rewrite(event) diff --git a/src/scoping/lifetimes.jl b/src/scoping/lifetimes.jl index b5b88469c..1a82aee5e 100644 --- a/src/scoping/lifetimes.jl +++ b/src/scoping/lifetimes.jl @@ -38,17 +38,24 @@ function rewrite_eager_lifetimes(scope) broadcast_assignment = _broadcast_assignment(expr) if !isnothing(broadcast_assignment) (; lhs, rhs) = broadcast_assignment - new_lhs, lhs_temps = rewrite(lhs) + op = expr.head + # Indexed broadcast assignment needs a view even when getindex + # copies (notably all-colon indexing). + new_lhs, lhs_temps = if isnothing(_reference(lhs)) + rewrite(lhs) + else + fresh_tmp(:(Base.@view $lhs)) + end # Do not hoist the top-level call of the RHS to preserve fusion. call = _call(rhs) if !isnothing(call) new_rhs_args, rhs_temps = _maphoist(rewrite, call.args) new_rhs = Expr(:call, call.f, new_rhs_args...) - return Expr(:(.=), new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) + return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) end new_rhs, rhs_temps = rewrite(rhs) - return Expr(:(.=), new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) + return Expr(op, new_lhs, new_rhs), vcat(lhs_temps, rhs_temps) end reference = _reference(expr) @@ -70,6 +77,6 @@ function rewrite_eager_lifetimes(scope) return _prepend_statements(rewritten, temps), assigned_vars end -function process_lifetime_scope(scope) - return _process_lifetime_scope(scope, rewrite_eager_lifetimes) +function process_lifetime_scope(scope; protected_roots=Set{Symbol}()) + return _process_lifetime_scope(scope, rewrite_eager_lifetimes; protected_roots) end diff --git a/src/scoping/scoping.jl b/src/scoping/scoping.jl index 676708b30..eac7b7744 100644 --- a/src/scoping/scoping.jl +++ b/src/scoping/scoping.jl @@ -1,4 +1,4 @@ -export @analyze_lifetimes, @show_lifetimes +export @accelerate, @show_lifetimes # Include generic syntax layers before the cuNumeric-specific lifetime passes. include("util.jl") @@ -44,9 +44,7 @@ function _normalize_return(block) for (i, stmt) in enumerate(stmts) stmt isa Expr && stmt.head === :return || continue i == length(stmts) || throw( - ArgumentError( - "@analyze_lifetimes: `return` is only allowed as the block's final statement" - ), + ArgumentError("`return` is only allowed as the final statement") ) value = isempty(stmt.args) ? :nothing : only(stmt.args) return Expr(block.head, stmts[1:(end - 1)]..., value) @@ -54,47 +52,6 @@ function _normalize_return(block) return block end -@doc""" - @analyze_lifetimes expr - -Wraps a block of code so that all temporary `NDArray` allocations -(e.g. from slicing or function calls) are tracked and safely freed -at the end of the block. Ensures proper cleanup of GPU memory by -inserting `maybe_insert_delete` calls automatically. - -Assignments created inside the macro are scoped to its lexical region. Existing -arrays can still be mutated in place, and the final value of the block is -returned, but internal bindings do not leak into the surrounding scope. - -The block's final statement determines what leaves the region. Any binding it -returns (a bare name or the elements of a returned tuple) is both protected from -the automatic free and, under fusion, kept materialized rather than inlined into -its consumer, so a real `NDArray` escapes rather than a lazy broadcast tree: - - x, y = @analyze_lifetimes begin - x = e1 .+ e2 - c = x .* e1 # not returned, single-use -> fused into y - y = c .^ 2 - (x, y) # returned -> x and y stay materialized - end - -A trailing `return expr` is accepted as an explicit spelling of the final -statement (`return (x, y)` above); a `return` anywhere else is an error. - -When broadcast fusion is enabled (`FUSE_BROADCAST_EXPRS`), dotted operators -(`.+`, `.*`, etc.) form a lazy `Base.Broadcast.Broadcasted` tree compiled into -a single PTX kernel; intermediate nodes are not real `NDArray` allocations and -are not individually hoisted. The macro automatically selects the -broadcast-aware analysis in that case and the plain analysis otherwise. -""" -macro analyze_lifetimes(block) - block = _normalize_return(_expand_dot_macros(block, __module__)) - on_rewrite = BCAST_FUSION_DEBUG[] ? InterBroadcastFusion.log_rewrite : nothing - rewritten = process_ndarray_scope(block; on_rewrite) - bindings = union(_assigned_symbols(block), _assigned_symbols(rewritten)) - return esc(_lexical_scope(rewritten, bindings)) -end - const counter = Ref(0) function maybe_insert_delete(var::NDArray) @@ -103,10 +60,8 @@ end maybe_insert_delete(x) = x -# `@analyze_lifetimes` is an ownership region, analogous to a C++ `{ ... }` -# block. Bind every source and generated assignment explicitly so it cannot -# accidentally reuse or leak a caller local with the same name. Indexed and -# broadcast assignments are mutations, not new bindings, and remain visible. +# Symbols bound by an assignment anywhere in `expr` (let-form locals; soft-form +# protected bindings). function _assigned_symbols(expr) assigned = Set{Symbol}() @@ -124,9 +79,7 @@ function _assigned_symbols(expr) function visit(node) node isa Expr || return nothing assignment = _assignment(node) - if !isnothing(assignment) - collect_binding(assignment.lhs) - end + isnothing(assignment) || collect_binding(assignment.lhs) foreach(visit, node.args) return nothing end @@ -136,24 +89,10 @@ function _assigned_symbols(expr) end function _lexical_scope(body, bindings::Set{Symbol}) - ordered = sort!(collect(bindings); by=string) + ordered = Base.sort!(collect(bindings); by=string) return Expr(:let, Expr(:block, ordered...), body) end -function _register_scoping_error_hint!() - isdefined(Base.Experimental, :register_error_hint) || return nothing - Base.Experimental.register_error_hint(UndefVarError) do io, exc - return print( - io, - "\nHint: bindings assigned inside `@analyze_lifetimes` are local to its " * - "block. If `", - exc.var, - "` was created there, return it from the block to use it afterward.", - ) - end - return nothing -end - function _hoist_temporary(expr, assigned_vars) counter[] += 1 temporary = Symbol(:tmp, counter[]) @@ -202,7 +141,9 @@ end insert_finalizers(stmts::Vector) Insert `cuNumeric.maybe_insert_delete(var)` after the last use of each temporary variable. """ -function insert_finalizers(exprs::Vector, assigned_vars::Set{Symbol}) +function insert_finalizers( + exprs::Vector, assigned_vars::Set{Symbol}; protected_roots::Set{Symbol}=Set{Symbol}() +) last_use = Dict{Symbol,Int}() alias_map = Dict{Symbol,Symbol}() @@ -261,7 +202,8 @@ function insert_finalizers(exprs::Vector, assigned_vars::Set{Symbol}) # return `nothing` rather than leak it or hand back a dangling handle. terminal_indexed = n > 0 && is_indexed_assign(stmts[n]) - protected = Set{Symbol}() + # Roots (function args) are protected regardless of the terminal statement. + protected = Set{Symbol}(canon(root) for root in protected_roots) if n > 0 && !terminal_indexed for result in _result_symbols(stmts[n]) push!(protected, canon(result)) @@ -315,16 +257,20 @@ end insert_finalizers(block::Expr) Apply finalizer insertion to a `begin ... end` or `:block` expression. """ -function insert_finalizers(block::Expr, assigned_vars::Set{Symbol}) +function insert_finalizers( + block::Expr, assigned_vars::Set{Symbol}; protected_roots::Set{Symbol}=Set{Symbol}() +) stmts = _scope_statements(block) isnothing(stmts) && error("Expected a begin/block expression") - return Expr(:block, insert_finalizers(stmts, assigned_vars)...) + return Expr(:block, insert_finalizers(stmts, assigned_vars; protected_roots)...) end -function _process_lifetime_scope(scope, rewrite_lifetimes) +function _process_lifetime_scope( + scope, rewrite_lifetimes; protected_roots::Set{Symbol}=Set{Symbol}() +) try rewritten, assigned_vars = rewrite_lifetimes(scope) - return insert_finalizers(rewritten, assigned_vars) + return insert_finalizers(rewritten, assigned_vars; protected_roots) finally counter[] = 0 end @@ -335,13 +281,15 @@ end include("lifetimes.jl") include("broadcast_lifetimes.jl") -function process_ndarray_scope(scope; on_rewrite=nothing) +function process_ndarray_scope( + scope; on_rewrite=nothing, protected_roots::Set{Symbol}=Set{Symbol}() +) # Broadcast expressions stay lazy only when fusion is enabled; otherwise # every call is analyzed as an eager allocation. @static if FUSE_BROADCAST_EXPRS - return process_broadcast_lifetime_scope(scope; on_rewrite) + return process_broadcast_lifetime_scope(scope; on_rewrite, protected_roots) end - return process_lifetime_scope(scope) + return process_lifetime_scope(scope; protected_roots) end # Return the deleted value for a generated finalizer call. @@ -354,12 +302,27 @@ function _delete_argument(expr) return only(call.args) end -function print_lifetime_analysis(block; io::IO=stdout) +# Header + body statements per form, so the printout mirrors the real expansion. +function _analysis_parts(ex) + ex = _strip_lines(ex) + if ex isa Expr && ex.head === :function + return "function " * string(first(ex.args)), _flatten_statements(ex.args[2]) + elseif ex isa Expr && ex.head === :let + binds = _strip_lines(first(ex.args)) + bindstr = binds isa Expr ? join(binds.args, ", ") : string(binds) + return "let " * bindstr, _flatten_statements(ex.args[2]) + end + return nothing, _flatten_statements(ex) +end + +# Pretty-print `expansion` (the exact `@accelerate` output), highlighting frees. +function print_lifetime_analysis(expansion; io::IO=stdout) rule = "-"^60 - stmts = _flatten_statements(process_ndarray_scope(block)) + header, stmts = _analysis_parts(expansion) mode = FUSE_BROADCAST_EXPRS ? "fusion-aware" : "plain" - println(io, "@analyze_lifetimes expansion ($mode analysis)\n", rule) + println(io, "@accelerate expansion ($mode)\n", rule) + isnothing(header) || println(io, header) n = 0 for s in stmts @@ -368,24 +331,13 @@ function print_lifetime_analysis(block; io::IO=stdout) printstyled(io, lpad("✗ free ", 11), deleted, "\n"; color=:red) else n += 1 - println(io, lpad(n, 4), " ", s) + println(io, lpad(n, 4), " ", _strip_lines(s)) end end + isnothing(header) || println(io, "end") println(io, rule) return nothing end -@doc""" - @show_lifetimes expr - -Print the lifetime-analysis rewrite of `expr` — the same transformation -[`@analyze_lifetimes`](@ref) applies — without running it. Every statement is -shown in source order and each inserted `maybe_insert_delete` is highlighted so -you can see exactly where each temporary is freed. Pure AST work, so it runs on -CPU-only checkouts. -""" -macro show_lifetimes(block) - block = _normalize_return(_expand_dot_macros(block, __module__)) - return :(print_lifetime_analysis($(QuoteNode(block)))) -end +include("accelerate.jl") diff --git a/src/scoping/util.jl b/src/scoping/util.jl index 64f68366e..985adce2c 100644 --- a/src/scoping/util.jl +++ b/src/scoping/util.jl @@ -25,9 +25,12 @@ function _assignment(expr) end function _broadcast_assignment(expr) - MacroTools.isexpr(expr, :(.=)) || return nothing - MacroTools.@capture(expr, lhs_ .= rhs_) || return nothing - return (; lhs, rhs) + expr isa Expr && length(expr.args) == 2 || return nothing + op = expr.head + op isa Symbol || return nothing + spelling = string(op) + startswith(spelling, ".") && endswith(spelling, "=") || return nothing + return (; lhs=expr.args[1], rhs=expr.args[2]) end function _call(expr) diff --git a/src/utilities/version.jl b/src/utilities/version.jl index 6f05603c6..8c3f42667 100644 --- a/src/utilities/version.jl +++ b/src/utilities/version.jl @@ -28,7 +28,9 @@ end versioninfo() Prints the cuNumeric build configuration summary, including package -metadata, Julia and compiler version, and paths to core dependencies. +metadata, Julia and compiler version, paths to core dependencies, and +cuSolverMp availability and linear algebra tuning constants. Runtime GPU +eligibility is reported separately from the per-operation size/shape policy. """ function versioninfo(io::IO=stdout) name = string(Base.nameof(@__MODULE__)) @@ -48,12 +50,23 @@ function versioninfo(io::IO=stdout) dirs2 = Legate.find_dependency_paths(typeof(legate_mode)) other_dirs = merge(dirs1, dirs2) - hardware_str = HAS_CUDA ? "CPU + GPU" : "CPU Only" + active = runtime_started() + hardware_str = + active ? + (_has_gpu_target() ? "CPU + GPU" : "CPU Only") : + "not queried (runtime inactive)" legate_auto_config = get(ENV, "LEGATE_AUTO_CONFIG", "1") is_auto_config = legate_auto_config != "0" ? true : false legate_config = is_auto_config ? "auto" : get(ENV, "LEGATE_CONFIG", "not set") + # versioninfo is also called by the test driver with LEGATE_SKIP_RUNTIME. + # Do not start the runtime or query its machine just to print diagnostics. + not_queried = "not queried (runtime inactive)" + mp_available = active ? _LINALG_RUNTIME[].available : not_queried + active_gpus = active ? _LINALG_RUNTIME[].gpus : not_queried + mp_eligible = active ? _LINALG_RUNTIME[].mp_eligible : not_queried + str = """ ─────────────────────────────────────────────── cuNumeric Build Configuration @@ -68,6 +81,18 @@ function versioninfo(io::IO=stdout) Brodcast Fusion: $(FUSE_BROADCAST_EXPRS) Brodcast Min Ops: $(FUSE_BROADCAST_MIN_OPS) + cuSolverMp / Linear Algebra: + Library support: $mp_available + Active GPUs: $active_gpus + MP eligible before size/shape checks: $mp_eligible + MIN_SOLVE_MATRIX_SIZE: $MIN_SOLVE_MATRIX_SIZE (dimension) + MIN_SOLVE_TILE_SIZE: $MIN_SOLVE_TILE_SIZE + MIN_CHOLESKY_MATRIX_SIZE: $MIN_CHOLESKY_MATRIX_SIZE (dimension) + MIN_CHOLESKY_TILE_SIZE: $MIN_CHOLESKY_TILE_SIZE + MIN_QR_MATRIX_SIZE: $MIN_QR_MATRIX_SIZE (elements) + QR_TILE_SIZE: $QR_TILE_SIZE + MAX_CHOLESKY_TILES_PER_PROC: $MAX_CHOLESKY_TILES_PER_PROC + Hostname: $hostname Julia Version: $(VERSION) C++ Compiler: $compiler diff --git a/src/warnings.jl b/src/warnings.jl index db0ae37f4..c2cfc99b7 100644 --- a/src/warnings.jl +++ b/src/warnings.jl @@ -13,7 +13,7 @@ function repl_frontend_task() if !isassigned(_repl_frontend_task) _repl_frontend_task[] = get_repl_frontend_task() end - _repl_frontend_task[] + return _repl_frontend_task[] end @noinline function get_repl_frontend_task() if isdefined(Base, :active_repl) @@ -69,7 +69,7 @@ function assertscalar(op::String) return nothing end - _assertscalar(op, behavior) + return _assertscalar(op, behavior) end """ @@ -94,7 +94,7 @@ function assertpromotion(op, ::Type{FROM}, ::Type{TO}) where {FROM,TO} return nothing end - _assertpromotion(op, behavior, FROM, TO) + return _assertpromotion(op, behavior, FROM, TO) end @noinline function _assertscalar(op, behavior) @@ -119,26 +119,193 @@ end return nothing end +const _CUNUMERIC_MODULE = @__MODULE__ + +# Sentinel for stack frames with no recoverable module. Prefer this over `nothing` +# so `_module_of_stackframe` is type-stable as `Module`. Must not match Base / +# LinearAlgebra / cuNumeric checks below. +const _UNKNOWN_STACK_MODULE = Module(:__cuNumeric_unknown_stack_module__, false, false) + +@inline function _is_cunumeric_module(m::Module) + m === _UNKNOWN_STACK_MODULE && return false + m === _CUNUMERIC_MODULE && return true + pm = parentmodule(m) + while pm !== m + pm === _CUNUMERIC_MODULE && return true + m = pm + pm = parentmodule(m) + end + return false +end + +@inline function _is_linalg_module(m::Module) + m === _UNKNOWN_STACK_MODULE && return false + m === LinearAlgebra && return true + nameof(m) === :LinearAlgebra && return true + pm = parentmodule(m) + while pm !== m + (pm === LinearAlgebra || nameof(pm) === :LinearAlgebra) && return true + m = pm + pm = parentmodule(m) + end + return false +end + +# Note: LinearAlgebra (and other stdlibs) often have parentmodule === Base, so callers +# must check `_is_linalg_module` before treating a frame as Base. +@inline function _is_base_module(m::Module) + m === _UNKNOWN_STACK_MODULE && return false + m === Base && return true + nameof(m) === :Base && return true + pm = parentmodule(m) + while pm !== m + (pm === Base || nameof(pm) === :Base) && return true + m = pm + pm = parentmodule(m) + end + return false +end + +_module_from_def(def::Method) = def.module +_module_from_def(def::Module) = def +_module_from_def(_) = _UNKNOWN_STACK_MODULE + +_module_of_linfo(linfo::Core.MethodInstance) = _module_from_def(linfo.def) +_module_of_linfo(linfo::Method) = linfo.module +# Julia 1.12+ often stores a CodeInstance on stack frames; unwrap to MethodInstance. +_module_of_linfo(linfo::Core.CodeInstance) = _module_of_linfo(linfo.def) +_module_of_linfo(_) = _UNKNOWN_STACK_MODULE + +# When `frame.linfo` is missing (common for inlined frames), recover module from +# the source path so LinearAlgebra callers are not skipped and the walk does not +# fall through to loader frames like `Base.include_string`. +function _module_from_file(file) + (file === nothing || file === :none) && return _UNKNOWN_STACK_MODULE + f = string(file) + # Match stdlib path segments; avoid false positives on user paths when possible. + if occursin(r"(?:^|[/\\])LinearAlgebra(?:[/\\]|$)", f) + return LinearAlgebra + end + if occursin(r"(?:^|[/\\])cuNumeric(?:\.jl)?(?:[/\\]|$)", f) + return _CUNUMERIC_MODULE + end + # Base frames commonly appear as `./abstractarray.jl`, `./set.jl`, etc. + if startswith(f, "./") || occursin(r"(?:^|[/\\])[Bb]ase(?:[/\\]|$)", f) + return Base + end + return _UNKNOWN_STACK_MODULE +end + +function _module_of_stackframe(frame::Base.StackTraces.StackFrame) + m = _module_of_linfo(frame.linfo) + m !== _UNKNOWN_STACK_MODULE && return m + return _module_from_file(frame.file) +end + +# Keyword bodies often look like `#cholesky!#272`; surface `cholesky!`. +function _clean_stack_func_name(fname) + fname_sym = ifelse(fname isa Symbol, fname, Symbol(string(fname))) + s = string(fname_sym) + m = match(r"^#([^#]+)#\d+$", s) + return m === nothing ? fname_sym : Symbol(m.captures[1]) +end + +# Frames that are never the user-facing "triggering" API for enrichment: +# - loaders / client entry (`include_string` / Julia 1.12 `IncludeInto` from +# `include`ing tests, etc.) +# - keyword-call wrappers (`kwcall`) that would otherwise outrank cholesky/svd +# - AbstractArray iteration/indexing plumbing between NDArray getindex and the +# real stdlib caller (e.g. LinearAlgebra.cholesky / Base.unique) +const _SKIP_STACK_FUNCS = Set{Symbol}(( + :include_string, + :include, + :include_relative, + :_include, + :IncludeInto, # Julia 1.12+ callable include wrapper (Base.IncludeInto) + :eval, + :exec_options, + :_start, + :invokelatest, + :error, + :stacktrace, + :kwcall, # Base keyword-call wrapper; do not steal blame from cholesky/svd + :iterate, + :getindex, + :setindex!, + :indexed_iterate, + Symbol("macro expansion"), + Symbol("top-level scope"), +)) + +""" +Best-effort: outermost Base or LinearAlgebra frame above cuNumeric scalar-index +frames. Walk innermost-first, skip cuNumeric/Core (and frames with unknown +module), indexing/iteration plumbing, keyword-call wrappers (`kwcall`), and +Base loader frames (`include_string`, `IncludeInto` on Julia 1.12+, etc.). +Keep updating the candidate while still in Base/LinearAlgebra (last one wins) +so attribution names the user-facing API (`LinearAlgebra.svd`) rather than an +inner helper (`Base.lt`) or a loader (`Base.include_string`). Stop at the first +user/other-package frame and return that candidate, or `nothing` for the plain +message (e.g. user `Main`). Check LinearAlgebra before Base — stdlibs often +parent to Base. +""" +@noinline function _scalar_indexing_stdlib_caller() + caller = nothing + for frame in stacktrace() + m = _module_of_stackframe(frame) + # Skip frames with unknown module (same as previous `nothing` skip). + m === _UNKNOWN_STACK_MODULE && continue + (m === Core || _is_cunumeric_module(m)) && continue + clean_name = _clean_stack_func_name(frame.func) + clean_name in _SKIP_STACK_FUNCS && continue + if _is_linalg_module(m) + caller = (:LinearAlgebra, clean_name) + elseif _is_base_module(m) + caller = (:Base, clean_name) + else + # User / other package code — keep the last stdlib candidate, if any. + break + end + end + return caller +end + +# Returns (enriched::Bool, desc::String). Enriched = Base or LinearAlgebra stdlib caller. function scalardesc(op) - desc = """Invocation of $op resulted in scalar indexing of an NDArray. + caller = _scalar_indexing_stdlib_caller() + if caller !== nothing + modname, fname = caller + # Base/LinearAlgebra AbstractArray fallback — name the outer API first. + # No "Scalar indexing is disallowed." header (not part of the enriched template). + return true, + "`$modname.$fname` fell back to an AbstractArray implementation, which scalar-indexed an `NDArray`. " * + "This $modname path is probably not implemented yet for `NDArray`. " * + "Using `allowscalar` or `@allowscalar` might allow this function to work slowly, but it has not been tested." + end + + # Plain user-level scalar indexing (unchanged). + return false, """Invocation of $op resulted in scalar indexing of an `NDArray`. This is typically caused by calling an iterating implementation of a method. - This is very slow and should be avoided. + This is very slow and should be avoided. This can also happen if an external + method (i.e., LinearAlgebra.kron) is not re-implemented in cuNumeric.jl. Because + `NDArray`s subtype `AbstractArray`, the method call will dispatch to the + `AbstractArray` implementation, which often iterates over the array. If you want to allow scalar iteration, use `allowscalar` or `@allowscalar` to enable scalar iteration globally or for the operations in question.""" end function promotiondesc(op, ::Type{FROM}, ::Type{TO}) where {FROM,TO} - desc = """Invocation of $op resulted in implicit promotion of an NDArray from $(FROM) to - wider type: $(TO). This is typically caused by mixing NDArrays or literals - with different precision. This can cause extra copies of data and is slow. + return desc = """Invocation of $op resulted in implicit promotion of an NDArray from $(FROM) to + wider type: $(TO). This is typically caused by mixing NDArrays or literals + with different precision. This can cause extra copies of data and is slow. - If you want to allow implicit promotion to wider types, use `allowpromotion` or `@allowpromotion` - to enable implicit promotion.""" + If you want to allow implicit promotion to wider types, use `allowpromotion` or `@allowpromotion` + to enable implicit promotion.""" end @noinline function warnscalar(op) - desc = scalardesc(op) + _, desc = scalardesc(op) @warn("""Performing scalar indexing on task $(current_task()). $desc""") end @@ -150,9 +317,14 @@ end end @noinline function errorscalar(op) - desc = scalardesc(op) - error("""Scalar indexing is disallowed. - $desc""") + enriched, desc = scalardesc(op) + if enriched + error(desc) + else + # Plain path keeps the historical disallow header. + error("""Scalar indexing is disallowed. + $desc""") + end end @noinline function errordouble(op, ::Type{FROM}, ::Type{TO}) where {FROM,TO} @@ -165,7 +337,7 @@ end # NOTE: This is deprecated and should not be used from user logic. A proper solution to # this problem will be introduced in https://github.com/JuliaLang/julia/pull/39217 macro __tryfinally(ex, fin) - Expr(:tryfinally, + return Expr(:tryfinally, :($(esc(ex))), :($(esc(fin))), ) @@ -185,7 +357,7 @@ See also: [`@allowscalar`](@ref). allowscalar function allowscalar(f::Base.Callable) - task_local_storage(f, :ScalarIndexing, ScalarAllowed) + return task_local_storage(f, :ScalarIndexing, ScalarAllowed) end function allowscalar(allow::Bool=true) diff --git a/test/Project.toml b/test/Project.toml index 73f4b047f..8bff086c3 100644 --- a/test/Project.toml +++ b/test/Project.toml @@ -1,15 +1,22 @@ [deps] +Krylov = "ba0b0d4f-ebba-5204-a429-3ac8c609bfb7" CNPreferences = "3e078157-ea10-49d5-bf32-908f777cd46f" CUDA = "052768ef-5323-5732-b1bb-66c8b64840ba" -CUDA_Driver_jll = "4ee394cb-3365-5eb0-8335-949819d2adfc" +FFTW = "7a1cc6ca-52ef-59f5-83cd-3a7055c09341" InteractiveUtils = "b77e0a4c-d291-57a0-90e8-8db25a27a240" LinearAlgebra = "37e2e46d-f89d-539d-b4ee-838fcccc9c8e" ParallelTestRunner = "d3525ed8-44d0-4b2c-a655-542cee43accc" Pkg = "44cfe95a-1eb2-52ea-b672-e2afdf69b78f" Random = "9a3f8284-a2c9-5f02-9a11-845980a1fd5c" StatsBase = "2913bbd2-ae8a-5f71-8c99-4fb6c76f3a91" +StructArrays = "09ab397b-f2b6-538f-b94a-2f83cf4a842a" +TensorOperations = "6aa20fa7-93e2-5fca-9bc0-fbd0db3c71a2" Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40" cuNumeric = "0fd9ffd4-7e84-4cd0-b8f8-645bd8c73620" +[compat] +CUDA = "6.4" +StructArrays = "0.7" + [sources] cuNumeric = {path = ".."} diff --git a/test/analysis/accelerate.jl b/test/analysis/accelerate.jl new file mode 100644 index 000000000..2c0038895 --- /dev/null +++ b/test/analysis/accelerate.jl @@ -0,0 +1,226 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +# Coverage of the four `@accelerate` forms and their scope contracts: +# @accelerate function f(...) ... end -> function scope, frees non-returned +# @accelerate begin ... end -> 1:1 Julia scope, bindings stay alive +# @accelerate let ... end -> hard scope, combine + free non-returned +# @accelerate expr -> materialized result, temps freed + +using InteractiveUtils: code_typed + +@testset "@accelerate respects rebindings and alias writes" begin + @accelerate function _acc_rebind(a, b) + t = a .+ 1f0 + a = b .+ 2f0 + t .+ a + end + @accelerate function _acc_aliaswrite(a, b) + t = a .+ 1f0 + b[1] = 9f0 + t .+ 0f0 + end + for make in (identity, NDArray) + @test Array(_acc_rebind(make(Float32[1]), make(Float32[10]))) == Float32[14] + a = make(Float32[1, 2]) + @allowscalar result = _acc_aliaswrite(a, a) + @test Array(result) == Float32[2, 3] + end + # Adjacent chains still inline their single-use intermediates. + ex = cuNumeric.InterBroadcastFusion.rewrite_scope(quote + t = a .+ 1f0 + u = t .* 2f0 + u .+ 3f0 + end) + @test !(:t in cuNumeric.ScopingUtils.walk_symbols(ex)) + @test !(:u in cuNumeric.ScopingUtils.walk_symbols(ex)) +end + +@testset "@accelerate — four forms" begin + T = Float32 + N = 64 + _nd(v) = @allowscalar NDArray(v) + approx(x, ref) = isapprox(Array(x), ref; rtol=1.0f-4) + ja = my_rand(T, N) + jb = my_rand(T, N) + + @testset "1. function form" begin + @accelerate function _acc_fsq(a, b) + c = a .* b + return c .^ 2 + end + a = _nd(ja) + b = _nd(jb) + @test approx(_acc_fsq(a, b), (ja .* jb) .^ 2) + # Arguments are caller-owned: a second call on the same inputs still works. + @test approx(_acc_fsq(a, b), (ja .* jb) .^ 2) + end + + @testset "3. let form (hard scope)" begin + function _acc_let(a, b) + s = @accelerate let + r = a .+ b + s = r .* T(2) + s + end + return s, @isdefined(r) + end + s, r_leaked = _acc_let(_nd(ja), _nd(jb)) + @test approx(s, (ja .+ jb) .* T(2)) + @test r_leaked == false # `r` must not escape the let scope + end + + @testset "4. expr form" begin + a = _nd(ja) + b = _nd(jb) + res = @accelerate (a .+ b) .^ 2 + @test res isa NDArray + @test approx(res, (ja .+ jb) .^ 2) + end + + @testset "2. begin form (bindings stay alive)" begin + function _acc_begin(a, b) + q = @accelerate begin + p = a .* b + q = p .+ one(T) + q + end + return p, q # both must be defined in this scope + end + p, q = _acc_begin(_nd(ja), _nd(jb)) + @test approx(p, ja .* jb) + @test approx(q, (ja .* jb) .+ one(T)) + + # A nested `let` keeps its intermediate private while the outer block + # can consume and return the value it produces. + a = _nd(ja) + b = _nd(jb) + one_t = one(T) + shifted, x = @accelerate begin + shifted = let + product = @. a * b + @. product + one_t + end + x = @. shifted * 2 + (shifted, x) + end + @test approx(shifted, (ja .* jb) .+ one(T)) + @test approx(x, ((ja .* jb) .+ one(T)) .* 2) + end + + @testset "expansion contracts (white-box)" begin + expand(ex) = cuNumeric._accelerate_expand(ex, @__MODULE__) + hasfree(ex) = occursin("maybe_insert_delete", string(expand(ex))) + + # Scope shape per form. + @test expand(:(function f(a) + ;c = a .* a; + c .^ 2; + end)).head === :function + @test expand(:( + begin + C .= a[2:end] .+ b[2:end] + end + )).head === :block + @test expand(:( + let + r = a .+ b; + r .* 2 + end + )).head === :let + + # Slices are freed in every non-`let` form (uniform cleanup). + @test hasfree(:(function f(a) + ;s = a[2:end]; + s .+ 1; + end)) + @test hasfree(:( + begin + C .= a[2:end] .+ b[2:end] + end + )) + + # Compound dotted assignments must remain one broadcast tree. Hoisting + # their RHS would add a full-size temporary and a second GPU launch. + compound = string(expand(:(function update!(x, alpha, p) + x .+= alpha .* p + x + end))) + @test occursin("x .+= alpha .* p", compound) + @test !occursin(r"tmp\d+ = alpha \.\* p", compound) + + if cuNumeric.FUSE_BROADCAST_EXPRS + # A same-shape chain fuses into one multi-output launch and still + # frees the hoisted slice temporaries. + mo = string(expand(:( + begin + p = a[2:end] .* b[2:end] + q = p .+ 1 + q + end + ))) + @test occursin("copyto_fused_multi_alloc!", mo) + @test occursin("maybe_insert_delete", mo) + else + # Multi-output fusion is GPU-only; CPU expansion must use the + # ordinary broadcast path even when fusion is enabled in preferences. + cpu = string(expand(:( + begin + p = a .* b + q = p .+ 1 + q + end + ))) + @test !occursin("copyto_fused_multi_alloc!", cpu) + end + end + + @testset "multi-output segment runner is fully unrolled" begin + # GPU compilation requires every chained segment call to be statically + # dispatched. This three-segment shape crossed Julia 1.10's recursive + # inference limit when `_run_segments` recursed over `Base.tail`. + segs = ( + (+, (cuNumeric.RuntimeBroadcastArg{1}(), cuNumeric.RuntimeBroadcastArg{2}())), + (*, (cuNumeric.LocalBroadcastArg{1}(), cuNumeric.RuntimeBroadcastArg{1}())), + (^, (cuNumeric.LocalBroadcastArg{2}(), cuNumeric.RuntimeBroadcastArg{3}())), + ) + outs = ntuple(_ -> zeros(T, 2, 2), 3) + runtime_args = (ones(T, 2, 2), ones(T, 2, 2), 2) + + @test @inferred( + cuNumeric._run_segments( + segs, outs, runtime_args, (), (), CartesianIndex(1, 1) + ) + ) === nothing + @test getindex.(outs, Ref(CartesianIndex(1, 1))) == (T(2), T(2), T(4)) + + argtypes = ( + typeof(segs), + typeof(outs), + typeof(runtime_args), + Tuple{}, + Tuple{}, + CartesianIndex{2}, + ) + typed = only(code_typed(cuNumeric._run_segments, argtypes; optimize=true)).first + @test !occursin( + "_run_segments", sprint(show, MIME("text/plain"), typed) + ) + end +end diff --git a/test/analysis/inter_broadcast_fusion.jl b/test/analysis/inter_broadcast_fusion.jl new file mode 100644 index 000000000..aebe96c94 --- /dev/null +++ b/test/analysis/inter_broadcast_fusion.jl @@ -0,0 +1,143 @@ +using Test + +const IBF = cuNumeric.InterBroadcastFusion +const SU = cuNumeric.ScopingUtils + +@testset "Nonadjacent producer elimination" begin + ex = IBF.rewrite_scope(quote + t = a .+ 1f0 + r = b .* 2f0 + out .= t .+ r + end) + @test !(:t in SU.walk_symbols(ex)) + @test !(:r in SU.walk_symbols(ex)) + + # Expanding t into s must retain a as a dependency of s. + ex = IBF.rewrite_scope(quote + t = a .+ 1f0 + s = t .* 2f0 + a = b .+ 2f0 + s .+ a + end) + @test !(:t in SU.walk_symbols(ex)) + @test :s in SU.walk_symbols(ex) + + # Calls can mutate inputs without a bang suffix, including inside a broadcast. + for barrier in (:(touch(a)), :(unused = touch.(a)), :(a[1] = 9f0)) + ex = IBF.rewrite_scope(quote + t = a .+ 1f0 + $barrier + out .= t .* 2f0 + end) + @test :t in SU.walk_symbols(ex) + end + ex = IBF.rewrite_scope(quote + t = a .+ 1f0 + r = b .* 2f0 + t .+ r + end; protected=Set([:t])) + @test :t in SU.walk_symbols(ex) + + ex = IBF.rewrite_scope(quote + t = a .+ 1f0 + out[touch(a)] .= t .* 2f0 + end) + @test :t in SU.walk_symbols(ex) + ex = IBF.rewrite_scope(quote + t = sin.(a) + sin = cos + t .+ 1f0 + end) + @test :t in SU.walk_symbols(ex) + + # A compact Gray-Scott-shaped sequence: four producers and two array writes. + body = quote + F_u = u .* v + F_v = u .+ v + u_lap = u .* 2f0 + v_lap = v .* 3f0 + u_new[:] = F_u .+ u_lap + v_new[:] = F_v .+ v_lap + nothing + end + checks = Tuple{Symbol,Symbol}[] + ex = IBF.rewrite_scope(body; guard_roots=Set([:u, :v, :u_new, :v_new]), + alias_checks=checks) + @test all(s -> !(s in SU.walk_symbols(ex)), (:F_u, :F_v, :u_lap, :v_lap)) + @test Set(checks) == Set([(:u_new, :u), (:u_new, :v)]) + fallback = IBF.rewrite_scope(body) + @test :F_v in SU.walk_symbols(fallback) + @test :v_lap in SU.walk_symbols(fallback) + + # Guards cannot be hoisted past unknown calls that could change storage. + prefixed = Expr(:block, :(prepare(u_new, u)), SU._scope_statements(body)...) + checks = Tuple{Symbol,Symbol}[] + IBF.rewrite_scope(prefixed; guard_roots=Set([:u, :v, :u_new, :v_new]), alias_checks=checks) + @test isempty(checks) +end + +@accelerate function _acc_two_updates!(u, v, u_new, v_new) + F_u = u .* v + F_v = u .+ v + u_lap = u .* 2f0 + v_lap = v .* 3f0 + u_new[:] = F_u .+ u_lap + v_new[:] = F_v .+ v_lap + nothing +end +function _plain_two_updates!(u, v, u_new, v_new) + F_u = u .* v + F_v = u .+ v + u_lap = u .* 2f0 + v_lap = v .* 3f0 + u_new[:] = F_u .+ u_lap + v_new[:] = F_v .+ v_lap + nothing +end + +@testset "Guarded fusion preserves shared inputs" begin + for make in (identity, NDArray), alias in (:none, :u, :v, :shifted) + host = Float32[1, 2, 3, 4, 5] + hu = view(host, 1:4) + hv = Float32[5, 6, 7, 8] + ho = alias === :none ? zeros(Float32, 4) : + alias === :u ? hu : alias === :v ? hv : view(host, 2:5) + hout = zeros(Float32, 4) + parent = make(copy(host)) + u = view(parent, 1:4) + v = make(copy(hv)) + unew = alias === :none ? make(zeros(Float32, 4)) : + alias === :u ? u : alias === :v ? v : view(parent, 2:5) + vnew = make(zeros(Float32, 4)) + _plain_two_updates!(hu, hv, ho, hout) + @test _acc_two_updates!(u, v, unew, vnew) === nothing + @test Array(unew) == ho + @test Array(vnew) == hout + @test Array(parent) == host + if make === NDArray + foreach(cuNumeric.destroy!, (unew, vnew, u, v, parent)) + end + end +end + +@testset "Transitive rebindings and unknown calls remain barriers" begin + @accelerate function transitive(a, b) + t = a .+ 1f0 + s = t .* 2f0 + a = b .+ 2f0 + s .+ a + end + function touch(a) + fill!(a, 9f0) + nothing + end + @accelerate function unknown_effect(a) + t = a .+ 1f0 + touch(a) + t .* 2f0 + end + for make in (identity, NDArray) + @test Array(transitive(make(Float32[1, 2]), make(Float32[10, 20]))) == Float32[16, 28] + @test Array(unknown_effect(make(Float32[1, 2]))) == Float32[4, 6] + end +end diff --git a/test/analysis/lifetime.jl b/test/analysis/lifetime.jl index c6253ae8a..d21a1459f 100644 --- a/test/analysis/lifetime.jl +++ b/test/analysis/lifetime.jl @@ -17,6 +17,64 @@ * Ethan Meitz =# +@testset "linear algebra partition lifetime" begin + a = cuNumeric.NDArray(reshape(Float64.(1:15), 5, 3)) + for fail in (false, true) + released = Ref(0) + function use_partitions(p, q) + for part in (p, q) + finalizer(part.handle) do _ + released[] += 1 + end + end + @test released[] == 0 + fail && error("partition callback failed") + return 7 + end + if fail + @test_throws "partition callback failed" cuNumeric._with_linalg_partitions( + use_partitions, (a, (3, 3)), (a, (3, 3), (2, 1)) + ) + else + @test (@inferred cuNumeric._with_linalg_partitions( + use_partitions, (a, (3, 3)), (a, (3, 3), (2, 1)) + )) == 7 + end + # No GC is needed to release either partition owner, even on failure. + @test released[] == 2 + end + @allowscalar @test Array(a) == reshape(Float64.(1:15), 5, 3) +end + +@testset "Host-copy handles do not reach off-thread GC" begin + # Run with both thread pools enabled to exercise the original abort. All + # array operations stay on the runtime thread; only GC runs on the other pool. + if Threads.nthreads(:interactive) > 0 + runtime_thread = Threads.threadid() + function copy_and_release() + x = cuNumeric.ones(Float64, 8) + @test Array(x) == ones(8) + cuNumeric.destroy!(x) + end + function collect_elsewhere() + @test Threads.threadid() != runtime_thread + GC.gc(true) + end + for _ in 1:10 + copy_and_release() + task = if Threads.threadpool() === :interactive + Threads.@spawn :default collect_elsewhere() + else + Threads.@spawn :interactive collect_elsewhere() + end + fetch(task) + cuNumeric.drain_pending_frees!() + end + else + @test_skip false # Requires an additional thread pool. + end +end + @testset "Array ↔ NDArray value roundtrip (row-major attach)" begin A = rand(Float64, 4, 4) NA = NDArray(A) @@ -29,45 +87,49 @@ end end -@testset "Zero-Copy Verification (1D)" begin - # 1D attach remains zero-copy; N>=2 copies into a C-ordered buffer. - A = rand(Float64, 16) +@testset "Independent Storage (1D)" begin + # Conversion copies into Legate-owned storage before returning. + A = Float64.(1:16) + expected = copy(A) NA = NDArray(A) @allowscalar begin @test all(A .== Array(NA)) end - @test pointer(A) == cuNumeric.get_ptr(NA) + GC.@preserve A NA begin + @test pointer(A) != cuNumeric.get_ptr(NA) + end - # modify julia array, verify ndarray sees it + # Mutating the Julia source must not change the NDArray. A[1] = 99.0 @allowscalar begin - @test NA[1] == 99.0 + @test NA[1] == expected[1] end - # modify ndarray, verify julia array sees it + # Mutating the NDArray must not change the Julia source. @allowscalar begin NA[2] = 88.0 + @test NA[2] == 88.0 end - @test A[2] == 88.0 + @test A[2] == expected[2] end @testset "Lifetime Protection" begin - function create_attached_ndarray() + function create_owned_ndarray() local_A = rand(Float32, 100) local_A[1] = 1.23f0 return NDArray(local_A), local_A[1] end - NA, expected_val = create_attached_ndarray() + NA, expected_val = create_owned_ndarray() # force gc to try and collect the local array GC.gc(true) GC.gc(true) # do it again GC.gc(true) # and again lol - # data should still be intact via parent reference + # Data survives in Legate-owned storage after the Julia source is collected. @allowscalar begin @test NA[1] == expected_val end @@ -87,7 +149,7 @@ end NA = create_typed_ndarray() - # the temporary Float64 array should be kept alive by NA.parent + # The temporary Float64 source is no longer needed after construction. GC.gc(true) GC.gc(true) GC.gc(true) @@ -97,7 +159,7 @@ end @test NA[10] == 10.0 end - # modification should work on the attached temporary + # Modification should work on the independently owned storage. @allowscalar begin NA[1] = 42.0 @test NA[1] == 42.0 diff --git a/test/analysis/promotion.jl b/test/analysis/promotion.jl index 7b1fb39a7..3c3d5ddd4 100644 --- a/test/analysis/promotion.jl +++ b/test/analysis/promotion.jl @@ -19,3 +19,25 @@ @test safe_compare(r1, r2, atol(Float64), rtol(Float64)) end end + +@testset "Flattened associative broadcast promotion" begin + @test @inferred(cuNumeric.__checked_promote_op(+, NTuple{5,Float64})) === Float64 + @test @inferred(cuNumeric.__checked_promote_op(*, NTuple{4,Int32})) === Int32 +end + +@testset "Ternary broadcast promotion" begin + @test @inferred(cuNumeric.__checked_promote_op(muladd, Tuple{Float32,Float32,Float32})) === Float32 +end + +_promotion_test_norm(x, t) = abs(x) +_promotion_test_residual(e, u0, u1, atol, rtol, norm, t) = + e / (atol + max(norm(u0, t), norm(u1, t)) * rtol) + +@testset "Custom broadcast promotion with a function argument" begin + argtypes = Tuple{ + Float32,Float32,Float32,Float32,Float32,typeof(_promotion_test_norm),Float32 + } + @test @inferred(cuNumeric._numeric_broadcast_types(argtypes)) == + (Float32, Float32, Float32, Float32, Float32, Float32) + @test @inferred(cuNumeric.__checked_promote_op(_promotion_test_residual, argtypes)) === Float32 +end diff --git a/test/analysis/type_stability.jl b/test/analysis/type_stability.jl index b000bd522..86147913c 100644 --- a/test/analysis/type_stability.jl +++ b/test/analysis/type_stability.jl @@ -17,6 +17,32 @@ * Ethan Meitz =# +@accelerate function _type_stable_accelerate_function(a, b) + intermediate = @. a + b + return @. intermediate * 2.0f0 +end + +function _type_stable_accelerate_begin(a, b) + return @accelerate begin + intermediate = @. a + b + result = @. intermediate * 2.0f0 + (intermediate, result) + end +end + +function _type_stable_accelerate_let(a, b) + return @accelerate let + intermediate = @. a + b + @. intermediate * 2.0f0 + end +end + +function _type_stable_accelerate_expr(a, b) + return @accelerate (@. (a + b) * 2.0f0) +end + +_type_stable_cuda_argtypes(task::cuNumeric.CUDATask) = task.argtypes + @testset verbose = true "core" begin a = cuNumeric.zeros(5) b = cuNumeric.zeros(Float64, 3, 4) @@ -55,12 +81,29 @@ end @test @inferred(cuNumeric.rand(4, 3)) !== nothing @test @inferred(cuNumeric.rand(Float32, 5)) !== nothing + @test @inferred(cuNumeric.randn(Float64, 5)) !== nothing + @test @inferred(cuNumeric.randexp(Float32, 5)) !== nothing + @test @inferred(cuNumeric.rand(0:3, 3)) !== nothing + @test @inferred(cuNumeric.rand(ComplexF32, 5)) !== nothing + @test @inferred(cuNumeric.randn(ComplexF64, 5)) !== nothing # NDArray from Julia Array (Parent-stable attachment) @test @inferred(cuNumeric.NDArray(rand(10))) !== nothing @test @inferred(cuNumeric.NDArray(rand(Float32, 3, 3))) !== nothing end +@testset verbose = true "custom CUDA metadata" begin + task = cuNumeric.CUDATask("kernel", (Float32, Int32)) + @test isconcretetype(typeof(task)) + @test all(isconcretetype, fieldtypes(typeof(task))) + @test @inferred(_type_stable_cuda_argtypes(task)) == DataType[Float32, Int32] + + storage_type = cuNumeric.PaddedStorage{Float32,1} + @test all(isconcretetype, Base.uniontypes(fieldtype(storage_type, :backing))) + @test all(isconcretetype, Base.uniontypes(fieldtype(storage_type, :staging))) + @test all(isconcretetype, Base.uniontypes(fieldtype(storage_type, :shape))) +end + @testset verbose = true "conversion" begin # cast to array, as_type a = cuNumeric.zeros(Float64, 5, 5) @@ -99,6 +142,23 @@ end @test @inferred(((a .* b) .+ a) .* 2.0f0) !== nothing end +@testset verbose = true "@accelerate forms" begin + a = cuNumeric.ones(Float32, 3, 3) + b = cuNumeric.ones(Float32, 3, 3) + + function_result = @inferred _type_stable_accelerate_function(a, b) + @test function_result isa NDArray{Float32,2} + + begin_result = @inferred _type_stable_accelerate_begin(a, b) + @test begin_result isa Tuple{NDArray{Float32,2},NDArray{Float32,2}} + + let_result = @inferred _type_stable_accelerate_let(a, b) + @test let_result isa NDArray{Float32,2} + + expr_result = @inferred _type_stable_accelerate_expr(a, b) + @test expr_result isa NDArray{Float32,2} +end + @testset verbose = true "solve" begin # native float/complex, 2D and 1D rhs @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) @@ -122,14 +182,14 @@ end @testset verbose = true "svd" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A = cuNumeric.NDArray(T[1 0; 0 1]) - @test @inferred(cuNumeric.svd(A)) !== nothing - @test @inferred(cuNumeric.svd(A, false)) !== nothing + @test @inferred(LinearAlgebra.svd(A)) !== nothing + @test @inferred(LinearAlgebra.svd(A; full=true)) !== nothing end @testset "promote $(T)" for T in (Int32, Int64, Bool) A = cuNumeric.NDArray(T[1 0; 0 1]) allowpromotion() do - @test @inferred(cuNumeric.svd(A)) !== nothing + @test @inferred(LinearAlgebra.svd(A)) !== nothing end end end @@ -137,26 +197,129 @@ end @testset verbose = true "qr" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) A = cuNumeric.NDArray(T[1 0; 0 1]) - @test @inferred(cuNumeric.qr(A)) !== nothing + @test @inferred(LinearAlgebra.qr(A)) !== nothing + end + + @testset "promote $(T)" for T in (Int32, Int64, Bool) + A = cuNumeric.NDArray(T[1 0; 0 1]) + allowpromotion() do + @test @inferred(LinearAlgebra.qr(A)) !== nothing + end + end +end + +@testset verbose = true "cholesky" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + A = cuNumeric.NDArray(Matrix{T}(I, 2, 2)) + @test @inferred(LinearAlgebra.cholesky(A)) !== nothing + B = cuNumeric.NDArray(reshape(Matrix{T}(I, 2, 2), 1, 2, 2)) + @test @inferred(cuNumeric.batched_cholesky(B)) !== nothing end @testset "promote $(T)" for T in (Int32, Int64, Bool) A = cuNumeric.NDArray(T[1 0; 0 1]) allowpromotion() do - @test @inferred(cuNumeric.qr(A)) !== nothing + @test @inferred(LinearAlgebra.cholesky(A)) !== nothing end end end +# A synthetic configuration exercises MP selection without changing the live +# runtime cache or requiring multiple GPUs. Size remains a runtime argument. +function dl_mp_backend(op, shape) + return cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(true, 4, 4)) +end + +@testset "distributed linear algebra inference" begin + cn = cuNumeric + mp = cn._CuSolverMpLinalg + single = cn._SingleProcLinalg + tiled = cn._TiledCholesky + @test (@inferred cn._LinalgRuntime(true, 4, 4)).mp_eligible + @test (@inferred cn._mp_row_partition(33, 4)) == (9, (4, 1)) + + # Dynamic sizes legitimately infer a small union of backend tags, never Any. + for (op, fallback, shape) in ( + (:solve, single, (cn.MIN_SOLVE_MATRIX_SIZE, cn.MIN_SOLVE_MATRIX_SIZE)), + (:qr, single, (cn.MIN_QR_MATRIX_SIZE, 1)), + (:cholesky, tiled, (cn.MIN_CHOLESKY_MATRIX_SIZE, cn.MIN_CHOLESKY_MATRIX_SIZE)), + ) + @test dl_mp_backend(Val(op), shape) isa mp + @test only(Base.return_types(dl_mp_backend, Tuple{Val{op},NTuple{2,Int}})) == + Union{mp,fallback} + end + + # Infer the real MP launchers without executing collectives on this machine. + for T in (Float32, Float64, ComplexF32, ComplexF64) + a = cn.NDArray{T,2,Nothing} + @test only(Base.return_types(cn._solve!, Tuple{mp,a,a,a})) == a + @test only(Base.return_types(cn._qr!, Tuple{mp,a,a,a})) == Nothing + @test only(Base.return_types(cn._cholesky!, Tuple{mp,a,a})) == a + end +end + +@testset verbose = true "eigen" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + A = cuNumeric.NDArray(Matrix{T}(I, 2, 2)) + @test @inferred(LinearAlgebra.eigen(A)) !== nothing + @test @inferred(LinearAlgebra.eigvals(A)) !== nothing + B = cuNumeric.NDArray(reshape(Matrix{T}(I, 2, 2), 1, 2, 2)) + @test @inferred(cuNumeric.batched_eigen(B)) !== nothing + @test @inferred(cuNumeric.batched_eigvals(B)) !== nothing + end + + @testset "promote $(T)" for T in (Int32, Int64, Bool) + A = cuNumeric.NDArray(T[1 0; 0 1]) + allowpromotion() do + @test @inferred(LinearAlgebra.eigen(A)) !== nothing + end + end +end + +@testset verbose = true "fft" begin + if cuNumeric._has_gpu_target() + a = cuNumeric.zeros(ComplexF32, 8) + b = cuNumeric.zeros(ComplexF32, 4, 6) + @test @inferred(fft(a)) !== nothing + @test @inferred(ifft(a)) !== nothing + @test @inferred(fft(b, 1)) !== nothing + @test @inferred(cuNumeric.batched_fft(b)) !== nothing + end +end + +@testset verbose = true "batched_solve" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) + A = cuNumeric.NDArray(reshape(T[2 1; 5 7], 1, 2, 2)) + b = cuNumeric.NDArray(reshape(T[11, 13], 1, 2, 1)) + @test @inferred(cuNumeric.batched_solve(A, b)) !== nothing + end +end + @testset verbose = true "linalg ops" begin @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) M = cuNumeric.zeros(T, 4, 3) sq = cuNumeric.zeros(T, 5, 5) v = cuNumeric.zeros(T, 8) - @test @inferred(cuNumeric.eye(T, 5)) !== nothing + @test @inferred(NDArray{T}(I, 5, 5)) !== nothing @test @inferred(cuNumeric.transpose(M)) !== nothing @test @inferred(cuNumeric.trace(sq)) !== nothing @test @inferred(cuNumeric.diag(sq)) !== nothing + end +end + +@testset verbose = true "sort" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) + v = cuNumeric.zeros(T, 8) + @test @inferred(cuNumeric.sort(v)) !== nothing + if !(T <: Complex) + @test @inferred(cuNumeric.searchsortedfirst(v, zero(T))) !== nothing + end + end +end + +@testset verbose = true "unique" begin + @testset "$(T)" for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) + v = cuNumeric.zeros(T, 8) @test @inferred(cuNumeric.unique(v)) !== nothing end end diff --git a/test/array/batched_linalg.jl b/test/array/batched_linalg.jl new file mode 100644 index 000000000..6621fdd76 --- /dev/null +++ b/test/array/batched_linalg.jl @@ -0,0 +1,190 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): Ethan Meitz +=# + +# Batched (3D+) linear algebra. These entry points are deliberately not +# LinearAlgebra methods: Base has no batched equivalent. + +const N_BATCH = 3 + +function batched_spd(::Type{T}, b, n) where {T} + RT = real(T) + out = zeros(T, b, n, n) + for i in 1:b + B = my_rand(T, n, n; L=RT(-1), R=RT(1)) + out[i, :, :] = B * B' + T(n) * Matrix{T}(I, n, n) + end + return out +end + +function batched_random(::Type{T}, dims...) where {T} + RT = real(T) + return my_rand(T, dims...; L=RT(-1), R=RT(1)) +end + +@testset "batched_solve" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) + n, nrhs = 4, 2 + A_ref = batched_spd(T, N_BATCH, n) + b_ref = batched_random(T, N_BATCH, n, nrhs) + + x = cuNumeric.batched_solve(cuNumeric.NDArray(A_ref), cuNumeric.NDArray(b_ref)) + + allowscalar() do + X = Array(x) + @test size(X) == (N_BATCH, n, nrhs) + for i in 1:N_BATCH + @test isapprox( + A_ref[i, :, :] \ b_ref[i, :, :], X[i, :, :]; atol=atol(T), rtol=rtol(T) + ) + end + end + end +end + +@testset "batched_solve promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + A = cuNumeric.NDArray(reshape(T[1, 0, 0, 1], 1, 2, 2)) + b = cuNumeric.NDArray(reshape(T[1, 1], 1, 2, 1)) + + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" cuNumeric.batched_solve(A, b) + else + @test cuNumeric.batched_solve(A, b) isa NDArray{Float64} + end + + allowpromotion() do + x = cuNumeric.batched_solve(A, b) + allowscalar() do + @test safe_compare( + reshape(Float64[1, 1], 1, 2, 1), x, atol(Float64), rtol(Float64) + ) + end + end + end +end + +@testset "batched_solve rejects 2D input" begin + A = cuNumeric.zeros(Float64, 3, 3) + b = cuNumeric.zeros(Float64, 3, 1) + @test_throws "requires shape (b,m,m)" cuNumeric.batched_solve(A, b) +end + +@testset "batched_cholesky" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + n = 4 + A_ref = batched_spd(T, N_BATCH, n) + out = cuNumeric.batched_cholesky(cuNumeric.NDArray(A_ref)) + + allowscalar() do + L = Array(out) + @test size(L) == (N_BATCH, n, n) + for i in 1:N_BATCH + Li = L[i, :, :] + @test istril(Li) + @test isapprox(A_ref[i, :, :], Li * Li'; atol=atol(T), rtol=rtol(T)) + end + end + end +end + +@testset "batched_cholesky promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + vals = reshape(T[1, 0, 0, 1], 1, 2, 2) + A = cuNumeric.NDArray(vals) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" cuNumeric.batched_cholesky(A) + else + @test cuNumeric.batched_cholesky(A) isa NDArray{Float64} + end + allowpromotion() do + out = cuNumeric.batched_cholesky(A) + allowscalar() do + @test safe_compare(Float64.(vals), out, atol(Float64), rtol(Float64)) + end + end + end +end + +function batched_eigen_residual(A_ref, values, vectors, i) + C = eltype(values) + Ai = C.(A_ref[i, :, :]) + Vi = vectors[i, :, :] + return maximum(abs.(Ai * Vi .- Vi * Diagonal(values[i, :]))) +end + +@testset "batched_eigen" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 4 + A_ref = batched_random(T, N_BATCH, n, n) + values, vectors = cuNumeric.batched_eigen(cuNumeric.NDArray(A_ref)) + + allowscalar() do + vals, vecs = Array(values), Array(vectors) + @test size(vals) == (N_BATCH, n) + @test size(vecs) == (N_BATCH, n, n) + @test eltype(vals) == complex(T) + for i in 1:N_BATCH + @test batched_eigen_residual(A_ref, vals, vecs, i) <= + max(atol(T), rtol(T) * n) + @test isapprox( + sort(vals[i, :]; by=x -> (real(x), imag(x))), + sort(LinearAlgebra.eigvals(A_ref[i, :, :]); by=x -> (real(x), imag(x))), + atol=atol(T), + rtol=rtol(T), + ) + end + end + end +end + +@testset "batched_eigvals" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 4 + A_ref = batched_random(T, N_BATCH, n, n) + values = cuNumeric.batched_eigvals(cuNumeric.NDArray(A_ref)) + + allowscalar() do + vals = Array(values) + @test size(vals) == (N_BATCH, n) + for i in 1:N_BATCH + @test isapprox( + sort(vals[i, :]; by=x -> (real(x), imag(x))), + sort(LinearAlgebra.eigvals(A_ref[i, :, :]); by=x -> (real(x), imag(x))), + atol=atol(T), + rtol=rtol(T), + ) + end + end + end +end + +@testset "batched dimension limits" begin + # Exactly one batch dimension: 2D belongs to the LinearAlgebra entry points, + # and 4D exceeds both the POTRF task and Legate's launch-domain construction. + @testset "$f" for f in + (cuNumeric.batched_cholesky, cuNumeric.batched_eigen, + cuNumeric.batched_eigvals) + @test_throws ArgumentError f(cuNumeric.zeros(Float64, 3, 3)) + @test_throws ArgumentError f(cuNumeric.zeros(Float64, 2, 2, 3, 3)) + @test_throws ArgumentError f(cuNumeric.zeros(Float64, 2, 3, 4)) + end + + @test_throws ArgumentError cuNumeric.batched_solve( + cuNumeric.zeros(Float64, 2, 2, 3, 3), cuNumeric.zeros(Float64, 2, 2, 3, 1) + ) +end diff --git a/test/array/binary/float.jl b/test/array/binary/float.jl index 490d92773..9198fd216 100644 --- a/test/array/binary/float.jl +++ b/test/array/binary/float.jl @@ -25,3 +25,14 @@ run_binary_ops_tests( Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES), ), ) +run_array_equal_tests() + +@testset "map!" begin + lhs = ComplexF64[1 + 2im, 3 + 4im] + rhs = ComplexF64[5 + 6im, 7 + 8im] + dest = cuNumeric.zeros(ComplexF64, 2) + @test map!(*, dest, cuNumeric.NDArray(lhs), cuNumeric.NDArray(rhs)) === dest + allowscalar() do + @test Array(dest) == lhs .* rhs + end +end diff --git a/test/array/binary/tests.jl b/test/array/binary/tests.jl index 794c2e19a..b43ee5966 100644 --- a/test/array/binary/tests.jl +++ b/test/array/binary/tests.jl @@ -36,11 +36,33 @@ function test_binary_operation(func, julia_arr1, julia_arr2, cunumeric_arr1, cun end function test_binary_function_set(func_dict, T, N) - skip = (Base.lcm, Base.gcd) + # Random data includes zeros / huge shift amounts; these are tested separately. + skip = ( + Base.lcm, Base.gcd, Base.fld, Base.mod, Base.rem, Base.:(%), Base.:(<<), Base.:(>>) + ) # not defined for complex. skip_on_complex = ( - Base.:(<), Base.:(<=), Base.:(>), Base.:(>=), Base.max, Base.min, Base.atan, Base.hypot + Base.:(<), + Base.:(<=), + Base.:(>), + Base.:(>=), + Base.max, + Base.min, + Base.atan, + Base.hypot, + Base.:(&), + Base.:(|), + Base.:(⊻), + Base.copysign, + Base.fld, + Base.mod, + Base.rem, + Base.:(%), + Base.:(<<), + Base.:(>>), ) + skip_on_float = (Base.:(&), Base.:(|), Base.:(⊻), Base.:(<<), Base.:(>>)) + skip_on_integer = (Base.copysign,) # kernel is float-only; Bool <: Integer @testset "$func" for func in keys(func_dict) @@ -51,6 +73,14 @@ function test_binary_function_set(func_dict, T, N) continue end + if T <: AbstractFloat && (func in skip_on_float) + continue + end + + if T <: Integer && (func in skip_on_integer) + continue + end + (func in skip) && continue arrs_jl = make_julia_arrays(T, N, :uniform; count=2) @@ -98,12 +128,65 @@ function run_binary_ops_tests(types) end end + # fld/mod/rem/% need a non-zero divisor. FLOOR_DIVIDE is fld, not div. + if (T <: cuNumeric.SUPPORTED_INT_TYPES && T != Bool) || T <: AbstractFloat + if T <: Integer + arr_jl_div = my_rand(T, N; L=(T <: Unsigned ? 1 : -20), R=20) + arr_jl_den = my_rand(T, N; L=1, R=20) + else + arr_jl_div = my_rand(T, N) + arr_jl_den = my_rand(T, N) + arr_jl_den = map( + x -> abs(x) < T(1) ? copysign(one(T), iszero(x) ? one(T) : x) : x, + arr_jl_den, + ) + end + arr_cn_div = @allowscalar NDArray(arr_jl_div) + arr_cn_den = @allowscalar NDArray(arr_jl_den) + + allowscalar() do + for func in (fld, mod, rem, Base.:%) + @test safe_compare( + func.(arr_jl_div, arr_jl_den), func.(arr_cn_div, arr_cn_den), + atol(T), rtol(T), + ) + end + end + end + + # Shifts: small non-negative amounts. Left shift uses a small lhs to + # avoid C++ undefined behavior on signed overflow. + if T <: cuNumeric.SUPPORTED_INT_TYPES && T != Bool + max_shift = T(min(7, 8 * sizeof(T) - 1)) + arr_jl_sh = my_rand(T, N; L=0, R=max_shift) + arr_jl_lshift_lhs = my_rand(T, N; L=0, R=7) + arr_jl_rshift_lhs = my_rand(T, N) + arr_cn_sh = @allowscalar NDArray(arr_jl_sh) + arr_cn_lshift_lhs = @allowscalar NDArray(arr_jl_lshift_lhs) + arr_cn_rshift_lhs = @allowscalar NDArray(arr_jl_rshift_lhs) + + allowscalar() do + @test safe_compare( + arr_jl_lshift_lhs .<< arr_jl_sh, arr_cn_lshift_lhs .<< arr_cn_sh, + atol(T), rtol(T), + ) + @test safe_compare( + arr_jl_rshift_lhs .>> arr_jl_sh, arr_cn_rshift_lhs .>> arr_cn_sh, + atol(T), rtol(T), + ) + end + end + allowscalar() do - @test unwrap(arr_cn == arr_cn) - @test !unwrap(arr_cn == arr_cn2) - @test unwrap(arr_cn != arr_cn2) - @test !unwrap(arr_cn != arr_cn) - @test unwrap(all(arr_cn .== arr_cn)) + eq = arr_cn == arr_cn + neq = arr_cn != arr_cn2 + @test ndims(eq) == 0 + @test ndims(neq) == 0 + @test fetch(eq) + @test !fetch(arr_cn == arr_cn2) + @test fetch(neq) + @test !fetch(arr_cn != arr_cn) + @test fetch(all(arr_cn .== arr_cn)) end end end @@ -117,3 +200,34 @@ function run_binary_copyto_tests() @test is_same(a, b) end end + +function run_array_equal_tests() + @testset "0-d == / !=" begin + a32 = Float32[1, 2, 3] + a64 = Float64[1, 2, 3] + n32 = @allowscalar NDArray(a32) + n64 = @allowscalar NDArray(a64) + + allowscalar() do + @test ndims(n32 == n64) == 0 + @test fetch(n32 == n64) == (a32 == a64) + @test fetch(n32 != n64) == (a32 != a64) + + b64 = Float64[1, 2, 4] + m64 = NDArray(b64) + @test fetch(n32 == m64) == (a32 == b64) + @test fetch(n32 != m64) == (a32 != b64) + end + + same = cuNumeric.ones(2, 2) + other_shape = cuNumeric.ones(3, 3) + other_rank = cuNumeric.ones(4) + allowscalar() do + @test ndims(same == other_shape) == 0 + @test !fetch(same == other_shape) + @test fetch(same != other_shape) + @test !fetch(same == other_rank) + @test fetch(same != other_rank) + end + end +end diff --git a/test/array/broadcast_basic.jl b/test/array/broadcast_basic.jl index e9eb95c04..c0afa6516 100644 --- a/test/array/broadcast_basic.jl +++ b/test/array/broadcast_basic.jl @@ -68,6 +68,18 @@ end result_cpu = zeros(dims) @test result == result_cpu + # Plain broadcast assignment lowers to identity.(source). It must copy + # into both dense NDArrays and writable views. + result .= arrA + @test result == arrA_cpu + + parent = cuNumeric.zeros(Float64, N + 2, N + 2) + center = parent[2:(N + 1), 2:(N + 1)] + center .= arrA + expected_parent = zeros(N + 2, N + 2) + expected_parent[2:(N + 1), 2:(N + 1)] .= arrA_cpu + @test parent == expected_parent + # where the real testing starts arrA = 13.74 .- arrA arrA_cpu = 13.74 .- arrA_cpu @@ -125,6 +137,21 @@ end result_cpu = arrA_cpu .* arrB_cpu @test result == result_cpu + # `@.` lowers associative chains to n-ary Broadcasted nodes. Both the + # fused GPU path and pairwise unfused CPU path must accept them. + result = @. arrA + arrB + arrA + arrB + arrA + result_cpu = @. arrA_cpu + arrB_cpu + arrA_cpu + arrB_cpu + arrA_cpu + @test result == result_cpu + + result = @. 0.5 * arrA * arrB * arrA + result_cpu = @. 0.5 * arrA_cpu * arrB_cpu * arrA_cpu + @test result == result_cpu + + dx = 0.1 + result = @. arrA / dx^2 + result_cpu = @. arrA_cpu / dx^2 + @test result == result_cpu + operator(arrA, arrB) operator(arrA_cpu, arrB_cpu) @test arrA == arrA_cpu @@ -132,6 +159,16 @@ end end end +@testset "muladd broadcast" begin + for T in (Float32, Float64) + a = NDArray(T[1, 2, 3]) + b = NDArray(T[4, 5, 6]) + out = cuNumeric.zeros(T, 3) + out .= muladd.(T(2), a, b) + @test Array(out) == muladd.(T(2), T[1, 2, 3], T[4, 5, 6]) + end +end + #TODO LOOP BINARY OPS WITH SCALARS @testset verbose = true "Scalars" begin N = 10 diff --git a/test/array/cnscalar.jl b/test/array/cnscalar.jl new file mode 100644 index 000000000..d0d33ce8a --- /dev/null +++ b/test/array/cnscalar.jl @@ -0,0 +1,157 @@ +using Test, LinearAlgebra, cuNumeric + +struct CNScalarRealSlot{T<:Real} + value::T +end +struct CNScalarNumberSlot{T<:Number} + value::T +end + +@testset "CNScalar storage and arithmetic" begin + allowautofetch(false) + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + a = NDArray(one(T)) + x = cnscalar(a) + @test x.value === a + @test @inferred(typeof(x)(x)) === x + @test fetch(oneunit(typeof(x))) == one(T) + @test x isa Number + @test x isa supertype(T) + @test isconcretetype(fieldtype(typeof(x), :value)) + @test x isa CNScalar{T} + @test CNScalarNumberSlot(x).value === x + if T <: Real + @test x isa Real + @test CNScalarRealSlot(x).value === x + end + @test fetch(x) == one(T) + @test fetch(a) == one(T) + @test only(x) == fetch(x) + @test_throws ErrorException x[] + @test_throws ErrorException (@allowautofetch x[]) + cuNumeric.allowscalar() do + @test x[] === a[] + @test x[] == one(T) + @test_throws ArgumentError x == one(T) + end + for op in (+, -, *, /, ^) + for (l, r) in ((x, x), (x, one(T)), (one(T), x)) + result = @inferred op(l, r) + @test result isa CNScalar + @test fetch(result) ≈ op(one(T), one(T)) + end + end + @test fetch(zero(x)) == zero(T) + @test fetch(one(x)) == one(T) + for op in (abs, abs2, sqrt, conj, real, imag, inv) + @test fetch(op(x)) ≈ op(one(T)) + end + @test_throws ArgumentError convert(T, x) + @test allowautofetch(() -> convert(T, x)) == one(T) + @test_throws ArgumentError x == one(T) + @test (@allowautofetch x == one(T)) + @test_throws ArgumentError x == one(T) + end + x = sum(NDArray([1.0, 2.0, 3.0])) + @test x isa CNReal{Float64} + @test fetch(x^2) == 36.0 + @test fetch(sqrt(x)) ≈ sqrt(6.0) + @test Array(NDArray([1.0, 2.0]) .* x) == [6.0, 12.0] + @test fetch(max(x, 2.0)) == 6.0 + @test (@allowautofetch x > 2.0) + @test (@allowautofetch isless(2.0, x)) + @test (@allowautofetch 2x) isa CNScalar + p, q = promote(x, 2.0) + @test p isa CNScalar && q isa CNScalar + @test fetch(p) == 6.0 && fetch(q) == 2.0 + for host in (3.0, 1.0 + 2.0im) + p, q = promote(cnscalar(NDArray(2.0f0)), host) + @test p isa CNScalar && q isa CNScalar + @test fetch(p) == 2.0 && fetch(q) == host + end + fractional = cnscalar(NDArray(1.5)) + @test_throws ArgumentError CNInt{Int64}(fractional) + @test_throws InexactError @allowautofetch CNInt{Int64}(fractional) + imaginary = cnscalar(NDArray(1.0 + 2.0im)) + @test_throws ArgumentError CNFloat{Float64}(imaginary) + @test_throws InexactError @allowautofetch CNFloat{Float64}(imaginary) + end +end + +@testset "Device scalar parent storage" begin + for T in (Float64, Int64, UInt64, Bool, ComplexF64) + host = fill(one(T)) + a = NDArray(host) + x = @inferred cnscalar(a) + @test x.value === a + @test x.value.parent === host + @test isconcretetype(fieldtype(typeof(x), :value)) + @test fetch(x) == one(T) + end +end + +@testset "fill! host and device scalars" begin + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + a = cuNumeric.zeros(T, 3) + @test fill!(a, false) === a + @test Array(a) == fill(zero(T), 3) + @test fill!(a, one(T)) === a + @test Array(a) == fill(one(T), 3) + device_zero = cnscalar(NDArray(zero(T))) + for value in (device_zero, device_zero.value) + @test fill!(a, value) === a + @test Array(a) == fill(zero(T), 3) + end + end + a = cuNumeric.zeros(Float32, 2) + @test fill!(a, 2) === a + @test Array(a) == Float32[2, 2] +end + +@testset "Scoped allowautofetch" begin + x = cnscalar(NDArray(2.0)) + @test allowautofetch(() -> 42) == 42 + @test (@allowautofetch 43) == 43 + @test_throws ErrorException allowautofetch() do + error("scope test") + end + @test_throws ArgumentError Float64(x) + @allowautofetch begin + @test Float64(x) == 2.0 + allowautofetch(false) do + @test_throws ArgumentError Float64(x) + end + @test Float64(x) == 2.0 + # Independent tasks do not inherit this permission. + @test fetch(@async get(task_local_storage(), :cuNumericAllowAutoFetch, false)) == false + end + @test_throws ArgumentError Float64(x) + @test_throws ErrorException @allowautofetch error("macro scope test") + @test_throws ArgumentError Float64(x) + allowautofetch(true) + @test Float64(x) == 2.0 + allowautofetch(false) + @test_throws ArgumentError Float64(x) +end + +@testset "Reduction return boundaries" begin + x = NDArray([1.0, 2.0, 3.0]) + for f in (sum, prod, minimum, maximum, cuNumeric.mean, cuNumeric.var, cuNumeric.std) + result = f(x) + @test result isa CNReal + @test fetch(result) ≈ f([1.0, 2.0, 3.0]) + @test f(x; dims=1) isa NDArray{<:Any,1} + end + @test all(NDArray([true, true])) isa CNReal{Bool} + @test fetch(any(NDArray([false, true]))) + @test dot(x,x) isa CNReal + @test fetch(dot(x,x)) == 14.0 + z = NDArray(ComplexF64[1+2im, 3-im]) + @test dot(z,z) isa CNComplex + @test fetch(dot(z,z)) ≈ dot(ComplexF64[1+2im, 3-im], ComplexF64[1+2im, 3-im]) + @test (x == x) isa CNReal{Bool} + @test fetch(x != NDArray([3.0, 2.0, 1.0])) + @test cuNumeric.trace(NDArray([1.0 2.0; 3.0 4.0])) isa CNReal +end diff --git a/test/array/contract.jl b/test/array/contract.jl new file mode 100644 index 000000000..066693b71 --- /dev/null +++ b/test/array/contract.jl @@ -0,0 +1,283 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +using TensorOperations + +# Contractions sum products; TBLIS/cuTENSOR/BLAS may associate differently than +# the host ref. Default (`cancel=true`): Higham floor like unary reductions +# (`n` = contracted length, `scale` ≈ max|A| * max|B|). All-positive inputs +# set `cancel=false` and drop that floor. No-sum cases keep n=1. +function _host_contract_compare( + ref, out, ::Type{T}; n::Integer=1, scale=1, cancel::Bool=true +) where {T} + atolv, rtolv = if n > 1 && cancel + reduction_atol(T, n, scale), reduction_rtol(T, n) + elseif n > 1 + atol(T) * n, reduction_rtol(T, n) + else + atol(T), rtol(T) + end + allowscalar() do + @test safe_compare(ref, out, atolv, rtolv) + end +end + +_contract_scale(A, B) = maximum(abs, A) * maximum(abs, B) + +# TensorOperations forbids an index on both inputs and the output (batched / +# Hadamard). Those refs are explicit loops or broadcasting. +function _batched_ref(A::Array{T,3}, B::Array{T,3}) where {T} + ni, nj, nk = size(A) + nl = size(B, 3) + C = zeros(T, ni, nj, nl) + for i in 1:ni, j in 1:nj, l in 1:nl + s = zero(T) + for k in 1:nk + s += A[i, j, k] * B[i, k, l] + end + C[i, j, l] = s + end + return C +end + +@testset "contract GEMM" begin + @testset for T in (Float32, Float64, ComplexF32, ComplexF64) + A = my_rand(T, 5, 4) + B = my_rand(T, 4, 6) + nda = NDArray(A) + ndb = NDArray(B) + nk = size(A, 2) + scale = _contract_scale(A, B) + @tensor ref_ij[i, j] := A[i, k] * B[k, j] + @tensor ref_ji[j, i] := A[i, k] * B[k, j] + + C = contract(nda, "ik", ndb, "kj") + @test size(C) == (5, 6) + _host_contract_compare(ref_ij, C, T; n=nk, scale) + + out = NDArray{T}(undef, 5, 6) + contract!(out, "ij", nda, "ik", ndb, "kj") + _host_contract_compare(ref_ij, out, T; n=nk, scale) + + C_int = contract(nda, (1, 2), ndb, (2, 3)) + _host_contract_compare(ref_ij, C_int, T; n=nk, scale) + + # Every MM layout: A/B axis order, operand swap, output permutation. + ndaT = permutedims(nda) + ndbT = permutedims(ndb) + for (Ause, Am) in ((nda, "ik"), (ndaT, "ki")) + for (Buse, Bm) in ((ndb, "kj"), (ndbT, "jk")) + for swap in (false, true) + X, Xm, Y, Ym = swap ? (Buse, Bm, Ause, Am) : (Ause, Am, Buse, Bm) + Cij = cuNumeric.zeros(T, 5, 6) + contract!(Cij, "ij", X, Xm, Y, Ym) + _host_contract_compare(ref_ij, Cij, T; n=nk, scale) + Cji = cuNumeric.zeros(T, 6, 5) + contract!(Cji, "ji", X, Xm, Y, Ym) + _host_contract_compare(ref_ji, Cji, T; n=nk, scale) + end + end + end + end +end + +@testset "contract MV / VV / outer" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 5, 4) + x = my_rand(T, 4) + y = my_rand(T, 5) + nda = NDArray(A) + ndx = NDArray(x) + ndy = NDArray(y) + + @tensor ref_mv[i] := A[i, j] * x[j] + _host_contract_compare( + ref_mv, contract(nda, "ij", ndx, "j"), T; n=length(x), scale=_contract_scale(A, x) + ) + _host_contract_compare( + ref_mv, contract(ndx, "j", nda, "ij"), T; n=length(x), scale=_contract_scale(A, x) + ) + + @tensor ref_vm[i] := A[j, i] * y[j] + _host_contract_compare( + ref_vm, contract(nda, "ji", ndy, "j"), T; n=length(y), scale=_contract_scale(A, y) + ) + _host_contract_compare( + ref_vm, contract(ndy, "j", nda, "ji"), T; n=length(y), scale=_contract_scale(A, y) + ) + + u = my_rand(T, 6) + v = my_rand(T, 6) + @tensor ref_dot[] := u[i] * v[i] + dot = contract(NDArray(u), "i", NDArray(v), "i") + @test ndims(dot) == 0 + _host_contract_compare(ref_dot, dot, T; n=length(u), scale=_contract_scale(u, v)) + + @tensor ref_outer[i, j] := u[i] * v[j] + outer = contract(NDArray(u), "i", NDArray(v), "j") + @test size(outer) == (6, 6) + _host_contract_compare(ref_outer, outer, T) + end +end + +@testset "contract batched" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 3, 4, 5) + B = my_rand(T, 3, 5, 6) + nda = NDArray(A) + ndb = NDArray(B) + ref = _batched_ref(A, B) + nk = size(A, 3) + scale = _contract_scale(A, B) + # All 6 output orders of (i, j, l) + for (cm, p) in ( + ("ijl", (1, 2, 3)), + ("ilj", (1, 3, 2)), + ("jil", (2, 1, 3)), + ("jli", (2, 3, 1)), + ("lij", (3, 1, 2)), + ("lji", (3, 2, 1)), + ) + C = cuNumeric.zeros(T, map(d -> size(ref, d), p)...) + contract!(C, cm, nda, "ijk", ndb, "ikl") + _host_contract_compare(permutedims(ref, p), C, T; n=nk, scale) + end + # Input axis orders (general path, not MM) + C = cuNumeric.zeros(T, 3, 4, 6) + contract!(C, "ijl", permutedims(nda, (2, 1, 3)), "jik", ndb, "ikl") + _host_contract_compare(ref, C, T; n=nk, scale) + contract!(C, "ijl", nda, "ijk", permutedims(ndb, (2, 1, 3)), "kil") + _host_contract_compare(ref, C, T; n=nk, scale) + end +end + +@testset "contract Hadamard" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 4, 5) + B = my_rand(T, 4, 5) + ref = A .* B + C = cuNumeric.zeros(T, 4, 5) + contract!(C, "ij", NDArray(A), "ij", NDArray(B), "ij") + _host_contract_compare(ref, C, T) + Cji = cuNumeric.zeros(T, 5, 4) + contract!(Cji, "ji", NDArray(A), "ij", NDArray(B), "ij") + _host_contract_compare(permutedims(ref), Cji, T) + end +end + +@testset "contract alpha/beta" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 4, 3) + B = my_rand(T, 3, 5) + nda = NDArray(A) + ndb = NDArray(B) + α2, α3, β3 = T(2), T(3), T(3) + nk = size(A, 2) + scale = _contract_scale(A, B) + @tensor prod[i, j] := A[i, k] * B[k, j] + + C = contract(nda, "ik", ndb, "kj"; α=α2) + _host_contract_compare(α2 * prod, C, T; n=nk, scale=α2 * scale) + + out = NDArray{T}(undef, 4, 5) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α3, β=0) + _host_contract_compare(α3 * prod, out, T; n=nk, scale=α3 * scale) + + seed = my_rand(T, 4, 5) + out = NDArray(copy(seed)) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α2, β=β3) + _host_contract_compare(β3 * seed + α2 * prod, out, T; n=nk, scale=α2 * scale) + + # α/β on a non-canonical MM layout + out = NDArray(copy(seed)) + contract!(out, "ij", permutedims(nda), "ki", ndb, "kj"; α=α2, β=β3) + _host_contract_compare(β3 * seed + α2 * prod, out, T; n=nk, scale=α2 * scale) + + seed_ji = permutedims(seed) + out_ji = NDArray(copy(seed_ji)) + contract!(out_ji, "ji", nda, "ik", ndb, "kj"; α=α2, β=β3) + _host_contract_compare( + β3 * seed_ji + α2 * permutedims(prod), out_ji, T; n=nk, scale=α2 * scale + ) + + α0 = NDArray(α2) + out = NDArray{T}(undef, 4, 5) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α0, β=0) + _host_contract_compare(α2 * prod, out, T; n=nk, scale=α2 * scale) + + β0 = NDArray(β3) + out = NDArray(copy(seed)) + contract!(out, "ij", nda, "ik", ndb, "kj"; α=α0, β=β0) + _host_contract_compare(β3 * seed + α2 * prod, out, T; n=nk, scale=α2 * scale) + end +end + +@testset "tensordot" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 3, 4, 5) + B = my_rand(T, 5, 6) + nda = NDArray(A) + ndb = NDArray(B) + @tensor ref[i, j, l] := A[i, j, k] * B[k, l] + + C = tensordot(nda, ndb, 1) + @test size(C) == (3, 4, 6) + _host_contract_compare(ref, C, T; n=size(A, 3), scale=_contract_scale(A, B)) + + D = tensordot(nda, ndb, ([3], [1])) + _host_contract_compare(ref, D, T; n=size(A, 3), scale=_contract_scale(A, B)) + end +end + +@testset "contract with empty reduction axis" begin + A = cuNumeric.zeros(Float32, 2, 0) + B = cuNumeric.zeros(Float32, 0, 3) + C = contract(A, "ik", B, "kj") + @test Array(C) == zeros(Float32, 2, 3) +end + +@testset "contract no cancellation" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 5, 4; L=one(T), R=T(1000)) + B = my_rand(T, 4, 6; L=one(T), R=T(1000)) + nk = size(A, 2) + @tensor ref[i, j] := A[i, k] * B[k, j] + _host_contract_compare( + ref, contract(NDArray(A), "ik", NDArray(B), "kj"), T; n=nk, cancel=false + ) + + A3 = my_rand(T, 3, 4, 5; L=one(T), R=T(1000)) + B3 = my_rand(T, 3, 5, 6; L=one(T), R=T(1000)) + C = cuNumeric.zeros(T, 3, 4, 6) + contract!(C, "ijl", NDArray(A3), "ijk", NDArray(B3), "ikl") + _host_contract_compare(_batched_ref(A3, B3), C, T; n=size(A3, 3), cancel=false) + end +end + +@testset "contract errors" begin + A = cuNumeric.ones(Float32, 3, 4) + B = cuNumeric.ones(Float32, 4, 5) + @test_throws ArgumentError contract(A, "ii", B, "jk") + @test_throws ArgumentError contract(A, "ik", B, "kjx") + @test_throws DimensionMismatch contract(A, "ik", cuNumeric.ones(Float32, 3, 5), "kj") + C = cuNumeric.zeros(Float64, 3, 5) + @test_throws ArgumentError contract!(C, "ij", A, "ik", B, "kj") + @test_throws ArgumentError contract!(A, "ij", A, "ik", B, "kj") +end diff --git a/test/array/conversion_lifetimes.jl b/test/array/conversion_lifetimes.jl new file mode 100644 index 000000000..34b0fb65f --- /dev/null +++ b/test/array/conversion_lifetimes.jl @@ -0,0 +1,92 @@ +using Test + +@testset "Direct host-vector copy ownership" begin + for T in (Float32, Float64, ComplexF32, Int32, Bool), n in (0, 1, 4) + source = fill(one(T), n) + dest = cuNumeric.zeros(T, n) + @test copyto!(dest, source) === dest + fill!(source, zero(T)) + GC.gc(true) + @test Array(dest) == fill(one(T), n) + @test_throws DimensionMismatch copyto!(dest, fill(one(T), n + 1)) + end + parent = cuNumeric.zeros(Float32, 6) + dest = view(parent, 2:5) + source = Float32[1, 2, 3, 4] + copyto!(dest, source) + fill!(source, 9f0) + @test Array(parent) == Float32[0, 1, 2, 3, 4, 0] +end + +@testset "copyto! from Array" begin + expected = reshape(ComplexF64.(1:8), 2, 2, 2) + source = copy(expected) + dest = cuNumeric.zeros(ComplexF64, size(source)) + try + @test copyto!(dest, source) === dest + fill!(source, 0) + @test Array(dest) == expected + finally + cuNumeric.destroy!(dest) + end +end + +@testset "1D conversion ownership" begin + for T in (Float32, ComplexF32) + expected = T[1, 2, 3, 4] + source = copy(expected) + a = cuNumeric.NDArray(source) + try + # Construction must finish reading source before returning. + fill!(source, T(99)) + source = nothing + GC.gc(true) + @test Array(a) == expected + + # The returned Julia vector must not alias the NDArray. + converted = Array(a) + converted[1] = T(77) + @test Array(a) == expected + + # It must also survive explicit destruction of the source owner. + survivor = Array(a) + cuNumeric.destroy!(a) + cuNumeric.issue_execution_fence(; block=true) + GC.gc(true) + @test survivor == expected + finally + cuNumeric.destroy!(a) + end + end +end + +@testset "Singleton vector host conversion" begin + # Runtime-created singletons can use scalar futures; attached input + # vectors do not exercise the same storage representation. + for (T, S, value) in ((Float32, Float64, -0.0f0), + (ComplexF32, ComplexF64, ComplexF32(Inf, 0)), + (Bool, Int32, true)) + a = cuNumeric.fill(value, (1,)) + try + converted = Array(a) + widened = Array{S}(a) + @test isequal(converted, T[value]) + @test isequal(widened, S[value]) + converted[1] = zero(T) + @test isequal(Array(a), T[value]) + cuNumeric.destroy!(a) + cuNumeric.issue_execution_fence(; block=true) + GC.gc(true) + @test isequal(widened, S[value]) + finally + cuNumeric.destroy!(a) + end + end + a = cuNumeric.zeros(Float32, 0) + try + @test Array(a) == Float32[] + @test Array{Float64}(a) == Float64[] + finally + cuNumeric.destroy!(a) + end +end diff --git a/test/array/diagonal.jl b/test/array/diagonal.jl new file mode 100644 index 000000000..bcc02f303 --- /dev/null +++ b/test/array/diagonal.jl @@ -0,0 +1,829 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +# Coverage for src/ndarray/diagonal.jl: diag/_eye/trace, Diagonal, UniformScaling. + +const DIAGONAL_NUMERIC_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) +const DIAGONAL_ARRAY_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) +const DIAGONAL_FLOAT_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_FLOAT_TYPES) +const DIAGONAL_COMPLEX_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_COMPLEX_TYPES) + +_nonzero_diag(::Type{T}, n) where {T<:AbstractFloat} = abs.(my_rand(T, n)) .+ one(T) +function _nonzero_diag(::Type{T}, n) where {T<:Complex} + return Complex.(abs.(real(my_rand(T, n))) .+ one(real(T)), zero(real(T))) +end +_nonzero_diag(::Type{T}, n) where {T<:Integer} = T.(collect(2:(n + 1))) +_nonzero_diag(::Type{Bool}, n) = fill(true, n) + +function _host_diag_compare(ref, out, ::Type{T}) where {T} + allowscalar() do + @test cuNumeric.compare(ref, out, atol(T), rtol(T)) + end +end + +function _host_scalar_compare(ref, out::CNScalar, ::Type{T}) where {T} + allowscalar() do + @test ref ≈ fetch(out) atol=atol(T) rtol=rtol(T) + end +end + +function _host_bool_compare(ref::Bool, out::CNReal{Bool}) + allowscalar() do + @test fetch(out) == ref + end +end + +function _host_matrix_compare(ref::AbstractMatrix, out::NDArray, ::Type{T}) where {T} + allowscalar() do + @test cuNumeric.compare(ref, out, atol(T), rtol(T)) + end +end + +###### diag / identity / trace ###### + +@testset "diag" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + A = my_rand(T, 6, 5) + nda = NDArray(A) + @testset "k=$k" for k in (-2, 0, 2) + ref = diag(A, k) + _host_diag_compare(ref, cuNumeric.diag(nda; k=k), T) + _host_diag_compare(ref, LinearAlgebra.diag(nda, k), T) + end + end +end + +@testset "N-D diagonal extract" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 4, 4, 3) + nda = NDArray(A) + ref = [A[i, i, k] for k in 1:3, i in 1:4] + _host_diag_compare(ref, cuNumeric.diagonal(nda; dims=(1, 2)), T) + A2 = copy(A[:, :, 1]) + _host_diag_compare(diag(A2), cuNumeric.diagonal(NDArray(A2)), T) + + B = my_rand(T, 3, 5, 5) + ndb = NDArray(B) + ref_b = [B[i, j, j] for i in 1:3, j in 1:5] + _host_diag_compare(ref_b, cuNumeric.diagonal(ndb; dims=(2, 3)), T) + end + @test_throws ArgumentError cuNumeric.diagonal(cuNumeric.ones(4); dims=(1, 1)) + @test_throws ArgumentError cuNumeric.diagonal(cuNumeric.ones(2, 3); dims=(1, 1)) +end + +@testset "identity via I / _eye" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + n = 4 + ref = Matrix{T}(I, n, n) + _host_matrix_compare(ref, NDArray{T}(I, n, n), T) + _host_matrix_compare(ref, cuNumeric._eye(T, n), T) + end + # Untyped NDArray(I, ...) uses Bool; typed / _eye default to Float32 densify + ref_bool = Matrix{Bool}(I, 3, 3) + _host_matrix_compare(ref_bool, NDArray(I, 3, 3), Bool) + ref = Matrix{Float32}(I, 3, 3) + _host_matrix_compare(ref, NDArray{Float32}(I, 3, 3), Float32) + _host_matrix_compare(ref, cuNumeric._eye(3), Float32) +end + +@testset "trace" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + A = my_rand(T, 5, 5) + nda = NDArray(A) + @testset "offset=$k" for k in (-2, -1, 0, 1, 2) + ref = sum(diag(A, k)) + out = cuNumeric.trace(nda; offset=k) + @test out isa CNScalar + _host_scalar_compare(ref, out, eltype(ref)) + end + end +end + +###### Diagonal constructors / densify / show ###### + +@testset "Diagonal constructors" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 4) + v = NDArray(d) + D = Diagonal(v) + @test D isa Diagonal{T,<:NDArray{T,1}} + @test D.diag === v + allowscalar() do + @test cuNumeric.compare(d, D.diag, atol(T), rtol(T)) + @test Matrix(D) ≈ Matrix(Diagonal(d)) atol=atol(T) rtol=rtol(T) + @test Matrix{T}(D) ≈ Matrix(Diagonal(d)) atol=atol(T) rtol=rtol(T) + end + + A = my_rand(T, 4, 4) + D2 = Diagonal(NDArray(A)) + allowscalar() do + @test cuNumeric.compare(diag(A), D2.diag, atol(T), rtol(T)) + end + end +end + +@testset "Diagonal show" begin + D = Diagonal(NDArray(Float32[1, 2, 3])) + s = sprint(show, D) + @test occursin("1.0", s) && occursin("2.0", s) + plain = sprint(show, MIME"text/plain"(), D) + # Summary must reflect the real Diagonal{T,<:NDArray} type, not Vector. + @test occursin("NDArray", plain) + @test occursin(string(typeof(D)), plain) + @test occursin("1.0", plain) + # Match Base Diagonal formatting (⋅ off-diagonals), not a dense Matrix dump. + @test occursin("⋅", plain) + dense = sprint(show, MIME"text/plain"(), Matrix(D)) + @test plain != dense +end + +###### Diagonal operators ###### + +@testset "Diagonal * Diagonal / ± / scalar" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + allowscalar() do + @test Matrix(D * D) ≈ Matrix(Dh * Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D + D) ≈ Matrix(Dh + Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D - D) ≈ Matrix(Dh - Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(T(3) * D) ≈ Matrix(T(3) * Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * T(3)) ≈ Matrix(Dh * T(3)) atol=atol(T) rtol=rtol(T) + end + end +end + +@testset "Diagonal broadcast on diag" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 5) + c = T(5) + Dh = Diagonal(d) + D = Diagonal(NDArray(copy(d))) + + ref = collect((Dh .* c).diag) + + # Out-of-place: stays Diagonal with NDArray diag + D2 = D .* c + @test D2 isa Diagonal{T,<:NDArray{T,1}} + @test D2.diag isa NDArray{T,1} + _host_diag_compare(ref, D2.diag, T) + + # In-place fused assign + D3 = Diagonal(NDArray(copy(d))) + D3 .= D3 .* c + @test D3 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref, D3.diag, T) + + # In-place .*= + D4 = Diagonal(NDArray(copy(d))) + D4 .*= c + @test D4 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref, D4.diag, T) + + # Zero-preserving Diagonal .+ Diagonal (matches Base structure) + ref_add = collect((Dh .+ Dh).diag) + D_add = D .+ D + @test D_add isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref_add, D_add.diag, T) + + D_add2 = Diagonal(NDArray(copy(d))) + D_add2 .+= D_add2 + @test D_add2 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref_add, D_add2.diag, T) + + D_add3 = Diagonal(NDArray(copy(d))) + D_add3 .= D_add3 .+ D_add3 + @test D_add3 isa Diagonal{T,<:NDArray{T,1}} + _host_diag_compare(ref_add, D_add3.diag, T) + end + + # Exact repro from the bug report + a = Diagonal(cuNumeric.ones(Int32, 5)) + a .*= Int32(5) + @test a isa Diagonal{Int32,<:NDArray{Int32,1}} + @test Array(a.diag) == fill(Int32(5), 5) +end + +@testset "scalar indexing message: LinearAlgebra / Base vs plain" begin + # Unsupported LA/Base fallbacks enrich; intentional user scalar indexing stays plain. + allowscalar(false) + + function _assert_enriched_fallback(msg, modfunc) + @test occursin("`$modfunc` fell back to an AbstractArray implementation", msg) + @test occursin("which scalar-indexed an `NDArray`", msg) + @test occursin("path is probably not implemented yet for `NDArray`", msg) + @test occursin("allowscalar", msg) + @test occursin("@allowscalar", msg) + @test occursin("might allow this function to work slowly", msg) + @test occursin("it has not been tested", msg) + # Enriched path replaces the generic iterating-method lead and omits the plain header. + @test !occursin("typically caused by calling an iterating implementation", msg) + @test !occursin("Scalar indexing is disallowed", msg) + @test !occursin("If you want to allow scalar iteration", msg) + @test !occursin("triggered via", msg) + @test startswith(lstrip(msg), "`$modfunc`") + end + + err = @test_throws ErrorException cholesky(Diagonal(NDArray(Float32[2, 3, 4]))) + msg = sprint(showerror, err.value) + _assert_enriched_fallback(msg, "LinearAlgebra.cholesky") + + # sortperm-based LA path must blame svd, not the inner Base.lt helper. + err_svd = @test_throws ErrorException svd(Diagonal(NDArray(Float32[2, 3, 4]))) + msg_svd = sprint(showerror, err_svd.value) + _assert_enriched_fallback(msg_svd, "LinearAlgebra.svd") + @test !occursin("Base.lt", msg_svd) + + # Base AbstractArray fallback (unique) should enrich with Base.. + err_base = @test_throws ErrorException unique(NDArray(Float32[1, 2, 1])) + msg_base = sprint(showerror, err_base.value) + _assert_enriched_fallback(msg_base, "Base.unique") + + # Call through a Main function so the first non-cuNumeric/Core frame is user code, + # not Base.include_string / Base.IncludeInto (Julia 1.12+) / client frames from + # `include`ing this test file. + function _plain_scalar_index_probe() + a = NDArray(Float32[1, 2, 3]) + return a[1] + end + err_plain = try + _plain_scalar_index_probe() + nothing + catch e + e + end + @test err_plain isa ErrorException + msg_plain = sprint(showerror, err_plain) + @test occursin("Scalar indexing is disallowed", msg_plain) + @test occursin("typically caused by calling an iterating implementation", msg_plain) + @test occursin("If you want to allow scalar iteration", msg_plain) + @test !occursin("triggered via", msg_plain) + @test !occursin("fell back to an AbstractArray", msg_plain) + @test !occursin("probably not implemented yet", msg_plain) + @test !occursin("might allow this function to work slowly", msg_plain) + @test !occursin("it has not been tested", msg_plain) +end + +@testset "NDArray iszero / isone" begin + z = cuNumeric.zeros(Float32, 3) + o = cuNumeric.ones(Float32, 3) + _host_bool_compare(true, iszero(z)) + _host_bool_compare(false, iszero(o)) + I = one(cuNumeric.zeros(Float32, 3, 3)) + _host_bool_compare(true, isone(I)) + _host_bool_compare(false, isone(cuNumeric.ones(Float32, 3, 3))) + _host_bool_compare(false, isone(cuNumeric.zeros(Float32, 3, 3))) + _host_bool_compare(false, isone(cuNumeric.ones(Float32, 2, 3))) +end + +@testset "Diagonal densifying broadcast" begin + D = Diagonal(NDArray(Float32[1, 2, 3])) + expected = Float32[2 1 1; 1 3 1; 1 1 4] + + # Out-of-place densifies to dense NDArray (Base densifies to Matrix). + R = D .+ Float32(1) + @test R isa NDArray{Float32,2} + @test Array(R) == expected + + A = cuNumeric.ones(Float32, 3, 3) + R2 = D .+ A + @test R2 isa NDArray{Float32,2} + @test Array(R2) == expected + R3 = A .+ D + @test R3 isa NDArray{Float32,2} + @test Array(R3) == expected + + # In-place still rejects off-diagonal writes (Base-style ArgumentError). + err_inp = @test_throws ArgumentError D .+= Float32(1) + @test occursin("off-diagonal", sprint(showerror, err_inp.value)) + + D_inp = Diagonal(NDArray(Float32[1, 2, 3])) + err_mat_inp = @test_throws ArgumentError D_inp .+= A + @test occursin("off-diagonal", sprint(showerror, err_mat_inp.value)) + + # Structure-preserving scale still stays Diagonal without @allowscalar. + D2 = Diagonal(NDArray(Float32[1, 2, 3])) + D2 .*= Float32(4) + @test Array(D2.diag) == Float32[4, 8, 12] + D3 = D2 .* Float32(2) + @test D3 isa Diagonal{Float32,<:NDArray{Float32,1}} + @test Array(D3.diag) == Float32[8, 16, 24] + + # 1×1 has no off-diagonals: Base allows in-place densifying-classified ops. + D1 = Diagonal(NDArray(Float32[5])) + D1 .+= Float32(1) + _host_diag_compare(Float32[6], D1.diag, Float32) + D1 .*= Float32(2) + _host_diag_compare(Float32[12], D1.diag, Float32) + + # 1×1 out-of-place densifies to NDArray (Base returns Matrix). + R1 = Diagonal(NDArray(Float32[5])) .+ Float32(1) + @test R1 isa NDArray{Float32,2} + @test size(R1) == (1, 1) + _host_matrix_compare(Float32[6;;], R1, Float32) +end + +@testset "Diagonal * NDArray / mul! / lmul! / rmul!" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + n, m = 3, 4 + d = my_rand(T, n) + dm = my_rand(T, m) + Ah = my_rand(T, n, m) # n×m + As = my_rand(T, n, n) # n×n + vh = my_rand(T, n) + Dh = Diagonal(d) + Dm = Diagonal(dm) + D = Diagonal(NDArray(d)) + D_m = Diagonal(NDArray(dm)) + A = NDArray(Ah) + A_nd = NDArray(As) + v = NDArray(vh) + + _host_diag_compare(Dh * vh, D * v, T) + _host_matrix_compare(Dh * Ah, D * A, T) # D (n)× A (n×m) + _host_matrix_compare(Ah * Dm, A * D_m, T) # A (n×m) × D (m) + _host_matrix_compare(As * Dh, A_nd * D, T) # square A*D + + C = cuNumeric.zeros(T, n, m) + mul!(C, D, A) + _host_matrix_compare(Dh * Ah, C, T) + + Cs = cuNumeric.zeros(T, n, n) + mul!(Cs, A_nd, D) + _host_matrix_compare(As * Dh, Cs, T) + + B = copy(A_nd) + lmul!(D, B) + _host_matrix_compare(Dh * As, B, T) + + B = copy(A_nd) + rmul!(B, D) + _host_matrix_compare(As * Dh, B, T) + end +end + +@testset "Diagonal \\ / / inv / ldiv! / rdiv!" begin + # Floats: full path including singular CUDA-style behavior + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + + _host_diag_compare(Dh \ vh, D \ v, T) + _host_matrix_compare(Dh \ Ah, D \ A, T) + _host_matrix_compare(Ah / Dh, A / D, T) + allowscalar() do + @test Matrix(inv(D)) ≈ Matrix(inv(Dh)) atol=atol(T) rtol=rtol(T) + end + + B = copy(A) + ldiv!(D, B) + _host_matrix_compare(Dh \ Ah, B, T) + + B = copy(A) + rdiv!(B, D) + _host_matrix_compare(Ah / Dh, B, T) + + # Singular: zeros become Inf/NaN (no SingularException; that needs a host Bool) + d0 = copy(d) + d0[2] = zero(T) + D0 = Diagonal(NDArray(d0)) + r = D0 \ v + Di = inv(D0) + Q = A / D0 + allowscalar() do + @test any(isinf, Array(r)) || any(isnan, Array(r)) + @test any(isinf, Array(Di.diag)) || any(isnan, Array(Di.diag)) + @test any(isinf, Array(Q)) || any(isnan, Array(Q)) + end + end + + # Complex: \ works via ./ ; inv/A/D blocked by missing __recip_type + @testset verbose=true for T in DIAGONAL_COMPLEX_TYPES + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + _host_diag_compare(Dh \ vh, D \ v, T) + _host_matrix_compare(Dh \ Ah, D \ A, T) + end + + # Integers with explicit allowpromotion for inv / \ + # `\` uses ./ → typically float(T) (Float64 for Int32/Int64). + # `inv` / `A / D` use cuNumeric.__recip_type (Float32 for Int32, Float64 for Int64), + # which is intentional NDArray promotion — not Base's float(Int32)==Float64. + @testset verbose=true for T in (Int32, Int64) + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + FT = float(T) + RT = cuNumeric.__recip_type(T) + Dh_div = Diagonal(FT.(d)) + Dh_inv = Diagonal(RT.(d)) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + allowpromotion() do + _host_diag_compare(Dh_div \ FT.(vh), D \ v, FT) + _host_matrix_compare(Dh_div \ FT.(Ah), D \ A, FT) + allowscalar() do + @test Matrix(inv(D)) ≈ Matrix(inv(Dh_inv)) atol=atol(RT) rtol=rtol(RT) + end + _host_matrix_compare(RT.(Ah) / Dh_inv, A / D, RT) + end + end +end + +@testset "Diagonal det" begin + # Floats: det is prod of the diagonal (0D NDArray, not a Julia scalar) + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + @test det(D) isa CNScalar + _host_scalar_compare(det(Dh), det(D), T) + _host_scalar_compare(det(Matrix(D)), det(D), T) + + # 1×1 + d1 = T[T(5)] + D1 = Diagonal(NDArray(d1)) + _host_scalar_compare(det(Diagonal(d1)), det(D1), T) + end + + # Integers: prod may widen (e.g. Int32 → Int64); needs allowpromotion + @testset verbose=true for T in (Int32, Int64) + d = _nonzero_diag(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + allowpromotion() do + _host_scalar_compare(det(Dh), det(D), T) + _host_scalar_compare(det(Matrix(D)), det(D), T) + end + # 1×1 + D1 = Diagonal(NDArray(T[7])) + allowpromotion() do + _host_scalar_compare(T(7), det(D1), T) + end + end +end + +@testset "NDArray ± Diagonal" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + n = 3 + d = my_rand(T, n) + Ah = my_rand(T, n, n) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + _host_matrix_compare(Ah + Dh, A + D, T) + _host_matrix_compare(Dh + Ah, D + A, T) + _host_matrix_compare(Ah - Dh, A - D, T) + _host_matrix_compare(Dh - Ah, D - A, T) + end + + # Bool promotes under + + @testset "Bool with allowpromotion" begin + Ah = Bool[1 0; 0 1] + d = Bool[true, true] + A = NDArray(Ah) + D = Diagonal(NDArray(d)) + allowpromotion() do + allowscalar() do + @test Array(A + D) == Ah + Diagonal(d) + @test Array(A - D) == Ah - Diagonal(d) + end + end + end +end + +###### UniformScaling ###### + +# Prefer typed UniformScaling (one(T)*I) so narrow integers do not promote against +# Int64 λ from 2I / -I. Plain I (Bool λ) is fine for +/* on numeric arrays. + +@testset "UniformScaling constructors / copyto!" begin + @testset verbose=true for T in DIAGONAL_ARRAY_TYPES + n = 3 + J1 = one(T) * I + ref = Matrix{T}(J1, n, n) + E = NDArray{T}(J1, n, n) + _host_matrix_compare(ref, E, T) + E2 = NDArray{T}(I, (n, n)) # Bool λ → ones on diagonal still + _host_matrix_compare(Matrix{T}(I, n, n), E2, T) + + C = cuNumeric.zeros(T, n, n) + copyto!(C, J1) + _host_matrix_compare(ref, C, T) + + C = cuNumeric.ones(T, n, n) + copyto!(C, zero(T) * I) + _host_matrix_compare(zeros(T, n, n), C, T) + + # Rectangular scaled identity (skip Bool: Bool(2) is inexact) + if T != Bool + J2 = T(2) * I + R = NDArray{T}(J2, 2, 3) + allowscalar() do + @test Array(R) == Matrix{T}(J2, 2, 3) + end + C = cuNumeric.zeros(T, 2, 3) + copyto!(C, J2) + allowscalar() do + @test Array(C) == Matrix{T}(J2, 2, 3) + end + else + R = NDArray{Bool}(I, 2, 3) + allowscalar() do + @test Array(R) == Matrix{Bool}(I, 2, 3) + end + end + end + + # Untyped NDArray(I, ...) uses λ's type (Bool for I) + E = NDArray(I, 2, 2) + @test eltype(E) == Bool + allowscalar() do + @test Array(E) == Bool[1 0; 0 1] + end + E = NDArray(I, (2, 2)) + allowscalar() do + @test Array(E) == Bool[1 0; 0 1] + end +end + +@testset "NDArray ± / * UniformScaling / one / oneunit" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + Ah = my_rand(T, 3, 3) + A = NDArray(Ah) + J1 = one(T) * I + J2 = T(2) * I + + _host_matrix_compare(Ah + J1, A + J1, T) + _host_matrix_compare(J1 + Ah, J1 + A, T) + _host_matrix_compare(Ah * I, A * I, T) + _host_matrix_compare(I * Ah, I * A, T) + _host_matrix_compare(Ah * J2, A * J2, T) + _host_matrix_compare(J2 * Ah, J2 * A, T) + _host_matrix_compare(Matrix{T}(I, 3, 3), one(A), T) + _host_matrix_compare(Matrix{T}(I, 3, 3), oneunit(A), T) + + # Subtraction / A+2I: signed & float/complex only (unsigned -one wraps; skip) + if T <: Union{AbstractFloat,Complex} || (T <: Signed) + _host_matrix_compare(Ah - J1, A - J1, T) + _host_matrix_compare(J1 - Ah, J1 - A, T) + _host_matrix_compare(Ah + J2, A + J2, T) + end + end + + # Plain I / 2I (Bool/Int64 λ) — natural API for floats & complex + @testset verbose=true for T in (DIAGONAL_FLOAT_TYPES..., DIAGONAL_COMPLEX_TYPES...) + Ah = my_rand(T, 3, 3) + A = NDArray(Ah) + _host_matrix_compare(Ah + I, A + I, T) + _host_matrix_compare(I + Ah, I + A, T) + _host_matrix_compare(Ah - I, A - I, T) + _host_matrix_compare(I - Ah, I - A, T) + _host_matrix_compare(Ah + 2I, A + 2I, T) + _host_matrix_compare(Ah * (2I), A * (2I), T) + _host_matrix_compare((2I) * Ah, (2I) * A, T) + end + + # Bool: * I keeps Bool; ± I needs promotion + @testset "Bool UniformScaling" begin + Ah = Bool[1 0; 0 1] + A = NDArray(Ah) + allowscalar() do + @test Array(A * I) == Ah + @test eltype(A * I) == Bool + @test Array(one(A)) == Ah + @test Array(oneunit(A)) == Ah + end + allowpromotion() do + allowscalar() do + @test Array(A + I) == Ah + I + @test Array(I - A) == I - Ah + end + end + end +end + +###### Diagonal ↔ UniformScaling ###### + +@testset "Diagonal ± / * UniformScaling / copyto!" begin + @testset verbose=true for T in DIAGONAL_NUMERIC_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + J1 = one(T) * I + J2 = T(2) * I + + Di = D + J1 + @test Di isa Diagonal + allowscalar() do + @test Matrix(Di) ≈ Matrix(Dh + J1) atol=atol(eltype(Di)) rtol=rtol(eltype(Di)) + @test Matrix(J1 + D) ≈ Matrix(J1 + Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * I) ≈ Matrix(Dh * I) atol=atol(T) rtol=rtol(T) + @test Matrix(I * D) ≈ Matrix(I * Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * J2) ≈ Matrix(Dh * J2) atol=atol(T) rtol=rtol(T) + end + + if T <: Union{AbstractFloat,Complex} || (T <: Signed) + allowscalar() do + @test Matrix(D - J1) ≈ Matrix(Dh - J1) atol=atol(T) rtol=rtol(T) + @test Matrix(J1 - D) ≈ Matrix(J1 - Dh) atol=atol(eltype((J1 - D).diag)) rtol=rtol( + eltype((J1 - D).diag) + ) + end + end + + copyto!(D, J1) + allowscalar() do + @test Array(D.diag) == ones(T, 3) + end + copyto!(D, zero(T) * I) + allowscalar() do + @test Array(D.diag) == zeros(T, 3) + end + end + + # Plain I on floats + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = my_rand(T, 3) + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + allowscalar() do + @test Matrix(D + I) ≈ Matrix(Dh + I) atol=atol(T) rtol=rtol(T) + @test Matrix(D - I) ≈ Matrix(Dh - I) atol=atol(T) rtol=rtol(T) + @test Matrix(I - D) ≈ Matrix(I - Dh) atol=atol(T) rtol=rtol(T) + @test Matrix(D * (2I)) ≈ Matrix(Dh * (2I)) atol=atol(T) rtol=rtol(T) + end + end + + @testset "Bool Diagonal + I with allowpromotion" begin + D = Diagonal(NDArray(Bool[true, false])) + allowpromotion() do + Di = D + I + @test Di isa Diagonal + allowscalar() do + @test Array(Di.diag) == [2, 1] + end + end + end +end + +###### Structured Diagonal vs densified Matrix path ###### + +# Compare Diagonal{NDArray} ops to the same math on densified host Matrices, +# and check that D + I stays Diagonal while A + I densifies to NDArray. + +@testset "Diagonal vs dense" begin + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + vh = my_rand(T, n) + Dh = Diagonal(d) + Md = Matrix(Dh) # densified host counterpart + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + v = NDArray(vh) + + _host_matrix_compare(Md * Ah, D * A, T) + _host_matrix_compare(Ah * Md, A * D, T) + _host_diag_compare(Md \ vh, D \ v, T) + _host_matrix_compare(Ah / Md, A / D, T) + allowscalar() do + @test Matrix(inv(D)) ≈ inv(Md) atol=atol(T) rtol=rtol(T) + end + _host_matrix_compare(Ah + Md, A + D, T) + + Di = D + I + @test Di isa Diagonal + allowscalar() do + @test Matrix(Di) ≈ Md + I atol=atol(T) rtol=rtol(T) + end + + Ai = A + I + @test Ai isa NDArray + @test !(Ai isa Diagonal) + _host_matrix_compare(Ah + I, Ai, T) + _host_matrix_compare(Ah * I, A * I, T) + end + + # Mul / add / structure for remaining numeric types (skip inv / div) + @testset verbose=true for T in (DIAGONAL_COMPLEX_TYPES..., Int32, Int64) + n = 3 + d = _nonzero_diag(T, n) + Ah = my_rand(T, n, n) + Dh = Diagonal(d) + Md = Matrix(Dh) + D = Diagonal(NDArray(d)) + A = NDArray(Ah) + J1 = one(T) * I + + _host_matrix_compare(Md * Ah, D * A, T) + _host_matrix_compare(Ah * Md, A * D, T) + _host_matrix_compare(Ah + Md, A + D, T) + + Di = D + J1 + @test Di isa Diagonal + allowscalar() do + @test Matrix(Di) ≈ Md + Matrix{T}(J1, n, n) atol=atol(T) rtol=rtol(T) + end + + Ai = A + J1 + @test Ai isa NDArray + @test !(Ai isa Diagonal) + _host_matrix_compare(Ah + J1, Ai, T) + end +end + +###### Native LinearAlgebra API (supported Diagonal paths) ###### + +@testset "Diagonal eigen / eigvals native" begin + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = T[T(3), T(1), T(2)] + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + + # eigvals is a copy of the diagonal (NDArray) + λ = eigvals(D) + @test λ isa NDArray{T,1} + _host_diag_compare(eigvals(Dh), λ, T) + @test λ !== D.diag + + # unsorted eigen: values == diag copy, vectors == I (NDArray) + F = eigen(D) + @test F.values isa NDArray{T,1} + _host_diag_compare(d, F.values, T) + @test F.vectors isa NDArray{T,2} + _host_matrix_compare(Matrix{T}(I, 3, 3), F.vectors, T) + + @test eigvecs(D) isa NDArray{T,2} + _host_matrix_compare(Matrix{T}(I, 3, 3), eigvecs(D), T) + end +end + +@testset "Diagonal native reductions / predicates / norms" begin + @testset verbose=true for T in DIAGONAL_FLOAT_TYPES + d = abs.(my_rand(T, 3)) .+ one(T) # positive → isposdef + Dh = Diagonal(d) + D = Diagonal(NDArray(d)) + + _host_scalar_compare(tr(Dh), tr(D), T) + _host_scalar_compare(sum(Dh), sum(D), T) + _host_scalar_compare(zero(T), prod(D), T) # n>1 off-diagonals + _host_scalar_compare(maximum(Dh), maximum(D), T) + _host_scalar_compare(minimum(Dh), minimum(D), T) + _host_bool_compare(isposdef(Dh), isposdef(D)) + _host_bool_compare(true, issymmetric(D)) + _host_bool_compare(true, ishermitian(D)) + @test isdiag(D) + _host_bool_compare(false, iszero(D)) + _host_bool_compare(false, isone(D)) + _host_bool_compare(true, iszero(Diagonal(cuNumeric.zeros(T, 3)))) + _host_bool_compare(true, isone(Diagonal(cuNumeric.ones(T, 3)))) + _host_bool_compare(true, istriu(D)) + _host_bool_compare(true, istril(D)) + _host_bool_compare(false, istriu(D, 1)) + _host_bool_compare(false, istril(D, -1)) + + _host_scalar_compare(opnorm(Dh), opnorm(D), T) + _host_scalar_compare(norm(Dh), norm(D), T) + _host_scalar_compare(cond(Dh), cond(D), T) + _host_scalar_compare(logdet(Dh), logdet(D), T) + + # matrix functions via f.(diag) broadcast (Base Diagonal methods) + allowscalar() do + @test Matrix(sqrt(D)) ≈ Matrix(sqrt(Dh)) atol=atol(T) rtol=rtol(T) + @test Matrix(exp(D)) ≈ Matrix(exp(Dh)) atol=atol(T) rtol=rtol(T) + end + end +end diff --git a/test/array/diagonal_updates.jl b/test/array/diagonal_updates.jl new file mode 100644 index 000000000..6e4999914 --- /dev/null +++ b/test/array/diagonal_updates.jl @@ -0,0 +1,167 @@ +using Test, LinearAlgebra + +@testset "Diagonal division writes existing destination storage" begin + @allowpromotion for T in (Float32, Float64, ComplexF32, ComplexF64) + dh = T <: Complex ? T[2 + im, 3 - im, 4 + 2im] : T[2, 3, 4] + D = Diagonal(NDArray(dh)) + for ah in (T[2, 6, 12], reshape(T.(1:6), 3, 2)) + a = NDArray(ah) + alias = view(a, ntuple(_ -> Colon(), ndims(a))...) + expected = ah ./ (ndims(ah) == 1 ? dh : reshape(dh, 3, 1)) + @test Array(D \ a) ≈ expected + @test ldiv!(D, a) === a + @test Array(alias) ≈ expected + end + ah = reshape(T.(1:6), 2, 3) + a = NDArray(ah) + alias = view(a, :, :) + expected = ah ./ reshape(dh, 1, 3) + @test Array(a / D) ≈ expected + @test rdiv!(a, D) === a + @test Array(alias) ≈ expected + @test Array(D.diag) == dh + end + + # A shared diagonal must be read completely before its storage is overwritten. + for divide! in (ldiv!, rdiv!) + host = reshape(Float32.(1:9), 3, 3) + a = NDArray(host) + # NDArray slicing retains singleton dimensions; Diagonal needs a vector. + D = Diagonal(cuNumeric.reshape(view(a, :, 1), (3,))) + @test cuNumeric.nda_overlaps(a, D.diag) + dh = copy(host[:, 1]) + expected = host ./ reshape(dh, divide! === ldiv! ? (3, 1) : (1, 3)) + divide! === ldiv! ? ldiv!(D, a) : rdiv!(a, D) + @test Array(a) ≈ expected + @test Array(D.diag) ≈ expected[:, 1] + end + parent = NDArray(Float32[2, 4, 8, 16]) + @test ldiv!(Diagonal(view(parent, 1:3)), view(parent, 2:4)) isa NDArray + @test Array(parent) == Float32[2, 2, 2, 2] + + @allowpromotion for (AType, DType) in ((Float32, Float64), (Float64, Float32), (Int32, Float32)) + dh = DType[2, 4] + ah = AType[4 8; 8 16] + D = Diagonal(NDArray(dh)) + a, b = NDArray(ah), NDArray(ah) + @test ldiv!(D, a) === a + @test rdiv!(b, D) === b + @test Array(a) == AType.(ah ./ reshape(dh, 2, 1)) + @test Array(b) == AType.(ah ./ reshape(dh, 1, 2)) + end + + for T in (Float32, Float64) + dh = T[0, -0.0, Inf, NaN] + ah = reshape(T[1, 0, -1, Inf, 0, 1, Inf, NaN], 2, 4) + D = Diagonal(NDArray(dh)) + a = NDArray(ah) + expected = ah ./ reshape(dh, 1, 4) + @test isequal(Array(a / D), expected) + @test rdiv!(a, D) === a + @test isequal(Array(a), expected) + b = NDArray(copy(permutedims(ah))) + @test ldiv!(D, b) === b + @test isequal(Array(b), permutedims(expected)) + end + + D = Diagonal(NDArray(Float32[2])) + @test_throws DimensionMismatch ldiv!(D, cuNumeric.ones(Float32, 3)) + @test_throws DimensionMismatch ldiv!(D, cuNumeric.ones(Float32, 3, 2)) + @test_throws DimensionMismatch rdiv!(cuNumeric.ones(Float32, 2, 3), D) + empty = Diagonal(NDArray(Float32[])) + @test size(ldiv!(empty, cuNumeric.zeros(Float32, 0, 2))) == (0, 2) + @test size(rdiv!(cuNumeric.zeros(Float32, 2, 0), empty)) == (2, 0) +end + +@testset "Diagonal products write existing destination storage" begin + @allowpromotion for T in (Float32, Float64, ComplexF32, ComplexF64) + dh = T[2, 3, 4] + D = Diagonal(NDArray(dh)) + for ah in (T[1, 2, 3], reshape(T.(1:6), 3, 2)) + a = NDArray(ah) + c = cuNumeric.zeros(T, size(ah)) + v = view(c, ntuple(_ -> Colon(), ndims(c))...) + @test mul!(c, D, a) === c + @test Array(v) ≈ Diagonal(dh) * ah + @test lmul!(D, a) === a + @test Array(a) ≈ Diagonal(dh) * ah + end + ah = reshape(T.(1:6), 2, 3) + a = NDArray(ah) + c = cuNumeric.zeros(T, 2, 3) + v = view(c, :, :) + @test mul!(c, a, D) === c + @test Array(v) ≈ ah * Diagonal(dh) + @test rmul!(a, D) === a + @test Array(a) ≈ ah * Diagonal(dh) + @test_throws DimensionMismatch mul!(cuNumeric.zeros(T, 2), D, NDArray(T[1, 2, 3])) + @test_throws DimensionMismatch mul!(cuNumeric.zeros(T, 2, 2), cuNumeric.ones(T, 2, 2), D) + end + + # Partially overlapping vector input and output must use a temporary. + a = NDArray(Float32[1, 2, 3, 4]) + D = Diagonal(NDArray(Float32[2, 3, 4])) + mul!(view(a, 2:4), D, view(a, 1:3)) + @test Array(a) == Float32[1, 2, 6, 12] + + @allowpromotion begin + c = cuNumeric.zeros(Float64, 3) + mul!(c, D, NDArray(Float32[1, 2, 3])) + @test Array(c) == [2.0, 6.0, 12.0] + end +end + +@testset "Diagonal norm dispatch and autofetch policy" begin + dh = Float32[3, 4] + D = Diagonal(NDArray(dh)) + for p in (0, 0f0, -0.0, 1, 1f0, 2, 2f0, 2.0, 3, 1.5, Inf, Inf32, -Inf32) + actual = fetch(norm(D, p)) + expected = norm(Diagonal(dh), p) + @test actual ≈ expected + @test typeof(actual) == typeof(expected) + end + for p in (NDArray(2), cnscalar(NDArray(2))) + @test_throws "Implicit CNScalar host extraction is disabled" norm(D, p) + @allowautofetch @test fetch(norm(D, p)) ≈ 5f0 + @test fetch(norm(D, fetch(p))) ≈ 5f0 + end +end + +@testset "Diagonal condition results survive temporary cleanup" begin + @allowpromotion for T in (Float32, Float64, ComplexF32, ComplexF64) + dh = T <: Complex ? T[1 + im, 2 - im, 4 + im] : T[1, 2, 4] + D = Diagonal(NDArray(dh)) + orders = (1, 2, Inf) + results = map(p -> cond(D, p), orders) + # Fetch only after temporary handles have been released and collected. + GC.gc() + cuNumeric.drain_pending_frees!() + for (p, result) in zip(orders, results) + @test fetch(result) ≈ cond(Diagonal(dh), p) + end + @test Array(D.diag) == dh + end +end + +@testset "Diagonal zero and negative norms" begin + @allowpromotion for T in (Float32, Float64, ComplexF32, ComplexF64) + for dh in (T[], T[2], T[0], T[2, 3], T[2, 0], T[NaN, 2], T[Inf, 2]) + D = Diagonal(NDArray(dh)) + for p in (0, -1, -2, -Inf) + expected = norm(Diagonal(dh), p) + actual = fetch(norm(D, p)) + @test isapprox(actual, expected; nans=true) + @test typeof(actual) == typeof(expected) + end + end + end +end + +@testset "Diagonal product includes structural zeros" begin + @allowpromotion for T in (Float32, Float64, ComplexF32, ComplexF64) + for dh in (T[], T[2], T[0], T[2, 3], T[NaN, 2], T[Inf, 2], + T[floatmax(real(T)), floatmax(real(T))]) + @test isequal(fetch(prod(Diagonal(NDArray(dh)))), prod(Diagonal(dh))) + end + end +end diff --git a/test/array/distributed_linalg.jl b/test/array/distributed_linalg.jl new file mode 100644 index 000000000..11350981a --- /dev/null +++ b/test/array/distributed_linalg.jl @@ -0,0 +1,165 @@ +using Test, LinearAlgebra, Random +using cuNumeric: cuNumeric + +dl_host(a) = cuNumeric.allowscalar() do + return Array(a) +end +dl_tol(::Type{T}) where {T} = 200 * eps(real(T)) +function dl_backend(op, shape, available, gpus, procs) + return cuNumeric._linalg_backend(op, shape, cuNumeric._LinalgRuntime(available, gpus, procs)) +end + +@testset "linear algebra task selection" begin + cn = cuNumeric + for (op, limit) in ( + (:solve, cn.MIN_SOLVE_MATRIX_SIZE), (:cholesky, cn.MIN_CHOLESKY_MATRIX_SIZE) + ) + for available in (false, true), gpus in (0, 1, 2, 4), n in (limit - 1, limit, limit + 1) + backend = dl_backend(Val(op), (n, n), available, gpus, max(2, gpus)) + @test (backend isa cn._CuSolverMpLinalg) == (available && gpus > 1 && n >= limit) + end + end + qr_volumes = ( + cn.MIN_QR_MATRIX_SIZE - 1, cn.MIN_QR_MATRIX_SIZE, cn.MIN_QR_MATRIX_SIZE + 1 + ) + for available in (false, true), gpus in (0, 1, 2, 4), volume in qr_volumes + backend = dl_backend(Val(:qr), (volume, 1), available, gpus, max(2, gpus)) + @test (backend isa cn._CuSolverMpLinalg) == ( + available && gpus > 1 && volume >= cn.MIN_QR_MATRIX_SIZE + ) + end + for op in (:solve, :cholesky) + @test dl_backend(Val(op), (4, 9000, 9000), true, 4, 4) isa cn._SingleProcLinalg + end + @test dl_backend(Val(:cholesky), (9000, 9000), true, 1, 1) isa cn._SingleProcLinalg + @test dl_backend(Val(:cholesky), (9000, 9000), false, 4, 4) isa cn._TiledCholesky + @test dl_backend(Val(:qr), (0, 10), true, 4, 4) isa cn._SingleProcLinalg + @test cn._mp_row_partition(33, 4) == (9, (4, 1)) + @test cn._mp_row_partition(2, 4) == (1, (2, 1)) + n = max(cn.MIN_CHOLESKY_MATRIX_SIZE + 1, 100 * cn.MIN_CHOLESKY_TILE_SIZE) + colors = cn._cholesky_color_shape(n, 4) + @test colors[1] == colors[2] + @test 4 <= colors[1] <= 4 * cn.MAX_CHOLESKY_TILES_PER_PROC + @test cn._cholesky_color_shape(cn.MIN_CHOLESKY_MATRIX_SIZE, 4) == (1, 1) +end + +@testset "cached runtime configuration" begin + rt = cuNumeric._LINALG_RUNTIME[] + @test rt.available == cuNumeric.cusolvermp_available() + @test rt.gpus == Int(cuNumeric.Legate.num_gpus()) + @test cuNumeric.HAS_CUDA[] == (rt.gpus > 0) + @test cuNumeric._has_gpu_target() == (rt.gpus > 0) + @test rt.procs == Int(cuNumeric.Legate.num_procs()) + @test rt.mp_eligible == (rt.available && rt.gpus > 1) + @test cuNumeric.choose_nd_color_shape((33, 33)) == (1, 1) + @test cuNumeric.choose_nd_color_shape((5, 33, 33)) == (rt.procs, 1, 1) + tiles, colors = cuNumeric.prepare_manual_task_for_batched_matrices((5, 33, 33)) + @test tiles == (cld(5, rt.procs), 33, 33) + @test colors == (cld(5, tiles[1]), 1, 1) +end + +function dl_check_solve(T) + cn = cuNumeric + rng = MersenneTwister(71) + n = 33 + a = randn(rng, T, n, n) + T(n) * I + da = cn.NDArray(a) + for nrhs in (1, 3) + b = randn(rng, T, n, nrhs) + db = cn.NDArray(b) + x = da \ db + hx = dl_host(x) + residual = norm(a * hx - b) / (norm(a) * norm(hx) + norm(b)) + @test residual <= dl_tol(T) + @test dl_host(da) == a + @test dl_host(db) == b + end + # Exercise the public vector-RHS reshape path on the selected backend. + b = randn(rng, T, n) + db = cn.NDArray(b) + x = da \ db + @test size(x) == (n,) + @test isapprox(dl_host(x), a \ b; rtol=dl_tol(T)) +end + +function dl_check_cholesky(T; backend=nothing) + cn = cuNumeric + rng = MersenneTwister(72) + n = 33 + z = randn(rng, T, n, n) + a = z * z' + T(n) * I + # Only the lower triangle is meaningful, including for complex input. + input = copy(a) + for j in 1:n, i in 1:(j - 1) + input[i, j] = T(123) + end + da = cn.NDArray(input) + factors = if backend === nothing + f = cholesky(da) + @test f isa Cholesky + f.factors + else + cn._cholesky!(backend, cn.zeros(T, n, n), da) + end + l = dl_host(factors) + residual = norm(a - l * l') / norm(a) + @test residual <= dl_tol(T) + @test istril(l) + @test dl_host(da) == input +end + +function dl_check_qr(T) + cn = cuNumeric + rng = MersenneTwister(73) + @testset "QR shape ($m, $n)" for (m, n) in ( + (33, 33), (65, 17), (17, 65), (7, 1), (1, 7) + ) + a = randn(rng, T, m, n) + da = cn.NDArray(a) + f = qr(da) + @test f isa cn.NDArrayQR + q, r = f.Q, f.R + k = min(m, n) + @test size(q) == (m, k) + @test size(r) == (k, n) + hq, hr = dl_host(q), dl_host(r) + residual = norm(a - hq * hr) / norm(a) + @test residual <= dl_tol(T) + @test norm(hq' * hq - I) / sqrt(k) <= dl_tol(T) + @test istriu(hr) + @test dl_host(da) == a + end +end + +@testset "distributed linear algebra numerics" begin + for T in (Float32, Float64, ComplexF32, ComplexF64) + @testset "$T" begin + dl_check_solve(T) + dl_check_cholesky(T) + dl_check_qr(T) + dl_check_cholesky(T; backend=cuNumeric._TiledCholesky()) + end + end +end + +@testset "empty and invalid solves" begin + cn = cuNumeric + @test size(cn.zeros(Float64, 0, 0) \ cn.zeros(Float64, 0)) == (0,) + @test size(cn.zeros(Float64, 0, 0) \ cn.zeros(Float64, 0, 3)) == (0, 3) + @test size(cn.zeros(Float64, 3, 3) \ cn.zeros(Float64, 3, 0)) == (3, 0) + @test_throws ArgumentError cn.zeros(Float64, 2, 3) \ cn.zeros(Float64, 2) + @test_throws ArgumentError cn.zeros(Float64, 3, 3) \ cn.zeros(Float64, 2) + # Reject mismatched batches before constructing partitions, including empties. + for (a_batch, b_batch) in ((2, 3), (3, 2), (0, 2), (2, 0)) + @test_throws "matching batch dimensions" cn.batched_solve( + cn.zeros(Float64, a_batch, 3, 3), cn.zeros(Float64, b_batch, 3, 1) + ) + end + @test size(cn.batched_solve(cn.zeros(Float64, 0, 3, 3), cn.zeros(Float64, 0, 3, 1))) == + (0, 3, 1) + for (m, n) in ((0, 0), (0, 3), (3, 0)) + f = qr(cn.zeros(Float64, m, n)) + @test size(f.Q) == (m, min(m, n)) + @test size(f.R) == (min(m, n), n) + end +end diff --git a/test/array/initialization.jl b/test/array/initialization.jl new file mode 100644 index 000000000..df8b0f11b --- /dev/null +++ b/test/array/initialization.jl @@ -0,0 +1,54 @@ +using Test, cuNumeric + +@testset "undef constructor types and ranks" begin + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES), N in 0:6 + dims = ntuple(_ -> 1, N) + for a in (NDArray{T}(undef, dims), NDArray{T}(undef, dims...)) + @test eltype(a) === T + @test size(a) == dims + cuNumeric.destroy!(a) + end + end +end + +@testset "uninitialized NDArray construction" begin + a = NDArray{Float32}(undef, 2, 3) + @test size(a) == (2, 3) + fill!(a, 2f0) + @test Array(a) == fill(2f0, 2, 3) + + b = NDArray{Float64}(undef, (3, 2)) + @test size(b) == (3, 2) + fill!(b, 4.0) + @test Array(b) == fill(4.0, 3, 2) + + scalar = NDArray{Int32}(undef) + @test size(scalar) == () + fill!(scalar, Int32(7)) + @test Array(scalar)[] == 7 + + scalar_like = similar(NDArray{Int32}, ()) + fill!(scalar_like, Int32(9)) + @test Array(scalar_like)[] == 9 + + same = similar(a) + @test size(same) == size(a) + fill!(same, 3f0) + @test Array(same) == fill(3f0, size(a)) + + typed = similar(a, Float64, (3, 2)) + @test size(typed) == (3, 2) + fill!(typed, 5.0) + @test Array(typed) == fill(5.0, 3, 2) + + by_type = similar(NDArray{Float32}, (Base.OneTo(2), Base.OneTo(3))) + @test size(by_type) == (2, 3) + fill!(by_type, 6f0) + @test Array(by_type) == fill(6f0, 2, 3) + + empty = similar(a, Float32, (0, 3)) + @test size(empty) == (0, 3) + @test size(Array(empty)) == (0, 3) + + @test all(iszero, Array(cuNumeric.zeros(Float32, 2, 3))) +end diff --git a/test/array/krylov.jl b/test/array/krylov.jl new file mode 100644 index 000000000..069b1ba70 --- /dev/null +++ b/test/array/krylov.jl @@ -0,0 +1,5 @@ +using Krylov + +@testset "Krylov extension loading" begin + @test Base.get_extension(cuNumeric, :cuNumericKrylovExt) !== nothing +end diff --git a/test/array/linalg.jl b/test/array/linalg.jl index 836d47249..161772402 100644 --- a/test/array/linalg.jl +++ b/test/array/linalg.jl @@ -111,7 +111,7 @@ end @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) n = 5 ref = Matrix{T}(I, n, n) - out = cuNumeric.eye(T, n) + out = NDArray{T}(I, n, n) allowscalar() do @test safe_compare(ref, out, atol(T), rtol(T)) end @@ -126,7 +126,7 @@ end ref = sum(diag(A)) # widens ints like trace's accumulator out = cuNumeric.trace(nda) allowscalar() do - @test ref ≈ out[1] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) + @test ref ≈ out[] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) end end end @@ -140,7 +140,7 @@ end ref = sum(diag(A, k)) out = cuNumeric.trace(nda; offset=k) allowscalar() do - @test ref ≈ out[1] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) + @test ref ≈ out[] atol=atol(eltype(ref)) rtol=rtol(eltype(ref)) end end end @@ -174,18 +174,6 @@ end # end # end -@testset "unique" begin - @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES) - A = T[1, 2, 2, 3, 4, 4, 4, 5] - nda = cuNumeric.NDArray(A) - - ref = unique(A) - out = cuNumeric.unique(nda) - - @test Set(Array(out)) == Set(ref) - end -end - @testset "solve diagonal" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) n = 4 @@ -247,8 +235,12 @@ end A = cuNumeric.NDArray(T[1 0; 0 1]) b = cuNumeric.NDArray(reshape(T[1, 1], 2, 1)) - # int/bool requires promotion to float. Will throw without allowpromtion() - @test_throws "Implicit promotion" cuNumeric.solve(A, b) + # int/bool converts to float; only a widening conversion needs allowpromotion() + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" cuNumeric.solve(A, b) + else + @test cuNumeric.solve(A, b) isa NDArray{Float64} + end # ...allowed under @allowpromotion, result is Float64 allowpromotion() do @@ -261,32 +253,51 @@ end end end -function check_svd_reconstruction(ref_A::AbstractMatrix, u, s, vh, tol_a, tol_r) - U = Array(u) - S = Array(s) - Vh = Array(vh) - A_rec = U * Diagonal(S) * Vh +@testset "solve rejects batched input" begin + A = cuNumeric.zeros(Float64, 2, 3, 3) + b = cuNumeric.zeros(Float64, 2, 3, 1) + @test_throws "batched_solve" cuNumeric.solve(A, b) +end + +@testset "backslash" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SOLVE_TYPES) + A_ref = T[2 1; 5 7] + b_ref = T[11, 13] + A = cuNumeric.NDArray(A_ref) + + x_vec = A \ cuNumeric.NDArray(b_ref) + x_mat = A \ cuNumeric.NDArray(reshape(b_ref, 2, 1)) + + allowscalar() do + @test safe_compare(A_ref \ b_ref, x_vec, atol(T), rtol(T)) + @test safe_compare(A_ref \ reshape(b_ref, 2, 1), x_mat, atol(T), rtol(T)) + end + end +end + +function check_svd_reconstruction(ref_A::AbstractMatrix, F, tol_a, tol_r) + A_rec = Array(F.U) * Diagonal(Array(F.S)) * Array(F.Vt) return isapprox(ref_A, A_rec; atol=tol_a, rtol=tol_r) end -function check_svd_orthonormality(u, vh, tol_a, tol_r) - U = Array(u) - Vh = Array(vh) +function check_svd_orthonormality(F, tol_a, tol_r) + U = Array(F.U) + Vt = Array(F.Vt) ku = size(U, 2) - kv = size(Vh, 1) + kv = size(Vt, 1) ok_u = isapprox(U' * U, Matrix{eltype(U)}(I, ku, ku); atol=tol_a, rtol=tol_r) - ok_vh = isapprox(Vh * Vh', Matrix{eltype(Vh)}(I, kv, kv); atol=tol_a, rtol=tol_r) - return ok_u && ok_vh + ok_vt = isapprox(Vt * Vt', Matrix{eltype(Vt)}(I, kv, kv); atol=tol_a, rtol=tol_r) + return ok_u && ok_vt end @testset "svd square matrix" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A_ref = my_rand(T, 5, 5) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) + @test F isa LinearAlgebra.SVD allowscalar() do - @test check_svd_reconstruction(A_ref, u, s, vh, atol(T), rtol(T)) - @test check_svd_orthonormality(u, vh, atol(T), rtol(T)) + @test check_svd_reconstruction(A_ref, F, atol(T), rtol(T)) + @test check_svd_orthonormality(F, atol(T), rtol(T)) end end end @@ -294,41 +305,43 @@ end @testset "svd tall matrix (m > n)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A_ref = my_rand(T, 6, 4) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda, false) # thin SVD for reconstruction test + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) # thin, for reconstruction allowscalar() do - @test check_svd_reconstruction(A_ref, u, s, vh, atol(T), rtol(T)) - @test check_svd_orthonormality(u, vh, atol(T), rtol(T)) + @test check_svd_reconstruction(A_ref, F, atol(T), rtol(T)) + @test check_svd_orthonormality(F, atol(T), rtol(T)) end end end -@testset "svd thin output shapes (full_matrices=false)" begin +@testset "svd wide matrix (m < n) throws" begin + A = cuNumeric.NDArray(my_rand(Float32, 3, 5)) + @test_throws "m >= n" LinearAlgebra.svd(A) +end + +@testset "svd thin output shapes (full=false)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) m, n = 6, 4 k = min(m, n) A_ref = my_rand(T, m, n) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda, false) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - @test size(Array(u)) == (m, k) - @test size(Array(s)) == (k,) - @test size(Array(vh)) == (k, n) - @test check_svd_reconstruction(A_ref, u, s, vh, atol(T), rtol(T)) + @test size(Array(F.U)) == (m, k) + @test size(Array(F.S)) == (k,) + @test size(Array(F.Vt)) == (k, n) + @test check_svd_reconstruction(A_ref, F, atol(T), rtol(T)) end end end -@testset "svd full output shapes (full_matrices=true)" begin +@testset "svd full output shapes (full=true)" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) m, n = 6, 4 A_ref = my_rand(T, m, n) - nda = cuNumeric.NDArray(A_ref) - u, s, vh = cuNumeric.svd(nda, true) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref); full=true) allowscalar() do - @test size(Array(u)) == (m, m) - @test size(Array(s)) == (min(m, n),) - @test size(Array(vh)) == (n, n) + @test size(Array(F.U)) == (m, m) + @test size(Array(F.S)) == (min(m, n),) + @test size(Array(F.Vt)) == (n, n) end end end @@ -336,10 +349,9 @@ end @testset "svd singular values non-negative and sorted" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) A_ref = my_rand(T, 5, 5) - nda = cuNumeric.NDArray(A_ref) - _, s, _ = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - sv = Array(s) + sv = Array(F.S) @test all(sv .>= 0) @test issorted(sv; rev=true) end @@ -350,10 +362,9 @@ end @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_SVD_TYPES) n = 4 A_ref = Matrix{T}(I, n, n) - nda = cuNumeric.NDArray(A_ref) - _, s, _ = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - @test safe_compare(ones(T, n), s, atol(T), rtol(T)) + @test safe_compare(ones(T, n), F.S, atol(T), rtol(T)) end end end @@ -365,10 +376,9 @@ end v1 = T.(collect(1:5)) v2 = T.(collect(1:4)) A_ref = v1 * v2' - nda = cuNumeric.NDArray(A_ref) - _, s, _ = cuNumeric.svd(nda) + F = LinearAlgebra.svd(cuNumeric.NDArray(A_ref)) allowscalar() do - sv = Array(s) + sv = Array(F.S) @test sv[1] > atol(T) @test all(sv[2:end] .< sqrt(atol(T)) * 100) end @@ -379,10 +389,14 @@ end @testset verbose=true for T in (Int32, Int64, Bool) vals = T == Bool ? T[1 0; 0 1] : reshape(T.(collect(1:4)), 2, 2) A = cuNumeric.NDArray(vals) - @test_throws "Implicit promotion" cuNumeric.svd(A) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.svd(A) + else + @test LinearAlgebra.svd(A) isa LinearAlgebra.SVD + end allowpromotion() do - u, s, vh = cuNumeric.svd(A) - @test eltype(Array(u)) == Float64 + F = LinearAlgebra.svd(A) + @test eltype(Array(F.U)) == Float64 end end end @@ -390,8 +404,7 @@ end @testset "qr reconstruction" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) A_ref = my_rand(T, 6, 4) - nda = cuNumeric.NDArray(A_ref) - q, r = cuNumeric.qr(nda) + q, r = LinearAlgebra.qr(cuNumeric.NDArray(A_ref)) allowscalar() do Q = Array(q) R = Array(r) @@ -406,8 +419,7 @@ end @testset "qr square matrix" begin @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) A_ref = my_rand(T, 5, 5) - nda = cuNumeric.NDArray(A_ref) - q, r = cuNumeric.qr(nda) + q, r = LinearAlgebra.qr(cuNumeric.NDArray(A_ref)) allowscalar() do Q = Array(q) R = Array(r) @@ -422,9 +434,13 @@ end @testset verbose=true for T in (Int32, Int64, Bool) vals = T == Bool ? T[1 0; 0 1] : reshape(T.(collect(1:4)), 2, 2) A = cuNumeric.NDArray(vals) - @test_throws "Implicit promotion" cuNumeric.qr(A) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.qr(A) + else + @test LinearAlgebra.qr(A) isa cuNumeric.NDArrayQR + end allowpromotion() do - q, r = cuNumeric.qr(A) + q, r = LinearAlgebra.qr(A) allowscalar() do @test eltype(Array(q)) == Float64 @test isapprox( @@ -434,3 +450,222 @@ end end end end + +@testset "qr returns an NDArrayQR factorization" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_QR_TYPES) + A_ref = my_rand(T, 6, 4; L=real(T)(-1), R=real(T)(1)) + + F = LinearAlgebra.qr(cuNumeric.NDArray(A_ref)) + @test F isa cuNumeric.NDArrayQR{T} + @test F isa LinearAlgebra.Factorization{T} + @test size(F) == (6, 4) + @test size(F, 1) == 6 + @test size(F, 2) == 4 + @test size(F, 3) == 1 + @test sprint(show, MIME("text/plain"), F) isa String + + allowscalar() do + @test isapprox(A_ref, Array(F.Q) * Array(F.R); atol=atol(T), rtol=rtol(T)) + end + end +end + +# Hermitian positive-definite, with a diagonal shift to keep it well conditioned. +function spd_matrix(::Type{T}, n) where {T} + RT = real(T) + B = my_rand(T, n, n; L=RT(-1), R=RT(1)) + return B * B' + T(n) * Matrix{T}(I, n, n) +end + +@testset "cholesky reconstruction" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + n = 6 + A_ref = spd_matrix(T, n) + F = LinearAlgebra.cholesky(cuNumeric.NDArray(A_ref)) + + @test F isa LinearAlgebra.Cholesky + @test F.uplo == 'L' + + allowscalar() do + L = Array(F.factors) + @test size(L) == (n, n) + @test istril(L) # `zeroout` clears the upper triangle in-task + @test isapprox(A_ref, L * L'; atol=atol(T), rtol=rtol(T)) + end + end +end + +@testset "cholesky of the identity" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_CHOLESKY_TYPES) + n = 4 + A_ref = Matrix{T}(I, n, n) + F = LinearAlgebra.cholesky(cuNumeric.NDArray(A_ref)) + allowscalar() do + @test safe_compare(A_ref, F.factors, atol(T), rtol(T)) + end + end +end + +@testset "cholesky factorization object" begin + n = 5 + T = Float64 + A_ref = spd_matrix(T, n) + F = LinearAlgebra.cholesky(cuNumeric.NDArray(A_ref)) + + L = F.L + @test L isa LowerTriangular + @test parent(L) === F.factors + @test size(F) == (n, n) + + allowscalar() do + @test isapprox(A_ref, Array(parent(L)) * Array(parent(L))'; atol=atol(T), rtol=rtol(T)) + end +end + +@testset "cholesky rejects bad shapes and types" begin + @test_throws ArgumentError LinearAlgebra.cholesky(cuNumeric.zeros(Float64, 3, 4)) + @test_throws ArgumentError LinearAlgebra.cholesky(cuNumeric.zeros(ComplexF64, 0, 0)) +end + +# The POTRF task raises this itself. It only surfaces as a catchable error +# because the launcher marks the task as throwing; without that Legate aborts +# the process. Not a PosDefException: the pivot index is not reported. +@testset "cholesky of a non-positive-definite matrix throws" begin + A = cuNumeric.NDArray(Float64[1.0 2.0; 2.0 1.0]) + @test_throws "Matrix is not positive definite" begin + F = LinearAlgebra.cholesky(A) + allowscalar() do + sum(abs.(Array(F.factors))) + end + end +end + +@testset "cholesky promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + vals = T == Bool ? T[1 0; 0 1] : T[2 0; 0 2] + A = cuNumeric.NDArray(vals) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.cholesky(A) + else + @test LinearAlgebra.cholesky(A) isa LinearAlgebra.Cholesky + end + allowpromotion() do + F = LinearAlgebra.cholesky(A) + allowscalar() do + L = Array(F.factors) + @test eltype(L) == Float64 + @test isapprox(Float64.(vals), L * L'; atol=atol(Float64), rtol=rtol(Float64)) + end + end + end +end + +# Eigenvectors are only unique up to sign/phase, so compare the residual +# `A*v - λ*v` rather than the vectors themselves. +function eigen_residual(A_ref, values, vectors) + C = eltype(values) + return maximum(abs.(C.(A_ref) * vectors .- vectors * Diagonal(values))) +end + +sort_spectrum(v) = sort(v; by=x -> (real(x), imag(x))) + +@testset "eigen residual" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 5 + A_ref = my_rand(T, n, n; L=real(T)(-1), R=real(T)(1)) + F = LinearAlgebra.eigen(cuNumeric.NDArray(A_ref)) + + @test F isa LinearAlgebra.Eigen + + allowscalar() do + values, vectors = Array(F.values), Array(F.vectors) + # geev always produces complex output, even for a real input matrix + @test eltype(values) == complex(T) + @test eltype(vectors) == complex(T) + @test size(values) == (n,) + @test size(vectors) == (n, n) + @test eigen_residual(A_ref, values, vectors) <= max(atol(T), rtol(T) * n) + @test isapprox( + sort_spectrum(values), + sort_spectrum(LinearAlgebra.eigvals(A_ref)), + atol=atol(T), + rtol=rtol(T), + ) + end + end +end + +@testset "eigen of a diagonal matrix" begin + @testset verbose=true for T in Base.uniontypes(cuNumeric.SUPPORTED_EIG_TYPES) + n = 4 + A_ref = Matrix{T}(Diagonal(T.(1:n))) + values = LinearAlgebra.eigvals(cuNumeric.NDArray(A_ref)) + allowscalar() do + @test isapprox( + sort_spectrum(Array(values)), + complex(T).(1:n), + atol=atol(T), + rtol=rtol(T), + ) + end + end +end + +@testset "eigen factorization object" begin + n = 5 + T = Float64 + A_ref = my_rand(T, n, n; L=-1.0, R=1.0) + F = LinearAlgebra.eigen(cuNumeric.NDArray(A_ref)) + + # an Eigen destructures as (values, vectors) + values, vectors = F + @test values === F.values + @test vectors === F.vectors + + allowscalar() do + @test eigen_residual(A_ref, Array(values), Array(vectors)) <= rtol(T) * n + end +end + +@testset "eigvals and eigvecs agree with eigen" begin + n = 4 + T = Float64 + A_ref = my_rand(T, n, n; L=-1.0, R=1.0) + nda = cuNumeric.NDArray(A_ref) + + values = LinearAlgebra.eigvals(nda) + vectors = LinearAlgebra.eigvecs(nda) + + allowscalar() do + @test eigen_residual(A_ref, Array(values), Array(vectors)) <= rtol(T) * n + end +end + +@testset "eigen rejects bad shapes and types" begin + @test_throws ArgumentError LinearAlgebra.eigen(cuNumeric.zeros(Float64, 3, 4)) + @test_throws ArgumentError LinearAlgebra.eigvals(cuNumeric.zeros(Float64, 0, 0)) +end + +@testset "eigen promotion" begin + @testset verbose=true for T in (Int32, Int64, Bool) + vals = T == Bool ? T[1 0; 0 1] : T[2 0; 0 3] + A = cuNumeric.NDArray(vals) + if promotion_is_gated(T, Float64) + @test_throws "Implicit promotion" LinearAlgebra.eigen(A) + else + @test LinearAlgebra.eigen(A) isa LinearAlgebra.Eigen + end + allowpromotion() do + F = LinearAlgebra.eigen(A) + allowscalar() do + @test eltype(Array(F.values)) == ComplexF64 + @test isapprox( + sort_spectrum(Array(F.values)), + sort_spectrum(ComplexF64.(LinearAlgebra.eigvals(Float64.(vals)))), + atol=atol(Float64), + rtol=rtol(Float64), + ) + end + end + end +end diff --git a/test/array/linalg_edge_cases.jl b/test/array/linalg_edge_cases.jl new file mode 100644 index 000000000..30b7ca97a --- /dev/null +++ b/test/array/linalg_edge_cases.jl @@ -0,0 +1,150 @@ +using Test, LinearAlgebra, Random +using cuNumeric: cuNumeric + +le_host(a) = cuNumeric.allowscalar() do + return Array(a) +end +le_tol(::Type{T}) where {T} = 200 * eps(real(T)) +le_residual(a, b) = norm(a - b) / max(norm(a), one(real(eltype(a)))) + +# Keep the padded parent so we can check that the operation preserves both +# its input and the elements outside the view. Do not copy the view before +# passing it to the operation: the backend must handle its layout. +function le_input(a, layout) + if layout == :slice + m, n = size(a) + parent = fill(eltype(a)(19), m + 2, n + 2) + parent[2:(m + 1), 2:(n + 1)] = a + dp = cuNumeric.NDArray(parent) + return view(dp, 2:(m + 1), 2:(n + 1)), dp, parent + elseif layout == :transpose + parent = copy(transpose(a)) + dp = cuNumeric.NDArray(parent) + return cuNumeric.transpose(dp), dp, parent + end + dp = cuNumeric.NDArray(a) + return dp, dp, a +end + +function le_qr(a, da) + m, n = size(a) + k = min(m, n) + f = qr(da) + q, r = le_host(f.Q), le_host(f.R) + @test size(q) == (m, k) + @test size(r) == (k, n) + @test le_residual(a, q * r) <= le_tol(eltype(a)) + @test norm(q' * q - I) <= le_tol(eltype(a)) * k + @test istriu(r) +end + +function le_svd(a, da) + m, n = size(a) + for full in (false, true) + f = svd(da; full) + u, s, vt = le_host(f.U), le_host(f.S), le_host(f.Vt) + @test size(u) == (m, full ? m : n) + @test size(s) == (n,) + @test size(vt) == (n, n) + @test le_residual(a, u[:, 1:n] * Diagonal(s) * vt) <= le_tol(eltype(a)) + @test norm(u' * u - I) <= le_tol(eltype(a)) * size(u, 2) + @test norm(vt * vt' - I) <= le_tol(eltype(a)) * n + @test all(s .>= 0) + @test issorted(s; rev=true) + @test isapprox(s, svdvals(a); atol=le_tol(eltype(a)), rtol=le_tol(eltype(a))) + end +end + +function le_eigen(a, da) + f = eigen(da) + w, v = le_host(f.values), le_host(f.vectors) + n = size(a, 1) + @test size(w) == (n,) + @test size(v) == (n, n) + @test norm(a * v - v * Diagonal(w)) / max(norm(a) * norm(v), 1) <= le_tol(eltype(a)) + @test all(j -> isapprox(norm(v[:, j]), 1; atol=le_tol(eltype(a))), 1:n) + # The fixtures are Hermitian, so their spectra are real, including repeated + # zero eigenvalues. Sorting by real part avoids arbitrary eigenvector order. + expected = eigvals(Hermitian(a)) + for values in (w, le_host(eigvals(da))) + @test maximum(abs, imag.(values)) <= le_tol(eltype(a)) * max(norm(a), 1) + @test isapprox( + sort(real.(values)), expected; atol=le_tol(eltype(a)), rtol=le_tol(eltype(a)) + ) + end +end + +@testset "linear algebra degenerate inputs" begin + @testset "$T" for T in (Float32, Float64, ComplexF32, ComplexF64) + @testset "QR/SVD $kind ($m, $n)" for kind in (:zero, :rank_one), + (m, n) in ((4, 4), (6, 4), (4, 6)) + + u = T.(1:m) + T <: Complex && (u .+= im .* reverse(u)) + a = kind == :zero ? zeros(T, m, n) : u * transpose(T.(1:n)) + da = cuNumeric.NDArray(a) + le_qr(a, da) + m >= n && le_svd(a, da) + @test le_host(da) == a + end + @testset "square $kind" for kind in (:zero, :rank_one) + a = zeros(T, 4, 4) + kind == :rank_one && (a[1, 1] = 3) + da = cuNumeric.NDArray(a) + le_eigen(a, da) + @test le_host(da) == a + # A zero pivot is exact here. Materialize inside @test_throws so + # asynchronous task errors are observed by the assertion. + for b in (ones(T, 4), ones(T, 4, 2)) + db = cuNumeric.NDArray(b) + @test_throws "Singular matrix" le_host(da \ db) + @test le_host(db) == b + @test le_host(da) == a + end + @test_throws "Matrix is not positive definite" le_host(cholesky(da).factors) + @test le_host(da) == a + end + end +end + +@testset "linear algebra input layouts" begin + @testset "$T $layout" for T in (Float32, Float64, ComplexF32, ComplexF64), + layout in (:slice, :transpose) + + rng = MersenneTwister(81) + @testset "QR/SVD ($m, $n)" for (m, n) in ((4, 4), (6, 4), (4, 6)) + a = randn(rng, T, m, n) + da, dp, parent = le_input(a, layout) + le_qr(a, da) + m >= n && le_svd(a, da) + @test le_host(dp) == parent + end + z = randn(rng, T, 4, 4) + a = z * z' + T(4) * I + da, dp, parent = le_input(a, layout) + l = le_host(cholesky(da).factors) + @test le_residual(a, l * l') <= le_tol(T) + @test istril(l) + @test le_host(dp) == parent + le_eigen(a, da) + @test le_host(dp) == parent + # Also exercise a non-Hermitian solve so transpose/conjugation mistakes + # cannot be hidden by the Cholesky/eigen fixture's symmetry. + a = z + T(8) * I + da, dp, parent = le_input(a, layout) + @testset "solve $nrhs RHS" for nrhs in (1, 2) + b = randn(rng, T, 4, nrhs) + db, bp, bparent = le_input(b, layout) + x = le_host(da \ db) + @test size(x) == size(b) + @test le_residual(b, a * x) <= le_tol(T) + @test le_host(bp) == bparent + @test le_host(dp) == parent + end + # A zero RHS is valid even though a zero coefficient matrix is not. + for b in (zeros(T, 4), zeros(T, 4, 2)) + @test le_host(da \ cuNumeric.NDArray(b)) == b + end + @test le_host(dp) == parent + end +end diff --git a/test/array/mapreduce_policy.jl b/test/array/mapreduce_policy.jl new file mode 100644 index 000000000..eaf52142e --- /dev/null +++ b/test/array/mapreduce_policy.jl @@ -0,0 +1,77 @@ +struct ReductionPointerNumber <: Number + ptr::Ptr{Float32} +end +(f::ReductionPointerNumber)(x) = x + +@testset "Mapped reduction policies" begin + CN = cuNumeric + noinit = CN.NoReductionInit() + @test CN._mr_dims((2, 3), :) == ((true, true), ()) + @test CN._mr_dims((2, 3), (1, 1, 4)) == ((true, false), (1, 3)) + @test CN._mr_dims((2, 3), ()) == ((false, false), (2, 3)) + @test_throws ArgumentError CN._mr_dims((2, 3), 0) + @test_throws ArgumentError CN._mr_dims((2, 3), (1, 1.5)) + @test_throws ArgumentError CN._mr_operator(-) + @test CN._mr_redop(+, Float32) isa CN.MapReduceOp + @test CN._mr_redop(+, Float32) == CN.MAPREDUCE_ADD + @test CN._mr_redop(Base.add_sum, Int64) == CN.MAPREDUCE_ADD + @test CN._mr_redop(*, Float32) == CN.MAPREDUCE_MUL + @test CN._mr_redop(Base.mul_prod, Int64) == CN.MAPREDUCE_MUL + @test CN._mr_redop(min, UInt32) == CN.MAPREDUCE_MIN + @test CN._mr_redop(max, UInt64) == CN.MAPREDUCE_MAX + for op in (*, Base.mul_prod, min, max), a in (false, true), b in (false, true) + @test CN._mr_storage(op, Bool) === UInt8 + encoded = CN._mr_combine(op)(CN._mr_encode(op, a), CN._mr_encode(op, b)) + @test CN._mr_decode(op, Bool, encoded) === op(a, b) + end + @test_throws ArgumentError CN._mr_accumulator(min, ComplexF32) + for op in (*, Base.mul_prod) + @test_throws ArgumentError CN._mr_accumulator(op, ComplexF64) + @test_throws ArgumentError CN._mr_accumulator(op, ComplexF32, one(ComplexF64), 1) + end + @test_throws ArgumentError CN._mr_accumulator(+, Float64, 0f0, 1) + @test_throws ArgumentError CN._mr_accumulator(min, Float32, ComplexF32(0), 1) + @test_throws ArgumentError CN._mr_mapped_type(ReductionPointerNumber(Ptr{Float32}(0)), Float32) + + for T in (Bool, Int8, UInt16, Int32, UInt64, Float32, Float64, ComplexF32), + op in (+, *, Base.add_sum, Base.mul_prod) + @test CN._mr_accumulator(op, T) === typeof(Base.reduce_first(op, one(T))) + end + @testset "Empty reduction f=$f op=$op dims=$dims" for + f in (identity, abs, abs2, x -> x*x), op in (+, *, min, max), dims in (:, 1, (1,)) + reference = try + mapreduce(f, op, Float32[]; dims) + catch e + e + end + if reference isa Exception + @test_throws typeof(reference) CN._mr_empty(f, op, Float32, Float32, noinit, dims) + else + scalar = reference isa AbstractArray ? only(reference) : reference + @test isequal(CN._mr_empty(f, op, Float32, Float32, noinit, dims), scalar) + end + end + @test CN._mr_output_type(+, Float32, 0.0, :) === Float64 + @test CN._mr_output_type(+, Float32, noinit, 1) === Float32 + + # Verify ordering against Julia, including NaNs and signed zeros. + for T in (Float32, Float64), op in (min, max) + xs = T[-Inf, -2, -0.0, 0.0, 2, Inf, NaN] + for a in xs, b in xs + encoded = op(CN._mr_encode(op, a), CN._mr_encode(op, b)) + @test isequal(CN._mr_decode(op, T, encoded), op(a, b)) + end + end + captured = let a = [1.0f0] + x -> x + a[1] + end + @test_throws ArgumentError CN._mr_mapped_type(captured, Float32) + @test CN._mr_mapped_type(CN._mr_callable(Float64), Float32) === Float64 + + if !CN._has_gpu_target() + A = CN.ones(Float32, 4) + @test_throws ArgumentError mapreduce(identity, +, A) + @test_throws ArgumentError sum(abs2, A) + CN.destroy!(A) + end +end diff --git a/test/array/permutedims.jl b/test/array/permutedims.jl new file mode 100644 index 000000000..526d0fa43 --- /dev/null +++ b/test/array/permutedims.jl @@ -0,0 +1,67 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +@testset "permutedims" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 3, 4) + nda = NDArray(A) + allowscalar() do + @test safe_compare(permutedims(A), cuNumeric.transpose(nda), atol(T), rtol(T)) + @test safe_compare(permutedims(A, (2, 1)), permutedims(nda, (2, 1)), atol(T), rtol(T)) + @test safe_compare(permutedims(A), permutedims(nda), atol(T), rtol(T)) + end + + B = my_rand(T, 2, 3, 4) + ndb = NDArray(B) + allowscalar() do + @test safe_compare( + permutedims(B, (3, 1, 2)), permutedims(ndb, (3, 1, 2)), atol(T), rtol(T) + ) + @test safe_compare( + permutedims(B, (1, 2, 3)), permutedims(ndb, (1, 2, 3)), atol(T), rtol(T) + ) + end + end + @test_throws ArgumentError permutedims(cuNumeric.ones(2, 3), (1,)) + @test_throws ArgumentError permutedims(cuNumeric.ones(2, 3), (1, 1)) + @test_throws ArgumentError permutedims(cuNumeric.ones(2, 3), (1, 3)) +end + +@testset "squeeze / dropdims" begin + @testset for T in (Float32, Float64) + A = my_rand(T, 2, 3) + nda = NDArray(reshape(A, 2, 1, 3)) + out = squeeze(nda) + @test size(out) == (2, 3) + allowscalar() do + @test safe_compare(A, out, atol(T), rtol(T)) + end + @test size(squeeze(NDArray(A))) == (2, 3) + + dropped = dropdims(nda; dims=2) + @test size(dropped) == (2, 3) + allowscalar() do + @test safe_compare(A, dropped, atol(T), rtol(T)) + end + @test size(squeeze(nda, 2)) == (2, 3) + @test_throws DimensionMismatch squeeze(nda, 1) + @test_throws ArgumentError dropdims(nda; dims=4) + end +end diff --git a/test/array/random.jl b/test/array/random.jl new file mode 100644 index 000000000..9049c3307 --- /dev/null +++ b/test/array/random.jl @@ -0,0 +1,296 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +#= Purpose of test: random + -- dtype / shape of rand, randn, randexp, and in-place fills + -- complex rand / randn (independent real/imag); randexp throws + -- uniforms land in [0, 1) with mean 1/2 and variance 1/12 + -- normals match loc / scale moments (default and shifted) + -- exponentials match scale moments (mean = var = scale) + -- Monte-Carlo integral of exp(-x^2) recovers √π + -- rand(a:b) is inclusive and a small discrete range has the right mean / var + -- XORWOW / MRG32k3a / PHILOX4_32_10 all draw valid uniforms +=# + +function _host(arr) + return allowscalar() do + return Array(arr) + end +end + +function _moments(arr) + h = _host(arr) + return mean(h), var(h) +end + +@testset verbose = true "constructors" begin + for T in (Float32, Float64) + A = cuNumeric.rand(T, 8, 4) + @test eltype(A) === T + @test size(A) == (8, 4) + @test all(0 .<= _host(A) .< 1) + + B = cuNumeric.zeros(T, 5, 5) + cuNumeric.rand!(B) + @test all(0 .<= _host(B) .< 1) + + N = cuNumeric.randn(T, 32) + @test eltype(N) === T + @test size(N) == (32,) + + C = cuNumeric.zeros(T, 16) + Random.randn!(C) + @test eltype(C) === T + + E = cuNumeric.randexp(T, 16) + @test eltype(E) === T + @test size(E) == (16,) + Random.randexp!(C) + @test eltype(C) === T + end + + @test eltype(cuNumeric.randexp(8)) === Float32 + @test_throws MethodError cuNumeric.randexp(Int32, 4) + @test_throws MethodError cuNumeric.randexp(ComplexF32, 4) + @test_throws MethodError cuNumeric.randexp(ComplexF64, 4) + @test_throws MethodError cuNumeric.rand(Complex{Int64}, 4) + @test_throws MethodError cuNumeric.randn(Complex{Int64}, 4) + @test_throws MethodError cuNumeric.randexp(Complex{Int64}, 4) + + for T in (ComplexF32, ComplexF64) + A = cuNumeric.rand(T, 8, 4) + @test eltype(A) === T + @test size(A) == (8, 4) + h = _host(A) + @test all(0 .<= real.(h) .< 1) + @test all(0 .<= imag.(h) .< 1) + + B = cuNumeric.zeros(T, 5, 5) + cuNumeric.rand!(B) + hb = _host(B) + @test all(0 .<= real.(hb) .< 1) + @test all(0 .<= imag.(hb) .< 1) + + N = cuNumeric.randn(T, 32) + @test eltype(N) === T + @test size(N) == (32,) + + C = cuNumeric.zeros(T, 16) + Random.randn!(C) + @test eltype(C) === T + + @test_throws ErrorException Random.randexp!(C) + g = cuNumeric.default_rng() + @test_throws ErrorException randexp!(g, C) + @test_throws MethodError cuNumeric.randexp(g, T, (4,)) + end + + R = cuNumeric.rand(3, 2) + @test eltype(R) === Float32 + @test size(R) == (3, 2) + + @test_throws MethodError cuNumeric.randn(Int32, 4) + + Coin = cuNumeric.rand(Bool, 8, 4) + @test eltype(Coin) === Bool + @test size(Coin) == (8, 4) + @test all(x -> x == true || x == false, _host(Coin)) + Mask = cuNumeric.falses(16) + cuNumeric.rand!(Mask) + @test eltype(Mask) === Bool +end + +@testset verbose = true "uniform moments" begin + n = 65_536 + for T in (Float32, Float64) + μ, v = _moments(cuNumeric.rand(T, n)) + @test abs(μ - 0.5) < 0.03 + @test abs(v - 1 / 12) < 0.01 + + μ2, v2 = _moments(cuNumeric.rand(T, 128, 128)) + @test abs(μ2 - 0.5) < 0.03 + @test abs(v2 - 1 / 12) < 0.01 + end + + μb, vb = _moments(cuNumeric.rand(Bool, 65_536)) + @test abs(μb - 0.5) < 0.03 + @test abs(vb - 0.25) < 0.02 + + n = 65_536 + for T in (ComplexF32, ComplexF64) + Z = _host(cuNumeric.rand(T, n)) + @test abs(mean(real.(Z)) - 0.5) < 0.03 + @test abs(mean(imag.(Z)) - 0.5) < 0.03 + @test abs(var(real.(Z)) - 1 / 12) < 0.01 + @test abs(var(imag.(Z)) - 1 / 12) < 0.01 + end +end + +@testset verbose = true "normal moments" begin + n = 65_536 + for T in (Float32, Float64) + μ, v = _moments(cuNumeric.randn(T, n)) + @test abs(μ) < 0.05 + @test abs(v - 1) < 0.08 + + g = cuNumeric.default_rng() + A = cuNumeric.zeros(T, n) + cuNumeric._randn!(g, A; loc=T(3), scale=T(2)) + μs, vs = _moments(A) + @test abs(μs - 3) < 0.1 + @test abs(vs - 4) < 0.3 + end + + # Julia randn(Complex): Var(re) = Var(im) = 1/2, E[|z|²] = 1. + n = 65_536 + for T in (ComplexF32, ComplexF64) + Z = _host(cuNumeric.randn(T, n)) + @test abs(mean(real.(Z))) < 0.05 + @test abs(mean(imag.(Z))) < 0.05 + @test abs(var(real.(Z)) - 0.5) < 0.08 + @test abs(var(imag.(Z)) - 0.5) < 0.08 + @test abs(mean(abs2.(Z)) - 1) < 0.08 + end +end + +@testset verbose = true "exponential moments" begin + n = 65_536 + for T in (Float32, Float64) + μ, v = _moments(cuNumeric.randexp(T, n)) + @test abs(μ - 1) < 0.08 + @test abs(v - 1) < 0.12 + @test all(>=(zero(T)), _host(cuNumeric.randexp(T, 1024))) + + g = cuNumeric.default_rng() + A = cuNumeric.zeros(T, n) + randexp!(g, A; scale=T(2)) + μs, vs = _moments(A) + @test abs(μs - 2) < 0.12 + @test abs(vs - 4) < 0.4 + end + + @test_throws ArgumentError randexp!(cuNumeric.default_rng(), cuNumeric.zeros(8); scale=0) +end + +@testset verbose = true "monte carlo" begin + # ∫_{-∞}^{∞} exp(-x^2) dx = √π, truncated to [-10, 10] as in the docs example. + n = 131_072 + for T in (Float32, Float64) + xmax = T(10) + Ω = T(2) * xmax + samples = Ω .* cuNumeric.rand(T, n) .- xmax + integrand = (x) -> @. exp(-x^2) + estimate = fetch((Ω / n) * sum(integrand(samples))) + @test isapprox(estimate, T(sqrt(π)); atol=T(0.08)) + end + + # Area of the unit disk via darts in [0, 1]^2 → π. + n = 131_072 + for T in (Float32, Float64) + x = _host(cuNumeric.rand(T, n)) + y = _host(cuNumeric.rand(T, n)) + π_estimate = 4 * mean(x .^ 2 .+ y .^ 2 .< one(T)) + @test isapprox(π_estimate, T(π); atol=T(0.05)) + end +end + +@testset verbose = true "integers" begin + A = cuNumeric.rand(0:9, 64) + @test eltype(A) === Int + @test all(0 .<= _host(A) .<= 9) + + B = cuNumeric.rand(Int32(-4):Int32(4), 8, 8) + @test eltype(B) === Int32 + @test size(B) == (8, 8) + @test all(-4 .<= _host(B) .<= 4) + + C = cuNumeric.rand(Int16, 32) + @test eltype(C) === Int16 + @test all(typemin(Int16) .<= _host(C) .< typemax(Int16)) + + D = cuNumeric.zeros(Int32, 16) + cuNumeric.rand!(D) + @test all(typemin(Int32) .<= _host(D) .< typemax(Int32)) + + # singleton range is constant + Z = _host(cuNumeric.rand(0:0, 256)) + @test all(==(0), Z) + F = _host(cuNumeric.rand(Int32(5):Int32(5), 128)) + @test all(==(5), F) + + # discrete uniform on 0:9: mean 4.5, var (n^2-1)/12 with n=10 + n = 65_536 + h = Float64.(_host(cuNumeric.rand(0:9, n))) + @test abs(mean(h) - 4.5) < 0.08 + @test abs(var(h) - (100 - 1) / 12) < 0.2 + + @test_throws ArgumentError cuNumeric.rand(1:0, 4) +end + +@testset verbose = true "default_rng" begin + g = cuNumeric.default_rng() + @test g isa cuNumeric.Generator + A = cuNumeric.random(g, Float32, (4, 4)) + @test eltype(A) === Float32 + @test size(A) == (4, 4) + @test all(0 .<= _host(A) .< 1) + + g2 = cuNumeric.default_rng(1234) + @test g2 isa cuNumeric.Generator + @test g2.bit_generator isa cuNumeric.XORWOW + @test g2.bit_generator.seed == UInt64(1234) + B = cuNumeric.randn(g2, Float64, (32,)) + @test eltype(B) === Float64 + + C = cuNumeric.random(g2, ComplexF32, (8, 8)) + @test eltype(C) === ComplexF32 + @test size(C) == (8, 8) + Ch = _host(C) + @test all(0 .<= real.(Ch) .< 1) + @test all(0 .<= imag.(Ch) .< 1) + + s1 = cuNumeric.get_static_generator() + s2 = cuNumeric.get_static_generator() + @test s1 === s2 + @test s1 !== g2 +end + +@testset verbose = true "bitgenerators" begin + n = 16_384 + for B in (cuNumeric.XORWOW, cuNumeric.MRG32k3a, cuNumeric.PHILOX4_32_10) + @testset "$(B)" begin + g = cuNumeric.default_rng(B, 42) + @test g isa cuNumeric.Generator{B} + @test g.bit_generator isa B + @test g.bit_generator.seed == UInt64(42) + + U = cuNumeric.random(g, Float32, (n,)) + @test eltype(U) === Float32 + Uh = _host(U) + @test all(0 .<= Uh .< 1) + @test abs(mean(Uh) - 0.5) < 0.05 + @test abs(var(Uh) - 1 / 12) < 0.02 + + Nrm = cuNumeric.randn(g, Float64, (n,)) + μ, v = _moments(Nrm) + @test abs(μ) < 0.08 + @test abs(v - 1) < 0.12 + end + end +end diff --git a/test/array/slicing.jl b/test/array/slicing.jl index 2e2a539a4..f189c418d 100644 --- a/test/array/slicing.jl +++ b/test/array/slicing.jl @@ -143,3 +143,29 @@ end end end end + +# Issue #211 +@testset "Range assignment" begin + A = NDArray(Float32[1, 2, 3, 4]) + A[2:3] = NDArray(Float32[10, 20]) + @test Array(A) == Float32[1, 10, 20, 4] + A[1:4] = NDArray(Float32[5, 6, 7, 8]) + @test Array(A) == Float32[5, 6, 7, 8] + A[3:2] = NDArray(Float32[]) + @test Array(A) == Float32[5, 6, 7, 8] + A[2:3] = NDArray(Int64[1, 2]) + @test Array(A) == Float32[5, 1, 2, 8] + @test_throws DimensionMismatch (A[2:3] = NDArray(Float32[1, 2, 3])) + @test_throws BoundsError (A[0:1] = NDArray(Float32[1, 2])) + + M = NDArray(Float32[1 2; 3 4; 5 6]) + M[2:3, :] = NDArray(Float32[30 40; 50 60]) + @test Array(M) == Float32[1 2; 30 40; 50 60] + M[1, :] = NDArray(Float32[7, 8]) + M[:, 2] = NDArray(Float32[0, 0, 0]) + M[2:3, 1] = NDArray(Float32[9, 9]) + @test Array(M) == Float32[7 0; 9 0; 9 0] + @test_throws DimensionMismatch (M[2:3, :] = NDArray(Float32[1 2 3; 4 5 6])) + @test_throws DimensionMismatch (M[:, 1] = NDArray(Float32[1, 2])) + @test Array(M) == Float32[7 0; 9 0; 9 0] +end diff --git a/test/array/sort.jl b/test/array/sort.jl new file mode 100644 index 000000000..18ead1138 --- /dev/null +++ b/test/array/sort.jl @@ -0,0 +1,159 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz + * Nader Rahhal +=# + +const SORT_TYPES = (Bool, Base.uniontypes(cuNumeric.SUPPORTED_NUMERIC_TYPES)...) + +# Julia has no isless(::Complex, ::Complex). Use this only as Base.sort's `by` +# so the reference matches cupynumeric's (real, imag) order. +_lex_complex(x) = (real(x), imag(x)) + +function _sort_fixture(::Type{T}) where {T} + T <: Bool && return Bool[1, 0, 1, 0, 0, 1, 1, 0, 1, 0, 0, 1] + T <: Complex && + return T[ + 10 + 3im, + 2 + 4im, + 1 + 5im, + 8 + 9im, + 3 + 1im, + 2 + 4im, + 10 + 3im, + 1, + 5 + 2im, + 0, + 4, + 7 + im, + ] + return T[10, 3, 12, 5, 2, 4, 8, 9, 7, 6, 11, 1] +end + +function _search_haystack(::Type{T}) where {T} + T <: Bool && return Bool[0, 0, 1, 1] + return T[1, 2, 2, 4, 5, 7] +end + +function _search_needles(::Type{T}) where {T} + T <: Bool && return Bool[0, 1] + return T[0, 1, 2, 3, 7, 8] +end + +function _unique_fixture(::Type{T}) where {T} + T <: Bool && return Bool[1, 0, 1, 0, 1] + T <: Complex && return T[1, 2, 2, 3 + im, 3 + im, 1] + return T[1, 2, 2, 3, 4, 4, 4, 5] +end + +@testset "sort 1-d" begin + @testset verbose = true for T in SORT_TYPES + A = _sort_fixture(T) + nda = cuNumeric.NDArray(A) + @test Array(cuNumeric.sort(nda)) == + (T <: Complex ? Base.sort(A; by=_lex_complex) : Base.sort(A)) + @test Array(nda) == A # cuNumeric.sort is not in-place + end +end + +@testset "sort! 1-d" begin + @testset verbose = true for T in SORT_TYPES + A = _sort_fixture(T) + nda = cuNumeric.NDArray(copy(A)) + cuNumeric.sort!(nda) + @test Array(nda) == (T <: Complex ? Base.sort(A; by=_lex_complex) : Base.sort(A)) + end +end + +@testset "sort dims" begin + @testset verbose = true for T in SORT_TYPES + A = reshape(_sort_fixture(T), 3, 4) + nda = cuNumeric.NDArray(A) + @test Array(cuNumeric.sort(nda; dims=1)) == + (T <: Complex ? Base.sort(A; dims=1, by=_lex_complex) : Base.sort(A; dims=1)) + @test Array(cuNumeric.sort(nda; dims=2)) == + (T <: Complex ? Base.sort(A; dims=2, by=_lex_complex) : Base.sort(A; dims=2)) + @test_throws "invalid for a" cuNumeric.sort(nda; dims=3) + @test_throws "invalid for a" cuNumeric.sort(nda; dims=0) + @test_throws UndefKeywordError cuNumeric.sort(nda) # dims required for N>1 + end +end + +@testset "unsupported kwargs and not Base.sort" begin + v = cuNumeric.NDArray(Int32[3, 1, 2]) + @test_throws MethodError cuNumeric.sort(v; alg=Base.QuickSort) + @test_throws MethodError cuNumeric.sort(v; rev=true) + @test_throws MethodError cuNumeric.sort(v; lt=(!)) + # cuNumeric.sort is not Base.sort: Julia arrays are a MethodError + @test_throws MethodError cuNumeric.sort([3, 1, 2]) + @test_throws MethodError cuNumeric.sort!([3, 1, 2]) + @test_throws MethodError cuNumeric.searchsortedfirst([1, 2, 3], 2) + @test_throws MethodError cuNumeric.unique([1, 1, 2]) +end + +@testset "searchsorted" begin + @testset verbose = true for T in SORT_TYPES + A = _search_haystack(T) + nda = cuNumeric.sort(cuNumeric.NDArray(A)) + if T <: Complex + @test_throws "not supported for complex" cuNumeric.searchsortedfirst(nda, zero(T)) + @test_throws "not supported for complex" cuNumeric.searchsortedlast(nda, zero(T)) + @test_throws "not supported for complex" cuNumeric.searchsorted(nda, zero(T)) + continue + end + for x in _search_needles(T) + @test cuNumeric.fetch(cuNumeric.searchsortedfirst(nda, x)) == + Base.searchsortedfirst(A, x) + @test cuNumeric.fetch(cuNumeric.searchsortedlast(nda, x)) == + Base.searchsortedlast(A, x) + @test cuNumeric.searchsorted(nda, x) == Base.searchsorted(A, x) + end + needles = _search_needles(T) + firsts = Array(cuNumeric.searchsortedfirst(nda, cuNumeric.NDArray(needles))) + lasts = Array(cuNumeric.searchsortedlast(nda, cuNumeric.NDArray(needles))) + @test firsts == Base.searchsortedfirst.(Ref(A), needles) + @test lasts == Base.searchsortedlast.(Ref(A), needles) + M = cuNumeric.NDArray(reshape(A, 2, :)) + @test_throws MethodError cuNumeric.searchsortedfirst(M, zero(T)) + end +end + +@testset "Scalar search preserves query precision" begin + for (h, queries) in ((Int64[1, 3, 5], (2.5, -0.5, 5.5)), + (Float32[1, 2, 3], (1.0 + eps(Float64), 2.0 - eps(Float64), 4.0))) + a = NDArray(h) + for q in queries + @test fetch(cuNumeric.searchsortedfirst(a, q)) == Base.searchsortedfirst(h, q) + @test fetch(cuNumeric.searchsortedlast(a, q)) == Base.searchsortedlast(h, q) + @test cuNumeric.searchsorted(a, q) == Base.searchsorted(h, q) + end + end +end + +@testset "unique" begin + @testset verbose = true for T in SORT_TYPES + A = _unique_fixture(T) + out = Array(cuNumeric.unique(cuNumeric.NDArray(A))) + @test Set(out) == Set(Base.unique(A)) + if !(T <: Complex) + @test Base.issorted(out) + end + B = reshape(_sort_fixture(T), 3, 4) + @test Set(Array(cuNumeric.unique(cuNumeric.NDArray(B)))) == Set(Base.unique(vec(B))) + end + @test_throws MethodError cuNumeric.unique(cuNumeric.NDArray(Int32[1, 1]); dims=1) +end diff --git a/test/array/storage_semantics.jl b/test/array/storage_semantics.jl new file mode 100644 index 000000000..50443222e --- /dev/null +++ b/test/array/storage_semantics.jl @@ -0,0 +1,95 @@ +using Test + +@testset "Shape queries across ranks and shared handles" begin + for dims in ((), (0,), (5,), (2, 3), (2, 1, 3)) + a = cuNumeric.zeros(Float32, dims) + @test size(a) === dims + @test axes(a) == map(Base.OneTo, dims) + @test length(a) == prod(dims) + cuNumeric.destroy!(a) + end + a = cuNumeric.zeros(Float32, 4, 5) + sliced = view(a, 2:3, 1:4) + reshaped = cuNumeric.reshape(a, (2, 2, 5)) + @test size(sliced) === (2, 4) + @test size(reshaped) === (2, 2, 5) + @test size(a) === (4, 5) + foreach(cuNumeric.destroy!, (sliced, reshaped, a)) +end + +@testset "Full-colon views share storage and own their handles" begin + for shape in ((4,), (2, 3), (2, 2, 3)) + a = cuNumeric.zeros(Float32, shape) + inds = ntuple(_ -> Colon(), length(shape)) + v = view(a, inds...) + @test v !== a + @test size(v) == shape + fill!(v, 3f0) + @test Array(a) == fill(3f0, shape) + fill!(a, 4f0) + @test Array(v) == fill(4f0, shape) + cuNumeric.destroy!(v) + a[inds...] .= 5f0 + @test Array(a) == fill(5f0, shape) + copied = a[inds...] + fill!(copied, 9f0) + @test Array(a) == fill(5f0, shape) + end + a = cuNumeric.zeros(Float32, 0) + @test size(view(a, :)) == (0,) + + a = NDArray(2f0) + v = view(a) + @test v !== a + @test size(v) == () + fill!(v, 3f0) + @test fetch(a) == 3f0 + cuNumeric.destroy!(v) + @test fetch(a) == 3f0 + + @accelerate function _full_view_update!(a) + a[:, :] .= 7f0 + a + end + a = cuNumeric.zeros(Float32, 2, 3) + @test _full_view_update!(a) === a + @test Array(a) == fill(7f0, 2, 3) +end + +@testset "In-place broadcast preserves existing views" begin + a = NDArray(Float32[1, 2, 3, 4]) + v = view(a, 1:2) + a .+= 1f0 + @test Array(v) == Float32[2, 3] + a .= a .* 2f0 .+ 1f0 + @test Array(v) == Float32[5, 7] + + reshaped = cuNumeric.reshape(a, (2, 2)) + reshaped .+= 1f0 + @test Array(a) == Float32[6, 8, 10, 12] + @test Array(v) == Float32[6, 8] + + # Changing result type must copy back into the existing destination store. + ints = NDArray(Int32[1, 2, 3, 4]) + a .= ints .+ Int32(1) + @test Array(v) == Float32[2, 3] + + # Shifted inputs need a temporary, including identity broadcasts. + a = NDArray(Float32[1, 2, 3, 4]) + a[2:4] .= a[1:3] + @test Array(a) == Float32[1, 1, 2, 3] + a[2:4] .= a[1:3] .+ 10f0 + @test Array(a) == Float32[1, 11, 11, 12] +end + +@testset "Reject unsupported multidimensional linear indexing" begin + a = NDArray(Float32[1 3 5; 2 4 6]) + @test_throws ArgumentError a[1:2] + @test_throws ArgumentError a[:] + @test_throws ArgumentError view(a, :) + @test_throws ArgumentError (a[1:2] = NDArray(Float32[9, 9])) + @test_throws ArgumentError (a[:] = cuNumeric.ones(Float32, 2, 3)) + @test_throws ArgumentError (a[:] .= 9f0) + @test Array(a[1:2, 2:3]) == Float32[3 5; 4 6] + @test Array(a[:, 2]) == reshape(Float32[3, 4], 2, 1) +end diff --git a/test/array/struct_guards.jl b/test/array/struct_guards.jl new file mode 100644 index 000000000..b9e57bef8 --- /dev/null +++ b/test/array/struct_guards.jl @@ -0,0 +1,78 @@ +using StructArrays + +# Struct element storage needs the fused GPU broadcast kernel. Without a GPU or +# with fusion disabled, every operation that touches element data must raise an +# ArgumentError instead of reaching the unfused path or a missing GPU variant. + +struct GuardPair + a::Float32 + b::Int32 +end + +_guard_pair(x) = GuardPair(Float32(x + 1), Int32(2x)) +_guard_pair_b(p::GuardPair) = p.b + +if cuNumeric._struct_kernel_available() + @testset "Struct guards (skipped: struct kernel available)" begin + @test_skip false + end +else + @testset "Struct storage without the fused kernel" begin + template = cuNumeric.zeros(Int64, 6) + value = GuardPair(3.0f0, Int32(4)) + arr = similar(template, GuardPair, (6,)) + other = similar(template, GuardPair, (6,)) + input = NDArray(collect(0:5)) + empty_arr = similar(template, GuardPair, (0,)) + try + # Allocation, fills, scalar writes and views need no kernel. + @test arr isa NDArray{GuardPair,1} + @test fill!(arr, value) === arr + cuNumeric.allowscalar() do + arr[2] = GuardPair(9.0f0, Int32(9)) + end + view = arr[2:4] + @test size(view) == (3,) + cuNumeric.destroy!(view) + + # Empty arrays have no elements to move. + @test isempty(Array(empty_arr)) + @test isempty(Array(NDArray(GuardPair[]))) + + @test_throws ArgumentError NDArray([value, value]) + @test_throws ArgumentError Array(arr) + @test_throws ArgumentError cuNumeric.allowscalar(() -> arr[1]) + @test_throws ArgumentError copy(arr) + @test_throws ArgumentError copyto!(other, arr) + @test_throws ArgumentError cuNumeric.reshape(arr, 2, 3) + @test_throws ArgumentError (arr == other) + @test_throws ArgumentError (arr .= _guard_pair.(input)) + @test_throws ArgumentError _guard_pair_b.(arr) + + err = try + Array(arr) + catch e + e + end + @test occursin("requires GPU broadcast fusion", sprint(showerror, err)) + finally + foreach(cuNumeric.destroy!, (template, arr, other, input, empty_arr)) + end + end + + @testset "StructArray broadcast without the fused kernel" begin + previous_experimental = get(task_local_storage(), :Experimental, false) + input = NDArray(collect(0:2)) + dest = StructArray{GuardPair}((a=cuNumeric.zeros(Float32, 3), b=cuNumeric.zeros(Int32, 3))) + try + cuNumeric.Experimental(true) + @test_throws ArgumentError (dest .= _guard_pair.(input)) + @test all(iszero, Array(dest.a)) + # Field storage is ordinary NDArrays. + @test Array(dest.a .+ 1.0f0) == ones(Float32, 3) + finally + cuNumeric.Experimental(previous_experimental) + foreach(cuNumeric.destroy!, (input, dest.a, dest.b)) + end + end +end diff --git a/test/array/tensoroperations.jl b/test/array/tensoroperations.jl new file mode 100644 index 000000000..ab75475e5 --- /dev/null +++ b/test/array/tensoroperations.jl @@ -0,0 +1,263 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): Ethan Meitz +=# + +using TensorOperations +using TensorOperations: TensorOperations as TO + +@testset "TensorOperations extension" begin + @test Base.get_extension(cuNumeric, :cuNumericTensorOperationsExt) !== nothing + + @testset "@tensor permutations and accumulation" begin + hostA = reshape(collect(Float64, 1:24), 2, 3, 4) + hostB = reshape(collect(Float64, 1:24), 3, 4, 2) + A = NDArray(hostA) + B = NDArray(hostB) + + @tensor permuted[k, i, j] := A[i, j, k] + @test permuted isa NDArray + @test Array(permuted) == permutedims(hostA, (3, 1, 2)) + + @tensor combined[k, i, j] := 2 * A[i, j, k] - 3 * B[j, k, i] + ref = 2 .* permutedims(hostA, (3, 1, 2)) .- + 3 .* permutedims(hostB, (2, 3, 1)) + @test Array(combined) == ref + + accumulated = cuNumeric.zeros(Float64, 4, 2, 3) + @tensor accumulated[k, i, j] = A[i, j, k] + @tensor accumulated[k, i, j] += 2 * B[j, k, i] + @tensor accumulated[k, i, j] -= 3 * A[i, j, k] + @test Array(accumulated) == + -2 .* permutedims(hostA, (3, 1, 2)) .+ + 2 .* permutedims(hostB, (2, 3, 1)) + end + + @testset "@tensor traces and output permutations" begin + host = reshape(collect(Float64, 1:(2 * 3 * 4 * 3 * 5 * 4)), 2, 3, 4, 3, 5, 4) + A = NDArray(host) + + @tensor traced[d, a] := A[a, b, c, b, d, c] + @tensor ref[d, a] := host[a, b, c, b, d, c] + @test traced isa NDArray + @test Array(traced) == ref + + destination = cuNumeric.ones(Float64, 5, 2) + @tensor destination[d, a] += 0.5 * A[a, b, c, b, d, c] + @test Array(destination) == ones(5, 2) .+ 0.5 .* ref + end + + @testset "@tensor contractions and outer products" begin + hostA = reshape(collect(Float64, 1:(2 * 3 * 4 * 5)), 2, 3, 4, 5) + hostB = reshape(collect(Float64, 1:(4 * 6 * 3 * 7)), 4, 6, 3, 7) + A = NDArray(hostA) + B = NDArray(hostB) + + @tensor contracted[a, g, d, f] := A[a, b, c, d] * B[c, f, b, g] + @tensor ref[a, g, d, f] := hostA[a, b, c, d] * hostB[c, f, b, g] + @test contracted isa NDArray + @test Array(contracted) == ref + + hostX = reshape( + ComplexF64.(1:6) .+ im .* ComplexF64.(6:-1:1), 2, 3 + ) + hostY = reshape( + ComplexF64.(7:26) .- im .* ComplexF64.(1:20), 4, 5 + ) + X = NDArray(hostX) + Y = NDArray(hostY) + @tensor outer[a, c, b, d] := conj(X[a, b]) * conj(Y[c, d]) + @test Array(outer) == reshape(conj(hostX), 2, 1, 3, 1) .* + reshape(conj(hostY), 1, 4, 1, 5) + end + + @testset "@tensor block and tensor network" begin + hostA = reshape( + collect(Float64, 1:(2 * 3 * 4 * 5 * 4 * 6)), 2, 3, 4, 5, 4, 6 + ) + hostB = reshape(collect(Float64, 1:(6 * 7 * 3)), 6, 7, 3) + hostC = reshape(collect(Float64, 1:(5 * 2 * 7)), 5, 2, 7) + A = NDArray(hostA) + B = NDArray(hostB) + C = NDArray(hostC) + D = cuNumeric.zeros(Float64, 2, 7, 5) + α = 0.25 + + @tensor begin + D[a, b, c] = A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + E[a, b, c] := A[a, e, f, c, f, g] * B[g, b, e] + α * C[c, a, b] + end + @tensor ref[a, b, c] := hostA[a, e, f, c, f, g] * hostB[g, b, e] + + α * hostC[c, a, b] + @test D isa NDArray + @test E isa NDArray + @test Array(D) == ref + @test Array(E) == ref + + hostA1 = reshape(collect(Float64, 1:24), 2, 3, 4) + hostA2 = reshape(collect(Float64, 1:120), 4, 5, 6) + hostL = reshape(collect(Float64, 1:4), 2, 2) + hostR = reshape(collect(Float64, 1:36), 6, 6) + hostH = reshape(collect(Float64, 1:225), 3, 5, 3, 5) + A1, A2 = NDArray(hostA1), NDArray(hostA2) + L, R, H = NDArray(hostL), NDArray(hostR), NDArray(hostH) + + @tensor network[a, s1, s2, c] := + L[a, ap] * A1[ap, t1, b] * + A2[b, t2, cp] * R[cp, c] * + H[s1, s2, t1, t2] + @tensor network_ref[a, s1, s2, c] := + hostL[a, ap] * hostA1[ap, t1, b] * + hostA2[b, t2, cp] * hostR[cp, c] * + hostH[s1, s2, t1, t2] + @test network isa NDArray + @test Array(network) == network_ref + end + + @testset "tensoradd!" begin + host = ComplexF64.(reshape(1:6, 2, 3)) .+ im .* reshape(7:12, 2, 3) + A = NDArray(host) + C = NDArray(fill(ComplexF64(2 - im), 3, 2)) + + TO.tensoradd!(C, A, ((2, 1), ()), true, 2, 3) + @test Array(C) ≈ 3 .* fill(ComplexF64(2 - im), 3, 2) .+ + 2 .* permutedims(conj(host)) + + @tensor allocated[j, i] := conj(A[i, j]) + @test allocated isa NDArray + @test Array(allocated) == permutedims(conj(host)) + + overwrite = NDArray(fill(ComplexF64(NaN), 3, 2)) + TO.tensoradd!(overwrite, A, ((2, 1), ()), false, 1, 0) + @test Array(overwrite) == permutedims(host) + end + + @testset "tensortrace!" begin + host = reshape(collect(Float64, 1:(2 * 3 * 3)), 2, 3, 3) + A = NDArray(host) + @tensor traced[i] := A[i, j, j] + ref = [sum(host[i, j, j] for j in axes(host, 2)) for i in axes(host, 1)] + @test traced isa NDArray + @test Array(traced) == ref + + host5 = reshape(collect(Float64, 1:(2 * 3 * 3 * 4 * 4)), 2, 3, 3, 4, 4) + A5 = NDArray(host5) + @tensor traced2[i] := A5[i, j, j, k, k] + ref2 = [ + sum(host5[i, j, j, k, k] for j in axes(host5, 2), k in axes(host5, 4)) + for i in axes(host5, 1) + ] + @test Array(traced2) == ref2 + + matrix = reshape(ComplexF64.(1:9) .+ im .* ComplexF64.(9:-1:1), 3, 3) + M = NDArray(matrix) + scalar_trace = @tensor(t = M[i, i]) + @test scalar_trace isa NDArray + @test ndims(scalar_trace) == 0 + @test only(Array(scalar_trace)) == sum(matrix[i, i] for i in axes(matrix, 1)) + + conjugated_trace = TO.tensortrace(M, ((), ()), ((1,), (2,)), true) + ts = TO.tensorscalar(conjugated_trace) + @test ts isa NDArray + @test ndims(ts) == 0 + @test only(Array(ts)) == sum(conj(matrix[i, i]) for i in axes(matrix, 1)) + TO.tensorfree!(conjugated_trace) + end + + @testset "tensorcontract!" begin + hostA = ComplexF64.(reshape(1:12, 3, 4)) .+ im .* reshape(13:24, 3, 4) + hostB = ComplexF64.(reshape(1:20, 4, 5)) .- im .* reshape(21:40, 4, 5) + A = NDArray(hostA) + B = NDArray(hostB) + + @tensor C[i, j] := conj(A[i, k]) * B[k, j] + @test C isa NDArray + @test Array(C) ≈ conj(hostA) * hostB + + seed = fill(ComplexF64(1 + 2im), 3, 5) + accumulated = NDArray(copy(seed)) + TO.tensorcontract!( + accumulated, + A, + ((1,), (2,)), + true, + B, + ((1,), (2,)), + false, + ((1, 2), ()), + 2, + 3, + ) + @test Array(accumulated) ≈ 3 .* seed .+ 2 .* (conj(hostA) * hostB) + + uhost = ComplexF64.(1:4) .+ im .* ComplexF64.(4:-1:1) + vhost = ComplexF64.(5:8) .- im .* ComplexF64.(1:4) + u = NDArray(uhost) + v = NDArray(vhost) + scalar_product = @tensor(s = u[k] * v[k]) + @test scalar_product isa NDArray + @test ndims(scalar_product) == 0 + @test only(Array(scalar_product)) ≈ sum(uhost .* vhost) + end + + @testset "Device scalar scale factors" begin + hostC = reshape(collect(Float64, 1:6), 2, 3) + C = NDArray(hostC) + reduced = sum(C) + @test reduced isa CNScalar + @test ndims(reduced) == 0 + allowautofetch(false) do + for α in (NDArray(sum(hostC)), reduced) + @tensor scaled[i, j] := α * C[i, j] + @test Array(scaled) ≈ sum(hostC) .* hostC + + dest = NDArray(fill(2.0, 2, 3)) + @tensor dest[i, j] += α * C[i, j] + @test Array(dest) ≈ fill(2.0, 2, 3) .+ sum(hostC) .* hostC + end + end + + @tensor s = C[i, j] * C[i, j] + @test s isa NDArray + @test ndims(s) == 0 + @tensor scaled2[i, j] := s * C[i, j] + @test Array(scaled2) ≈ only(Array(s)) .* hostC + end + + @testset "temporary destruction" begin + hostA = reshape(collect(Float64, 1:12), 3, 4) + hostB = reshape(collect(Float64, 1:20), 4, 5) + hostD = reshape(collect(Float64, 1:10), 5, 2) + A = NDArray(hostA) + B = NDArray(hostB) + D = NDArray(hostD) + GC.gc() + cuNumeric.drain_pending_frees!() + current_bytes = + cuNumeric._has_gpu_target() ? + cuNumeric.current_device_bytes : + cuNumeric.current_host_bytes + baseline = current_bytes[] + + @tensor C[i, l] := A[i, j] * B[j, k] * D[k, l] + @test current_bytes[] == baseline + C.nbytes + @test Array(C) ≈ hostA * hostB * hostD + + TO.tensorfree!(C) + @test C.ptr == C_NULL + @test current_bytes[] == baseline + end +end diff --git a/test/array/unary/tests.jl b/test/array/unary/tests.jl index 558a2c877..44dfd9d29 100644 --- a/test/array/unary/tests.jl +++ b/test/array/unary/tests.jl @@ -47,14 +47,19 @@ function test_unary_operation(func, julia_arr, cunumeric_arr, T) end end -skip_on_integer = (Base.acosh, Base.atanh, Base.atan, Base.acos, Base.asin) -skip_on_bool = (Base.:(-), skip_on_integer...) +skip_on_integer = ( + Base.acosh, Base.atanh, Base.atan, Base.acos, Base.asin, + Base.ceil, Base.floor, Base.trunc, Base.round, Base.signbit, +) +skip_on_bool = skip_on_integer skip_on_complex = ( Base.tanh, Base.deg2rad, Base.rad2deg, Base.sign, Base.cbrt, Base.exp2, Base.expm1, Base.log10, Base.log1p, Base.log2, Base.acos, Base.asin, Base.atan, Base.acosh, Base.asinh, Base.atanh, + Base.ceil, Base.floor, Base.trunc, Base.signbit, Base.:(~), ) +skip_on_float = (Base.:(~),) function test_unary_function_set(func_dict, T, N) default_generator = (T == Bool) ? :uniform : :unit_interval @@ -73,6 +78,10 @@ function test_unary_function_set(func_dict, T, N) continue end + if func in skip_on_float && (T <: AbstractFloat) + continue + end + domain_type = get(SPECIAL_DOMAINS, func, default_generator) # :uniform is the only generator capable of generating bits @@ -106,10 +115,132 @@ function test_unary_reduction_dims( end end - # we are testing a multi axis reduction. This will throw a runtime error. - # https://github.com/nv-legate/cupynumeric/blob/main/src/cupynumeric/ndarray.cc#L1132 + # Multi-axis is sequential single-axis keepdims reductions (C++ allows one axis). + # Integer `prod` of 10×10 random values overflows Int64; association then + # disagrees with Julia, so skip that case (1D/single-axis prod is still tested). + if N >= 2 && !(func === Base.prod && T <: Base.BitInteger) + julia_res = func(julia_arr; dims=(1, 2)) + cunumeric_res = func(cunumeric_arr; dims=(1, 2)) + n = size(julia_arr, 1) * size(julia_arr, 2) + scale = maximum(abs, julia_arr) + allowscalar() do + @test safe_compare( + julia_res, + cunumeric_res, + reduction_atol(T, n, scale), + reduction_rtol(T, n), + ) + end + end + end +end + +function test_mean_var_std( + julia_arr::AbstractArray{T,N}, cunumeric_arr::NDArray{T,N} +) where {T,N} + n = length(julia_arr) + scale = maximum(abs, julia_arr) + atolv = reduction_atol(T, n, scale) + rtolv = reduction_rtol(T, n) + allowpromotion(true) do + allowscalar() do + @test isapprox(mean(julia_arr), fetch(mean(cunumeric_arr)); atol=atolv, rtol=rtolv) + if T <: Real + @test isapprox(var(julia_arr), fetch(var(cunumeric_arr)); atol=atolv, rtol=rtolv) + @test isapprox(std(julia_arr), fetch(std(cunumeric_arr)); atol=atolv, rtol=rtolv) + end + end + for d in 1:N + nd = size(julia_arr, d) + atol_d = reduction_atol(T, nd, scale) + rtol_d = reduction_rtol(T, nd) + allowscalar() do + @test safe_compare( + mean(julia_arr; dims=d), mean(cunumeric_arr; dims=d), atol_d, rtol_d + ) + if T <: Real + @test safe_compare( + var(julia_arr; dims=d), var(cunumeric_arr; dims=d), atol_d, rtol_d + ) + @test safe_compare( + std(julia_arr; dims=d), std(cunumeric_arr; dims=d), atol_d, rtol_d + ) + end + end + end if N >= 2 - @test_throws Exception func(cunumeric_arr, dims=(1, 2)) + allowscalar() do + @test safe_compare( + mean(julia_arr; dims=(1, 2)), + mean(cunumeric_arr; dims=(1, 2)), + atolv, + rtolv, + ) + if T <: Real + @test safe_compare( + var(julia_arr; dims=(1, 2)), + var(cunumeric_arr; dims=(1, 2)), + atolv, + rtolv, + ) + @test safe_compare( + std(julia_arr; dims=(1, 2)), + std(cunumeric_arr; dims=(1, 2)), + atolv, + rtolv, + ) + end + end + end + end +end + +# count API is commented out for now (see src/ndarray/unary.jl). +# function test_count(julia_arr::AbstractArray{T,N}, cunumeric_arr::NDArray{T,N}) where {T,N} +# allowpromotion(true) do +# allowscalar() do +# @test count(!iszero, julia_arr) == fetch(count(!iszero, cunumeric_arr)) +# if T == Bool +# @test count(julia_arr) == fetch(count(cunumeric_arr)) +# end +# end +# for d in 1:N +# allowscalar() do +# @test safe_compare( +# count(!iszero, julia_arr; dims=d), +# count(!iszero, cunumeric_arr; dims=d), +# 0, +# 0, +# ) +# end +# end +# if N >= 2 +# allowscalar() do +# @test safe_compare( +# count(!iszero, julia_arr; dims=(1, 2)), +# count(!iszero, cunumeric_arr; dims=(1, 2)), +# 0, +# 0, +# ) +# end +# end +# end +# end + +function test_argmax_argmin() + # 1-d only. Fixtures, not random: ties must pick the first extremum. + v = Int32[1, 5, 3, 5, 2] + allowpromotion(true) do + allowscalar() do + vn = NDArray(v) + @test fetch(argmax(vn)) == 2 + @test fetch(argmin(vn)) == 1 + @test fetch(argmax(vn)) == argmax(v) + @test fetch(argmin(vn)) == argmin(v) + + c = NDArray(ComplexF32[1, 2]) + @test_throws ArgumentError argmax(c) + @test_throws ArgumentError argmin(c) end end end @@ -126,6 +257,13 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) allowpromotion(T == Bool) do return test_unary_function_set(cuNumeric.unary_op_map_no_args, T, N) end + + if T <: AbstractFloat + @testset "round is 1-arg only" begin + a = cuNumeric.ones(T, 4) + @test_throws ArgumentError round.(a; digits=1) + end + end # Special cases for unary ops that dont use . syntax @testset "- (Negation)" begin arr = my_rand(T, N) @@ -195,14 +333,52 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) end end + # All-positive: |sum| ≈ Σ|xᵢ|, so the Higham input-magnitude floor is + # unnecessary. Keep O(n) rtol for association; this still flags a + # wrong kernel that cancellation tols would swallow. + @testset "sum no cancellation" begin + @testset for T in types + T <: AbstractFloat || continue + x = my_rand(T, N; L=one(T), R=T(1000)) + ndx = @allowscalar NDArray(x) + allowscalar() do + @test isapprox( + sum(x), + fetch(sum(ndx)); + atol=atol(T) * N, + rtol=reduction_rtol(T, N), + ) + end + x2 = my_rand(T, isqrt(N), isqrt(N); L=one(T), R=T(1000)) + nd2 = @allowscalar NDArray(x2) + n1 = size(x2, 1) + allowpromotion(true) do + allowscalar() do + @test safe_compare( + sum(x2; dims=1), + sum(nd2; dims=1), + atol(T) * n1, + reduction_rtol(T, n1), + ) + end + end + end + end + if include_bool_reductions # Test things that only work on Booleans julia_bools = rand(Bool, N) + julia_bools_2D = rand(Bool, isqrt(N), isqrt(N)) allowscalar() do cunumeric_bools = NDArray(julia_bools) @test any(julia_bools) == any(cunumeric_bools)[] @test all(julia_bools) == all(cunumeric_bools)[] end + allowpromotion(true) do + cn2 = @allowscalar NDArray(julia_bools_2D) + test_unary_reduction_dims(any, julia_bools_2D, cn2) + test_unary_reduction_dims(all, julia_bools_2D, cn2) + end end end @@ -227,13 +403,22 @@ function run_unary_tests(types; include_bool_reductions::Bool=false) end ## TODO Int8 min/max along an axis is broken on GPU - if cuNumeric.HAS_CUDA && T == Int8 && (func == Base.minimum || func == Base.maximum) + if cuNumeric._has_gpu_target() && T == Int8 && (func == Base.minimum || func == Base.maximum) continue end test_unary_reduction_dims(func, julia_arr_1D, cunumeric_arr_1D) test_unary_reduction_dims(func, julia_arr_2D, cunumeric_arr_2D) end + + @testset "mean/var/std" begin + test_mean_var_std(julia_arr_1D, cunumeric_arr_1D) + test_mean_var_std(julia_arr_2D, cunumeric_arr_2D) + end + end + + @testset "argmax/argmin" begin + test_argmax_argmin() end end end diff --git a/test/array/unfused_destinations.jl b/test/array/unfused_destinations.jl new file mode 100644 index 000000000..acdbf434d --- /dev/null +++ b/test/array/unfused_destinations.jl @@ -0,0 +1,67 @@ +using Test + +@testset "Native n-ary broadcasts reuse the destination" begin + a = NDArray(Float32[1, 2, 3, 4]) + b = NDArray(Float32[2, 3, 4, 5]) + c = NDArray(Float32[3, 4, 5, 6]) + dest = cuNumeric.zeros(Float32, 4) + for f in (+, *) + for args in ((a, b, c), (a, b, c, a), (2f0, 3f0, a), + (Base.broadcasted(+, a, 1f0), b, c)) + bc = Base.Broadcast.instantiate(Base.broadcasted(f, args...)) + host_args = map(args) do arg + arg isa NDArray && return Array(arg) + arg isa Base.Broadcast.Broadcasted && return Array(a) .+ 1f0 + return arg + end + expected = f.(host_args...) + # Exercise the native route regardless of the fusion preference. + result = cuNumeric.unravel_broadcast_tree(bc, dest) + @test result === dest + @test Array(dest) == expected + @test Array(a) == Float32[1, 2, 3, 4] + @test Array(b) == Float32[2, 3, 4, 5] + @test Array(c) == Float32[3, 4, 5, 6] + + allocated = cuNumeric.unravel_broadcast_tree(bc) + @test allocated !== dest + @test Array(allocated) == expected + cuNumeric.destroy!(allocated) + end + # The destination can be read both before and during the final operation. + fill!(dest, 2f0) + bc = Base.Broadcast.instantiate(Base.broadcasted(f, dest, a, dest)) + @test cuNumeric._unfused_into!(dest, bc) === dest + @test Array(dest) == f.(2f0, Float32[1, 2, 3, 4], 2f0) + end + foreach(cuNumeric.destroy!, (a, b, c, dest)) +end + +@testset "Native n-ary broadcasts preserve overlap and conversion guards" begin + for f in (+, *) + parent = NDArray(Float32[1, 2, 3, 4, 5]) + source = view(parent, 1:4) + dest = view(parent, 2:5) + a = cuNumeric.fill(2f0, 4) + b = cuNumeric.fill(3f0, 4) + bc = Base.Broadcast.instantiate(Base.broadcasted(f, a, b, source)) + result = cuNumeric.unravel_broadcast_tree(bc, dest) + @test result !== dest + @test Array(parent) == Float32[1, 2, 3, 4, 5] + @test cuNumeric._copyto_unfused!(dest, result) === dest + @test Array(parent) == vcat(1f0, f.(2f0, 3f0, Float32[1, 2, 3, 4])) + foreach(cuNumeric.destroy!, (source, dest, parent, a, b)) + end + + ints = NDArray(Int32[1, 2, 3, 4]) + dest = cuNumeric.zeros(Float32, 4) + alias = view(dest, 1:2) + bc = Base.Broadcast.instantiate(Base.broadcasted(+, ints, Int32(2), Int32(3))) + result = cuNumeric.unravel_broadcast_tree(bc, dest) + @test result !== dest + @test eltype(result) === Int32 + @test cuNumeric._copyto_unfused!(dest, result) === dest + @test Array(dest) == Float32[6, 7, 8, 9] + @test Array(alias) == Float32[6, 7] + foreach(cuNumeric.destroy!, (alias, dest, ints)) +end diff --git a/test/array/vector_linalg.jl b/test/array/vector_linalg.jl new file mode 100644 index 000000000..1e422d11b --- /dev/null +++ b/test/array/vector_linalg.jl @@ -0,0 +1,194 @@ +using Test, LinearAlgebra, Random, cuNumeric + +# Host extraction belongs to validation, not the NDArray implementation. +la_value(x::cuNumeric.CNScalar) = fetch(x) + +@testset "LinearAlgebra vector interface" begin + cuNumeric.allowscalar(false) + begin + Random.seed!(912) + for T in Base.uniontypes( + Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_COMPLEX_TYPES} + ) + @testset "$T" begin + R = real(T) + ah, xh, yh = randn(T, 7, 5), randn(T, 5), randn(T, 7) + a, x, y = cuNumeric.NDArray(ah), cuNumeric.NDArray(xh), cuNumeric.NDArray(yh) + @test mul!(y, a, x) === y + @test Array(y) ≈ ah * xh + @test Array(a * x) ≈ ah * xh + copyto!(y, cuNumeric.NDArray(yh)) + @test mul!(y, a, x, T(2), T(-3)) === y + @test Array(y) ≈ 2ah * xh - 3yh + fill!(y, T(NaN)) + mul!(y, a, x, T(2), zero(T)) + @test Array(y) ≈ 2ah * xh + bad = cuNumeric.NDArray(fill(T(NaN), 7, 5)) + mul!(y, bad, x, zero(T), zero(T)) + @test all(iszero, Array(y)) + @test_throws DimensionMismatch mul!(similar(x), a, x) + square = cuNumeric.NDArray(randn(T, 5, 5)) + @test_throws ArgumentError mul!(x, square, x) + bh = randn(T, 5, 3) + ch = randn(T, 7, 3) + c = cuNumeric.NDArray(ch) + mul!(c, a, cuNumeric.NDArray(bh), T(2), T(3)) + @test Array(c) ≈ 2ah * bh + 3ch + + zh = randn(T, 5) + z = cuNumeric.NDArray(zh) + @test dot(x, z) isa (T <: Real ? CNReal{T} : CNComplex{T}) + @test la_value(dot(x, z)) ≈ dot(xh, zh) + @test_throws DimensionMismatch dot(x, y) + empty = cuNumeric.zeros(T, 0) + @test la_value(dot(empty, empty)) === zero(T) + @test axpy!(T(2), x, z) === z + @test Array(z) ≈ 2xh + zh + @test axpby!(T(-2), x, T(3), z) === z + @test Array(z) ≈ -2xh + 3(2xh + zh) + @test axpy!(one(T), x, x) === x + @test Array(x) ≈ 2xh + @test axpby!(T(2), x, T(3), x) === x + @test Array(x) ≈ 10xh + @test rmul!(x, T(2)) === x + @test lmul!(T(3), x) === x + @test Array(x) ≈ 60xh + @test_throws DimensionMismatch axpy!(one(T), x, y) + @test_throws DimensionMismatch axpby!(one(T), x, one(T), y) + d = Diagonal(cuNumeric.NDArray(T[1, 2, 3, 4, 5])) + ldiv!(z, d, x) + @test Array(z) ≈ 60xh ./ T[1, 2, 3, 4, 5] + + # Vector views must stay on the backend for updates and products. + storage = cuNumeric.zeros(T, 9) + dest = view(storage, 2:8) + mul!(dest, a, cuNumeric.NDArray(xh)) + @test Array(storage)[2:8] ≈ ah * xh + end + end + empty_a = cuNumeric.zeros(Float64, 3, 0) + out = cuNumeric.ones(Float64, 3) + mul!(out, empty_a, cuNumeric.zeros(Float64, 0)) + @test all(iszero, Array(out)) + cuNumeric.allowpromotion() do + a32 = cuNumeric.NDArray(Float32[2 1; 0 3]) + x64 = cuNumeric.NDArray([2.0, 4.0]) + y64 = similar(x64) + mul!(y64, a32, x64) + @test Array(y64) ≈ [8.0, 12.0] + @test la_value(dot(cuNumeric.NDArray(Float32[1, 2]), x64)) ≈ 10.0 + @test_throws ArgumentError mul!(cuNumeric.zeros(Float32, 2), a32, x64) + end + end +end + +@testset "LinearAlgebra promotion across numeric types" begin + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + types = Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + for T in types + @testset "updates $T" begin + h = ones(T, 2) + y = cuNumeric.ones(T,2) + z = cuNumeric.zeros(T,2) + @test axpy!(one(T),z,y) === y + @test Array(y) == h + @test axpby!(one(T),z,one(T),y) === y + @test Array(y) == h + @test lmul!(one(T),y) === y + @test rmul!(y,one(T)) === y + @test Array(y) == h + end + end + for T in types, S in types + @testset "$T / $S" begin + a = cuNumeric.ones(T,2,2) + b = cuNumeric.ones(S,2,2) + x, z = cuNumeric.ones(S,2), cuNumeric.ones(T,2) + expected_dot = dot(ones(T,2),ones(S,2)) + @test la_value(dot(z,x)) == expected_dot + # Only matrix kernels reject integer-integer inputs. + if !(T <: Integer && S <: Integer) + R = promote_type(T,S) + @test Array(a*x) ≈ fill(R(2),2) + y = cuNumeric.zeros(R,2) + @test mul!(y,a,x) === y + @test Array(y) ≈ fill(R(2),2) + mul!(y,a,x,R(2),R(3)) + @test Array(y) ≈ fill(R(10),2) + c = cuNumeric.zeros(R,2,2) + mul!(c,a,b,R(2),R(0)) + @test Array(c) ≈ fill(R(4),2,2) + end + end + end + end +end + +@testset "Integer-integer matrix multiplication errors" begin + cuNumeric.allowscalar(false) + ints = Base.uniontypes(Union{Bool,cuNumeric.SUPPORTED_INT_TYPES}) + for T in ints, S in ints + A, x = cuNumeric.ones(T, 2, 2), cuNumeric.ones(S, 2) + B = cuNumeric.ones(S, 2, 2) + y, C = cuNumeric.zeros(Float64, 2), cuNumeric.zeros(Float64, 2, 2) + for call in (() -> A*x, () -> mul!(y,A,x), () -> mul!(y,A,x,1,0), + () -> mul!(C,A,B,1,0), () -> mul!(y,A,x,0,0)) + err = try + call() + nothing + catch e + e + end + @test err isa ArgumentError + if err isa ArgumentError + @test occursin("integer-integer", sprint(showerror,err)) + @test occursin("Convert an operand", sprint(showerror,err)) + end + end + end + A = cuNumeric.ones(Float64,2,2) + x = cuNumeric.ones(Int32,2) + @test_throws ArgumentError mul!(cuNumeric.zeros(Int32,2), A, x) + @test_throws DimensionMismatch A * cuNumeric.ones(Float64,3) + cuNumeric.allowpromotion(false) do + @test_throws "Implicit promotion" A * cuNumeric.ones(Int8,2) + end +end + +@testset "0D coefficient arithmetic" begin + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + types = Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + pairs = [(T, T) for T in types] + append!(pairs, [(Float32, Float64), (Int32, Float32), (Float64, Int8), + (ComplexF32, Float32), (ComplexF64, Int64), (Bool, Int8), (Int8, UInt8)]) + for (T, S) in pairs + a, b = NDArray(one(T)), NDArray(one(S)) + for op in (+, -, *, /, ^) + for (x, y) in ((a, b), (a, one(S)), (one(T), b)) + result = @inferred op(x, y) + expected = op(one(T), one(S)) + @test result isa NDArray{typeof(expected),0} + @test only(result) ≈ expected + end + end + end + for T in Base.uniontypes(Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_COMPLEX_TYPES}) + a, b = NDArray(T(6)), NDArray(T(2)) + @test only(a-b) ≈ T(4) + @test only(a/b) ≈ T(3) + @test only(T(12)/a) ≈ T(2) + @test only(T(12)-a) ≈ T(6) + @test only(a^2) ≈ T(36) + exponent = 3 + @test only(a^exponent) ≈ T(216) + @test only(a^b) ≈ T(36) + @test only(T(2)^b) ≈ T(4) + @test only(a^(-1)) ≈ inv(T(6)) + end + end + # Higher-dimensional products still mean matrix multiplication. + A = NDArray([1.0 2.0; 3.0 4.0]) + @test Array(A*A) ≈ [7.0 10.0; 15.0 22.0] +end diff --git a/test/build_cxxwrap.jl b/test/build_cxxwrap.jl new file mode 100644 index 000000000..38866a5d6 --- /dev/null +++ b/test/build_cxxwrap.jl @@ -0,0 +1,31 @@ +# Standalone build regression test; no CUDA, Legate, or network required. +# Run: julia --startup-file=no test/build_cxxwrap.jl +using Test +include(joinpath(@__DIR__, "..", "deps", "cxxwrap.jl")) + +@testset "libcxxwrap cache validation" begin + mktempdir() do tmp + root = replace(tmp, '\\' => '/') + headers = "$root/headers" + library = "$root/libcxxwrap.so" + override = "$root/override" + mkpath(override) + mkpath(headers) + write(library, "fixture") + write( + joinpath(override, "JlCxxConfig.cmake"), + """ +foreach(name cxxwrap_julia cxxwrap_julia_stl) + add_library(JlCxx::\${name} SHARED IMPORTED) + set_target_properties(JlCxx::\${name} PROPERTIES + INTERFACE_INCLUDE_DIRECTORIES "$headers" + IMPORTED_LOCATION "$library") +endforeach() +""", + ) + + @test cxxwrap_usable(override; log_dir=root) + rm(headers; recursive=true) + @test !cxxwrap_usable(override; log_dir=root) + end +end diff --git a/test/defunct/fusion_compare.jl b/test/cuda.jl/fusion_compare.jl similarity index 87% rename from test/defunct/fusion_compare.jl rename to test/cuda.jl/fusion_compare.jl index dd4934fca..8d4c556c3 100644 --- a/test/defunct/fusion_compare.jl +++ b/test/cuda.jl/fusion_compare.jl @@ -1,3 +1,8 @@ +using CUDA: CUDA, @cuda, blockDim, blockIdx, threadIdx +import CUDA: i32 + +cuNumeric.Experimental(true) + function unfused_cunumeric(u, v, f, k) F_u = ( ( @@ -101,8 +106,8 @@ function run_unfused_baseline(N, u, v) end function fusion_test(; N=1024, atol=1.0f-6, rtol=1.0f-6) - u = cuNumeric.as_type(cuNumeric.random(Float32, (N, N)), Float32) - v = cuNumeric.as_type(cuNumeric.random(Float32, (N, N)), Float32) + u = cuNumeric.rand(Float32, (N, N)) + v = cuNumeric.rand(Float32, (N, N)) # using CUDA u_base = CUDA.rand(Float32, (N, N)) @@ -117,8 +122,14 @@ function fusion_test(; N=1024, atol=1.0f-6, rtol=1.0f-6) Fu_fused, Fv_fused = run_fused_cunumeric(N, u, v) Fu_unfused, Fv_unfused = run_unfused_cunumeric(N, u, v) - @test isapprox(Fu_fused, Fu_unfused; atol=atol, rtol=rtol) - @test isapprox(Fv_fused, Fv_unfused; atol=atol, rtol=rtol) + @test isapprox(Array(Fu_fused), Array(Fu_unfused); atol=atol, rtol=rtol) + @test isapprox(Array(Fv_fused), Array(Fv_unfused); atol=atol, rtol=rtol) end -fusion_test() +try + @testset "2D fusion comparison" begin + fusion_test() + end +finally + cuNumeric.Experimental(false) +end diff --git a/test/defunct/fusion_compare_1d.jl b/test/cuda.jl/fusion_compare_1d.jl similarity index 85% rename from test/defunct/fusion_compare_1d.jl rename to test/cuda.jl/fusion_compare_1d.jl index 8163a5db1..1fa018308 100644 --- a/test/defunct/fusion_compare_1d.jl +++ b/test/cuda.jl/fusion_compare_1d.jl @@ -1,4 +1,9 @@ +using CUDA: CUDA, @cuda, blockDim, blockIdx, threadIdx +import CUDA: i32 + +cuNumeric.Experimental(true) + function unfused_cunumeric(u, v, f, k) F_u = ( ( @@ -102,8 +107,8 @@ function run_unfused_baseline(N, u, v) end function fusion_test(; N=1024*1024, atol=1.0f-6, rtol=1.0f-6) - u = cuNumeric.as_type(cuNumeric.rand(NDArray, N), Float32) - v = cuNumeric.as_type(cuNumeric.rand(NDArray, N), Float32) + u = cuNumeric.rand(Float32, N) + v = cuNumeric.rand(Float32, N) # using CUDA u_base = CUDA.rand(Float32, N) @@ -116,8 +121,14 @@ function fusion_test(; N=1024*1024, atol=1.0f-6, rtol=1.0f-6) # using cuNumeric Fu_fused, Fv_fused = run_fused_cunumeric(N, u, v) Fu_unfused, Fv_unfused = run_unfused_cunumeric(N, u, v) - @test isapprox(Fu_fused, Fu_unfused; atol=atol, rtol=rtol) - @test isapprox(Fv_fused, Fv_unfused; atol=atol, rtol=rtol) + @test isapprox(Array(Fu_fused), Array(Fu_unfused); atol=atol, rtol=rtol) + @test isapprox(Array(Fv_fused), Array(Fv_unfused); atol=atol, rtol=rtol) end -fusion_test() +try + @testset "1D fusion comparison" begin + fusion_test() + end +finally + cuNumeric.Experimental(false) +end diff --git a/test/cuda.jl/padding.jl b/test/cuda.jl/padding.jl new file mode 100644 index 000000000..e4ee31d05 --- /dev/null +++ b/test/cuda.jl/padding.jl @@ -0,0 +1,224 @@ +#= Copyright 2025 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +#= Purpose of test: cuda + -- Validate custom-kernel padding, synchronization, and lifetime management +=# + +using CUDA: blockDim, blockIdx, threadIdx +import CUDA: i32 + +cuNumeric.Experimental(true) + +function padding_add(a, b, c, N) + i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x + if i <= N + @inbounds c[i] = a[i] + b[i] + end + return nothing +end + +function padding_mul(a, c, b, N) + i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x + if i <= N + @inbounds b[i] = a[i] * c[i] + end + return nothing +end + +function cuda_padding_lifetime() + N = 1_000_000 + M = N - 2 + threads = 256 + blocks = cld(M, threads) + initial_bytes = cuNumeric.current_device_bytes[] + + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + c = cuNumeric.zeros(Float32, M) + task = cuNumeric.@cuda_task padding_add(a, b, c, UInt32(M)) + unpadded_bytes = cuNumeric.current_device_bytes[] + + try + @test @inferred( + cuNumeric.launch( + task, (a, b), c, UInt32(M); threads=threads, blocks=blocks + ) + ) === nothing + padded_bytes = cuNumeric.current_device_bytes[] + + @test @inferred(cuNumeric._launch_shape(c)) == (N,) + @test @inferred(cuNumeric._sync_to_launch_padding!(c)) === nothing + @test @inferred(cuNumeric._sync_from_launch_padding!(c)) === nothing + + accounting_ok = true + for _ in 1:15 + cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=UInt32( + M + ) + accounting_ok &= cuNumeric.current_device_bytes[] == padded_bytes + end + + @test padded_bytes > unpadded_bytes + @test accounting_ok + @test size(c) == (M,) + @test all(Array(c) .== 2.0f0) + finally + cuNumeric.destroy!(a) + cuNumeric.destroy!(b) + cuNumeric.destroy!(c) + end + @test cuNumeric.current_device_bytes[] == initial_bytes +end + +function cuda_padding_api_interop() + N = 4096 + M = N - 2 + threads = 256 + blocks = cld(M, threads) + initial_bytes = cuNumeric.current_device_bytes[] + + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + c = cuNumeric.zeros(Float32, M) + library_result = nothing + + try + task = cuNumeric.@cuda_task padding_add(a, b, c, UInt32(M)) + cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, b) outputs=c scalars=UInt32( + M + ) + + # The broadcast writes through c's logical view into its padded backing. + c .= c .* 2.0f0 .+ 0.0f0 + @test all(Array(c) .== 4.0f0) + + # Non-broadcasted operators also consume the logical shape. + library_result = c + c + @test size(library_result) == (M,) + @test all(Array(library_result) .== 8.0f0) + + # A later custom launch sees the values written by the regular API. + task = cuNumeric.@cuda_task padding_mul(a, c, b, UInt32(M)) + cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, c) outputs=b scalars=UInt32( + M + ) + result = Array(b) + @test all(result[1:M] .== 4.0f0) + @test all(result[(M + 1):N] .== 1.0f0) + finally + !isnothing(library_result) && cuNumeric.destroy!(library_result) + cuNumeric.destroy!(a) + cuNumeric.destroy!(b) + cuNumeric.destroy!(c) + end + @test cuNumeric.current_device_bytes[] == initial_bytes +end + +function cuda_padding_slice_output() + N = 4096 + M = N - 2 + threads = 256 + blocks = cld(M, threads) + initial_bytes = cuNumeric.current_device_bytes[] + + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + parent = cuNumeric.zeros(Float32, N) + output = parent[1:M] + task = cuNumeric.@cuda_task padding_add(a, b, output, UInt32(M)) + unpadded_bytes = cuNumeric.current_device_bytes[] + + try + @test @inferred( + cuNumeric.launch( + task, (a, b), output, UInt32(M); threads=threads, blocks=blocks + ) + ) === nothing + padded_bytes = cuNumeric.current_device_bytes[] + @test @inferred(cuNumeric._launch_shape(output)) == (N,) + @test @inferred(cuNumeric._sync_from_launch_padding!(output)) === nothing + + # Mutate the logical parent view before reusing it as a custom-kernel input. + output .= output .* 1.0f0 .+ 1.0f0 + library_result = output + output + @test all(Array(library_result) .== 6.0f0) + cuNumeric.destroy!(library_result) + + task = cuNumeric.@cuda_task padding_mul(a, output, b, UInt32(M)) + @test @inferred( + cuNumeric.launch( + task, (a, output), b, UInt32(M); threads=threads, blocks=blocks + ) + ) === nothing + values = Array(parent) + product = Array(b) + + @test padded_bytes > unpadded_bytes + @test cuNumeric.current_device_bytes[] == padded_bytes + @test all(values[1:M] .== 3.0f0) + @test all(product[1:M] .== 3.0f0) + @test values[end] == 0.0f0 + finally + cuNumeric.destroy!(output) + cuNumeric.destroy!(parent) + cuNumeric.destroy!(a) + cuNumeric.destroy!(b) + end + @test cuNumeric.current_device_bytes[] == initial_bytes +end + +Base.@noinline function drop_padded_arrays() + N = 4096 + M = N - 2 + a = cuNumeric.ones(Float32, N) + b = cuNumeric.ones(Float32, N) + c = cuNumeric.zeros(Float32, M) + task = cuNumeric.@cuda_task padding_add(a, b, c, UInt32(M)) + cuNumeric.@launch task=task threads=256 blocks=cld(M, 256) inputs=(a, b) outputs=c scalars=UInt32( + M + ) + return nothing +end + +function cuda_padding_finalizer() + GC.gc(true) + cuNumeric.drain_pending_frees!() + baseline = cuNumeric.current_device_bytes[] + + drop_padded_arrays() + allocated = cuNumeric.current_device_bytes[] + GC.gc(true) + GC.gc(true) + cuNumeric.drain_pending_frees!() + + @test allocated > baseline + @test cuNumeric.current_device_bytes[] == baseline +end + +try + @testset "Custom CUDA padding" begin + cuda_padding_lifetime() + cuda_padding_api_interop() + cuda_padding_slice_output() + cuda_padding_finalizer() + end +finally + cuNumeric.Experimental(false) +end diff --git a/test/defunct/vecadd.jl b/test/cuda.jl/vecadd.jl similarity index 81% rename from test/defunct/vecadd.jl rename to test/cuda.jl/vecadd.jl index ea01e7ab3..723d110c8 100644 --- a/test/defunct/vecadd.jl +++ b/test/cuda.jl/vecadd.jl @@ -21,6 +21,11 @@ -- Register various custom kernels using CUDA.jl =# +using CUDA: blockDim, blockIdx, threadIdx +import CUDA: i32 + +cuNumeric.Experimental(true) + function kernel_add(a, b, c, N) i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x if i <= N @@ -29,9 +34,8 @@ function kernel_add(a, b, c, N) return nothing end -# testing a second kernel -# on purpose switching inputs and outputs -function kernel_mul(a, b, c, N) +# Test a second kernel with `c` as an input and `b` as the output. +function kernel_mul(a, c, b, N) i = (blockIdx().x - 1i32) * blockDim().x + threadIdx().x if i <= N @inbounds b[i] = a[i] * c[i] @@ -70,18 +74,16 @@ function cuda_binaryop(max_diff) N ) - @test @allowscalar cuNumeric.compare(c, c_cpu, atol(Float32), rtol(Float32)) + @test @allowscalar cuNumeric.compare(c, c_cpu, max_diff, max_diff) - for i in 1:N - @allowscalar b[i] = a[i] * c[i] - end + b_cpu .= a_cpu .* c_cpu - task = cuNumeric.@cuda_task kernel_mul(a, b, c, UInt32(1)) + task = cuNumeric.@cuda_task kernel_mul(a, c, b, UInt32(1)) cuNumeric.@launch task=task threads=threads blocks=blocks inputs=(a, c) outputs=b scalars=UInt32( N ) - @test @allowscalar cuNumeric.compare(b, b_cpu, atol(Float32), rtol(Float32)) + @test @allowscalar cuNumeric.compare(b, b_cpu, max_diff, max_diff) end function kernel_sin(a, b, N) @@ -119,5 +121,14 @@ function cuda_unaryop(max_diff) # TODO explore getting inplace ops working. cuNumeric.@launch task=task threads=threads blocks=blocks inputs=a outputs=b scalars=UInt32(N) - @test @allowscalar cuNumeric.compare(b, b_cpu, atol(Float32), rtol(Float32)) + @test @allowscalar cuNumeric.compare(b, b_cpu, max_diff, max_diff) +end + +try + @testset "Custom CUDA kernels" begin + cuda_binaryop(1.0f-5) + cuda_unaryop(1.0f-5) + end +finally + cuNumeric.Experimental(false) end diff --git a/test/gpu_only/broadcast_fusion.jl b/test/gpu_only/broadcast_fusion.jl index 5c92911c8..1808be97b 100644 --- a/test/gpu_only/broadcast_fusion.jl +++ b/test/gpu_only/broadcast_fusion.jl @@ -30,6 +30,12 @@ =# _broadcast_fusion_user_add(x, y) = x + y +_broadcast_fusion_absnorm(x, t) = abs(x) +function _broadcast_fusion_residual(e, u0, u1, atol, rtol, norm, t) + return e / (atol + max(norm(u0, t), norm(u1, t)) * rtol) +end +_broadcast_fusion_bad_result(x) = string(x) +_broadcast_fusion_bad_kernel(x) = parse(Float32, string(x)) @testset "Broadcast Fusion" begin T=Float32 @@ -46,10 +52,99 @@ _broadcast_fusion_user_add(x, y) = x + y b = @allowscalar NDArray(julia_b) c = @allowscalar NDArray(julia_c) + @testset "identity broadcast between slices" begin + values = NDArray(Float64.(1:8)) + dst = values[1:2] + src = values[7:8] + @test !cuNumeric.nda_overlaps(dst, src) + dst .= src + @test Array(values) == Float64[7, 8, 3, 4, 5, 6, 7, 8] + cuNumeric.destroy!(dst) + cuNumeric.destroy!(src) + + values = NDArray(Float64.(1:8)) + dst = values[2:4] + src = values[1:3] + @test cuNumeric.nda_overlaps(dst, src) + dst .= src + @test Array(values) == Float64[1, 1, 2, 3, 5, 6, 7, 8] + cuNumeric.destroy!(dst) + cuNumeric.destroy!(src) + end + s1 = T(2.5) s2 = T(1.0) s3 = T(0.5) + @testset "single custom operation uses fusion" begin + dest = cuNumeric.zeros(T, N) + custom_bc = Base.broadcasted(_broadcast_fusion_user_add, a, b) + native_bc = Base.broadcasted(sin, a) + ref = Ref(_broadcast_fusion_absnorm) + + @test cuNumeric.__materialize(ref) === ref + @test cuNumeric._should_attempt_broadcast_fusion(dest, custom_bc) + if cuNumeric.FUSE_BROADCAST_MIN_OPS > 1 + @test !cuNumeric._should_attempt_broadcast_fusion(dest, native_bc) + @test !cuNumeric._should_attempt_broadcast_fusion(dest, Base.broadcasted(abs2, a)) + end + + if cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric._has_gpu_target() + result = _broadcast_fusion_user_add.(a, b) + @allowscalar @test safe_compare(julia_a .+ julia_b, result, atol, rtol) + + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + T <: Real || continue + values = T === Bool ? Bool[false, true] : T[0, 1, 2] + @test Array(abs2.(NDArray(values))) == abs2.(values) + @test Array(map(abs2, NDArray(values))) == abs2.(values) + end + + for T in (ComplexF32, ComplexF64) + z_host = T[1 + 2im, 2 - 3im] + z = NDArray(z_host) + @test Array(abs2.(z)) ≈ abs2.(z_host) + @test Array(map(abs2, z)) ≈ abs2.(z_host) + end + + residual = _broadcast_fusion_residual.( + a, b, c, s1, s2, ref, s3 + ) + expected = julia_a ./ (s1 .+ max.(abs.(julia_b), abs.(julia_c)) .* s2) + @allowscalar @test safe_compare(expected, residual, atol, rtol) + + err = try + dest .= _broadcast_fusion_bad_result.(a) + nothing + catch caught + caught + end + @test err isa ArgumentError + @test occursin("unsupported result type String", sprint(showerror, err)) + + nested_err = try + dest .= _broadcast_fusion_bad_result.(a) .+ s1 + nothing + catch caught + caught + end + @test nested_err isa ArgumentError + @test occursin( + "unsupported result type String", + sprint(showerror, nested_err), + ) + + compile_err = try + dest .= _broadcast_fusion_bad_kernel.(a) + nothing + catch caught + caught + end + @test compile_err isa ErrorException + @test occursin("GPU broadcast function failed to fuse", sprint(showerror, compile_err)) + end + end + @testset "Debug formatting" begin input_indices = Dict(objectid(a) => 0, objectid(b) => 1) tree = Base.broadcasted(+, Base.broadcasted(*, a, b), s1) @@ -203,7 +298,7 @@ _broadcast_fusion_user_add(x, y) = x + y @testset "z .= scalar * f.(A, B)" begin expected = T(2.0) .* (julia_a .+ julia_b) z = cuNumeric.zeros(T, (N,)) - @analyze_lifetimes begin + @accelerate begin z .= T(2.0) .* _broadcast_fusion_user_add.(a, b) end @allowscalar @test safe_compare(expected, z, atol, rtol) @@ -559,6 +654,18 @@ end b2d = @allowscalar NDArray(j2b) a2d .= a2d .* s1 .+ b2d @allowscalar @test safe_compare(j2a .* s1 .+ j2b, a2d, atol, rtol) + + original = T.(1:10) + parent = @allowscalar NDArray(copy(original)) + dst = parent[2:9] + src = parent[1:8] + @test cuNumeric.nda_overlaps(dst, src) + dst .= src .* s1 .+ s2 + expected = copy(original) + expected[2:9] .= original[1:8] .* s1 .+ s2 + @allowscalar @test safe_compare(expected, parent, atol, rtol) + cuNumeric.destroy!(dst) + cuNumeric.destroy!(src) end @testset "cross-statement fusion into a slice" begin @@ -566,7 +673,7 @@ end ja = reshape(T.(1:(N * N)), N, N) a = @allowscalar NDArray(ja) out = cuNumeric.zeros(T, (N + 2, N + 2)) - @analyze_lifetimes begin + @accelerate begin producer = a .* s1 out[2:(end - 1), 2:(end - 1)] = producer .+ s2 end @@ -574,21 +681,32 @@ end expected[2:(end - 1), 2:(end - 1)] = ja .* s1 .+ s2 @allowscalar @test safe_compare(expected, out, atol, rtol) end + + @testset "kernel values with data are rejected" begin + # The launcher passes no closure state to the device, so these would + # otherwise read garbage kernel parameters and abort the GPU stream. + a = @allowscalar NDArray(rand(T, 8)) + scale = s1 + @test_throws ArgumentError (x -> x * scale).(a) + @test_throws ArgumentError ((x, t) -> x * t[1]).(a, Ref((s1, s2))) + # Values passed as broadcast arguments stay supported. + @allowscalar @test safe_compare(Array(a) .* s1, ((x, c) -> x * c).(a, s1), atol, rtol) + end end #= Broadcast fusion PTX compilation cache. * Verifies `_BCAST_PTX_CACHE` grows on first fused launch of a signature and * is reused (no new entry) on a second launch of the same signature. - * Gated on `FUSE_BROADCAST_EXPRS` + `HAS_CUDA`; skips otherwise. - * With `FUSE_BROADCAST_MIN_OPS > 1`, single-op exprs are unfused — tests - * should set min ops to 1 (LocalPreferences / ENV) to exercise the cache. + * Gated on `FUSE_BROADCAST_EXPRS` and an active GPU target; skips otherwise. + * With `FUSE_BROADCAST_MIN_OPS > 1`, single native ops are unfused — tests + * should set min ops to 1 (LocalPreferences) to exercise the native-op cache. =# @testset "Broadcast Fusion PTX Cache" begin T=Float32 N=64 - if !(cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric.HAS_CUDA) - @info "Skipping PTX cache tests (need FUSE_BROADCAST_EXPRS && HAS_CUDA)" + if !(cuNumeric.FUSE_BROADCAST_EXPRS && cuNumeric._has_gpu_target()) + @info "Skipping PTX cache tests (need fusion and an active GPU target)" return nothing end if cuNumeric.FUSE_BROADCAST_MIN_OPS > 1 diff --git a/test/gpu_only/fft.jl b/test/gpu_only/fft.jl new file mode 100644 index 000000000..0564de16e --- /dev/null +++ b/test/gpu_only/fft.jl @@ -0,0 +1,182 @@ +#= Copyright 2026 Northwestern University, + * Carnegie Mellon University University + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + * Author(s): David Krasowska + * Ethan Meitz +=# + +using FFTW + +const _FFT_TEST_TYPES = Base.uniontypes(cuNumeric._FFT_ACCEPTED) +const _FFT_INPLACE_TYPES = Base.uniontypes(cuNumeric.SUPPORTED_COMPLEX_TYPES) + +_fft_out_eltype(::Type{T}) where {T} = cuNumeric._fft_eltype(T) + +function _reference_fft(x::AbstractArray, dims=ntuple(identity, ndims(x))) + return FFTW.fft(_fft_out_eltype(eltype(x)).(x), dims) +end + +function _reference_ifft(x::AbstractArray, dims=ntuple(identity, ndims(x))) + return FFTW.ifft(_fft_out_eltype(eltype(x)).(x), dims) +end + +function _fft_compare(got::NDArray, expected; rtol, atol) + return @allowscalar isapprox(Array(got), expected; rtol=rtol, atol=atol) +end + +@testset verbose = true "fft/ifft" begin + @testset verbose = true for T in _FFT_TEST_TYPES + OT = _fft_out_eltype(T) + rtol_t = rtol(OT) + atol_t = atol(OT) + + @testset "1d" begin + x_cpu = my_rand(T, 16) + x = cuNumeric.NDArray(x_cpu) + y = fft(x) + @test eltype(y) === OT + @test _fft_compare(y, _reference_fft(x_cpu); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y), _reference_ifft(_reference_fft(x_cpu)); rtol=rtol_t, atol=atol_t + ) + end + + @testset "2d all dims" begin + x_cpu = my_rand(T, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = fft(x) + @test eltype(y) === OT + @test _fft_compare(y, _reference_fft(x_cpu); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y), _reference_ifft(_reference_fft(x_cpu)); rtol=rtol_t, atol=atol_t + ) + end + + @testset "2d dims=1" begin + x_cpu = my_rand(T, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = fft(x, 1) + @test _fft_compare(y, _reference_fft(x_cpu, (1,)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y, 1), _reference_ifft(_reference_fft(x_cpu, (1,)), (1,)); + rtol=rtol_t, atol=atol_t, + ) + end + + @testset "2d dims=2" begin + x_cpu = my_rand(T, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = fft(x, 2) + @test _fft_compare(y, _reference_fft(x_cpu, (2,)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + ifft(y, 2), _reference_ifft(_reference_fft(x_cpu, (2,)), (2,)); + rtol=rtol_t, atol=atol_t, + ) + end + end +end + +@testset verbose = true "fft!/ifft!/bfft!" begin + @testset verbose = true for T in _FFT_INPLACE_TYPES + rtol_t = rtol(T) + atol_t = atol(T) + x_cpu = my_rand(T, 16) + expected = _reference_fft(x_cpu) + + x = cuNumeric.NDArray(copy(x_cpu)) + y = fft!(x) + @test y === x + @test _fft_compare(x, expected; rtol=rtol_t, atol=atol_t) + + z = ifft!(x) + @test z === x + @test _fft_compare(x, x_cpu; rtol=rtol_t, atol=atol_t) + + x = cuNumeric.NDArray(copy(x_cpu)) + z = bfft!(x) + @test z === x + @test _fft_compare(x, length(x_cpu) .* ifft(x_cpu); rtol=rtol_t, atol=atol_t) + + src_cpu = my_rand(T, 5, 8, 12) + src = cuNumeric.NDArray(src_cpu) + dest = cuNumeric.zeros(T, size(src)) + @test bfft!(dest, src) === dest + @test _fft_compare( + dest, prod(size(src_cpu)) .* _reference_ifft(src_cpu); + rtol=rtol_t, atol=atol_t, + ) + @test _fft_compare(src, src_cpu; rtol=rtol_t, atol=atol_t) + end + + @testset "fft! rejects non-complex" begin + for T in (Float32, Float64, Int32, Bool) + x = cuNumeric.NDArray(my_rand(T, 8)) + @test_throws ArgumentError fft!(x) + @test_throws ArgumentError ifft!(x) + @test_throws ArgumentError bfft!(x) + end + end +end + +@testset verbose = true "batched_fft" begin + @testset verbose = true for T in _FFT_TEST_TYPES + OT = _fft_out_eltype(T) + rtol_t = rtol(OT) + atol_t = atol(OT) + + @testset "1d signals (b, n)" begin + x_cpu = my_rand(T, 4, 16) + x = cuNumeric.NDArray(x_cpu) + y = batched_fft(x) + @test eltype(y) === OT + @test _fft_compare(y, _reference_fft(x_cpu, (2,)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + batched_ifft(y), _reference_ifft(_reference_fft(x_cpu, (2,)), (2,)); + rtol=rtol_t, atol=atol_t, + ) + end + + @testset "2d fields (b, n, m)" begin + x_cpu = my_rand(T, 3, 8, 12) + x = cuNumeric.NDArray(x_cpu) + y = batched_fft(x) + @test _fft_compare(y, _reference_fft(x_cpu, (2, 3)); rtol=rtol_t, atol=atol_t) + @test _fft_compare( + batched_ifft(y), _reference_ifft(_reference_fft(x_cpu, (2, 3)), (2, 3)); + rtol=rtol_t, atol=atol_t, + ) + end + end + + @testset verbose = true for T in _FFT_INPLACE_TYPES + rtol_t = rtol(T) + atol_t = atol(T) + x_cpu = my_rand(T, 4, 16) + expected = _reference_fft(x_cpu, (2,)) + x = cuNumeric.NDArray(copy(x_cpu)) + y = batched_fft!(x) + @test y === x + @test _fft_compare(x, expected; rtol=rtol_t, atol=atol_t) + z = batched_ifft!(x) + @test z === x + @test _fft_compare(x, x_cpu; rtol=rtol_t, atol=atol_t) + end + + @testset "rejects 1d" begin + x = cuNumeric.NDArray(my_rand(ComplexF32, 8)) + @test_throws ArgumentError batched_fft(x) + @test_throws ArgumentError batched_fft!(x) + end +end diff --git a/test/gpu_only/krylov.jl b/test/gpu_only/krylov.jl new file mode 100644 index 000000000..bcb067318 --- /dev/null +++ b/test/gpu_only/krylov.jl @@ -0,0 +1,52 @@ +using Krylov + +@testset "Krylov GPU integration" begin + @testset "$T" for T in (Float32, Float64, ComplexF32, ComplexF64) + R = real(T) + hostx = T[1, 2, 3] + hosty = T[4, 5, 6] + # Complex coefficients exercise the complex scalar wrapper as well. + a = T <: Complex ? T(2 + im) : T(2) + b = T <: Complex ? T(3 - im) : T(3) + α, β = sum(NDArray([a])), sum(NDArray([b])) + x = NDArray(hostx) + # Neither helpers nor coefficient dispatch should implicitly fetch. + allowautofetch(false) do + for s in (α, α.value) + y = NDArray(copy(hosty)) + @test Krylov.kaxpy!(3, s, x, y) === y + @test Array(y) ≈ a .* hostx .+ hosty + end + for s in (a, α, α.value), t in (b, β, β.value) + y = NDArray(copy(hosty)) + @test Krylov.kaxpby!(3, s, x, t, y) === y + @test Array(y) ≈ a .* hostx .+ b .* hosty + end + end + + n = 16 + offdiag = T <: Complex ? T(-1 + 0.25im) : T(-1) + hostA = Matrix(Tridiagonal(fill(conj(offdiag), n - 1), fill(T(4), n), fill(offdiag, n - 1))) + hostb = T.(1:n) + A, rhs = NDArray(hostA), NDArray(hostb) + workspace = Krylov.CgWorkspace(Krylov.KrylovConstructor(rhs)) + tol = 20 * eps(R) + # Complex CG has real coefficients; permit their promotion to complex. + @allowpromotion @allowautofetch Krylov.cg!(workspace, A, rhs; atol=zero(R), rtol=tol, itmax=100, history=true) + @test workspace.stats.solved + @test norm(hostA * Array(workspace.x) - hostb) / norm(hostb) <= 5tol + @test !isempty(workspace.stats.residuals) + solution, stats = @allowpromotion @allowautofetch Krylov.cg(A, rhs; atol=zero(R), rtol=tol, itmax=100) + @test stats.solved + @test solution isa NDArray + @test norm(hostA * Array(solution) - hostb) / norm(hostb) <= 5tol + + # BiCGSTAB exercises the same device-scalar hooks on a nonsymmetric operator. + nonsymmetric = Matrix(Tridiagonal(fill(T(-0.3), n - 1), fill(T(4), n), fill(T(-0.8), n - 1))) + B = NDArray(nonsymmetric) + biworkspace = Krylov.BicgstabWorkspace(B, rhs) + @allowpromotion @allowautofetch Krylov.bicgstab!(biworkspace, B, rhs; atol=zero(R), rtol=tol, itmax=100) + @test biworkspace.stats.solved + @test norm(nonsymmetric * Array(biworkspace.x) - hostb) / norm(hostb) <= 5tol + end +end diff --git a/test/gpu_only/mapreduce.jl b/test/gpu_only/mapreduce.jl new file mode 100644 index 000000000..9ab8fe14b --- /dev/null +++ b/test/gpu_only/mapreduce.jl @@ -0,0 +1,199 @@ +_mapped_reduction_eltype(x::cuNumeric.CNScalar) = eltype(x.value) +_mapped_reduction_eltype(x) = eltype(x) +_mapped_reduction_host(A) = @allowscalar ndims(A) == 0 ? cuNumeric.fetch(A) : Array(A) + +struct ReductionAffine + scale::Float32 + offset::Float64 + sign::Int8 +end +(f::ReductionAffine)(x) = f.scale * x + f.offset + f.sign + +function _mapped_reduction_tolerances(f, input; dims=:) + T = Base.promote_op(f, eltype(input)) + region = dims isa Integer ? (dims,) : dims + n = prod((size(input, d) for d in 1:ndims(input) if dims isa Colon || d in region); init=1) + # Use mapped magnitudes so cancellation and type-changing maps are covered. + scale = maximum(x -> abs(f(x)), input; init=zero(real(T))) + return (; rtol=reduction_rtol(T, max(n, 1)), atol=reduction_atol(T, max(n, 1), scale)) +end + +function _check_mapped_reduction(f, op, input; kwargs...) + A = @allowscalar NDArray(input) + result = nothing + try + expected = mapreduce(f, op, input; kwargs...) + result = mapreduce(f, op, A; kwargs...) + @test size(result) == size(expected) + @test _mapped_reduction_eltype(result) === (expected isa AbstractArray ? eltype(expected) : typeof(expected)) + actual = _mapped_reduction_host(result) + if op === min || op === max || _mapped_reduction_eltype(result) <: Integer + @test isequal(actual, expected) + else + tolerances = _mapped_reduction_tolerances(f, input; dims=get(kwargs, :dims, :)) + @test isapprox(actual, expected; rtol=tolerances.rtol, atol=tolerances.atol) + end + finally + cuNumeric.destroy!(A) + isnothing(result) || cuNumeric.destroy!(result) + end +end + +@testset "Fused mapped reductions" begin + @allowpromotion begin + for T in (Bool, Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, Float32, Float64) + # A typed comprehension keeps Bool inputs byte-addressable, unlike broadcast. + input = reshape(T[mod(i, 2) for i in 0:104], 7, 3, 5) + for op in (+, *, min, max), dims in (:, 1, 2, (1, 3), (3, 1), (1, 1), (), 4) + _check_mapped_reduction(identity, op, input; dims) + end + end + for T in (ComplexF32, ComplexF64), op in (+, *) + if T === ComplexF64 && op === (*) + A = cuNumeric.ones(T, 7, 9) + try + @test_throws ArgumentError mapreduce(identity, op, A; dims=1) + @test_throws ArgumentError prod(identity, A) + finally + cuNumeric.destroy!(A) + end + continue + end + _check_mapped_reduction(identity, op, fill(T(1 + 0im), 7, 9); dims=1) + _check_mapped_reduction(abs2, +, fill(T(1 + 2im), 7, 9)) + end + for shape in ((), (1,), (1, 1)), op in (+, *, min, max) + _check_mapped_reduction(abs2, op, fill(2f0, shape)) + end + # Full singletons use reduce_first; dimensional sums/products seed + # their result with zero/one. Check exact zeros and complex infinities. + for x in (-0f0, ComplexF32(Inf, 0), ComplexF32(0, Inf)), + op in (+, *), dims in (:, 1, ()) + input = fill(x, 1) + A = @allowscalar NDArray(input) + r = nothing + try + r = mapreduce(identity, op, A; dims) + @test isequal(_mapped_reduction_host(r), mapreduce(identity, op, input; dims)) + finally + cuNumeric.destroy!(A) + isnothing(r) || cuNumeric.destroy!(r) + end + end + for T in (Float32, Float64), op in (min, max) + for pair in ((-zero(T), zero(T)), (T(NaN), T(2)), (T(-Inf), T(Inf))) + input = fill(pair[2], 131071) + input[1] = pair[1] + _check_mapped_reduction(identity, op, input) + reverse!(input) + _check_mapped_reduction(identity, op, input) + end + end + + input = reshape(Float32.(1:17017) ./ 17017f0, 7, 11, 221) + _check_mapped_reduction(ReductionAffine(2f0, 0.5, Int8(-1)), +, input) + _check_mapped_reduction(Float64, +, input) + _check_mapped_reduction(identity, +, Int8[100, 100]) + for init in (0f0, 0.0), dims in (:, 1, (1, 3)) + _check_mapped_reduction(abs2, +, input; init, dims) + end + # Non-neutral seeds expose accidental replication across partitions. + _check_mapped_reduction(identity, +, ones(Float32, 131071); init=7f0) + _check_mapped_reduction(identity, *, ones(Float32, 131071); init=3f0) + _check_mapped_reduction(identity, max, ones(Float32, 131071); init=5f0) + _check_mapped_reduction(identity, min, ones(Float32, 131071); init=-5f0) + + for dims in (:, 1), f in (identity, abs, abs2) + _check_mapped_reduction(f, +, Float32[]; dims) + _check_mapped_reduction(f, *, Float32[]; dims) + end + _check_mapped_reduction(abs2, max, Float32[]) + _check_mapped_reduction(x -> x*x, +, Float32[]; init=0f0) + _check_mapped_reduction(x -> x*x, +, zeros(Float32, 0, 3); dims=1) + _check_mapped_reduction(identity, min, zeros(Float32, 3, 0); dims=1) + + A = @allowscalar NDArray(Int8[2, 3, 4]) + try + for (fn, op) in ((sum, Base.add_sum), (prod, Base.mul_prod), (minimum, min), (maximum, max)) + r = fn(identity, A) + @test _mapped_reduction_host(r) === fn(identity, Int8[2, 3, 4]) + cuNumeric.destroy!(r) + end + finally + cuNumeric.destroy!(A) + end + + # Same closure type, different capture values must reuse PTX correctly. + A = @allowscalar NDArray(input) + try + cache_size = nothing + for alpha in (1f0, 3f0, -2f0) + f = let alpha = alpha + x -> abs2(x - alpha) + end + r = mapreduce(f, +, A) + tolerances = _mapped_reduction_tolerances(f, input) + @test isapprox( + _mapped_reduction_host(r), mapreduce(f, +, input); + rtol=tolerances.rtol, atol=tolerances.atol, + ) + cuNumeric.destroy!(r) + current_size = length(cuNumeric._MR_PTX_CACHE) + isnothing(cache_size) || (@test current_size == cache_size) + cache_size = current_size + end + @test_throws ArgumentError mapreduce(identity, -, A) + @test_throws ArgumentError mapreduce(+, +, A, A) + @test_throws ArgumentError mapreduce(identity, +, A; dims=0) + @test_throws ArgumentError mapreduce(identity, +, A; dims=1, init=Int8(0)) + finally + cuNumeric.destroy!(A) + end + + empty = cuNumeric.zeros(Float32, 0) + try + # Full empty reductions delegate to Base.mapreduce_empty. Base + # throws MethodError on Julia 1.10 and ArgumentError on 1.11+. + # Require the exact host exception rather than accepting either. + for reduce_empty in (a -> mapreduce(x -> x*x, +, a), a -> minimum(identity, a)) + expected_error = try + reduce_empty(Float32[]) + catch err + err + end + @test expected_error isa Exception + @test_throws typeof(expected_error) reduce_empty(empty) + end + @test_throws ArgumentError maximum(identity, empty; dims=1) + finally + cuNumeric.destroy!(empty) + end + + # A sliced logical store can have a nonzero origin and non-dense strides. + parent = @allowscalar NDArray(reshape(Float32.(1:323), 17, 19)) + sliced = parent[2:16, 3:18] + r = mapreduce(abs2, +, sliced; dims=1) + @test _mapped_reduction_host(r) ≈ mapreduce(abs2, +, reshape(Float32.(1:323), 17, 19)[2:16, 3:18]; dims=1) + cuNumeric.destroy!(r) + cuNumeric.destroy!(sliced) + cuNumeric.destroy!(parent) + + # Drop the source before execution completes, then consume the result + # on the device without an intervening fetch or execution fence. + A = cuNumeric.ones(Float32, 131071) + r = mapreduce(abs2, +, A) + cuNumeric.destroy!(A) + next = r + r + cuNumeric.destroy!(r) + @test _mapped_reduction_host(next) == 2f0 * 131071 + cuNumeric.destroy!(next) + end + cuNumeric.allowpromotion(false) do + A = cuNumeric.ones(Int8, 3) + @test_throws Exception sum(identity, A) + r = mapreduce(identity, +, A) + @test _mapped_reduction_eltype(r) === Int8 + cuNumeric.destroy!(r) + cuNumeric.destroy!(A) + end +end diff --git a/test/gpu_only/mapreduce_full.jl b/test/gpu_only/mapreduce_full.jl new file mode 100644 index 000000000..3c0008c8f --- /dev/null +++ b/test/gpu_only/mapreduce_full.jl @@ -0,0 +1,249 @@ +using Test +import CUDA + +# One block for each possible nonempty prefix, so every warp boundary and tail +# is checked in a single launch. Invalid lanes deliberately contain nonidentity +# values to detect accidental participation in either level of the reduction. +function _test_block_reduction(src, dest, op) + tid = Int(CUDA.threadIdx().x) + active = Int(CUDA.blockIdx().x) + @inbounds value = src[tid, active] + value = cuNumeric._mr_reduce_block(op, value, active) + tid == 1 && (@inbounds dest[active] = value) + return nothing +end + +function _check_block_prefixes(op, values::Vector{T}) where {T} + host = fill(T(3), 256, 256) + for active in 1:256 + host[1:active, active] .= values[1:active] + end + src, dest = CUDA.CuArray(host), CUDA.zeros(T, 256) + try + CUDA.@cuda threads=256 blocks=256 _test_block_reduction(src, dest, op) + expected = [foldl(op, @view(values[1:n])) for n in 1:256] + @test isequal(Array(dest), expected) + finally + CUDA.synchronize() + CUDA.unsafe_free!(src) + CUDA.unsafe_free!(dest) + end +end + +@testset "Block reduction valid prefixes" begin + for T in (Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, + Float32, Float64, ComplexF32, ComplexF64) + ops = T <: Complex ? (T === ComplexF64 ? (+,) : (+, *)) : (+, *, min, max) + for op in ops + values = op === (*) ? fill(one(T), 256) : + op === min ? fill(T(5), 256) : T[isodd(i) for i in 1:256] + _check_block_prefixes(op, values) + end + end + for T in (Float32, Float64), op in (+, *, min, max), + value in (-zero(T), zero(T), T(NaN), T(Inf), -T(Inf)) + _check_block_prefixes(op, fill(value, 256)) + end + for value in (ComplexF32(Inf, 0), ComplexF64(0, Inf)) + _check_block_prefixes(+, fill(value, 256)) + end +end + +@testset "Runtime reduction axes" begin + for (shape, strides) in (((5,), (2,)), ((2, 3), (2, 7)), ((2, 2, 3), (2, 7, 19))) + N = length(shape) + descriptor = cuNumeric.CuStridedDeviceArray{Int32,N,CUDA.AS.Global}( + reinterpret(Core.LLVMPtr{Int32,CUDA.AS.Global}, UInt(0)), + 0, shape, strides, prod(shape), + ) + mapper_type = typeof(cuNumeric.MapReduceMap(identity, +, Int32, ntuple(_ -> true, N))) + for bits in 0:(2^N - 1) + mask = ntuple(d -> !iszero(bits & (1 << (d - 1))), N) + mapper = @inferred cuNumeric.MapReduceMap(identity, +, Int32, mask) + @test typeof(mapper) === mapper_type + @test mapper.mask === mask + reduced = CartesianIndices(ntuple(d -> mask[d] ? shape[d] : 1, N)) + retained = CartesianIndices(ntuple(d -> mask[d] ? 1 : shape[d], N)) + for (r, ri) in enumerate(reduced), (o, oi) in enumerate(retained) + expected = sum(d -> (ri[d] + oi[d] - 2) * strides[d], 1:N) + @test (@inferred cuNumeric._mr_offset(descriptor, o - 1, r - 1, mask)) == expected + end + end + end +end + +function _test_strided_partial(parent, scratch, mapper, origin, stride, n, chunks) + T = eltype(parent) + src = cuNumeric.CuStridedDeviceArray{T,1,CUDA.AS.Global}( + pointer(parent, origin + 1), (length(parent) - origin) * sizeof(T), (n,), (stride,), n, + ) + dst = cuNumeric.CuStridedDeviceArray{T,1,CUDA.AS.Global}( + pointer(scratch), length(scratch) * sizeof(T), (length(scratch),), (1,), length(scratch), + ) + cuNumeric._mr_partial_kernel(src, dst, mapper, 0, chunks) + return nothing +end + +@testset "1D reduction offsets" begin + mapper = cuNumeric.MapReduceMap(identity, +, Int32, (true,)) + for n in (1, 31, 256, 257, 1025, 4097), origin in (0, 3), stride in (1, 2, 3) + host = Int32[mod(i, 7) - 3 for i in 1:(3n + 7)] + chunks = cld(n, 1024) + parent, scratch = CUDA.CuArray(host), CUDA.zeros(Int32, chunks) + try + CUDA.@cuda threads=256 blocks=chunks _test_strided_partial( + parent, scratch, mapper, origin, stride, n, chunks, + ) + expected = sum(host[(origin + 1):stride:(origin + 1 + (n - 1)*stride)]) + @test sum(Array(scratch)) == expected + finally + CUDA.synchronize() + CUDA.unsafe_free!(parent) + CUDA.unsafe_free!(scratch) + end + end +end + +function _check_full_reduction(input, op; kwargs...) + A = @allowscalar cuNumeric.NDArray(input) + result = nothing + try + expected = mapreduce(identity, op, input; kwargs...) + result = mapreduce(identity, op, A; kwargs...) + actual = @allowscalar cuNumeric.fetch(result) + @test typeof(actual) === typeof(expected) + @test isequal(actual, expected) + finally + isnothing(result) || cuNumeric.destroy!(result) + cuNumeric.destroy!(A) + end +end + +@testset "Full reduction launch boundaries" begin + @allowpromotion begin + for T in (Bool, Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, + Float32, Float64, ComplexF32, ComplexF64) + ops = T <: Complex ? (T === ComplexF64 ? (+,) : (+, *)) : (+, *, min, max) + for n in (1, 31, 32, 33, 255, 256, 257, 1023, 1024, 1025), op in ops + _check_full_reduction(T[isodd(i) for i in 1:n], op) + end + end + for T in (Float32, Float64), op in (+, *, min, max), n in (1, 257, 1025) + for seed in (T(3), Float64(3)) + _check_full_reduction(fill(one(T), n), op; init=seed) + end + end + for T in (Float32, Float64), op in (min, max) + for x in (-zero(T), zero(T), T(NaN), T(Inf), -T(Inf)) + _check_full_reduction(fill(x, 257), op) + end + end + for x in (-0f0, ComplexF32(Inf, 0), ComplexF32(0, Inf)), op in (+, *) + _check_full_reduction([x], op) + end + end + cuNumeric.issue_execution_fence(; block=true) +end + +@testset "Full reduction of a sliced store" begin + host = reshape(Int32.(1:323), 17, 19) + parent = @allowscalar cuNumeric.NDArray(host) + sliced = parent[2:16, 3:18] + result = nothing + try + result = mapreduce(x -> x*x, +, sliced) + cuNumeric.destroy!(sliced) + sliced = nothing + cuNumeric.destroy!(parent) + parent = nothing + @test (@allowscalar cuNumeric.fetch(result)) == mapreduce(x -> x*x, +, host[2:16, 3:18]) + finally + isnothing(result) || cuNumeric.destroy!(result) + isnothing(sliced) || cuNumeric.destroy!(sliced) + isnothing(parent) || cuNumeric.destroy!(parent) + end +end + +# Exercise the combination independently of Legate's partitioning decisions. +function _test_full_contribution(scratch, dest, op, single, chunks) + S = eltype(scratch) + src = cuNumeric.CuStridedDeviceArray{S,1,CUDA.AS.Global}( + pointer(scratch), length(scratch) * sizeof(S), (length(scratch),), (1,), length(scratch), + ) + cuNumeric._mr_contribute_full_kernel(src, dest, op, single, 0, 1, chunks) + return nothing +end + +@testset "Cooperative full contribution" begin + for S in (Int8, Int16, Int32, Int64, UInt8, UInt16, UInt32, UInt64, + Float32, Float64, ComplexF32, ComplexF64) + ops = S <: Complex ? (S === ComplexF64 ? (+,) : (+, *)) : (+, *, min, max) + for op in ops, n in (1, 31, 32, 33, 63, 65, 255, 256, 257, 511, 513, + 1023, 1024, 1025, 4095, 4096), single in (false, true) + host = S[isodd(i) for i in 1:n] + seed = S(3) + expected = foldl(op, host) + single || (expected = op(seed, expected)) + scratch, dest = CUDA.CuArray(host), CUDA.CuArray([seed]) + try + CUDA.@cuda threads=256 blocks=1 _test_full_contribution(scratch, dest, op, single, n) + @test isequal(only(Array(dest)), expected) + finally + CUDA.synchronize() + CUDA.unsafe_free!(scratch) + CUDA.unsafe_free!(dest) + end + end + end +end + +struct ReductionSingletonOnly + offset::Float32 +end +(f::ReductionSingletonOnly)(x) = x + f.offset + +@testset "Singleton preparation and asynchronous output ownership" begin + A = cuNumeric.ones(Float32, 1, 33) + results = Any[] + try + before = Set(keys(cuNumeric._MR_PTX_CACHE)) + push!(results, mapreduce(ReductionSingletonOnly(2f0), +, A; dims=1)) + added = setdiff(Set(keys(cuNumeric._MR_PTX_CACHE)), before) + @test !isempty(added) + @test all(key -> key[1] === cuNumeric._mr_finish_kernel, added) + # Queue multiple freshly allocated outputs, including decode/seed + # finishers. Their input and temporary handles may die before execution. + for _ in 1:8 + push!(results, mapreduce(identity, min, A; dims=2)) + push!(results, @allowpromotion mapreduce(abs2, +, A; init=2.0)) + push!(results, mapreduce(identity, *, A; dims=())) + end + cuNumeric.destroy!(A) + @test (@allowscalar Array(results[1])) == fill(3f0, 1, 33) + for i in 2:3:length(results) + @test (@allowscalar Array(results[i])) == fill(1f0, 1, 1) + @test (@allowscalar cuNumeric.fetch(results[i + 1])) === 35.0 + @test (@allowscalar Array(results[i + 2])) == fill(1f0, 1, 33) + end + finally + cuNumeric.destroy!(A) + foreach(cuNumeric.destroy!, results) + cuNumeric.issue_execution_fence(; block=true) + end +end + +@testset "Full contribution special values" begin + for value in (-0f0, -0.0, ComplexF32(Inf, 0), ComplexF64(0, Inf)), + n in (1, 31, 256, 257, 4096) + host = fill(value, n) + scratch, dest = CUDA.CuArray(host), CUDA.CuArray([zero(value)]) + try + CUDA.@cuda threads=256 blocks=1 _test_full_contribution(scratch, dest, +, true, n) + @test isequal(only(Array(dest)), foldl(+, host)) + finally + CUDA.synchronize() + CUDA.unsafe_free!(scratch) + CUDA.unsafe_free!(dest) + end + end +end diff --git a/test/gpu_only/struct_storage.jl b/test/gpu_only/struct_storage.jl new file mode 100644 index 000000000..7f7ec0ef9 --- /dev/null +++ b/test/gpu_only/struct_storage.jl @@ -0,0 +1,421 @@ +struct StructStorageTriple{T} + a::T + b::T + c::T +end + +struct StructStorageParticle{T} + position::T + velocity::T +end + +_struct_storage_step(p, dt) = StructStorageParticle( + p.position + dt * p.velocity, p.velocity +) + +function _struct_storage_float(x) + return StructStorageTriple{Float32}( + Float32(x + 1), Float32(x + 2), Float32(x + 3) + ) +end +_struct_storage_uint(x) = StructStorageTriple{UInt64}(UInt64(x + 1), UInt64(x + 2), UInt64(x + 3)) + +struct StructStoragePacked + a::UInt8 + b::Bool + c::Int16 + d::UInt32 +end +_struct_storage_packed(x) = StructStoragePacked(UInt8(x + 1), isodd(x), Int16(x + 3), UInt32(x + 4)) + +struct StructStoragePadded + a::UInt8 + b::Float64 + c::Int16 + d::Float32 +end +function _struct_storage_padded(x) + return StructStoragePadded( + UInt8(x + 1), Float64(x + 2), Int16(x + 3), Float32(x + 4) + ) +end + +struct StructStorageComplex + a::ComplexF32 + b::Float64 + c::UInt8 +end +function _struct_storage_complex(x) + return StructStorageComplex( + ComplexF32(Float32(x + 1), Float32(x + 2)), Float64(x + 3), UInt8(x + 4) + ) +end +_struct_storage_real(x::StructStorageComplex) = real(x.a) +_struct_storage_imag(x::StructStorageComplex) = imag(x.a) + +_struct_storage_named(x) = (a=Float32(x + 1), b=UInt8(x + 2)) + +struct StructStorageNested + a::StructStorageTriple{Float32} + b::Float64 +end +_struct_storage_nested(x) = StructStorageNested(_struct_storage_float(x), Float64(x + 4)) + +struct StructStorageTupleField + a::NTuple{3,Float32} + b::Int32 +end +function _struct_storage_tuple(x) + return StructStorageTupleField( + (Float32(x + 1), Float32(x + 2), Float32(x + 3)), Int32(x + 4) + ) +end + +@inline _struct_storage_field(x, ::Val{I}) where {I} = getfield(x, I) + +function _check_struct_storage(f::F, ::Type{T}, input) where {F,T} + @test isbitstype(T) + @test cuNumeric._struct_storage_type(T) + output = f.(input) + try + @test output isa NDArray{T,1} + for j in 1:fieldcount(T) + # Complex-valued extraction still takes cuPyNumeric's unary path. + fieldtype(T, j) <: Complex && continue + field = _struct_storage_field.(output, Ref(Val(j))) + try + @test Array(field) == [getfield(f(Int64(i)), j) for i in 0:3] + finally + cuNumeric.destroy!(field) + end + end + if T == StructStorageComplex + for (projection, expected) in ( + (_struct_storage_real, Float32[1, 2, 3, 4]), + (_struct_storage_imag, Float32[2, 3, 4, 5]), + ) + field = projection.(output) + try + @test Array(field) == expected + finally + cuNumeric.destroy!(field) + end + end + end + finally + cuNumeric.destroy!(output) + end +end + +@testset "Struct element storage" begin + input = NDArray(Int64[0, 1, 2, 3]) + try + _check_struct_storage(_struct_storage_float, StructStorageTriple{Float32}, input) + _check_struct_storage(_struct_storage_uint, StructStorageTriple{UInt64}, input) + _check_struct_storage(_struct_storage_packed, StructStoragePacked, input) + _check_struct_storage(_struct_storage_padded, StructStoragePadded, input) + _check_struct_storage(_struct_storage_complex, StructStorageComplex, input) + _check_struct_storage(_struct_storage_named, typeof(_struct_storage_named(0)), input) + + @test fieldoffset(StructStoragePadded, 2) > sizeof(UInt8) + @test isbitstype(StructStorageNested) + @test isbitstype(StructStorageTupleField) + @test_throws ArgumentError _struct_storage_nested.(input) + @test_throws ArgumentError _struct_storage_tuple.(input) + @test_throws ArgumentError similar(input, StructStorageNested, size(input)) + + # Built-in complex arrays must keep Legate's numeric complex type. + @test !cuNumeric._struct_storage_type(ComplexF32) + complex_array = cuNumeric.zeros(ComplexF32, 4) + try + @test Array(complex_array) == zeros(ComplexF32, 4) + finally + cuNumeric.destroy!(complex_array) + end + finally + cuNumeric.destroy!(input) + end +end + +@testset "Struct host transfer and in-place broadcast" begin + initial = StructStorageParticle{Float32}[ + StructStorageParticle(0.0f0, 1.0f0), + StructStorageParticle(2.0f0, -0.5f0), + ] + particles = NDArray(initial) + try + @test particles isa NDArray{StructStorageParticle{Float32},1} + @test Array(particles) == initial + @test occursin("StructStorageParticle", sprint(show, MIME"text/plain"(), particles)) + + particles .= _struct_storage_step.(particles, 0.1f0) + @test Array(particles) == StructStorageParticle{Float32}[ + StructStorageParticle(0.1f0, 1.0f0), + StructStorageParticle(1.95f0, -0.5f0), + ] + finally + cuNumeric.destroy!(particles) + end + + mixed = reshape([_struct_storage_padded(i) for i in 0:5], 2, 3) + device_mixed = NDArray(mixed) + try + @test Array(device_mixed) == mixed + finally + cuNumeric.destroy!(device_mixed) + end +end + +@testset "Struct storage round trips across layouts" begin + for make_value in ( + _struct_storage_float, + _struct_storage_packed, + _struct_storage_padded, + _struct_storage_complex, + _struct_storage_named, + ) + T = typeof(make_value(0)) + host = reshape(T[make_value(i) for i in 0:7], 2, 2, 2) + device = NDArray(host) + copied = similar(device) + try + @test size(device) == size(host) + @test Array(device) == host + @test Array{T,3}(device) == host + @test copyto!(copied, device) === copied + @test Array(copied) == host + finally + cuNumeric.destroy!(copied) + cuNumeric.destroy!(device) + end + end +end + +@testset "Preallocated struct storage needs no experimental opt-in" begin + previous_experimental = get(task_local_storage(), :Experimental, false) + input = NDArray(Int64[0, 1, 2, 3]) + output = similar(input, StructStoragePacked, size(input)) + empty_output = similar(input, StructStoragePacked, (0, 2)) + try + cuNumeric.Experimental(false) + @test output isa NDArray{StructStoragePacked,1} + @test (output .= _struct_storage_packed.(input)) === output + @test Array(output) == [_struct_storage_packed(i) for i in 0:3] + @test size(empty_output) == (0, 2) + @test isempty(Array(empty_output)) + finally + cuNumeric.Experimental(previous_experimental) + foreach(cuNumeric.destroy!, (input, output, empty_output)) + end +end + +struct StructStorageRetagged + x::Float32 + y::UInt8 + z::Int16 + w::UInt32 +end +_struct_storage_retag(p::StructStoragePacked) = StructStorageRetagged(p.a, p.b, p.c, p.d) +_struct_storage_bump(p::StructStoragePadded, dx) = StructStoragePadded(p.a, p.b + dx, p.c, p.d) +function _struct_storage_add(p::StructStoragePadded, q::StructStoragePadded) + return StructStoragePadded(p.a + q.a, p.b + q.b, p.c + q.c, p.d + q.d) +end +_struct_storage_padded_b(p::StructStoragePadded) = p.b +# Wrap indices so large arrays stay inside each field's range. +function _struct_storage_host(dims...) + return reshape( + [_struct_storage_padded(i % 200) for i in 0:(prod(dims) - 1)], dims... + ) +end + +@testset "Struct storage across ranks" begin + # Rank 0 cannot use the fused kernel, so transfers use fills and reshapes. + scalar_host = fill(_struct_storage_padded(3)) + scalar = NDArray(scalar_host) + scalar_copy = copy(scalar) + try + @test scalar isa NDArray{StructStoragePadded,0} + @test Array(scalar) == scalar_host + @test Array(scalar_copy) == scalar_host + @test @allowscalar(scalar[]) == scalar_host[] + @allowscalar scalar[] = _struct_storage_padded(9) + @test Array(scalar)[] == _struct_storage_padded(9) + @test Array(scalar_copy) == scalar_host + @test !fetch(scalar == scalar_copy) + finally + foreach(cuNumeric.destroy!, (scalar, scalar_copy)) + end + + # Ranks above three pack struct stores through the generic dimension dispatch. + for dims in ((5,), (3, 4), (2, 3, 4), (2, 1, 3, 2), (1, 2, 1, 2, 3)) + host = _struct_storage_host(dims...) + device = NDArray(host) + bumped = _struct_storage_bump.(device, 0.5) + try + @test Array(device) == host + @test Array(bumped) == _struct_storage_bump.(host, 0.5) + finally + foreach(cuNumeric.destroy!, (device, bumped)) + end + end + + # A multi-block launch covers the grid-stride loop for every thread. + large_host = _struct_storage_host(64, 32, 17) + large = NDArray(large_host) + large_bumped = _struct_storage_bump.(large, 1.0) + try + @test Array(large_bumped) == _struct_storage_bump.(large_host, 1.0) + finally + foreach(cuNumeric.destroy!, (large, large_bumped)) + end +end + +@testset "Struct copies, slices, and views" begin + host = _struct_storage_host(4, 3) + device = NDArray(host) + try + whole = copy(device) + rows = device[2:3, :] + rows_copy = copy(rows) + try + @test Array(whole) == host + @test Array(rows) == host[2:3, :] + @test Array(rows_copy) == host[2:3, :] + @test Array(_struct_storage_bump.(rows, 1.0)) == + _struct_storage_bump.(host[2:3, :], 1.0) + @test Array(permutedims(device)) == permutedims(host) + finally + foreach(cuNumeric.destroy!, (whole, rows, rows_copy)) + end + + # Slice destinations are transformed stores, so they copy with a kernel. + replacement_host = reshape([_struct_storage_padded(100 + i) for i in 1:6], 2, 3) + replacement = NDArray(replacement_host) + try + device[2:3, :] = replacement + host[2:3, :] = replacement_host + @test Array(device) == host + finally + cuNumeric.destroy!(replacement) + end + + column = @view device[:, 2] + column .= _struct_storage_bump.(column, 2.0) + host[:, 2] .= _struct_storage_bump.(host[:, 2], 2.0) + @test Array(device) == host + + mismatched = NDArray(_struct_storage_host(3, 4)) + try + @test_throws DimensionMismatch copyto!(similar(device), mismatched) + finally + cuNumeric.destroy!(mismatched) + end + + # Struct reshapes copy each field, following numeric reshape ordering. + flat = cuNumeric.reshape(device, 3, 4) + flat_b = cuNumeric.reshape(_struct_storage_padded_b.(device), 3, 4) + try + @test size(flat) == (3, 4) + @test Array(_struct_storage_padded_b.(flat)) == Array(flat_b) + @test_throws DimensionMismatch cuNumeric.reshape(device, 5, 2) + finally + foreach(cuNumeric.destroy!, (flat, flat_b)) + end + finally + cuNumeric.destroy!(device) + end +end + +@testset "Struct scalar indexing and fills" begin + host = _struct_storage_host(3, 4) + device = NDArray(host) + filled = cuNumeric.fill(_struct_storage_padded(7), (2, 3)) + empty_filled = cuNumeric.fill(_struct_storage_padded(7), (0, 3)) + try + @test_throws ErrorException device[2, 3] + @test @allowscalar(device[2, 3]) == host[2, 3] + @allowscalar device[3, 4] = _struct_storage_padded(42) + host[3, 4] = _struct_storage_padded(42) + @test Array(device) == host + + @test filled isa NDArray{StructStoragePadded,2} + @test Array(filled) == fill(_struct_storage_padded(7), 2, 3) + @test size(empty_filled) == (0, 3) + @test fill!(device, _struct_storage_padded(1)) === device + @test Array(device) == fill(_struct_storage_padded(1), 3, 4) + finally + foreach(cuNumeric.destroy!, (device, filled, empty_filled)) + end +end + +@testset "Struct equality" begin + host = _struct_storage_host(2, 3) + changed_host = copy(host) + changed_host[2, 2] = _struct_storage_padded(50) + nan_host = [StructStorageTriple{Float32}(NaN32, 1, 2)] + device, same, changed = NDArray(host), NDArray(host), NDArray(changed_host) + nan_device = NDArray(nan_host) + packed = NDArray([_struct_storage_packed(i) for i in 0:3]) + retagged = _struct_storage_retag.(packed) + try + @test fetch(device == same) + @test !fetch(device == changed) + @test fetch(device != changed) + @test !fetch(device == NDArray(_struct_storage_host(3, 2))) + # Records without a custom `==` compare with `===`, as in Base. + @test fetch(nan_device == nan_device) == (nan_host == nan_host) + # Identical layouts share a Legate type but are still different records. + @test retagged isa NDArray{StructStorageRetagged,1} + @test Array(retagged) == _struct_storage_retag.(Array(packed)) + @test fetch(packed == retagged) == (Array(packed) == Array(retagged)) + finally + foreach(cuNumeric.destroy!, (device, same, changed, nan_device, packed, retagged)) + end +end + +@testset "Struct broadcasts" begin + host = _struct_storage_host(3, 2) + device = NDArray(host) + other = NDArray(reverse(host)) + empty_input = NDArray(Float32[]) + try + sums = _struct_storage_add.(device, other) + projected = _struct_storage_padded_b.(device) + try + @test Array(sums) == _struct_storage_add.(host, reverse(host)) + @test projected isa NDArray{Float64,2} + @test Array(projected) == _struct_storage_padded_b.(host) + finally + foreach(cuNumeric.destroy!, (sums, projected)) + end + empty_output = _struct_storage_float.(empty_input) + @test empty_output isa NDArray{StructStorageTriple{Float32},1} + @test isempty(Array(empty_output)) + finally + foreach(cuNumeric.destroy!, (device, other, empty_input)) + end +end + +@testset "Unsupported struct operations raise errors" begin + host = _struct_storage_host(2, 3) + device = NDArray(host) + column = NDArray(reshape(Float64[1, 2], 2, 1)) + scalar = NDArray(fill(_struct_storage_padded(1))) + try + # Struct values cannot reach the kernel as broadcast scalars. + @test_throws ArgumentError _struct_storage_add.(device, Ref(_struct_storage_padded(1))) + # Struct results need the fused path: no size-1 extrusion and no rank 0. + @test_throws ArgumentError _struct_storage_bump.(device, column) + @test_throws ArgumentError _struct_storage_bump.(scalar, 1.0) + shift = 1.0 + @test_throws ArgumentError (p -> _struct_storage_bump(p, shift)).(device) + for reduction in (sum, prod, maximum, minimum) + @test_throws ArgumentError reduction(device) + end + @test_throws ArgumentError sum(device; dims=1) + # The host data survives each rejected operation. + @test Array(device) == host + finally + foreach(cuNumeric.destroy!, (device, column, scalar)) + end +end diff --git a/test/gpu_only/structarrays.jl b/test/gpu_only/structarrays.jl new file mode 100644 index 000000000..7bdc6f9a1 --- /dev/null +++ b/test/gpu_only/structarrays.jl @@ -0,0 +1,99 @@ +using StructArrays + +struct BroadcastPair + a::Float64 + b::Float64 +end + +struct MixedFields + a::Float32 + b::Int32 + c::Bool +end + +broadcast_pair(x) = BroadcastPair(x + 1.0, x + 2.0) +scaled_pair(x, s) = BroadcastPair(x * s, x + s) +swap_pair(a, b) = BroadcastPair(b + 10.0, a + 20.0) +mixed_fields(x) = MixedFields(x + 1.0f0, Int32(2x), x >= 2.0f0) + +@testset "StructArray broadcast (experimental)" begin + previous_experimental = get(task_local_storage(), :Experimental, false) + input = NDArray(reshape(Int64[0, 1, 2], 3, 1)) + dest = StructArray{BroadcastPair}(( + a=cuNumeric.zeros(Float64, 3, 1), + b=cuNumeric.zeros(Float64, 3, 1), + )) + mixed_input = NDArray(reshape(Float32[0, 1, 2, 3], 2, 2)) + mixed_dest = StructArray{MixedFields}(( + a=cuNumeric.zeros(Float32, 2, 2), + b=cuNumeric.zeros(Int32, 2, 2), + c=cuNumeric.zeros(Bool, 2, 2), + )) + wrong_shape = NDArray(reshape(Int64[0, 1, 2], 1, 3)) + row = NDArray(reshape(Float64[1, 2], 1, 2)) + wide_dest = StructArray{BroadcastPair}(( + a=cuNumeric.zeros(Float64, 3, 2), + b=cuNumeric.zeros(Float64, 3, 2), + )) + empty_input = cuNumeric.zeros(Int64, 0) + empty_dest = StructArray{BroadcastPair}(( + a=cuNumeric.zeros(Float64, 0), + b=cuNumeric.zeros(Float64, 0), + )) + host_dest = StructArray{BroadcastPair}((a=zeros(3, 1), b=zeros(3, 1))) + + try + @test Base.get_extension(cuNumeric, :cuNumericStructArraysExt) !== nothing + cuNumeric.Experimental(false) + disabled_error = try + dest .= broadcast_pair.(input) + nothing + catch err + err + end + @test disabled_error isa ArgumentError + @test occursin("Experimental features are disabled", sprint(showerror, disabled_error)) + @test_throws ArgumentError (empty_dest .= broadcast_pair.(empty_input)) + @test all(iszero, Array(dest.a)) + @test all(iszero, Array(dest.b)) + + cuNumeric.Experimental(true) + @test (dest .= broadcast_pair.(input)) === dest + @test vec(Array(dest.a)) == [1.0, 2.0, 3.0] + @test vec(Array(dest.b)) == [2.0, 3.0, 4.0] + + # Both outputs must read the old values of both destination fields. + dest .= swap_pair.(dest.a, dest.b) + @test vec(Array(dest.a)) == [12.0, 13.0, 14.0] + @test vec(Array(dest.b)) == [21.0, 22.0, 23.0] + + # Different field types and a nested broadcast share one struct result. + mixed_dest .= mixed_fields.(mixed_input .+ 1.0f0) + @test Array(mixed_dest.a) == reshape(Float32[2, 3, 4, 5], 2, 2) + @test Array(mixed_dest.b) == reshape(Int32[2, 4, 6, 8], 2, 2) + @test Array(mixed_dest.c) == reshape(Bool[false, true, true, true], 2, 2) + + # Numeric scalars stay runtime kernel arguments for every field. + dest .= scaled_pair.(input, 3.0) + @test vec(Array(dest.a)) == [0.0, 3.0, 6.0] + @test vec(Array(dest.b)) == [3.0, 4.0, 5.0] + dest .= scaled_pair.(input, 0.5) + @test vec(Array(dest.a)) == [0.0, 0.5, 1.0] + + # Field projections use the linear kernel, which cannot extrude size-1 axes. + @test_throws ArgumentError (wide_dest .= scaled_pair.(input, row)) + @test all(iszero, Array(wide_dest.a)) + + @test_throws DimensionMismatch (dest .= broadcast_pair.(wrong_shape)) + @test_throws ArgumentError (host_dest .= broadcast_pair.(input)) + @test (empty_dest .= broadcast_pair.(empty_input)) === empty_dest + cuNumeric.Experimental(false) + @test_throws ArgumentError (dest .= broadcast_pair.(input)) + finally + cuNumeric.Experimental(previous_experimental) + for arr in (dest, mixed_dest, empty_dest, wide_dest) + foreach(cuNumeric.destroy!, Tuple(StructArrays.components(arr))) + end + foreach(cuNumeric.destroy!, (input, mixed_input, wrong_shape, empty_input, row)) + end +end diff --git a/test/gpu_only/vector_norm.jl b/test/gpu_only/vector_norm.jl new file mode 100644 index 000000000..34a21745a --- /dev/null +++ b/test/gpu_only/vector_norm.jl @@ -0,0 +1,63 @@ +using Test, LinearAlgebra, Random, cuNumeric + +la_value(x::cuNumeric.CNScalar) = fetch(x) + +@testset "NDArray norms" begin + cuNumeric.allowscalar(false) + Random.seed!(912) + for T in Base.uniontypes(Union{cuNumeric.SUPPORTED_FLOAT_TYPES,cuNumeric.SUPPORTED_COMPLEX_TYPES}) + @testset "$T" begin + R = real(T) + ah, xh = randn(T, 7, 5), randn(T, 5) + a, x = cuNumeric.NDArray(ah), cuNumeric.NDArray(xh) + empty = cuNumeric.zeros(T, 0) + @test la_value(norm(empty)) === zero(R) + @test la_value(norm(cuNumeric.zeros(T, 5))) === zero(R) + for p in (0, 1, 2, 3, 0.5, -1, -2, Inf, -Inf) + @test la_value(norm(x, p)) ≈ norm(xh, p) + @test la_value(norm(a, p)) ≈ norm(ah, p) + end + @test norm(x) isa cuNumeric.CNReal{R} + @test la_value(norm(cuNumeric.NDArray(T[0, 2, 0, 3]), 0)) == R(2) + @test la_value(norm(cuNumeric.NDArray(T[0, 2]), -1)) == zero(R) + @test isnan(la_value(norm(cuNumeric.NDArray(T[NaN, 1])))) + @test isinf(la_value(norm(cuNumeric.NDArray(T[Inf, 1])))) + # Unscaled powers follow cuPyNumeric's overflow/underflow tradeoff. + @test isinf(la_value(norm(cuNumeric.NDArray(fill(T(floatmax(R)/4), 2))))) + @test iszero(la_value(norm(cuNumeric.NDArray(fill(T(floatmin(R)), 2))))) + + end + end + cuNumeric.allowpromotion() do + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + h = ones(T, 2) + x = cuNumeric.NDArray(h) + @test la_value(norm(x)) ≈ norm(h) + @test la_value(norm(x, 0)) ≈ norm(h, 0) + end + end +end + +# Prevent constant propagation of p from hiding branch-dependent return types. +Base.@noinline function check_norm_inference(x::cuNumeric.NDArray{T}, p::Real) where {T} + R = typeof(float(real(zero(T)))) + @test (@inferred norm(x, p)) isa cuNumeric.CNReal{R} +end + +@testset "NDArray norm inference" begin + cuNumeric.allowscalar(false) + cuNumeric.allowpromotion() do + for T in Base.uniontypes(cuNumeric.SUPPORTED_ARRAY_TYPES) + @testset "$T" begin + x = cuNumeric.ones(T, 3) + R = typeof(float(real(zero(T)))) + @test (@inferred norm(x)) isa cuNumeric.CNReal{R} + for p in (0, 1, 2, 3, -1, -2, 0.5, Inf, -Inf, NaN, 2.0f0) + check_norm_inference(x, p) + end + check_norm_inference(cuNumeric.zeros(T, 0), 2) + check_norm_inference(cuNumeric.ones(T, 2, 2), 2) + end + end + end +end diff --git a/test/io/hdf5.jl b/test/io/hdf5.jl index a35cfb59b..704dff55c 100644 --- a/test/io/hdf5.jl +++ b/test/io/hdf5.jl @@ -53,4 +53,38 @@ end test_hdf5_roundtrip(T, shape) end end + + @testset "host-side path checks" begin + mktempdir() do dir + arr = cuNumeric.ones(Float32, 4, 3) + path = joinpath(dir, "values.h5") + + @test_throws ArgumentError cuNumeric.h5read(joinpath(dir, "missing.h5"), "values") + @test_throws ArgumentError cuNumeric.h5read(dir, "values") + @test_throws ArgumentError cuNumeric.h5write(dir, "values", arr) + @test_throws ArgumentError cuNumeric.h5write(path, "", arr) + @test_throws ArgumentError cuNumeric.h5read(path, "") + + stub = joinpath(dir, "stub.h5") + write(stub, "not hdf5") + @test_throws ArgumentError cuNumeric.h5read(stub, "values") + + # A leftover empty file is what aborted HDF5CombineVDS. + leftover = joinpath(dir, "leftover.h5") + write(leftover, UInt8[]) + cuNumeric.h5write(leftover, "values", arr) + cuNumeric.Legate.runtime_sync() + @test cuNumeric._is_hdf5_file(leftover) + @test size(cuNumeric.h5read(leftover, "values")) == size(arr) + cuNumeric.Legate.runtime_sync() + + # A second write to the same path must replace, not abort. + arr2 = cuNumeric.zeros(Float32, 4, 3) + cuNumeric.h5write(leftover, "values", arr2) + cuNumeric.Legate.runtime_sync() + @allowscalar @test safe_compare( + zeros(Float32, 4, 3), cuNumeric.h5read(leftover, "values"), 0, 0 + ) + end + end end diff --git a/test/linalg_errors.jl b/test/linalg_errors.jl new file mode 100644 index 000000000..cd31ad363 --- /dev/null +++ b/test/linalg_errors.jl @@ -0,0 +1,33 @@ +# Opt-in: run each case in a separate process under an external timeout. +# A failed collective can poison its runtime. +using cuNumeric, LinearAlgebra, Test + +length(ARGS) == 1 && only(ARGS) in ("solve", "cholesky", "tiled_cholesky") || + error("Usage: julia --project test/linalg_errors.jl solve|cholesky|tiled_cholesky") +op = only(ARGS) +a = cuNumeric.zeros(Float64, 33, 33) +b = cuNumeric.ones(Float64, 33, 1) +cuNumeric.versioninfo() +backend = if op == "tiled_cholesky" + cuNumeric._TiledCholesky() +else + cuNumeric._linalg_backend(Val(Symbol(op)), a) +end +println("Testing numerical failure with ", typeof(backend)) + +@testset "$op numerical failure" begin + expected = op == "solve" ? r"singular"i : r"positive definite"i + @test_throws expected begin + out = if op == "solve" + a \ b + elseif op == "cholesky" + cholesky(a).factors + else + cuNumeric._cholesky!(backend, similar(a), a) + end + cuNumeric.allowscalar() do + Array(out) # Demand the result so deferred task failures surface. + end + cuNumeric.Legate.issue_execution_fence(true) + end +end diff --git a/test/linalg_preferences.jl b/test/linalg_preferences.jl new file mode 100644 index 000000000..dc5fbf752 --- /dev/null +++ b/test/linalg_preferences.jl @@ -0,0 +1,34 @@ +using Test, Pkg, CNPreferences + +@testset "linear algebra preferences" begin + mktempdir() do dir + # Preference writes must not alter the developer's active project. + write(joinpath(dir, "Project.toml"), + "[deps]\nCNPreferences = \"$(Base.PkgId(CNPreferences).uuid)\"\n") + previous = Base.active_project() + try + Pkg.activate(dir; io=devnull) + settings = ( + MIN_SOLVE_MATRIX_SIZE=32, MIN_SOLVE_TILE_SIZE=4, + MIN_CHOLESKY_MATRIX_SIZE=32, MIN_CHOLESKY_TILE_SIZE=4, + MIN_QR_MATRIX_SIZE=1, QR_TILE_SIZE=4, MAX_CHOLESKY_TILES_PER_PROC=2, + ) + CNPreferences.set_linalg!(; settings...) + for (key, value) in pairs(settings) + @test cuNumeric.load_preference(CNPreferences, string(key)) == value + end + path = joinpath(dir, "LocalPreferences.toml") + saved = read(path, String) + for bad in (0, -1, true, 1.5, "4") + @test_throws ArgumentError CNPreferences.set_linalg!(; + MIN_SOLVE_MATRIX_SIZE=100, QR_TILE_SIZE=bad + ) + @test read(path, String) == saved + end + @test_throws ArgumentError CNPreferences.set_linalg!(; QR_TIEL_SIZE=4) + @test read(path, String) == saved + finally + Pkg.activate(dirname(previous); io=devnull) + end + end +end diff --git a/test/regressions.jl b/test/regressions.jl new file mode 100644 index 000000000..9dcdc7b07 --- /dev/null +++ b/test/regressions.jl @@ -0,0 +1,16 @@ +using Test + +@testset "regressions" begin + @testset "array-size ABI preserves uint64_t" begin + a = cuNumeric.zeros(UInt8, 1) + try + # Check the actual C API binding without a multi-GiB allocation. + # An Int32 return truncates sizes above 2^31 - 1. + actual = cuNumeric.nda_array_size(a) + @test actual isa UInt64 + @test actual == length(a) + finally + cuNumeric.destroy!(a) + end + end +end diff --git a/test/runtests.jl b/test/runtests.jl index 279554fcd..997451dc4 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -7,32 +7,22 @@ using InteractiveUtils: versioninfo run_gpu_tests = CUDA.functional() @info "Julia information:\n" * sprint(io -> versioninfo(io)) -run_gpu_tests && @info "CUDA information:\n" * sprint(io -> CUDA.versioninfo(io)) @info "cuNumeric information:\n" * sprint(io -> cuNumeric.versioninfo(io)) # Forcibly precompile the current environment in parallel: Pkg sometimes ignores # dependencies pointed through via `[sources]` Pkg.precompile() -cuda_init = if run_gpu_tests - quote - using CUDA - import CUDA: i32 - end -else - :() -end - const init_code = quote using LinearAlgebra using Random + using StatsBase + using FFTW import Random: rand ENV["LEGATE_SKIP_RUNTIME"] = "false" using cuNumeric - $cuda_init - include("util.jl") end @@ -40,19 +30,36 @@ end testsuite = find_tests(@__DIR__) delete!(testsuite, "util") +# This standalone build test requires CMake and runs explicitly in developer CI. +delete!(testsuite, "build_cxxwrap") +# These cases deliberately fail a runtime; run individually under a timeout. +delete!(testsuite, "linalg_errors") delete!(testsuite, "array/unary/tests") delete!(testsuite, "array/binary/tests") -if !run_gpu_tests - @warn "CUDA GPU not available, skipping GPU-only tests" - filter!(test -> !startswith(first(test), "gpu_only/"), testsuite) -end +test_args = parse_args(ARGS) +if filter_tests!(testsuite, test_args) + if !run_gpu_tests + @warn "CUDA GPU not available, skipping GPU-only tests" + filter!( + test -> + !startswith(first(test), "gpu_only/") && + !startswith(first(test), "cuda.jl/"), + testsuite, + ) + end -if !run_gpu_tests || !cuNumeric.FUSE_BROADCAST_EXPRS - @warn "Broadcast fusion is disabled, skipping fusion tests" - filter!(test -> !startswith(first(test), "gpu_only/broadcast_fusion"), testsuite) + if !run_gpu_tests || !cuNumeric.FUSE_BROADCAST_EXPRS + @warn "Broadcast fusion is disabled, skipping fusion tests" + filter!( + test -> + !startswith(first(test), "gpu_only/broadcast_fusion") && + !startswith(first(test), "gpu_only/struct_storage") && + !startswith(first(test), "gpu_only/structarrays"), + testsuite, + ) + end end -filter!(test -> !startswith(first(test), "defunct/"), testsuite) - -runtests(cuNumeric, ARGS; testsuite, init_code) +cuda_tests = filter(test -> startswith(test, "cuda.jl/"), collect(keys(testsuite))) +runtests(cuNumeric, test_args; testsuite, init_code, serial=cuda_tests) diff --git a/test/util.jl b/test/util.jl index b9c3e596c..eb4657ab7 100644 --- a/test/util.jl +++ b/test/util.jl @@ -27,6 +27,10 @@ const DOMAIN_GENERATORS = Dict{Symbol,Function}( :positive => (T, N) -> (x=rand(T, N); T <: Signed ? abs.(max.(x, -typemax(T))) : x), ) +# The package only gates promotion when the target type is wider in bytes, so +# Int64/UInt64 -> Float64 needs no `@allowpromotion` while Int32/Bool do. +promotion_is_gated(::Type{FROM}, ::Type{TO}) where {FROM,TO} = sizeof(TO) > sizeof(FROM) + rtol(::Type{Float16}) = 1e-2 rtol(::Type{Float32}) = 1e-5 rtol(::Type{Float64}) = 1e-12 @@ -52,9 +56,9 @@ function reduction_atol(::Type{T}, n, scale=1) where {T} return max(atol(T) * n, n * eps(FT) * abs(scale)) end -is_same(arr1::NDArray, arr2::NDArray) = @allowscalar (arr1 == arr2)[1] -is_same(arr1::NDArray, arr2::Array) = @allowscalar (arr1 == arr2)[1] -is_same(arr1::Array, arr2::NDArray) = @allowscalar (arr1 == arr2)[1] +is_same(arr1::NDArray, arr2::NDArray) = fetch(arr1 == arr2) +is_same(arr1::NDArray, arr2::Array) = @allowscalar (arr1 == arr2) +is_same(arr1::Array, arr2::NDArray) = @allowscalar (arr1 == arr2) is_same(arr1::Array, arr2::Array) = (arr1 == arr2) function my_rand(::Type{F}, dims...; L=F(-1000), R=F(1000)) where {F<:AbstractFloat} diff --git a/test/workflows/grayscott.jl b/test/workflows/grayscott.jl index e81c364f6..0f863dbcd 100644 --- a/test/workflows/grayscott.jl +++ b/test/workflows/grayscott.jl @@ -33,64 +33,62 @@ struct ParamsGS{T<:AbstractFloat} end end -function step(u, v, u_new, v_new, args::ParamsGS) - @analyze_lifetimes begin - # calculate F_u and F_v functions - # currently we don't have NDArray^x working yet. - F_u = ( - ( - -u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) + args.f*(1 .- u[2:(end - 1), 2:(end - 1)]) - ) - F_v = ( - ( - u[2:(end - 1), 2:(end - 1)] .* - (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) - ) - (args.f+args.k)*v[2:(end - 1), 2:(end - 1)] - ) - # 2-D Laplacian of f using array slicing, excluding boundaries - # For an N x N array f, f_lap is the Nend x Nend array in the "middle" - u_lap = ( - ( - u[3:end, 2:(end - 1)] - 2*u[2:(end - 1), 2:(end - 1)] + - u[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 + - ( - u[2:(end - 1), 3:end] - 2*u[2:(end - 1), 2:(end - 1)] + - u[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) - v_lap = ( - ( - v[3:end, 2:(end - 1)] - 2*v[2:(end - 1), 2:(end - 1)] + - v[1:(end - 2), 2:(end - 1)] - ) ./ args.dx^2 + - ( - v[2:(end - 1), 3:end] - 2*v[2:(end - 1), 2:(end - 1)] + - v[2:(end - 1), 1:(end - 2)] - ) ./ args.dx^2 - ) +@accelerate function step(u, v, u_new, v_new, args::ParamsGS) + # calculate F_u and F_v functions + # currently we don't have NDArray^x working yet. + F_u = ( + ( + -u[2:(end - 1), 2:(end - 1)] .* + (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) + ) + args.f*(1 .- u[2:(end - 1), 2:(end - 1)]) + ) + F_v = ( + ( + u[2:(end - 1), 2:(end - 1)] .* + (v[2:(end - 1), 2:(end - 1)] .* v[2:(end - 1), 2:(end - 1)]) + ) - (args.f+args.k)*v[2:(end - 1), 2:(end - 1)] + ) + # 2-D Laplacian of f using array slicing, excluding boundaries + # For an N x N array f, f_lap is the Nend x Nend array in the "middle" + u_lap = ( + ( + u[3:end, 2:(end - 1)] - 2*u[2:(end - 1), 2:(end - 1)] + + u[1:(end - 2), 2:(end - 1)] + ) ./ args.dx^2 + + ( + u[2:(end - 1), 3:end] - 2*u[2:(end - 1), 2:(end - 1)] + + u[2:(end - 1), 1:(end - 2)] + ) ./ args.dx^2 + ) + v_lap = ( + ( + v[3:end, 2:(end - 1)] - 2*v[2:(end - 1), 2:(end - 1)] + + v[1:(end - 2), 2:(end - 1)] + ) ./ args.dx^2 + + ( + v[2:(end - 1), 3:end] - 2*v[2:(end - 1), 2:(end - 1)] + + v[2:(end - 1), 1:(end - 2)] + ) ./ args.dx^2 + ) - # # Forward-Euler time step for all points except the boundaries - u_new[2:(end - 1), 2:(end - 1)] = - ((args.c_u * u_lap) + F_u) * args.dt + u[2:(end - 1), 2:(end - 1)] - v_new[2:(end - 1), 2:(end - 1)] = - ((args.c_v * v_lap) + F_v) * args.dt + v[2:(end - 1), 2:(end - 1)] - - # Apply periodic boundary conditions - u_new[:, 1] = u[:, end - 1] - u_new[:, end] = u[:, 2] - u_new[1, :] = u[end - 1, :] - u_new[end, :] = u[2, :] - v_new[:, 1] = v[:, end - 1] - v_new[:, end] = v[:, 2] - v_new[1, :] = v[end - 1, :] - v_new[end, :] = v[2, :] - end + # # Forward-Euler time step for all points except the boundaries + u_new[2:(end - 1), 2:(end - 1)] = + ((args.c_u * u_lap) + F_u) * args.dt + u[2:(end - 1), 2:(end - 1)] + v_new[2:(end - 1), 2:(end - 1)] = + ((args.c_v * v_lap) + F_v) * args.dt + v[2:(end - 1), 2:(end - 1)] + + # Apply periodic boundary conditions + u_new[:, 1] = u[:, end - 1] + u_new[:, end] = u[:, 2] + u_new[1, :] = u[end - 1, :] + u_new[end, :] = u[2, :] + v_new[:, 1] = v[:, end - 1] + v_new[:, end] = v[:, 2] + v_new[1, :] = v[end - 1, :] + v_new[end, :] = v[2, :] end -# same as above but without @analyze_lifetimes macro +# same as above but without the @accelerate macro function step_base(u, v, u_new, v_new, args::ParamsGS) # calculate F_u and F_v functions # currently we don't have NDArray^x working yet. @@ -239,8 +237,8 @@ function run_slice_test(op, op_scoped, FT, N; f=0.04, k=0.06, dx=1.0) return base, scoped end -binary_scope(op) = (a, b, out) -> @analyze_lifetimes out[:, :] = op(a, b) -slice_scope(op) = (u, v, out, args) -> @analyze_lifetimes out[:, :] = op(u, v, args) +binary_scope(op) = (a, b, out) -> @accelerate out[:, :] = op(a, b) +slice_scope(op) = (u, v, out, args) -> @accelerate out[:, :] = op(u, v, args) const OPS = Dict( :add => (+), @@ -284,6 +282,8 @@ function test_scoping_rewrite_pipeline() @testset "Syntax helpers" begin @test utils._assignment(:(x = y)) == (lhs=:x, rhs=:y) @test utils._broadcast_assignment(:(A[:] .= x)).rhs == :x + @test utils._broadcast_assignment(:(A[:] .+= x)).rhs == :x + @test utils._broadcast_assignment(:(A[:] .-= x)).rhs == :x @test utils._call(:(f(x, y))) == (f=:f, args=Any[:x, :y]) @test utils._dotcall(:(f.(x, y))) == (f=:f, args=Any[:x, :y]) @test utils._reference(:(A[i, j])) == (array=:A, indices=Any[:i, :j]) @@ -358,7 +358,9 @@ function test_scoping_rewrite_pipeline() stmts = utils._flatten_statements(rewritten) @test assigned == Set([:tmp1, :tmp2, :tmp3]) - @test utils._assignment(stmts[1]).lhs == :tmp1 + destination = utils._assignment(stmts[1]) + @test destination.lhs == :tmp1 + @test occursin("@view", sprint(Base.show_unquoted, destination.rhs)) @test utils._assignment(stmts[2]) == (lhs=:tmp2, rhs=:(A[2:(end - 1), 2:(end - 1)])) @test utils._assignment(stmts[3]) == @@ -377,6 +379,32 @@ function test_scoping_rewrite_pipeline() end end + @testset "Compound broadcast assignments stay fused" begin + source = quote + x .+= alpha .* p + @views Ap[2:end] .-= lower[2:end] .* p[1:(end - 1)] + end + + for rewrite in ( + cuNumeric.rewrite_broadcast_lifetimes, cuNumeric.rewrite_eager_lifetimes + ) + cuNumeric.counter[] = 0 + try + rewritten, assigned = rewrite(source) + rendered = sprint(Base.show_unquoted, utils._strip_lines(rewritten)) + + @test occursin("x .+= alpha .* p", rendered) + @test occursin(".-=", rendered) + @test count(line -> occursin(".*", line), eachline(IOBuffer(rendered))) == 2 + @test !occursin(r"tmp\d+ = alpha \.\* p", rendered) + @test !occursin(r"tmp\d+ = .*lower.* \.\* .*p", rendered) + @test !isempty(assigned) # slice views are still lifetime-managed + finally + cuNumeric.counter[] = 0 + end + end + end + @testset "Scalar arithmetic stays inline" begin source = quote C .= A ./ args.dx^2 .+ (args.f + args.k) @@ -436,68 +464,30 @@ function test_scoping_rewrite_pipeline() @test isempty(freed) end - @testset "Lexical lifetime scope" begin - function hidden_binding() - @analyze_lifetimes begin - internal_result = 41 + @testset "Block form keeps Julia scope" begin + # Block form adds no scope: bindings stay live in the enclosing scope (1:1 Julia). + function visible_binding() + @accelerate begin + internal_result = 42 nothing end - return internal_result - end - - function shadowed_binding() - internal_result = :outer - @analyze_lifetimes begin - internal_result = :inner - nothing - end - return internal_result - end - - function hidden_destructured_bindings() - @analyze_lifetimes begin - internal_first, internal_second = (1, 2) - nothing - end - return internal_first, internal_second - end - - function unrelated_undefined_binding() - return unrelated_result - end - - function rendered_error(f) - try - f() - catch exc - return sprint(io -> showerror(io, exc, catch_backtrace())) - end - return "" + return internal_result # would be UndefVar under a `let` end + @test visible_binding() == 42 output = [0] - returned = @analyze_lifetimes begin + returned = @accelerate begin internal_result = 42 output[1] = internal_result internal_result end - - @test_throws UndefVarError hidden_binding() - @test_throws UndefVarError hidden_destructured_bindings() - @test occursin( - "If `internal_result` was created there", rendered_error(hidden_binding) - ) - @test occursin( - "If `unrelated_result` was created there", - rendered_error(unrelated_undefined_binding), - ) - @test shadowed_binding() == :outer @test output == [42] @test returned == 42 if cuNumeric.FUSE_BROADCAST_EXPRS - function hidden_fused_binding(a, b, destination) - @analyze_lifetimes begin + # A fused intermediate is materialized and also stays live. + function fused_binding(a, b, destination) + @accelerate begin fused_result = a .* b destination .= fused_result .+ 1 end @@ -505,9 +495,7 @@ function test_scoping_rewrite_pipeline() end destination = zeros(Int, 2) - @test_throws UndefVarError hidden_fused_binding( - [2, 3], [4, 5], destination - ) + @test fused_binding([2, 3], [4, 5], destination) == [8, 15] @test destination == [9, 16] end end @@ -519,7 +507,7 @@ function test_scoping_regressions(T, N) C = cuNumeric.zeros(T, (N, N)) @testset "In-place assignment" begin - @analyze_lifetimes begin + @accelerate begin result = A[1:end, :] .+ B[1:end, :] C .= result .* T(2.0) end @@ -529,7 +517,7 @@ function test_scoping_regressions(T, N) @testset "Macro as RHS" begin # Test values: (1+1)^2 = 4 - res = @analyze_lifetimes (A .+ B) .^ 2 + res = @accelerate (A .+ B) .^ 2 @test res isa cuNumeric.NDArray @test all(Array(res) .== T(4.0)) end @@ -537,7 +525,7 @@ function test_scoping_regressions(T, N) @testset "Returned bindings stay materialized" begin # A returned producer must come back as a materialized NDArray, not a # lazy broadcast tree; `c` stays a private intermediate that fuses away. - x, y = @analyze_lifetimes begin + x, y = @accelerate begin x = A .+ B c = x .* A y = c .^ 2 @@ -551,14 +539,14 @@ function test_scoping_regressions(T, N) @testset "Return forms yield materialized bindings" begin # `x = y` alias, tuple, and trailing `return` all return real NDArrays. - aliased = @analyze_lifetimes begin + aliased = @accelerate begin y = A .+ B x = y end @test aliased isa cuNumeric.NDArray @test all(Array(aliased) .== T(2)) - rx, ry = @analyze_lifetimes begin + rx, ry = @accelerate begin rx = A .+ B ry = rx .^ 2 return (rx, ry) @@ -570,7 +558,7 @@ function test_scoping_regressions(T, N) if cuNumeric.FUSE_BROADCAST_EXPRS @testset "Indexed fused assignment writes through NDArray slices" begin out = cuNumeric.zeros(T, (N + 2, N + 2)) - @analyze_lifetimes begin + @accelerate begin producer = A .* T(2) out[2:(end - 1), 2:(end - 1)] = producer .+ T(1) end @@ -582,7 +570,7 @@ function test_scoping_regressions(T, N) @testset "Nested @. macros fuse before lifetime analysis" begin multiplier = cuNumeric.ones(T, (N, N)) result = cuNumeric.zeros(T, (N, N)) - @analyze_lifetimes begin + @accelerate begin tmp = @. A + B result .= @. tmp * multiplier + T(1.0) end