diff --git a/.bazelrc b/.bazelrc index 0af36b72..32e24a9c 100644 --- a/.bazelrc +++ b/.bazelrc @@ -60,7 +60,7 @@ build:linux --host_cxxopt=-std=c++20 # These profiles are implementation details of helpers/cmd/develop. Keeping the # compiler and linker flags here makes sanitizer builds reproducible while the # command presents host-aware names and diagnostics to developers. -build:asan-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@/clang:-fsanitize=address +build:asan-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*,-^Compiler/Runtime/Freestanding/.*@/clang:-fsanitize=address build:asan-windows --repo_env=VXS_LLVM_SYSTEM_ALLOCATOR=1 # Clang's implicit absolute ignorelist otherwise appears as an undeclared input # to Bazel's Windows include scanner. The project currently owns no suppressions. @@ -82,14 +82,14 @@ build:asan-windows --linkopt=/WHOLEARCHIVE:clang_rt.asan_static_runtime_thunk-x8 # lld-link does not infer Clang's UBSan libraries. Link the runtime matching # clang-cl's resource directory; never silently substitute trap-only checks. -build:ubsan-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@/clang:-fsanitize=undefined +build:ubsan-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*,-^Compiler/Runtime/Freestanding/.*@/clang:-fsanitize=undefined build:ubsan-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@/clang:-fno-sanitize-recover=all build:ubsan-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@/clang:-fno-sanitize-ignorelist build:ubsan-windows --linkopt=/DEFAULTLIB:clang_rt.ubsan_standalone-x86_64.lib build:ubsan-windows --linkopt=/DEFAULTLIB:clang_rt.ubsan_standalone_cxx-x86_64.lib build:asan-ubsan-windows --config=asan-windows build:asan-ubsan-windows --config=ubsan-windows -build:asan-ubsan-windows '--per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@/clang:-fsanitize=address\\,undefined' +build:asan-ubsan-windows '--per_file_copt=^(Compiler|Interactive|Benchmarks)/.*,-^Compiler/Runtime/Freestanding/.*@/clang:-fsanitize=address\\,undefined' build:asan-macos --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@-fsanitize=address build:asan-macos --linkopt=-fsanitize=address @@ -114,7 +114,7 @@ build:asan-ubsan-linux --linkopt=-fsanitize=address,undefined # Coverage-guided decoder campaigns instrument owned native translation units, # then link libFuzzer's main only into the dedicated driver binary. -build:fuzz-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@/clang:-fsanitize=fuzzer-no-link +build:fuzz-windows --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*,-^Compiler/Runtime/Freestanding/.*@/clang:-fsanitize=fuzzer-no-link build:fuzz-windows --linkopt=/DEFAULTLIB:clang_rt.fuzzer-x86_64.lib build:fuzz-macos --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@-fsanitize=fuzzer-no-link build:fuzz-linux --per_file_copt=^(Compiler|Interactive|Benchmarks)/.*@-fsanitize=fuzzer-no-link diff --git a/.github/ISSUE_TEMPLATE/diagnostic.yml b/.github/ISSUE_TEMPLATE/diagnostic.yml index 87f1a8ec..e3f3c49e 100644 --- a/.github/ISSUE_TEMPLATE/diagnostic.yml +++ b/.github/ISSUE_TEMPLATE/diagnostic.yml @@ -48,7 +48,7 @@ body: attributes: label: Environment description: OS/version, Visual X# version, and relevant compiler or toolchain versions. - placeholder: Windows 11; vxs 0.4.1; Clang 22; GHC 9.10 + placeholder: Windows 11; vxs 0.5.0; Clang 22; GHC 9.10 validations: required: true - type: textarea diff --git a/.github/workflows/compiler-tier1.yml b/.github/workflows/compiler-tier1.yml index 1b707d4a..c340e584 100644 --- a/.github/workflows/compiler-tier1.yml +++ b/.github/workflows/compiler-tier1.yml @@ -87,6 +87,7 @@ jobs: go run ./helpers/cmd/verify-helpers go run ./helpers/cmd/verify-examples go run ./helpers/cmd/verify-benchmarks + go run ./helpers/cmd/execution-cases check - name: Verify project C++ formatting shell: pwsh @@ -198,6 +199,7 @@ jobs: go run ./helpers/cmd/verify-helpers go run ./helpers/cmd/verify-examples go run ./helpers/cmd/verify-benchmarks + go run ./helpers/cmd/execution-cases check - name: Test native compiler shell: bash diff --git a/.github/workflows/compiler-tier2.yml b/.github/workflows/compiler-tier2.yml index 0d838f75..16882578 100644 --- a/.github/workflows/compiler-tier2.yml +++ b/.github/workflows/compiler-tier2.yml @@ -63,6 +63,7 @@ jobs: go run ./helpers/cmd/verify-helpers go run ./helpers/cmd/verify-examples go run ./helpers/cmd/verify-benchmarks + go run ./helpers/cmd/execution-cases check - name: Build and run native compiler suites shell: bash diff --git a/.github/workflows/compiler-tier3.yml b/.github/workflows/compiler-tier3.yml index 19dbcbf8..f7a54acb 100644 --- a/.github/workflows/compiler-tier3.yml +++ b/.github/workflows/compiler-tier3.yml @@ -65,6 +65,7 @@ jobs: go run ./helpers/cmd/verify-helpers go run ./helpers/cmd/verify-examples go run ./helpers/cmd/verify-benchmarks + go run ./helpers/cmd/execution-cases check - name: Build and run native compiler suites shell: bash diff --git a/Analyzer/sources/main/haskell/Visual/Analyzer/Symbols.hs b/Analyzer/sources/main/haskell/Visual/Analyzer/Symbols.hs index 57ef0f76..f5aa83b8 100644 --- a/Analyzer/sources/main/haskell/Visual/Analyzer/Symbols.hs +++ b/Analyzer/sources/main/haskell/Visual/Analyzer/Symbols.hs @@ -38,6 +38,28 @@ declarationSymbol source tokens insideType declaration = typeSymbol spanValue name members TemplateTypeDeclaration spanValue name _ _ members -> typeSymbol spanValue name members + EnumDeclaration spanValue name _ _ cases -> + Lsp.DocumentSymbol + (Text.pack (identifierText name)) + Nothing + Lsp.SymbolKind_Enum + Nothing + Nothing + (spanRange source spanValue) + (selectionRange source tokens spanValue name) + ( Just + [ Lsp.DocumentSymbol + (Text.pack (identifierText (enumCaseName member))) + Nothing + Lsp.SymbolKind_EnumMember + Nothing + Nothing + (spanRange source (enumCaseSpan member)) + (spanRange source (enumCaseSpan member)) + Nothing + | member <- cases + ] + ) FunctionDeclaration spanValue name _ _ parameters _ isStatic _ -> let symbolRange = spanRange source spanValue selection = selectionRange source tokens spanValue name diff --git a/Analyzer/visual-analyzer.cabal b/Analyzer/visual-analyzer.cabal index 9ed8c8ee..fb39ed0f 100644 --- a/Analyzer/visual-analyzer.cabal +++ b/Analyzer/visual-analyzer.cabal @@ -29,8 +29,8 @@ library lsp-types >=2.4 && <2.5, text >=2.0 && <2.2, transformers >=0.6 && <0.7, - visual-xsharp-compiler >=0.4.0 && <0.5, - visual-xsharp-syntax >=0.4.0 && <0.5 + visual-xsharp-compiler >=0.5.0 && <0.6, + visual-xsharp-syntax >=0.5.0 && <0.6 default-language: GHC2024 executable visual-analyzer diff --git a/Benchmarks/2026-10-04-Nesting-And-Chains.md b/Benchmarks/2026-10-04-Nesting-And-Chains.md new file mode 100644 index 00000000..fbd07a48 --- /dev/null +++ b/Benchmarks/2026-10-04-Nesting-And-Chains.md @@ -0,0 +1,411 @@ + + +# Nesting depth, long chains and nested loops + +## Scope + +This record holds three sets of measurements taken before and after one +change set: + +- the stack each native Core stage needs per level of nesting, in an ordinary + build and in an AddressSanitizer and UndefinedBehaviorSanitizer build; +- the time of the Haskell Core operations that grew faster than their input: + the loop fixed point of the integer analysis on nested loops, and wire + encoding, CorePrep lowering and CorePrep verification on `else if` chains + and on sequences of `if` statements; +- the time of `vxs check` on whole programs of those shapes, and of the + sanitized `source_fuzz_smoke` program. + +"Before" is commit `5ffdbdff` for the Criterion measurements and `main` at +`d9bd54e4` for `vxs check` and the smoke program. "After" is commit +`35d9c1ab`. The stack tables and the operator chain table name their +revisions themselves; "now" is the working tree this file is committed with. + +## Environment + +- Date: 2026-10-04 +- OS: Windows NT 10.0.26200, x86-64 +- CPU: Intel Core i5-3210M @ 2.50 GHz +- GHC 9.10.3, Cabal `-O2` for Criterion; clang-cl 22.1.8 for the native code +- Native build: the default configuration of `develop build`; sanitizer + build: `develop sanitize address-undefined` + +## Stack per level of nesting + +`//Compiler/Support/Tests:stack_probe` builds a Core module of a given shape +and depth and runs one stage on a thread whose stack has a given size; the +process is ended by the operating system when the stack is too small. The +smallest size that completes was found by bisection, to within 0.5 percent or +4 KiB, at depths 64, 256, 1024 and 4096, and the table gives the slope +between the two largest depths measured. Windows rounds a stack to 64 KiB, so +small totals are coarse and the slopes are the meaningful figures. + +```powershell +go -C helpers run ./cmd/develop build -- //Compiler/Support/Tests:stack_probe +bazel-bin/Compiler/Support/Tests/stack_probe.exe statements 1024 decode 708 +``` + +The shapes are `statements`, an `if` nested in an `if`; `operands`, an +addition whose second operand is an addition; `expressions`, an addition +whose first operand is an addition; and `chain`, an `else if` chain. The +first two are nesting that every stage recurses along. The last two are +chains, which are as deep in the tree as they are long. `pipeline` is +`Pipeline::ConsumeCore`, the whole native route from Core bytes to LLVM. + +Three revisions were measured: `5ffdbdff`, before any of this work; +`35d9c1ab`, after the frames of the reader, the verifier and the adapter were +reduced; and the working tree this file is committed with, in which chains +are walked in a loop by every stage, modules are released from a list, and +the reader reads every expression into its place. KiB of stack per level: + +| Shape | Stage | `5ffdbdff` | `35d9c1ab` | Now | Now, sanitizers | +| --- | --- | ---: | ---: | ---: | ---: | +| statements | wire writer | 0.67 | 0.67 | 0.67 | 2.05 | +| statements | wire reader | 14.8 | 0.65 | 0.65 | 1.59 | +| statements | Core verifier | 1.70 | 0.67 | 0.67 | 1.70 | +| statements | CorePrep adapter | 3.81 | 0.31 | 0.31 | 1.17 | +| statements | pipeline | 14.9 | 0.67 | 0.67 | 1.67 | +| operands | wire writer | | 0.19 | 0.19 | 0.62 | +| operands | wire reader | | 7.13 | 0.37 | 1.06 | +| operands | Core verifier | | 0.91 | 0.91 | 2.52 | +| operands | CorePrep adapter | | 1.94 | 1.94 | 3.39 | +| operands | pipeline | | 7.13 | 1.94 | 3.39 | +| expressions | wire writer | 0.83 | 0.83 | 0 | not measured | +| expressions | wire reader | 10.6 | 0.42 | 0 | not measured | +| expressions | Core verifier | 2.34 | 0 | 0 | not measured | +| expressions | CorePrep adapter | 4.35 | 0 | 0 | not measured | +| expressions | pipeline | 10.6 | 0.42 | 0 | not measured | +| chain | wire reader | 0.38 | 0.38 | 0 | not measured | +| chain | CorePrep adapter | 0.25 | 0 | 0 | not measured | +| chain | pipeline | 0.38 | 0.38 | 0 | not measured | + +A zero means the stage completed on the smallest stack at every depth +measured: up to 1024 additions and 4096 links, and, for the working tree, +20000 of each on 512 KiB in every stage. An empty cell was not measured: the +`operands` shape was added to the probe after `5ffdbdff`, and until then the +cost of nesting in a second operand had been taken, wrongly, to be that of +the `expressions` shape. The sanitizer build is AddressSanitizer with +UndefinedBehaviorSanitizer. + +Totals that follow from the slopes, for the whole native pipeline: + +| Input | `5ffdbdff` | Now | Now, sanitizers | +| --- | ---: | ---: | ---: | +| statements nested 256 deep, the frontend limit | 3.7 MiB | 0.2 MiB | 0.4 MiB | +| operands nested 1024 deep, the frontend limit | | 1.9 MiB | 3.4 MiB | +| statements nested 4096 deep, the wire limit, writer | 2.7 MiB | 2.7 MiB | 8.2 MiB | +| operands nested 4000 deep, near the wire limit | | 7.6 MiB | 13.3 MiB | +| sum of 20000 operands | crash | under 0.1 MiB | not measured | +| `else if` chain of 20000 links | crash | under 0.1 MiB | not measured | + +The largest stack any measured stage needs on input the limits admit is +therefore 13.3 MiB, in a sanitizer build at the depth limit of the wire +codec, and 3.4 MiB at the limits of the frontend, against the 256 MiB the +compiler thread reserves. The adapter is now the stage that costs most per +level of real nesting. Copying a module still recurses once per level; the +pipeline copies only the bodies of closures. + +## Committed stack of whole compilations + +The slopes above come from one native stage at a time on synthetic Core. The +figures in this section are from whole compilations of source text: the +frontend and every native stage run on one thread, as in `vxs`, and the +stack that thread committed is read at the end. + +What is measured is the committed stack, not the bytes in use. It is an upper +bound on the deepest the thread's stack pointer went, rounded up to whole +pages, and it includes the guard page; it is not an exact high-water mark of +the bytes used. It counts only the thread's own stack. Memory that a build +keeps elsewhere for what would be stack frames is not in it: when +AddressSanitizer's use-after-return detection is active, it moves frames to +a fake stack on the heap, and those frames would be missing here. The +sanitizer builds measured ran with `halt_on_error=1:strict_string_checks=1` +and otherwise the defaults of the runtime, which on Windows leave that +detection off, so their figures are of instrumented frames on the real +stack; they say nothing about a run with the detection on. Measured +figures and figures derived from them are kept apart below: a derived figure +is called extrapolated where it appears. + +```powershell +go -C helpers run ./cmd/develop build -- //Compiler/Fuzzing:source_stack_probe +bazel-bin/Compiler/Fuzzing/source_stack_probe.exe program.vxs 262144 +``` + +The second argument is the reservation in KiB; 262144 is the 256 MiB of the +compiler thread. Windows commits a page of a stack when it is first touched. +Every program is one method; its conditions differ at every +level, so that the optimizer cannot remove the nesting before the native +stages see it. "Limit" is the deepest the frontend accepts: a statement at +level 256, an expression at level 1024. Committed stack, KiB: + +| Program | Ordinary build | ASan and UBSan | +| --- | ---: | ---: | +| `return value + 1;` | 72 | 76 | +| 255 nested `if`, limit | 196 | 464 | +| 255 nested `while`, limit | 172 | 416 | +| 255 nested `for`, limit | 176 | 416 | +| 255 nested `do`/`while`, limit | 168 | 416 | +| 255 nested `guard` blocks, limit | 188 | 464 | +| 255 nested `match` arms, limit | 196 | 464 | +| 255 nested `while` whose condition stores, limit | 172 | 416 | +| 255 nested blocks used as values, limit | 784 | 1348 | +| 256 nested `if`, rejected with `VXP0039` | 32 | 36 | +| 1023 additions nested to the right, limit | 2004 | 3484 | +| 1023 nested calls, limit | 2572 | 4720 | +| conditional chain of 1022 tests, limit | 548 | 1504 | +| 1023 unary operators, limit | 72 | 76 | +| 1024 additions nested to the right, rejected with `VXP0040` | 32 | 40 | +| sum of 7000 operands | 72 | 76 | +| 3000 comparisons joined by `&&` | 72 | 76 | +| `else if` chain of 1400 links | 68 | 76 | +| `match` of 2000 arms | 72 | 76 | + +The harness accepts at most 64 KiB of source, which bounds the chains +measured here; `vxs check` itself was run on a sum of 50000 operands and on an +`else if` chain of 4000 links. The frontend adds almost nothing: Haskell code +runs on stacks of its own, and a program the frontend rejects commits 32 to +40 KiB. Chains commit what the smallest program does. + +The most any program at the frontend limits committed is 2.5 MiB in an +ordinary build and 4.6 MiB under the sanitizers, for nested calls, which cost +2.5 and 4.6 KiB per level. Those are measurements. A function can combine a +statement nest with an expression nest at its bottom; adding the two worst +rows gives about 3.0 MiB and 5.2 MiB, which is derived, not measured. + +### What the lowering adds + +Core nesting per source level, counted on the unoptimized Core of each +construct nested 100 deep. Statement depth counts bodies, with an `else if` +chain as one level; expression depth counts every child except the first +operand of a primitive. + +| Source, 100 levels | Core statement depth | Core expression depth | +| --- | ---: | ---: | +| `if`, `while`, `for`, `do`/`while`, `guard`, `match` arm | 100 | 1 | +| block statement | 0 | 1 | +| `while` whose condition stores | 101 | 1 | +| `do`/`while` whose condition stores | 102 | 1 | +| blocks used as values, pure | 0 | 101 | +| blocks used as values that hold statements | 100 | 1 | +| additions nested to the right, nested calls | 0 | 100 | +| conditional chain | 0 | 101 | +| unary operators | 0 | 1 | +| right operands that store | 0 | 101 | +| `&&` whose right operands store, 200 expression levels | 101 | 2 | +| sum of 100 operands, `else if` chain of 100, `match` of 100 arms | 0 or 1 | 1 | + +The lowering adds at most two levels to a nest. It can turn a level of one +kind into a level of the other: a block used as a value is an expression +level when it is pure and a statement level when it holds statements, and a +short-circuit operator whose right operand stores becomes a conditional +statement. Core from a function at the frontend limits therefore has at most +about 256 + 1024 levels of either kind, inside the 4096 the wire codecs +admit. + +### Beyond the frontend limits + +Core that does not come from this frontend is bounded by the wire codecs at +4096 levels of each kind; deeper input is rejected by the reader before it is +walked. Measured near that bound with the stack probe: 8.2 MiB for 4096 +nested statements in the writer and 13.3 MiB for 4000 nested operands in the +adapter, both under the sanitizers. Nested calls were not measured at that +depth; from their slope of 4.6 KiB they would need about 18.4 MiB, and a +function that nests both kinds to the bound about 27 MiB. That last figure is +an extrapolation, not a measurement. + +### Reserve and commit + +The compiler thread reserves 256 MiB of address space and commits what it +touches. `vxs check` on this branch and on `main`, which has no compiler +thread, one process and eight at once; MiB, the largest of the processes +except where summed: + +| Program | Build | Processes | Peak virtual size | Peak commit | Commit, summed | Wall, s | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| small program | `main` | 1 | 4295.8 | 30.9 | 30.9 | 1.03 | +| | `main` | 8 | 4294.8 | 30.9 | 247.1 | 0.28 | +| | now | 1 | 4550.1 | 31.0 | 31.0 | 0.25 | +| | now | 8 | 4552.1 | 31.0 | 247.7 | 0.32 | +| 1023 nested calls | `main` | 1 | 4309.8 | 44.9 | 44.9 | crash | +| | now | 1 | 4568.1 | 50.5 | 50.5 | 0.50 | +| | now | 8 | 4567.1 | 50.4 | 403.3 | 1.43 | +| sum of 50000 operands | now | 1 | 5101.8 | 582.9 | 582.9 | 13.3 | +| | now | 8 | 5112.9 | 593.8 | 4673.7 | 34.8 | + +The reservation shows as 256 MiB of virtual size and as nothing else: the +commit of a small program is the same with and without it, alone and eight +at once. What a compilation commits is its heap; the stack is at most a few +megabytes of it. The first `main` row includes a cold start. Concurrency +inside one process was not measured, because the compiler has one compiler +thread per process. + +### What the measurements do and do not support + +The figures above are from one machine: Windows, x86-64, clang-cl, an +ordinary build and an AddressSanitizer build with UndefinedBehaviorSanitizer. +Frame sizes differ between compilers and ABIs, so one shape was measured on +the other platforms as well, in CI, on commit `a0b435a3`. + +### The other platforms + +`source_execution_smoke` compiles all of its programs on one compiler +thread and prints what that thread committed at the end. The largest of its +programs for the stack is 1023 calls nested in each other's arguments, the +worst shape of the table above, so the figure is the committed stack at the +expression limit. Each figure is one run on a hosted CI runner, in KiB: + +| Platform | Ordinary build | ASan and UBSan | TSan | +| --- | ---: | ---: | ---: | +| Windows Server, x86-64, clang-cl | 2580 | 4736 | not run | +| Ubuntu 26.04, x86-64, clang | 2588 | 3060 | 2880 | +| Fedora 43, x86-64, clang | 2588 | 3060 | not run | +| macOS 15, arm64, clang | 2480 | 5536 | 2752 | +| macOS 26, arm64, clang | 2480 | 5536 | 2768 | + +On Windows the figure is the committed part of the stack allocation, guard +page included; on Linux and macOS it is the resident pages of the stack +mapping. The Windows row agrees with the 2572 and 4720 KiB measured by hand +for the same shape. The ordinary builds are within 5 percent of each other +on three operating systems and two architectures. The Linux sanitizer figure +is lower than the others because the sanitizer runtime moves frames to a +fake stack on the heap by default there, which this figure leaves out, as +said above; it is not evidence that the sanitized compiler needs less stack +on Linux. The largest figure on any platform is 5536 KiB, 2.1 percent of the +256 MiB reservation. MemorySanitizer builds were not measured, and the +shapes other than nested calls were measured on Windows only. + +Against these figures a reservation of 64 MiB would be about 12 times the +largest commit at the frontend limits under the sanitizers and about 2.4 +times the extrapolated worst case of wire-fed Core. That is an argument for +a smaller reservation, not a decision: it rests on one extrapolation, the +cost of a reservation was measured to be address space only, and the values +in the code are unchanged. + +## Haskell Core operations + +```powershell +Set-Location Compiler +cabal bench visual-xsharp-core:core-benches --enable-benchmarks --benchmark-options="--match pattern NestedLoop Chain Sequence --time-limit 0.1" +``` + +Criterion estimates, one run of each revision. The sample limit is short and +the ranges are wide, so only the orders of magnitude and the growth per +doubling are meaningful. + +`Core/NestedLoopIntegerFacts`: the optimizer on loops nested to the given +depth. + +| Depth | Before | After | +| ---: | ---: | ---: | +| 2 | 251 us | 244 us | +| 4 | 876 us | 326 us | +| 8 | 14.2 ms | 650 us | +| 12 | 276 ms | 659 us | +| 16 | 4.89 s | 907 us | + +Before, four more levels multiplied the time by about 18. The analysis +repeated the fixed point of an inner loop on every pass over the loop around +it. It now iterates only loops that hold at most one further level of loops +and treats what a deeper loop assigns as unknown, which is sound. + +`else if` chains and sequences of `if` statements of the given length: + +| Benchmark | Size | Before | After | +| --- | ---: | ---: | ---: | +| `Core/EncodeChain` | 256 | 78.9 ms | 1.72 ms | +| | 512 | 289 ms | 5.01 ms | +| | 1024 | 968 ms | 14.9 ms | +| | 2048 | 4.95 s | 60.0 ms | +| `CorePrep/PrepareChain` | 256 | 23.8 ms | 1.94 ms | +| | 512 | 101 ms | 4.22 ms | +| | 1024 | 460 ms | 9.15 ms | +| | 2048 | 2.20 s | 20.2 ms | +| `CorePrep/PrepareSequence` | 256 | 3.60 ms | 3.21 ms | +| | 512 | 7.18 ms | 6.16 ms | +| | 1024 | 31.0 ms | 14.9 ms | +| | 2048 | 46.7 ms | 32.4 ms | +| `CorePrep/VerifyChain` | 256 | 11.2 ms | 1.36 ms | +| | 512 | 46.7 ms | 3.35 ms | +| | 1024 | 160 ms | 6.34 ms | +| | 2048 | 746 ms | 15.1 ms | +| `CorePrep/VerifySequence` | 256 | 16.4 ms | 1.85 ms | +| | 512 | 58.5 ms | 4.21 ms | +| | 1024 | 209 ms | 7.95 ms | +| | 2048 | 788 ms | 17.8 ms | + +Lowering and verification now double with the input. Encoding a chain still +grows by a factor of three to four per doubling, from a base about fifty +times lower than before; the cause was not investigated. + +## Whole programs + +`vxs check -File` on one method of the given shape, one run each, wall time +in seconds. `crash` is a stack overflow without a diagnostic. + +| Program | Size | Before | After | +| --- | ---: | ---: | ---: | +| `else if` chain | 500 | 1.40 | 0.64 | +| | 1000 | 2.52 | 1.28 | +| | 2000 | crash | 3.30 | +| | 4000 | crash | 10.4 | +| sequence of `if` statements | 500 | 0.76 | 0.70 | +| | 1000 | 1.51 | 1.32 | +| | 2000 | 3.58 | 2.79 | +| | 4000 | 12.0 | 7.11 | +| nested `for` loops | 8 | 0.19 | 0.17 | +| | 14 | 1.06 | 0.13 | +| | 50 | more than 120 | 0.28 | +| | 255 | more than 120 | 3.72 | +| `match` arms | 500 | 0.74 | 0.65 | +| | 1000 | 1.65 | 1.36 | +| | 2000 | 3.95 | 3.41 | + +Chains of operators, which the expression nesting limit rejected above 1024 +operands until this change, on `35d9c1ab` with the limit lifted and on the +working tree: + +| Program | Size | Limit lifted only | Now | +| --- | ---: | ---: | ---: | +| sum of operands | 5000 | 2.45 | 1.23 | +| | 20000 | 30.4 | 4.57 | +| | 50000 | 198 | 9.37 | +| comparisons joined by `&&` | 100 | 3.68 | not measured | +| | 200 | 47.6 | 1.13 | +| | 800 | more than 120 | 1.75 | +| | 3000 | not measured | 5.60 | + +The `&&` figures of the first column are the same on `main`: 3.66 seconds +for 100 comparisons. Constant propagation asked the integer facts about +every node of an expression, and the CorePrep lowering collected symbol +identities by appending lists; both are fixed, and the facts are consulted +only for conditions of at most 256 nodes. + +The time still grows faster than the program between 2000 and 4000 +statements: by 3.1 for the chain and by 2.5 for the sequence. Earlier +per-stage timing placed that growth in the frontend before Core and in the +native stages after CorePrep; it is not addressed here. + +## Sanitized smoke program + +`source_fuzz_smoke` built with AddressSanitizer and +UndefinedBehaviorSanitizer, run by hand on the same machine: + +| Revision | Wall time | Programs compiled for the execution tables | +| --- | ---: | --- | +| `main` at `d9bd54e4` | 291 s | one for every run | +| one program for every distinct body | 204 s | 134 for 237 runs | +| after: up to eight small bodies in a program | 160 s | 237 runs in fewer programs | + +In the last measurement the expression table takes 23 seconds, the branching +table with its four large programs 36 seconds, and the 342 generated +differential programs, which this change does not touch, 98 seconds. + +`main` exceeds the 240-second watchdog of `develop sanitize` on this machine. +No run was removed, and the large programs are run on more arguments than on +`main`: every run of a body is now a call in the one program compiled for +that body, which returns a bit for each run whose value differs from the expected +one and must return zero in both pipeline modes. Changing one expected value +by hand makes the smoke program fail and name the run. diff --git a/Benchmarks/2026-10-07-Many-Methods.md b/Benchmarks/2026-10-07-Many-Methods.md new file mode 100644 index 00000000..862a502c --- /dev/null +++ b/Benchmarks/2026-10-07-Many-Methods.md @@ -0,0 +1,132 @@ + + +# Compile time by number of methods + +## Scope + +The time of `vxs check` on one class with many small methods grew with the +square of their number: 1000 methods took 4 seconds and 4000 took 73. This +record holds the measurements that located the cause in five places, and the +same measurements after each was changed. + +"Before" is commit `805b0659`. "After" is the working tree this file is +committed with. + +## Environment + +- Date: 2026-10-07 +- OS: Windows NT 10.0.26200, x86-64 +- CPU: Intel Core i5-3210M @ 2.50 GHz +- GHC 9.10.3; clang-cl 22.1.8 for the native code +- Native build: the default configuration of `develop build` + +Other programs ran on the machine during the measurements, and single runs +of one program differed by up to a factor of two. Where a figure is the best +of several runs, the table says so. + +## Programs + +Each program is one class in one file. `methods` has the given number of +static methods of the form `public static int M7(_ int n) { return n + 7; }` +and one method that returns a local. `sequence` has one method with the +given number of statements `if (n > 7) { t += 1; }`. + +## Whole programs + +`vxs check -File`, wall time in seconds. + +| Program | Size | Before, one run | After, best of five | +| --- | ---: | ---: | ---: | +| `methods` | 1000 | 4.28 | 1.35 | +| | 2000 | 16.3 | 2.85 | +| | 4000 | 73.4 | 4.78 | +| | 8000 | not measured | 10.1 | +| `sequence` | 4000 | 5.85 | 6.06, best of two | + +After the change the time of `methods` doubles with the input, at about 1.25 +milliseconds for each method. `sequence` is not affected; it was measured to +confirm that. + +## Where the time went + +### Native stages + +The native stages were timed in the compiler itself, with a clock around each +stage that was added for the measurement and is not part of the change. Times +in milliseconds for `methods`. + +| Stage | 1000 before | 2000 before | 1000 after | 2000 after | 4000 after | 8000 after | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Core decode | 52 | 65 | 45 | not recorded | 117 | 236 | +| Core verify | 29 | 35 | 24 | not recorded | 66 | 133 | +| CorePrep prepare | 26 | 43 | 28 | not recorded | 76 | 149 | +| CorePrep verify | 3860 | 14388 | 25 | 40 | 74 | 149 | +| Xpp lower | 13 | 16 | 12 | not recorded | 30 | 62 | +| Xpp optimize | 96 | 145 | 104 | not recorded | 295 | 579 | +| Xpp ownership placement | 77 | 126 | 68 | not recorded | 251 | 511 | +| Xpp verify | 136 | 218 | 108 | 211 | 425 | 854 | +| Xmm lower | 31 | 28 | 14 | not recorded | 56 | 113 | +| Xmm optimize | 79 | 142 | 78 | not recorded | 286 | 570 | +| LLVM lower | 583 | 734 | 387 | 695 | 1340 | 2728 | + +The native CorePrep verifier took 14.4 of the 17.7 seconds of the run with +2000 methods. For every function it entered every function of the module +into a table of its own, building the function type of each, and it searched +the whole module for closures that capture into the function. The table and +the captures are now collected once for the module. + +`//Compiler/Core/CorePrep/Benches:coreprep_benches` has the case +`VerifyFunctions` for this. After the change, on modules of functions that +return a constant: + +| Functions | Time | +| ---: | ---: | +| 64 | 0.80 ms | +| 256 | 2.55 ms | +| 1024 | 6.77 ms | +| 4096 | 28.1 ms | + +Google Benchmark fits these to linear time with a deviation of 5 percent. +The case was not run on the verifier as it was before. + +### Frontend stages + +The Haskell stages were timed by a program that runs each stage of the +library on a source file and forces its whole result by rendering it as +text. Rendering is part of every figure, so the figures are larger than the +stages are in the compiler; they are comparable with one another. Seconds, +for `methods` with 8000 methods. + +| Stage | Before | After | +| --- | ---: | ---: | +| lexing and parsing | 2.91 | 2.75 | +| renaming, name resolution and type checking | 6.54 | 0.61 | +| Core verification | 1.03 | 0.31 | +| CorePrep lowering, which verifies its input first | 2.73 | 0.84 | + +Four places took time with the square of the number of members: + +- The renamer kept the names in scope as a list of pairs, so every lookup + and every check for a duplicate walked all members of the type. Renaming + alone took 3.91 seconds for 8000 methods; with a map it takes 0.4 for + methods of the same number that return their parameter. +- The type checker compared every method with every earlier member to find + two overloads with the same parameters. It now compares a method with the + earlier methods of its own name. Type checking took 4.30 seconds for 8000 + methods with empty bodies and takes 0.28. +- The Haskell Core verifier looked every function up in the list of source + owners. +- The CorePrep lowering appended the functions lifted out of closures to + the list of functions still waiting, once for every function, also when + there were none; each step then took time with the number of functions + before it. + +## What is not addressed + +`sequence` still takes more than twice as long for twice as many statements +between 2000 and 4000; that growth was recorded in +`2026-10-04-Nesting-And-Chains.md` and is unchanged. Lexing and parsing are +the largest frontend stage after this change and were not examined. diff --git a/Benchmarks/README.md b/Benchmarks/README.md index c17b4a09..6d0e39d8 100644 --- a/Benchmarks/README.md +++ b/Benchmarks/README.md @@ -64,3 +64,7 @@ statistical comparisons once it can provide fixed CPU frequency, warm-up policy, - `2026-09-27-Project-Artifacts.md` records per-source planning and staged output costs on a Windows development host. - `2026-09-27-Loop-Comparisons.md` records the loop-unrolling comparison and a compiler-owned CorePrep loop workload. - `2026-09-28-Core-Loop-Integer-Flow.md` records loop-header widening and loop-carried Core fact costs. +- `2026-10-04-Nesting-And-Chains.md` records the stack the native Core stages use per level of nesting, and the cost + of nested loops and of long `else if` chains, before and after they were reduced. +- `2026-10-07-Many-Methods.md` records the compile time of a class by its number of methods, which grew with the + square of that number in five places, before and after. diff --git a/CHANGELOG.md b/CHANGELOG.md index dee382f3..4f92de88 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,10 +5,112 @@ SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 # Changelog -## Unreleased +## 0.5.0 - 2026-10-09 + +### Why 0.5.0 and not 0.4.2 + +The third component of the version is for a release that a program and its +build artifacts written for the one before still work with. This release is +not one: + +- The language changed what existing programs mean. Evaluation is by need, + so a value nothing reads is no longer computed, and `==` on two strings + compares their characters where it compared the objects. +- `match`, `guard` and `enum` became reserved words. +- Every serialized stage changed incompatibly: Core went from version 8 to + 10, CorePrep to 8, Xpp and Xmm to 7. No artifact of 0.4.1 is read. +- A native executable is now linked with a runtime library that ships + beside the compiler, so an installation of 0.4.1 cannot be updated by + replacing the compiler alone. +- The Haskell packages of the formatter, linter and analyzer bound the + compiler below 0.5 on purpose, and their bounds move with this release. + +The language also grew by more than a patch carries: evaluation by need, +`match`, enums, constant expressions and console output. 0.4.2 was never +published; the number is skipped, not withdrawn. ### Language +- A method reference `Type::Method` is the static method as a callable + value: `auto f = Counter::Next;`, `Apply(Counter::Next)`. Where the method + has overloads, the type the place expects selects one, and `VXT0080` + reports a place that selects none. A reference through a value, + `counter::Next`, reports `VXT0081` until values have members. A selector + with a dot and no call, `Counter.Next`, stays rejected with `VXT0034`. +- A declaration may be named through the namespace it is declared in: + inside `namespace Demo;`, `Demo.Program.Run()` is `Program.Run()`, + `Demo.Color.Red` is `Color.Red` and `Demo.Program::Run` is + `Program::Run`. A namespace of several parts is written whole. A name of + the program spelled like the first part hides the namespace. Other + namespaces, and a qualified name in a type position, are not connected. +- The conditional expression selects strings: + `String kind = count > 0 ? "some" : "none";`. Only the selected operand + is evaluated. `?:` is unchanged, because its left operand is also its + test and a string is not one. +- Visual X# is a lazy language with call-by-need evaluation, and the + specification now says so in `Spec/Language/Evaluation.vxs`: a value is + computed when it is first needed and at most once, a value that is never + needed is never computed, effects happen where they are written, and + neither thunks nor effects have any notation in the source. The compiler + implements the first part of it: a local binding of `bool`, numeric or + enum type whose initializer has no effect is computed by its first read + and not at all when nothing reads it, so `int x = left / right; return 5;` + no longer divides. That holds in the body of a callable as in the body of + a method. An expression written as a statement, and a value assigned to + the discard, are evaluated. Results, other types and assigned + or captured variables are still computed where they are written; + `Documents/EVALUATION.md` lists what is pending. +- Arguments are passed by need. An argument of `bool`, numeric or enum type + that has no effect is computed when the method first needs it, at most + once however often the method reads it and however many methods it is + handed through, and not at all when no method needs it: + `Choose(true, 1, left / right)` no longer divides when `Choose` returns + its second parameter. A value the caller needs as well is computed once + for both. An argument of a call through a callable, an argument a closure + of the method captures, and arguments of other types are still computed + at the call. +- Programs write to the console. `Console.Print` and `Console.Println` write + a value, `Console.Printf` and `Console.Printfn` a format with its + conversions applied, `Console.Error`, `Errorln`, `Errorf` and `Errorfn` do + the same to standard error, and `Console.Format` returns the string and + writes nothing. `System` is imported implicitly, so `Console` needs no + qualification; `System.Console` is accepted. The example + `Examples/HelloWorld/HelloWorld.vxs` compiles to a native executable and + runs. +- A format is a string literal and is checked when the program is compiled. + The conversions are `%d`, `%u`, `%x`, `%f`, `%s`, `%c` and `%b`, with `%n` + for the line terminator of the platform and `%%` for a percent sign; the + flags are `-`, `0`, `+`, space, `#` and `'`; a width or a precision may be + written as `*` and is then an `int` argument before the value. A + conversion that does not exist, a flag a conversion does not take, a + missing or a surplus argument and an argument of the wrong type are + errors, `VXT0071` to `VXT0079`. Examples 69 to 85 of + `Spec/StandardLibrary/IO/ConsoleIO.vxs` state what the specification left + open: which flags each conversion takes, that `%b` writes a `bool`, that + `%f` rounds the exact value with a tie to the even digit, and that a line + ends with the terminator of the platform. +- `+` joins two strings, and writes a value that is not a string as text + first when the other side is one: an integer in decimal, a `bool` as + `true` or `false`, a `char` as itself. `+=` appends to a string variable. + `==` and `\=` on two strings compare the characters they hold; before, + they compared where the strings were kept. Examples 86 to 89 of + `Spec/Language/Operators.vxs`. +- Output is an effect, and effects are not lazy: an expression that writes, + itself or through a method it calls, is evaluated where it is written and + in the order it is written, whether or not its value is ever needed. A + value that only computes is still computed by need in the same program. + Examples 15 and 16 of `Spec/Language/Evaluation.vxs`. +- An integer quotient or remainder by zero and a shift by an amount that is + negative or not less than the width of the shifted value have no value, + and a program that needs one stops. Example 14 of + `Spec/Language/Evaluation.vxs` states the rule. Before, generated code gave + these operations no meaning and an optimized program could run on past + them with a result that was never computed. +- When one statement needs several values that cannot be computed, which + failure the program meets is not determined, and a value that runs without + end counts as one that cannot be computed. Whether the program fails is + determined, and so is the order of statements and of effects. Example 13 + of `Spec/Language/Evaluation.vxs` states the rule. - Added `match`. The statement `match (subject) { pattern -> body, ... }` runs the first arm whose patterns and guard accept the subject, and does nothing when no arm accepts. In operand position `match` is an expression: every @@ -16,14 +118,19 @@ SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 subject is. Several subjects are written `match (a), (b)` with one pattern for each in every arm. The subjects are evaluated once, left to right. - A pattern is a literal, `_`, or a type followed by a name or `_`. A type - pattern binds the subject for the guard and the body of its arm; the - binding is immutable. An arm may have a guard, `pattern if condition ->`, + pattern binds the value of the subject for the guard and the body of its + arm, as an ordinary local that may be assigned. An arm may have a guard, `pattern if condition ->`, which is evaluated only when the patterns of that arm accept. - An arm that can never be selected is an error: a second arm for the same literals, or any arm after one that accepts every value without a guard. - Added `if` as an expression: `int larger = if (a > b) { a } else { b };`. Both blocks are required and each ends with an expression that has no - semicolon. Only the selected block runs. + semicolon. Only the selected block runs. As the last item of a block that + is used as a value, an `if` with two such blocks and a `match` are the + value of that block. +- The comma after a match arm is optional, as in the grammar. A + parenthesized pattern that follows an expression body without a comma is + read as a call of that body; `VXP0038` is reported at the arrow after it. - Added `guard (condition) else { ... }`. The block runs when the condition is false and must leave the enclosing scope with `return`, `break` or `continue`. @@ -37,8 +144,76 @@ SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 such as `.Ready`, type patterns that name another type than their subject's, and a binding in the condition of `if`, `guard` or `while` are recognized and rejected with dedicated diagnostics until reference - subjects, enums, class hierarchies and optional values exist. `return`, - `break` and `continue` cannot leave a block that is used as a value. + subjects, enums, class hierarchies and optional values exist. +- A numeric constant pattern may be negative: `-1 -> ...`. The minus sign + belongs to the numeric literal; no other expression is a pattern. The + constant is checked against the type of its subject, so the most negative + value of a signed type is a pattern and a negative constant for an + unsigned subject is rejected. +- A block used as a value may leave instead of yielding a value. `return` + leaves the enclosing method and `break` and `continue` target the loop + around the expression, under the rules they have anywhere else. A block + that leaves has no value; the expression has the type of the blocks that + complete. An expression none of whose blocks completes is valid: it never + yields a value, what would have received the value is not lowered, and no + result slot or placeholder is created for it. +- Such a block may stand in a loop header and in a loop used as an + expression. The condition and the update clause belong to their loop: a + `break` in either leaves that loop; a `continue` in the condition + evaluates the condition again without running the body or the update; a + `continue` in the update clause of a `for` ends the update, and the + condition is tested next; a `break` may carry a value to the loop + expression around the block. A callable is not inside the loops around + the place that creates it. `Spec/Language/Iteration.vxs` gains examples + 79 to 83 for these rules, and `VXT0059` is retired. +- Added classic enums: `enum Name { A, B = 2, C }`, with an optional + underlying integer type written `enum Name = byte { ... }`. Members are + numbered from zero, a member without a value follows the one before it, + and two members may share a value. `Name.Member` is a value of the enum. + Values are compared with `==` and `\=` with values of the same enum and + take part in no other operation; there is no conversion between an enum + and an integer in either direction. `match` over an enum uses `.Member` + patterns and needs no catch-all arm when every value is named; a second + arm for the same value is unreachable. `Spec/Language/Decls.vxs` gains + examples 311 to 315 for the operations, the absence of conversions and + the target-typed `.Member` spelling. `enum` is now a reserved word. + The value of a member is a constant integer expression over integer + literals and earlier members of the same enum, as in `WRITE = READ << 1` + and `ALL = READ | WRITE`: inside its own declaration the name of a member + stands for its number, and nowhere else. The expression is computed in + the underlying type by the constant evaluator of the language; examples + 316 and 317 specify it, and `VXT0070` reports a value that is not such an + expression. `VXP0041`, which allowed only an integer literal, is retired. + `.Member` is accepted in an expression wherever an enum type is expected: + a declared type, a parameter, a return type, the right operand of a + comparison, the variable assigned to; elsewhere it is `VXT0069`. A + conditional, an `if` expression, a `match` and a loop expression may + yield an enum. +- A method declared with `auto` can be called. Return types are inferred + before any caller is checked, across classes and against declaration + order, through chains and mutual recursion; a method with no result + independent of itself is `VXT0063`, and one whose returns disagree is + `VXT0062`. Such calls used to pass the frontend and fail in the Core + verifier with `VXC1018`. +- A callable may be created inside a callable. The outer one captures what + the inner one reads from further out, and nothing that belongs to the + inner one; it used to capture the inner parameters and fail in the Core + verifier with `VXC1020`. +- `return` is accepted inside a loop used as an expression, and a loop + expression that returns and never breaks is valid. `VXT0045` and `VXT0047` + are retired. +- The return type of a callable is inferred from the returns inside its + expressions as well, up to nested callables. Returns of different types + are `VXT0062`. +- The `else` block of a `guard` is checked by its control flow instead of by + its last statement: a loop that cannot end and a statement `match` that + always selects an arm and all of whose arms leave are accepted, and a + block that leaves before its last statement is as well. +- A type pattern over a scalar applies no numeric conversion: `long n` does + not match an `int`. +- `Spec/Language/Decls.vxs` gains examples 295 to 310 for these rules, for + exhaustiveness and for the statement `match` that selects no arm, and the + grammar allows `-` before a numeric literal in a match pattern. ### Compiler pipeline @@ -47,10 +222,11 @@ SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 expression, `guard` to an `if` with an empty first branch, and a block statement to its statements in the enclosing sequence. Core, its wire format, the optimizer, CorePrep and the native pipeline are unchanged. -- A match with more than 16 arms is lowered in groups of 16 that follow each - other in one statement sequence, with a Boolean slot that records the taken - arm, so the nesting of the lowered Core does not grow with the number of - arms. Matches of 200 and of 2000 arms compile. +- The arms of a match are lowered to one chain, each arm in the false branch + of the one before it, which is the shape of an `else if` chain and is + walked in a loop by every stage. The names the arms bind are bound before + the chain, and a guard is the last operand of the test of its arm. Matches + of 200 and of 2000 arms compile. - Fixed a stack overflow that ended the compiler without a diagnostic on an `else if` chain of about 150 links. Such a chain reaches Core as one level of nesting per link, and the native Core wire reader and writer, the Core @@ -62,15 +238,169 @@ SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 statement. Closure analysis, template discovery, instantiation, freshening, member reachability and the specialization verifier handle them. +- Ownership placement. The compiler created AARC objects and never released + them: no stage wrote a retain or a release. A new Xpp pass gives every AARC + value an owner and writes the operations: parameters are borrowed, results + are owned, a local is released after its last use on each path, including + on the edge of a branch on which it dies, a copy retains or takes over the + reference, and a closure owns its captures. The ownership verifiers of Xpp + and Xmm check its output. +- A method used as a value is a closure without captures. Before, the value + was the address of the method's code, and calling it read an invoke + pointer out of that code. + +- Core has a new primitive, `Memoize`: a callable without parameters whose + result is `bool` or numeric becomes a callable of the same type that calls + it at most once and remembers the result. It is how a value by need + reaches another function. CorePrep, Xpp and Xmm carry it under their own + names, each verifier checks its rule, and LLVM lowers it to an AARC object + with an invoke thunk and a destructor of its own. The wire versions are + now Core 9, CorePrep 7, Xpp 6 and Xmm 6; artifacts of earlier versions are + rejected at the version field and have to be produced again. +- Core has a second new primitive, the runtime call: a call of a function of + the Visual X# runtime, named by a literal first operand that is the + function's identity in a catalog of thirteen. CorePrep, Xpp and Xmm carry + it, every verifier checks a call against the catalog (`VXC1075`, + `VXC0026`, `VXC1076`, `VXP1048`, `VXL1054`), and LLVM lowers it to a call + of the function's symbol, widening an integer or a floating-point argument + to 64 bits. The wire versions are now Core 10, CorePrep 8, Xpp 7 and + Xmm 7. +- The type checker rewrites a console call and the string operators into + runtime calls, so no stage after it knows what a format is. The frontend + infers which methods may write, from all methods of a program together, + and evaluates an expression that calls one where it stands. +- A native executable is linked with `vxs-runtime.lib`, the ownership + runtime and the new text and console runtime as one object that needs no + C runtime. It replaces the ownership functions the compiler generated into + each executable, and with them their limits: memory comes from the heap of + the process, and weak and unowned references, strings and type tests link. + The library stands beside the compiler, where the development helper and + the release bundle put it. It imports seven functions of kernel32, for + which the linker makes the import library from a list of names. +- Fixed a leak: a string literal passed directly as an argument was never + released. Ownership placement now gives such a literal a symbol of its + own. +- A method with a parameter that is passed by need has a second function + that takes suspended computations in place of those parameters. The + method's own function is unchanged, and a call that has nothing worth + suspending still uses it. +- Fixed native executables that stopped instead of ending. A method without + a result whose body ended without a `return` was closed with an + unreachable marker by both Core-to-CorePrep adapters, so a `Main` written + without a final `return;` executed an invalid instruction. Reaching the + end of such a body now returns. +- Fixed native executables that did not link. An executable is linked + without any library, and code that creates a closure calls the ownership + runtime. The module that holds the entry now defines allocation, retain + and release itself, over a 64 MiB arena whose released blocks are reused. + Weak and unowned handles, strings and type tests still need the runtime + library and do not link in an executable. +- Fixed `vxs run -File Name.vxs` with a relative path: the executable it had + just built was looked for on `PATH` and not found. +- Integer division, remainder and shifts are preceded by a check that stops + the program when the operation has no value. The least value of a signed + type divided by minus one is carried out so that it wraps like a sum that + does not fit, where the processor's instruction would stop the program. ### Verification and tooling +- `EffectTests.hs` holds the inference of which calls write from both + sides: programs whose writes lie one and two calls down, in mutual + recursion, in callables and in methods used as values must write in the + order they say, and programs that only compute must keep their deferred + bindings and suspended arguments. `FormatSweepTests.hs` writes every + combination of flags, width and precision for every conversion, 5184 + formats, and compares the reader with the rules stated a second time. + `TextRuntimeTests.cpp` compares `%d`, `%u` and `%x` with the C library of + the host over 44 400 fields of deterministic pseudo-random values, and + `%f` over 3000 numbers drawn from every exponent and 80 020 that lie on + or beside a rounding tie. +- Visual Formatter and Visual Linter are 0.1.1. Neither has a new rule: + both are rebuilt with this compiler, so they accept the forms above, and + their tests now hold them. The formatter keeps `::`, qualified names, + string conditionals and console formats as written while it indents; the + linter reports the new diagnostics under its `compiler` check. +- `QualifiedNameTests.hs` runs programs that name declarations through + their namespace and methods through `::`, before and after optimization, + and holds the forms that must be rejected. The leak-checked smoke program + and the tests that run a linked executable select strings in a loop of a + million passes and call methods through references. +- `ConsoleTests.hs` runs some 180 programs that write in the reference Core + evaluator, before and after optimization, against text written by hand, + and holds about a hundred programs that must be rejected. The evaluator's + text functions are a second implementation, in `RuntimeText.hs`. + `RuntimeCallTests.hs` pins the runtime catalog and the reading of formats. +- `text_runtime_tests` calls the runtime library as generated code does and + compares the digits of `%f` with those of the host's C library. + `RuntimeCallPipelineTests.cpp` and `RuntimeCallVerifierTests.cpp` break a + runtime call in each way it can be broken at each native stage. +- `source_console_smoke` compiles 83 programs that write through LLVM and + the JIT in both pipeline modes, reads their output through a sink, and + requires each to leave no object of the runtime behind. + `executable_run_tests` now runs executables with their standard streams + sent to files and compares the bytes. +- The expression and leaving tables run in a program of their own, + `source_expression_smoke`. With arguments passed by need the programs of + the three execution tables together ran past the 240 second watchdog under + the sanitizers. The watchdog is unchanged and no case was removed. +- `executable_run_tests` builds programs into native executables and runs + them as processes, which no test did before: every other execution test + runs generated code inside the compiler's process. It covers a `Main` + that ends without a return, arguments by need, closures, a quotient by + zero that is needed and one that is not, and two million objects created + and released in a loop. The linker is the Windows one, so these cases run + on Windows. +- `MemoizeTests.hs`, `MemoizePipelineTests.cpp`, `FallThroughTests.hs`, + `FallThroughTests.cpp` and `ComputabilityExecutionTests.cpp` cover the + remembering callable at every stage, the end of a body without a result + in both adapters, and the checks before division and shifts. + `source_feature_smoke` runs 23 programs that pass arguments by need + through LLVM in both pipeline modes and requires each to leave no object + of the runtime behind. +- `OwnershipPlacementTests.cpp` runs every placed function on an independent + reference-count model along all of its paths. `source_feature_smoke` + runs 26 closure programs through LLVM and the AARC runtime, one at a time, + and fails a program that leaves an allocation behind; the runtime counts + its live allocations for that purpose. - `ConditionalChainTests.cpp` pins the native handling of `else if` chains: wire round trips and a constant byte step per link, rejection of every truncated prefix, the verifier's checks in late links, the return analysis, and the block numbering of the adapter, at lengths up to 600 links. `source_fuzz_smoke` runs a chain of 300 links and a match of 200 arms through both pipeline modes. +- `NestingLimitTests.hs` nests every construct that holds statements or + operands to the limit and one level beyond, checks the position of the + diagnostic, and runs the accepted programs. `NestingLimitTests.cpp` pins + the wire statement depth limit and walks 1500 levels and a chain of 3000 + links through the native Core stages on the compiler stack. + `source_fuzz_smoke` runs 255 nested `if` statements + and a sum of 1024 operands through both pipeline modes, and the source + corpus has permanent seeds at and beyond both limits. +- `source_fuzz_smoke` compiles each distinct body of its execution tables + once, up to eight small bodies in a program, and checks all runs of a body + in that program, instead of compiling a program for every run. No run was + removed. Under sanitizers the program takes 160 seconds where it took 291 + on the same machine. +- The execution tables are a smoke program of their own, + `source_execution_smoke`, beside `source_fuzz_smoke`, each under the + unchanged process watchdog of 240 seconds. In the fuzzing configuration + the single program had grown to 221 seconds; apart, each takes 77 to 110 + seconds on the same machine. The new program also compiles and verifies + programs that own closures while control leaves through a value block. +- The tables written by hand for single features, which are inferred return + types, evaluation by need, enums and closures, are a third smoke program, + `source_feature_smoke`. With them `source_execution_smoke` ran past the + unchanged watchdog in the fuzzing configuration; no case was removed. +- An expression that is certain to read a value by need more than once + computes it once ahead of itself. Each read carried the whole computation, + so the Core of a chain of bindings that each read the one before twice + doubled with every link; it now grows with the length of the chain. +- The branching and leaving tables of both harnesses are generated from the + case files under `Compiler/Fuzzing/Cases` by the new Go helper + `execution-cases`, whose `check` command and tests fail on a stale table. + The reference Core evaluator of the frontend tests runs closures, and + `source_feature_smoke` links the AARC runtime and runs closures through + LLVM: created, called, nested, returned and alive across loop transfers. - `BranchingTests.hs` pins the grammar, every typing rule, the lowered shapes and the values of 83 program runs on the unoptimized and the optimized Core. `BranchingOracleTests.hs` generates 36 families of arm lists, writes @@ -91,22 +421,168 @@ SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 extension. The type definitions must match the Node.js runtime the extension runs on, so that update is made by hand. +### Nesting limits + +- Deep nesting no longer ends the compiler with a stack overflow. Before + this change a sum of 100 operands, 100 nested parentheses, an `if` nested + 100 levels deep and an `else if` chain of about 1500 links each overflowed + the one-megabyte stack a process starts with on Windows, with no + diagnostic. +- `vxs` and `vxsi` now run the compiler on a thread whose stack they choose + themselves, 256 MiB of reserved address space, so that how deep a program + may nest is the same on every platform and does not depend on the stack + the operating system gives the process. Pages are committed as they are + used. The fuzz targets and the smoke program use the same stack. Starting + that thread costs a few milliseconds per run: `vxs check` on small + programs measured 3 to 8 ms slower than before on Windows, and a program + of 300 statements measured the same. +- The frontend rejects a function body that nests statements more than 256 + levels deep (`VXP0039`) or expressions more than 1024 levels deep + (`VXP0040`), at the first node that is too deep. An `else if` chain is not + nesting, and the body of a `match` arm is one level below its match + however many arms the match has. A chain of a binary operator is not + nesting either: the left operand of a binary operator is at the level of + the operator, so `a + b + c + ...` is one level however long it is, and a + sum of 50000 operands compiles. The Core wire formats count the first + operand of a primitive at the level of the primitive for the same reason. +- Releasing a Core module no longer recurses along a chain of operators or + an `else if` chain: the destructors of the native Core expression and + statement release their operands and nested statements from a list. The + native wire writer walks operator chains in a loop like the reader. +- The native stages use far less stack per level of nesting. The Core wire + reader, the Core verifier and the Core-to-CorePrep adapter no longer hold + statements, instructions or diagnostic texts in the frames of the + functions that recurse, and they walk chains of operators, of conditional + expressions and of let bindings in a loop. Measured with the new + `stack_probe` program, the whole native pipeline went from 14.9 KiB to + 0.67 KiB of stack per statement level and from 7.1 KiB to 1.9 KiB per + level of nested operands; a sanitizer build uses 1.67 KiB and 3.4 KiB. + Chains of operators and `else if` chains use none per link. Whole + compilations at the frontend's limits commit at most 0.8 MiB of stack for + nested statements and 2.5 MiB for nested expressions, and 1.3 MiB and + 4.6 MiB in a sanitizer build, measured as committed stack, which is an + upper bound to the page and not an exact count of bytes in use, with the + new `source_stack_probe` program. The reservation of 256 MiB costs address space only: the commit + of a compilation is the same with and without it. The measurements, + including what the lowering adds to a nest and what was not measured, are + in `Benchmarks/2026-10-04-Nesting-And-Chains.md`. +- The native Core wire reader and writer bound the nesting of statement + bodies at 4096 levels, as they already bounded expression depth. A `.core` + file nested deeper is rejected as exceeding a limit instead of being + walked. +- The Core-to-CorePrep adapter no longer copies every function before + preparing it. Copying nested statements recurses once per level, which is + what overflowed the stack on long `else if` chains; chains of 5000 links + now compile. + +### Compile time + +- The native Core verifier no longer copies every visible definition when it + enters a branch, a loop body, a let or a closure. It did, so its time grew + with the square of a function's size: a function of 400 calls that each + pass four suspended arguments took 2.05 seconds to verify and takes 0.04. +- Nested loops no longer multiply the time of the integer analysis. It + repeated the fixed point of an inner loop on every pass over the loop + around it, so 14 nested loops took a second and 50 did not finish. It now + iterates only loops that hold at most one further level of loops and + treats what a deeper loop assigns as unknown; 255 nested loops compile in + under four seconds. +- The Haskell Core wire encoder, the Haskell CorePrep lowering and the + CorePrep verifier no longer copy what they have produced once per level + of nesting or per block. On an `else if` chain of 2048 links, encoding + went from 4.9 seconds to 60 milliseconds, lowering from 2.2 seconds to 20 + milliseconds and verification from 0.75 seconds to 15 milliseconds. +- `Compiler/Haskell/Core/Benches` has benchmarks for nested loops, `else if` + chains and sequences of `if` statements. +- Long chains of operators no longer take quadratic time or worse. Constant + propagation asked the integer facts about every node of an expression, + which re-evaluated the operands at each level, and the CorePrep lowering + collected symbol identities by appending lists. A sum of 20000 operands + went from 30 seconds to 4.6, and one of 50000 from 198 seconds to 9.4. + The truth of a condition is now looked up in the facts only while the + condition has at most 256 nodes; a chain of 200 comparisons joined by `&&` + went from 47 seconds to 1.1, on `main` as well as on this branch. + +- A class with many methods no longer compiles in time with the square of + their number. 1000 small methods took 4.3 seconds, 2000 took 16 and 4000 + took 73; they now take 1.4, 2.9 and 4.8 seconds, and 8000 take 10. The + native CorePrep verifier built a table of every function of the module, + with the function type of each, once for every function, and searched the + module for capturing closures as often; it collects both once. The + renamer kept the names in scope as a list of pairs and now keeps a map. + The type checker compared a method with every earlier member to find a + duplicate overload and now compares it with the methods of its name. The + Haskell Core verifier searched a list of source owners for every + function, and the CorePrep lowering re-wrapped the list of waiting + functions once for every function. What each stage accepts and reports is + unchanged. `Benchmarks/2026-10-07-Many-Methods.md` has the measurements, + `CorePrepVerifierTests.cpp` pins what a function sees of its module, and + the CorePrep benchmarks have a case by number of functions. + ### Known limitations -- An `else if` chain of about 2000 links still overflows the native stack, - now while a stage copies the nested Core statements, and so do statements - nested in any other way, such as an `if` inside the first branch of an `if`, - repeated: 60 levels compile and 100 do not. The compiler then ends without a - diagnostic. Neither is new in this version, and `match` is not affected. - The Core wire format bounds type and expression depth but has no statement - depth limit yet. +- An argument passed by need costs two objects of the runtime and two + functions of generated code where it is suspended. A program whose calls + pass many arguments that are themselves calls compiles to several times + the code it did before and takes correspondingly longer to compile. +- A program that stops because it needs a value that cannot be computed + ends with the status of a process that executed an invalid instruction. + It reports nothing about which value or where. +- Console input, the stream objects `Console.Stdout()` and its siblings, + `%A` and `%O`, a text form of a floating-point number outside `%f`, and + 128-bit numbers as text are specified and not implemented; each reports + `VXT0079`. `Documents/CONSOLE-IO.md` lists them. +- A string has `+`, `+=`, `==` and `\=` and nothing else yet: no length, no + indexing and no ordering. +- Native executables are linked on Windows only. +- Compile time still grows faster than the program on very long functions: + from 2000 to 4000 `else if` links the time of `vxs check` grows from 3.3 + to 10.4 seconds, and from 2000 to 4000 sequential `if` statements from 2.8 + to 7.1 seconds. Nothing bounds the number of statements of a function. +- Copying a native Core module recurses once per level of nesting; the + pipeline copies only closure bodies. +- Parts of the specified language are recognized and rejected as not + implemented. They are pending work, listed with their diagnostics under + "Pending branching and loop forms" in `Documents/IMPLEMENTATION.md`: a + binding in the condition of `if`, `guard` or `while`, which needs optional + values, a call that does not return as a way of leaving, enums declared + inside a class, and `enum class`. +- The nesting limits of 256 and 1024 and the compiler stack reservation of + 256 MiB are the limits this version ships with. `CommittedStackBytes` + reports on Linux and macOS as well, from the resident pages of the stack + mapping, and `source_execution_smoke` prints the figure after compiling a + program of 1023 nested calls. Measured in CI: about 2.5 MiB in an ordinary + build on Windows, Linux and macOS, and at most 5.5 MiB under a sanitizer. +- The statements after a statement that never completes are checked but no + longer lowered to Core. ### Upgrading from 0.4.1 -- Rename anything called `match` or `guard`. +- A method that writes and is used as a value, `auto f = Log;` or + `Run(Log)`, writes where its call stands, like a callable expression that + writes. No release behaved otherwise; it is listed because the rule in + `Documents/CONSOLE-IO.md` names it. +- A program that relied on an unused value being computed, for its failure + or for the time it takes, no longer gets either: a value nothing reads is + not computed. That now includes an argument the method does not read. +- Core, CorePrep, Xpp and Xmm artifacts written by an earlier compiler are + rejected. Build them again from source. +- `==` on two strings now compares their characters. A program that relied + on two equal strings comparing unequal because they were two objects + compares them equal. +- A program named a class `Console` or `System` keeps its own: the names the + language declares stand outside the program's. +- Rename anything called `match`, `guard` or `enum`. - `if (auto name = value)` and the same form in `while` now report `VXP0035` instead of a generic syntax error. They were not accepted before either. +### Release + +- Advanced compiler-owned Haskell packages, the CLI, Bazel module, Kotlin + project model, and compiler project version to 0.5.0, and the bounds of + the formatter, linter and analyzer packages on the compiler to + `>=0.5.0 && <0.6`. + ## 0.4.1 - 2026-10-03 ### Language diff --git a/Compiler/Backend/LLVM/BUILD.bazel b/Compiler/Backend/LLVM/BUILD.bazel index 4ee4e27e..2350a7da 100644 --- a/Compiler/Backend/LLVM/BUILD.bazel +++ b/Compiler/Backend/LLVM/BUILD.bazel @@ -5,7 +5,9 @@ package(default_visibility = ["//visibility:public"]) cc_library( name = "llvm_backend", srcs = ["Artifact.cpp", "Codegen.cpp", "JitSession.cpp", "Verifier.cpp"], - textual_hdrs = ["CodegenOperations.inc"], + textual_hdrs = [ + "CodegenOperations.inc", + ], deps = [ "//Compiler/Codegen/Xmm:xmm", "//Compiler/Core/Callable:callable", diff --git a/Compiler/Backend/LLVM/Codegen.cpp b/Compiler/Backend/LLVM/Codegen.cpp index 958da26a..2c12b285 100644 --- a/Compiler/Backend/LLVM/Codegen.cpp +++ b/Compiler/Backend/LLVM/Codegen.cpp @@ -38,6 +38,7 @@ #include "Visual/XSharp/Backend/LLVM.hpp" #include "Visual/XSharp/Core/Callable.hpp" #include "Visual/XSharp/Core/Ownership.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" #include "Visual/XSharp/Core/Scalar.hpp" namespace Visual::XSharp::Backend::LLVM @@ -898,6 +899,8 @@ namespace Visual::XSharp::Backend::LLVM // the owning generator retains one shared LLVM context and state. #include "CodegenOperations.inc" + // The ownership runtime an executable carries in its own module. + [[nodiscard]] auto DefineFunction(FunctionState &state) -> bool { @@ -970,6 +973,8 @@ namespace Visual::XSharp::Backend::LLVM builder.CreateCall(entry->type, entry->value); builder.CreateRet( llvm::ConstantInt::get(llvm::Type::getInt32Ty(context), 0)); + // What the program calls of the runtime, it finds in the + // runtime library the executable is linked with. return true; } }; diff --git a/Compiler/Backend/LLVM/CodegenOperations.inc b/Compiler/Backend/LLVM/CodegenOperations.inc index a38a4aec..2cbd631b 100644 --- a/Compiler/Backend/LLVM/CodegenOperations.inc +++ b/Compiler/Backend/LLVM/CodegenOperations.inc @@ -100,12 +100,31 @@ CreateClosureThunk(llvm::StructType *payload, return thunk; } +/// Whether destroying a closure has anything to release: a strong capture +/// of an object, or a weak or unowned handle. A closure over numbers and +/// Booleans alone owns nothing. +[[nodiscard]] static auto +ClosureOwnsCaptures(const xmm::Instruction &instruction, + const std::vector &captureTypes) -> bool +{ + for (std::size_t index = 0; index < instruction.operands.size(); ++index) + if (instruction.capture_modes[index] != core::CaptureMode::Strong + || captureTypes[index]->isPointerTy()) + return true; + return false; +} + +/// The destructor of a closure, or null for a closure that owns nothing. +/// The runtime runs no destructor for an object whose metadata names none, +/// so such a closure costs no function of its own. [[nodiscard]] auto CreateClosureDestructor(llvm::StructType *payload, const xmm::Instruction &instruction, const std::vector &captureTypes) -> llvm::Function * { + if (!ClosureOwnsCaptures(instruction, captureTypes)) + return nullptr; auto *pointer = llvm::PointerType::get(context, 0); auto *type = llvm::FunctionType::get(llvm::Type::getVoidTy(context), { pointer }, @@ -151,61 +170,49 @@ CreateClosureDestructor(llvm::StructType *payload, return destructor; } +/// The LLVM type of the metadata record an allocation is described by; it +/// mirrors `VxsAarcTypeMetadata` of the public runtime ABI. [[nodiscard]] auto -LowerClosure(llvm::IRBuilder<> &builder, - FunctionState &state, - const xmm::Instruction &instruction) -> llvm::Value * +ObjectMetadataType() -> llvm::StructType * { - const auto target = functions.find(instruction.closure_function); - if (target == functions.end() - || instruction.capture_modes.size() != instruction.operands.size()) - return nullptr; - auto *pointer = llvm::PointerType::get(context, 0); - std::vector captureTypes; - captureTypes.reserve(instruction.operands.size()); - std::vector fields{ pointer }; - for (std::size_t index = 0; index < instruction.operands.size(); ++index) - { - auto *type - = instruction.capture_modes[index] == core::CaptureMode::Strong - ? types.Lower(instruction.operands[index].type) - : pointer; - if (type == nullptr || type->isVoidTy()) - return nullptr; - captureTypes.push_back(type); - fields.push_back(type); - } - auto *payload = llvm::StructType::create( - context, - fields, - ".vxs.aarc.closure.payload." + std::to_string(closure_index)); - auto *thunk = CreateClosureThunk(payload, - instruction, - captureTypes, - target->second); - auto *destructor - = CreateClosureDestructor(payload, instruction, captureTypes); - if (thunk == nullptr || destructor == nullptr) - return nullptr; + auto *sizeType = llvm::Triple(module.getTargetTriple()).isArch64Bit() + ? llvm::Type::getInt64Ty(context) + : llvm::Type::getInt32Ty(context); + std::array fields{ llvm::Type::getInt32Ty(context), + llvm::Type::getInt32Ty(context), + llvm::Type::getInt64Ty(context), + sizeType, + sizeType, + pointer, + pointer }; + return llvm::StructType::get(context, fields, false); +} - // TypeMetadata mirrors the public runtime ABI. Size is - // expressed as an LLVM constant expression so the target data - // layout, not the host compiler, determines the closure payload - // size. +/// Allocates an AARC object with the given payload and destructor. +/// +/// TypeMetadata mirrors the public runtime ABI. Size is expressed as an +/// LLVM constant expression so the target data layout, not the host +/// compiler, determines the payload size. A null destructor describes an +/// object that owns nothing. A metadata record that the module already +/// holds under the given name is used again. +[[nodiscard]] auto +AllocateObject(llvm::IRBuilder<> &builder, + llvm::StructType *payload, + llvm::Function *destructor, + const std::string &metadataName, + const char *valueName) -> llvm::Value * +{ + auto *pointer = llvm::PointerType::get(context, 0); + if (auto *known = module.getNamedGlobal(metadataName)) + return builder.CreateCall( + RuntimeFunction("vxs_aarc_allocate", pointer, { pointer }), + { known }, + valueName); auto *sizeType = llvm::Triple(module.getTargetTriple()).isArch64Bit() ? llvm::Type::getInt64Ty(context) : llvm::Type::getInt32Ty(context); - std::array metadataFields{ - llvm::Type::getInt32Ty(context), - llvm::Type::getInt32Ty(context), - llvm::Type::getInt64Ty(context), - sizeType, - sizeType, - pointer, - pointer - }; - auto *metadataType = llvm::StructType::get(context, metadataFields, false); + auto *metadataType = ObjectMetadataType(); auto *payloadSize = llvm::ConstantExpr::getSizeOf(payload); if (payloadSize->getType() != sizeType) { @@ -226,7 +233,9 @@ LowerClosure(llvm::IRBuilder<> &builder, llvm::ConstantInt::get(llvm::Type::getInt64Ty(context), 0U), payloadSize, llvm::ConstantInt::get(sizeType, 16U), - destructor, + destructor != nullptr ? static_cast(destructor) + : static_cast( + llvm::ConstantPointerNull::get(pointer)), llvm::ConstantPointerNull::get(pointer) }; auto *metadata = new llvm::GlobalVariable( @@ -235,11 +244,203 @@ LowerClosure(llvm::IRBuilder<> &builder, true, llvm::GlobalValue::PrivateLinkage, llvm::ConstantStruct::get(metadataType, metadataValues), - ".vxs.aarc.closure.metadata." + std::to_string(closure_index++)); - auto *object = builder.CreateCall( + metadataName); + return builder.CreateCall( RuntimeFunction("vxs_aarc_allocate", pointer, { pointer }), { metadata }, - "closure"); + valueName); +} + +/// Lowers a callable that remembers its result. +/// +/// The value is an object of its own around the callable that computes the +/// result. Like every closure it starts with its invoke entry, so a call +/// site calls it as it calls any callable. Behind the entry it keeps +/// whether it has a result, the result, and the callable. The entry calls +/// the callable when there is no result yet, keeps what it returned, and +/// returns the kept result from then on. Every holder of the object sees +/// the same result, which is what lets a caller and the method it calls +/// share one computation. +/// +/// The object owns the callable and releases it when it is destroyed. The +/// result owns nothing: the verifiers admit only Bool and numeric results. +/// A call that does not return leaves the object without a result. +/// +/// Nothing in the object's code depends on which computation it holds, so a +/// module has one payload type, one entry, one destructor and one metadata +/// record for each result type, however many values it suspends. +[[nodiscard]] auto +LowerMemoize(llvm::IRBuilder<> &builder, + FunctionState &state, + const xmm::Instruction &instruction) -> llvm::Value * +{ + const auto signature + = ::Visual::XSharp::Core::Callable::Decompose(instruction.result_type); + if (!signature || !signature->parameters.empty() + || instruction.operands.size() != 1U) + return nullptr; + auto *resultType = types.Lower(signature->result); + if (resultType == nullptr || resultType->isVoidTy()) + return nullptr; + auto *computation = LoadValue(builder, state, instruction.operands.front()); + if (computation == nullptr || !computation->getType()->isPointerTy()) + return nullptr; + + constexpr unsigned kInvoke = 0U; + constexpr unsigned kKnown = 1U; + constexpr unsigned kResult = 2U; + constexpr unsigned kComputation = 3U; + std::string index; + { + llvm::raw_string_ostream spelling(index); + resultType->print(spelling); + } + auto *pointer = llvm::PointerType::get(context, 0); + auto *flagType = llvm::Type::getInt8Ty(context); + auto *payload + = llvm::StructType::getTypeByName(context, + ".vxs.aarc.memo.payload." + index); + if (payload == nullptr) + payload = llvm::StructType::create( + context, + { pointer, flagType, resultType, pointer }, + ".vxs.aarc.memo.payload." + index); + + auto *invokeType = llvm::FunctionType::get(resultType, { pointer }, false); + auto *invoke = module.getFunction(".vxs.aarc.memo.invoke." + index); + if (invoke == nullptr) + { + invoke = llvm::Function::Create(invokeType, + llvm::GlobalValue::InternalLinkage, + ".vxs.aarc.memo.invoke." + index, + module); + auto *entry = llvm::BasicBlock::Create(context, "entry", invoke); + auto *compute = llvm::BasicBlock::Create(context, "compute", invoke); + auto *done = llvm::BasicBlock::Create(context, "done", invoke); + auto *self = invoke->getArg(0U); + llvm::IRBuilder<> body(entry); + auto *known + = body.CreateLoad(flagType, + body.CreateStructGEP(payload, self, kKnown), + "memo.known"); + body.CreateCondBr( + body.CreateICmpNE(known, llvm::ConstantInt::get(flagType, 0U)), + done, + compute); + + body.SetInsertPoint(compute); + auto *held + = body.CreateLoad(pointer, + body.CreateStructGEP(payload, self, kComputation), + "memo.computation"); + // The callable is a closure: its payload begins with its own invoke + // entry, which takes the closure and returns the result. + auto *target = body.CreateLoad(pointer, held, "memo.target"); + auto *computed + = body.CreateCall(invokeType, target, { held }, "memo.computed"); + body.CreateStore(computed, + body.CreateStructGEP(payload, self, kResult)); + body.CreateStore(llvm::ConstantInt::get(flagType, 1U), + body.CreateStructGEP(payload, self, kKnown)); + body.CreateBr(done); + + body.SetInsertPoint(done); + body.CreateRet( + body.CreateLoad(resultType, + body.CreateStructGEP(payload, self, kResult), + "memo.result")); + } + + auto *destructor = module.getFunction(".vxs.aarc.memo.destroy." + index); + if (destructor == nullptr) + { + destructor = llvm::Function::Create( + llvm::FunctionType::get(llvm::Type::getVoidTy(context), + { pointer }, + false), + llvm::GlobalValue::InternalLinkage, + ".vxs.aarc.memo.destroy." + index, + module); + llvm::IRBuilder<> body( + llvm::BasicBlock::Create(context, "entry", destructor)); + auto *held + = body.CreateLoad(pointer, + body.CreateStructGEP(payload, + destructor->getArg(0U), + kComputation)); + body.CreateCall(RuntimeFunction("vxs_aarc_release_strong", + llvm::Type::getVoidTy(context), + { pointer }), + { held }); + body.CreateRetVoid(); + } + + auto *object = AllocateObject(builder, + payload, + destructor, + ".vxs.aarc.memo.metadata." + index, + "memo"); + builder.CreateStore(invoke, + builder.CreateStructGEP(payload, object, kInvoke)); + builder.CreateStore(llvm::ConstantInt::get(flagType, 0U), + builder.CreateStructGEP(payload, object, kKnown)); + builder.CreateStore(llvm::Constant::getNullValue(resultType), + builder.CreateStructGEP(payload, object, kResult)); + // The operand stays with its own owner; the object takes a reference + // of its own. + auto *owned = builder.CreateCall( + RuntimeFunction("vxs_aarc_retain_strong", pointer, { pointer }), + { computation }, + "memo.owned"); + builder.CreateStore(owned, + builder.CreateStructGEP(payload, object, kComputation)); + return object; +} + +[[nodiscard]] auto +LowerClosure(llvm::IRBuilder<> &builder, + FunctionState &state, + const xmm::Instruction &instruction) -> llvm::Value * +{ + const auto target = functions.find(instruction.closure_function); + if (target == functions.end() + || instruction.capture_modes.size() != instruction.operands.size()) + return nullptr; + + auto *pointer = llvm::PointerType::get(context, 0); + std::vector captureTypes; + captureTypes.reserve(instruction.operands.size()); + std::vector fields{ pointer }; + for (std::size_t index = 0; index < instruction.operands.size(); ++index) + { + auto *type + = instruction.capture_modes[index] == core::CaptureMode::Strong + ? types.Lower(instruction.operands[index].type) + : pointer; + if (type == nullptr || type->isVoidTy()) + return nullptr; + captureTypes.push_back(type); + fields.push_back(type); + } + auto *payload = llvm::StructType::create( + context, + fields, + ".vxs.aarc.closure.payload." + std::to_string(closure_index)); + auto *thunk = CreateClosureThunk(payload, + instruction, + captureTypes, + target->second); + auto *destructor + = CreateClosureDestructor(payload, instruction, captureTypes); + if (thunk == nullptr) + return nullptr; + + auto *object = AllocateObject(builder, + payload, + destructor, + ".vxs.aarc.closure.metadata." + + std::to_string(closure_index++), + "closure"); builder.CreateStore(thunk, builder.CreateStructGEP(payload, object, 0U)); for (std::size_t index = 0; index < instruction.operands.size(); ++index) @@ -273,11 +474,238 @@ LowerClosure(llvm::IRBuilder<> &builder, return object; } +// Values that cannot be computed. +// +// An integer quotient or remainder by zero, and a shift by an amount that is +// negative or not less than the width of the shifted value, have no value. +// A program that needs one fails: it stops there. LLVM gives the same +// instructions no meaning at all, so without a check an optimizer may remove +// the computation, or the code after it, as if the program had never needed +// the value. Each such instruction is therefore preceded by a check that +// stops the program, and the instruction itself only ever runs on operands +// it has a value for. + +/// The function that stops the program when its two operands are equal or, +/// with `below` unset, when the first is not below the second as unsigned +/// numbers. It is small and always inlined, so that the check costs a +/// comparison where the operand is not known and nothing where it is. +[[nodiscard]] auto +ComputabilityCheck(llvm::IntegerType *type, bool below) -> llvm::Function * +{ + const auto name + = std::string(below ? ".vxs.check.below.i" : ".vxs.check.differs.i") + + std::to_string(type->getBitWidth()); + if (auto *existing = module.getFunction(name)) + return existing; + auto *function = llvm::Function::Create( + llvm::FunctionType::get(llvm::Type::getVoidTy(context), + { type, type }, + false), + llvm::GlobalValue::InternalLinkage, + name, + module); + function->addFnAttr(llvm::Attribute::AlwaysInline); + function->addFnAttr(llvm::Attribute::NoUnwind); + auto *entry = llvm::BasicBlock::Create(context, "entry", function); + auto *fails = llvm::BasicBlock::Create(context, "fails", function); + auto *computable + = llvm::BasicBlock::Create(context, "computable", function); + llvm::IRBuilder<> body(entry); + auto *holds + = below ? body.CreateICmpULT(function->getArg(0U), function->getArg(1U)) + : body.CreateICmpNE(function->getArg(0U), function->getArg(1U)); + body.CreateCondBr(holds, computable, fails); + body.SetInsertPoint(fails); + body.CreateCall( + llvm::Intrinsic::getOrInsertDeclaration(&module, + llvm::Intrinsic::trap)); + body.CreateUnreachable(); + body.SetInsertPoint(computable); + body.CreateRetVoid(); + return function; +} + +/// Stops the program when an integer divisor is zero, and gives the divisor +/// the division is carried out with. +/// +/// The least value of a signed type divided by minus one has a quotient one +/// above the greatest value of the type. That is a result that does not fit, +/// not a value that cannot be computed, and generated code lets every other +/// integer result that does not fit wrap. Wrapped, the quotient is the least +/// value again and the remainder is zero, which is what dividing by one +/// gives; the division is carried out with one. The processor's own +/// instruction would stop the program instead. +[[nodiscard]] auto +ComputableDivisor(llvm::IRBuilder<> &builder, + llvm::Value *dividend, + llvm::Value *divisor, + bool isSigned) -> llvm::Value * +{ + auto *type = llvm::cast(divisor->getType()); + builder.CreateCall(ComputabilityCheck(type, false), + { divisor, llvm::ConstantInt::get(type, 0U) }); + if (!isSigned) + return divisor; + auto *least = llvm::ConstantInt::get( + type, + llvm::APInt::getSignedMinValue(type->getBitWidth())); + auto *wraps = builder.CreateAnd( + builder.CreateICmpEQ(dividend, least), + builder.CreateICmpEQ(divisor, llvm::ConstantInt::getSigned(type, -1)), + "quotient.wraps"); + return builder.CreateSelect(wraps, + llvm::ConstantInt::get(type, 1U), + divisor, + "divisor"); +} + +/// Stops the program when a shift amount is negative or not less than the +/// width of the shifted value. +void +RequireShiftAmount(llvm::IRBuilder<> &builder, llvm::Value *amount) +{ + auto *type = llvm::cast(amount->getType()); + builder.CreateCall( + ComputabilityCheck(type, true), + { amount, llvm::ConstantInt::get(type, type->getBitWidth()) }); +} + +/// Lowers a call of a function of the runtime. +/// +/// The verifiers have checked the call against its row of the catalog. An +/// argument that the row gives a family of types is widened here to the one +/// representation the runtime function is written for: a signed integer is +/// sign-extended and an unsigned one zero-extended to sixty-four bits, and a +/// floating-point number is extended to a `double`. None of these changes +/// the value. A string result arrives with a reference the caller owns; when +/// nothing takes it, it is released here. A Boolean result arrives as the +/// byte the C ABI returns and is narrowed to the one bit a Bool is here. +[[nodiscard]] auto +LowerRuntimeCall(llvm::IRBuilder<> &builder, + FunctionState &state, + const xmm::Instruction &instruction) -> bool +{ + namespace runtime = core::runtime; + if (instruction.operands.empty() + || instruction.operands.front().kind != xmm::Value::Kind::Immediate) + return false; + const auto identity + = runtime::IdentityOf(instruction.operands.front().immediate, + instruction.operands.front().type); + const auto *signature = identity ? runtime::Find(*identity) : nullptr; + if (signature == nullptr + || instruction.operands.size() != signature->parameters.size() + 1U) + return false; + + auto *pointer = llvm::PointerType::get(context, 0); + auto *word = llvm::Type::getInt64Ty(context); + std::vector parameterTypes; + std::vector arguments; + parameterTypes.reserve(signature->parameters.size()); + arguments.reserve(signature->parameters.size()); + for (std::size_t index = 0U; index < signature->parameters.size(); ++index) + { + auto *value + = LoadValue(builder, state, instruction.operands[index + 1U]); + if (value == nullptr) + return false; + llvm::Type *expected = word; + switch (signature->parameters[index]) + { + case runtime::Parameter::Signed: + case runtime::Parameter::Count: + if (!value->getType()->isIntegerTy()) + return false; + value + = builder.CreateSExtOrTrunc(value, word, "runtime.signed"); + break; + case runtime::Parameter::Unsigned: + if (!value->getType()->isIntegerTy()) + return false; + value = builder.CreateZExtOrTrunc(value, + word, + "runtime.unsigned"); + break; + case runtime::Parameter::Floating: + expected = llvm::Type::getDoubleTy(context); + if (!value->getType()->isFloatingPointTy()) + return false; + if (value->getType() != expected) + value = builder.CreateFPExt(value, + expected, + "runtime.floating"); + break; + case runtime::Parameter::Bool: + // The C ABI passes a `bool` as a byte that is zero or one. + expected = llvm::Type::getInt8Ty(context); + if (!value->getType()->isIntegerTy()) + return false; + value = builder.CreateZExtOrTrunc(value, + expected, + "runtime.bool"); + break; + case runtime::Parameter::Char: + expected = llvm::Type::getInt32Ty(context); + if (!value->getType()->isIntegerTy()) + return false; + value = builder.CreateZExtOrTrunc(value, + expected, + "runtime.char"); + break; + case runtime::Parameter::Text: + expected = pointer; + if (!value->getType()->isPointerTy()) + return false; + break; + } + parameterTypes.push_back(expected); + arguments.push_back(value); + } + + const auto yieldsText = signature->result == runtime::Result::Text; + const auto yieldsTruth = signature->result == runtime::Result::Truth; + auto callee = module.getOrInsertFunction( + std::string(signature->symbol), + llvm::FunctionType::get(yieldsText ? pointer + : yieldsTruth ? llvm::Type::getInt8Ty(context) + : llvm::Type::getVoidTy(context), + parameterTypes, + false)); + auto *result = builder.CreateCall(callee, + arguments, + yieldsText ? "runtime.text" + : yieldsTruth ? "runtime.truth" + : ""); + if (yieldsTruth) + { + if (instruction.has_result) + builder.CreateStore( + builder.CreateICmpNE( + result, + llvm::ConstantInt::get(result->getType(), 0U), + "runtime.bool"), + state.slots.at(instruction.destination)); + return true; + } + if (!yieldsText) + return true; + if (instruction.has_result) + builder.CreateStore(result, state.slots.at(instruction.destination)); + else + builder.CreateCall(RuntimeFunction("vxs_aarc_release_strong", + llvm::Type::getVoidTy(context), + { pointer }), + { result }); + return true; +} + [[nodiscard]] auto LowerInstruction(llvm::IRBuilder<> &builder, FunctionState &state, const xmm::Instruction &instruction) -> bool { + if (instruction.opcode == xmm::Opcode::RuntimeCall) + return LowerRuntimeCall(builder, state, instruction); if (instruction.opcode == xmm::Opcode::Call) { auto *result = LowerCall(builder, state, instruction); @@ -288,9 +716,12 @@ LowerInstruction(llvm::IRBuilder<> &builder, state.slots.at(instruction.destination)); return true; } - if (instruction.opcode == xmm::Opcode::MakeClosure) + if (instruction.opcode == xmm::Opcode::MakeClosure + || instruction.opcode == xmm::Opcode::Memoize) { - auto *result = LowerClosure(builder, state, instruction); + auto *result = instruction.opcode == xmm::Opcode::Memoize + ? LowerMemoize(builder, state, instruction) + : LowerClosure(builder, state, instruction); if (result == nullptr) return false; if (instruction.has_result) @@ -298,11 +729,11 @@ LowerInstruction(llvm::IRBuilder<> &builder, state.slots.at(instruction.destination)); else { - // Closure construction retains every strong capture. A - // discarded source closure still performs those - // observable ownership actions, but its temporary owner - // must be released immediately or a dead expression - // leaks both the environment and its captured values. + // Closure construction retains every strong capture, and a + // remembering callable retains its computation. A discarded + // one still performs those observable ownership actions, but + // its temporary owner must be released immediately or a dead + // expression leaks both the object and what it holds. builder.CreateCall( RuntimeFunction("vxs_aarc_release_strong", llvm::Type::getVoidTy(context), @@ -332,6 +763,24 @@ LowerInstruction(llvm::IRBuilder<> &builder, = core::is_unsigned_integer(operandType) || operandType.kind == core::Type::Kind::Character; switch (instruction.opcode) + { + case xmm::Opcode::Divide: + case xmm::Opcode::FloorDivide: + case xmm::Opcode::Remainder: + if (!floating) + operands[1] = ComputableDivisor(builder, + operands[0], + operands[1], + !unsignedInteger); + break; + case xmm::Opcode::ShiftLeft: + case xmm::Opcode::ShiftRight: + RequireShiftAmount(builder, operands[1]); + break; + default: + break; + } + switch (instruction.opcode) { case xmm::Opcode::LoadImmediate: case xmm::Opcode::Move: @@ -497,6 +946,8 @@ LowerInstruction(llvm::IRBuilder<> &builder, break; case xmm::Opcode::Call: case xmm::Opcode::MakeClosure: + case xmm::Opcode::Memoize: + case xmm::Opcode::RuntimeCall: case xmm::Opcode::RetainStrong: case xmm::Opcode::ReleaseStrong: case xmm::Opcode::MakeWeak: diff --git a/Compiler/Backend/LLVM/Tests/BUILD.bazel b/Compiler/Backend/LLVM/Tests/BUILD.bazel index ac678dee..dc24f5ae 100644 --- a/Compiler/Backend/LLVM/Tests/BUILD.bazel +++ b/Compiler/Backend/LLVM/Tests/BUILD.bazel @@ -6,6 +6,7 @@ cc_binary( name = "llvm_backend_tests", srcs = [ "CallableInvocationTests.cpp", + "ComputabilityExecutionTests.cpp", "ArithmeticExecutionTests.cpp", "ConditionalExecutionTests.cpp", "LLVMBackendTests.cpp", diff --git a/Compiler/Backend/LLVM/Tests/CallableInvocationTests.cpp b/Compiler/Backend/LLVM/Tests/CallableInvocationTests.cpp index bff20ef2..5e66722b 100644 --- a/Compiler/Backend/LLVM/Tests/CallableInvocationTests.cpp +++ b/Compiler/Backend/LLVM/Tests/CallableInvocationTests.cpp @@ -178,10 +178,16 @@ TEST_CASE("closure destruction remains independent from invocation") REQUIRE(result); const auto &ir = result.artifact->llvm_ir; - CHECK(Contains(ir, ".vxs.aarc.closure.destroy")); CHECK(Contains(ir, "@vxs_aarc_allocate")); // The capture is a scalar, so destruction must not invent an AARC release. CHECK_FALSE(Contains(ir, "call void @vxs_aarc_release_strong")); + // A closure that owns nothing has nothing to do when it is destroyed: + // its metadata names no destructor, which the runtime accepts, and the + // module holds no function for it. The entry it is called through is + // still its own. + CHECK_FALSE(Contains(ir, ".vxs.aarc.closure.destroy")); + CHECK(Contains(ir, ".vxs.aarc.closure.invoke")); + CHECK(Contains(ir, "i64 16, ptr null, ptr null }")); } TEST_CASE("Xmm preserves direct and indirect callees as different value kinds") diff --git a/Compiler/Backend/LLVM/Tests/ComputabilityExecutionTests.cpp b/Compiler/Backend/LLVM/Tests/ComputabilityExecutionTests.cpp new file mode 100644 index 00000000..489a5e74 --- /dev/null +++ b/Compiler/Backend/LLVM/Tests/ComputabilityExecutionTests.cpp @@ -0,0 +1,271 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// Values that cannot be computed, and the one that only seems not to be. +// +// An integer quotient or remainder by zero has no value, and neither has a +// shift by an amount that is negative or not less than the width of the +// shifted value. A program that needs such a value stops. The machine +// instructions for these operations give no such promise: LLVM assigns them +// no meaning on those operands, and an optimizer may then delete the +// computation together with whatever depended on it. The backend therefore +// checks the operand first. These cases pin that the check is there, that +// it does not disturb the values that can be computed, and that the least +// integer divided by minus one, which a processor refuses, wraps the way +// generated code lets every other integer result that does not fit wrap. +// The language has not fixed what such a result is; this pins that the +// backend treats the division like the sum and the product. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Pipeline = Visual::XSharp::Pipeline; + + constexpr auto kLeast = std::numeric_limits::min(); + constexpr auto kGreatest = std::numeric_limits::max(); + + /// `int left = ; int right = ; return left right;` + /// The operands are mutable locals so the operation is lowered as a + /// run-time instruction in the unoptimized pipeline. + [[nodiscard]] auto + Binary(Core::Primitive operation, std::int64_t left, std::int64_t right) + -> Core::Module + { + const auto local = + [](std::uint64_t id, std::u32string name, std::int64_t value) { + return Core::Statement::Bind( + { { id, std::move(name) }, + Core::Type::int64(), + true, + Core::Expression::Constant(value, Core::Type::int64()) }); + }; + const auto read = [](std::uint64_t id, std::u32string name) { + return Core::Expression::Variable({ id, std::move(name) }, + Core::Type::int64()); + }; + return { + { U"Computability" }, + { Core::Function{ + { 1U, U"Evaluate" }, + {}, + Core::Type::int64(), + { local(2U, U"left", left), + local(3U, U"right", right), + Core::Statement::Return(Core::Expression::InvokePrimitive( + operation, + { read(2U, U"left"), read(3U, U"right") }, + Core::Type::int64())) }, + } } + }; + } + + [[nodiscard]] auto + Lower(const Core::Module &module, bool optimize) -> Llvm::Artifact + { + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + Pipeline::Options options; + options.optimize_xpp = optimize; + options.optimize_xmm = optimize; + options.llvm.optimization = optimize ? Llvm::OptimizationLevel::Default + : Llvm::OptimizationLevel::Debug; + auto pipeline = Pipeline::ConsumeCore(encoded.bytes, options); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + return *pipeline.llvm; + } + + [[nodiscard]] auto + Run(const Core::Module &module, bool optimize) -> std::int64_t + { + const auto artifact = Lower(module, optimize); + constexpr std::string_view kSymbol = "Computability.Evaluate.1"; + Llvm::JitSession session; + REQUIRE_FALSE(session.AddModule(artifact.bitcode, + "computability-execution", + kSymbol, + Core::Type::int64())); + const auto result = session.InvokeScalar(kSymbol, Core::Type::int64()); + REQUIRE(result); + return std::get(result.value->payload); + } + + void + CheckBothPipelines(Core::Primitive operation, + std::int64_t left, + std::int64_t right, + std::int64_t expected) + { + CAPTURE(left, right); + const auto module = Binary(operation, left, right); + CHECK(Run(module, false) == expected); + CHECK(Run(module, true) == expected); + } + + /// Whether the code of the operation can stop the program. + [[nodiscard]] auto + Stops(Core::Primitive operation, + std::int64_t left, + std::int64_t right, + bool optimize) -> bool + { + const auto artifact = Lower(Binary(operation, left, right), optimize); + return artifact.llvm_ir.find("call void @llvm.trap()") + != std::string::npos; + } +} // namespace + +TEST_CASE("an integer quotient by zero stops the program", + "[llvm][computability]") +{ + for (const auto operation : { Core::Primitive::Divide, + Core::Primitive::FloorDivide, + Core::Primitive::Remainder }) + for (const std::int64_t dividend : { std::int64_t{ 0 }, + std::int64_t{ 6 }, + std::int64_t{ -6 }, + kLeast, + kGreatest }) + { + CAPTURE(dividend); + // Unoptimized, the check stands before the instruction. + // Optimized, the operands are known and what is left of the + // function is the stop itself: the division is not deleted as + // if it had never been needed. + CHECK(Stops(operation, dividend, 0, false)); + CHECK(Stops(operation, dividend, 0, true)); + } +} + +TEST_CASE("a quotient whose divisor is known not to be zero is not checked", + "[llvm][computability]") +{ + for (const auto operation : { Core::Primitive::Divide, + Core::Primitive::FloorDivide, + Core::Primitive::Remainder }) + for (const std::int64_t divisor : { std::int64_t{ 1 }, + std::int64_t{ -1 }, + std::int64_t{ 7 }, + kLeast, + kGreatest }) + { + CAPTURE(divisor); + CHECK_FALSE(Stops(operation, 42, divisor, true)); + } +} + +TEST_CASE("the check leaves every quotient that has a value unchanged", + "[llvm][computability][execution]") +{ + for (std::int64_t left = -7; left <= 7; ++left) + for (std::int64_t right = -3; right <= 3; ++right) + if (right != 0) + { + CheckBothPipelines(Core::Primitive::Divide, + left, + right, + left / right); + CheckBothPipelines(Core::Primitive::Remainder, + left, + right, + left % right); + } + CheckBothPipelines(Core::Primitive::Divide, kGreatest, 1, kGreatest); + CheckBothPipelines(Core::Primitive::Divide, kGreatest, -1, -kGreatest); + CheckBothPipelines(Core::Primitive::Divide, kLeast, 1, kLeast); + CheckBothPipelines(Core::Primitive::Divide, kLeast, 2, kLeast / 2); + CheckBothPipelines(Core::Primitive::Divide, kLeast, kLeast, 1); + CheckBothPipelines(Core::Primitive::Divide, kGreatest, kLeast, 0); + CheckBothPipelines(Core::Primitive::Remainder, kLeast, 2, 0); + CheckBothPipelines(Core::Primitive::Remainder, kLeast, 3, kLeast % 3); + CheckBothPipelines(Core::Primitive::Remainder, + kGreatest, + kLeast, + kGreatest); +} + +TEST_CASE("the least integer divided by minus one wraps", + "[llvm][computability][execution]") +{ + // The quotient is one above the greatest value. Generated code wraps + // it, so the result is the least value again, as it is for `0 - least` + // and for `least * -1`. + CheckBothPipelines(Core::Primitive::Subtract, 0, kLeast, kLeast); + CheckBothPipelines(Core::Primitive::Multiply, kLeast, -1, kLeast); + CheckBothPipelines(Core::Primitive::Divide, kLeast, -1, kLeast); + CheckBothPipelines(Core::Primitive::FloorDivide, kLeast, -1, kLeast); + // The division is exact, so nothing remains. + CheckBothPipelines(Core::Primitive::Remainder, kLeast, -1, 0); + // It is not a division by zero, and does not stop the program. + CHECK_FALSE(Stops(Core::Primitive::Divide, kLeast, -1, true)); + CHECK_FALSE(Stops(Core::Primitive::FloorDivide, kLeast, -1, true)); + CHECK_FALSE(Stops(Core::Primitive::Remainder, kLeast, -1, true)); +} + +TEST_CASE("a shift by an amount within the width has its value", + "[llvm][computability][execution]") +{ + CheckBothPipelines(Core::Primitive::ShiftLeft, 1, 0, 1); + CheckBothPipelines(Core::Primitive::ShiftLeft, 1, 3, 8); + CheckBothPipelines(Core::Primitive::ShiftLeft, 1, 62, kGreatest / 2 + 1); + CheckBothPipelines(Core::Primitive::ShiftLeft, 1, 63, kLeast); + CheckBothPipelines(Core::Primitive::ShiftLeft, -1, 63, kLeast); + CheckBothPipelines(Core::Primitive::ShiftRight, 8, 3, 1); + CheckBothPipelines(Core::Primitive::ShiftRight, -8, 1, -4); + CheckBothPipelines(Core::Primitive::ShiftRight, kLeast, 63, -1); + CheckBothPipelines(Core::Primitive::ShiftRight, kGreatest, 63, 0); + for (const auto operation : + { Core::Primitive::ShiftLeft, Core::Primitive::ShiftRight }) + for (const std::int64_t amount : + { std::int64_t{ 0 }, std::int64_t{ 1 }, std::int64_t{ 63 } }) + { + CAPTURE(amount); + CHECK_FALSE(Stops(operation, 5, amount, true)); + } +} + +TEST_CASE("a shift by an amount outside the width stops the program", + "[llvm][computability]") +{ + for (const auto operation : + { Core::Primitive::ShiftLeft, Core::Primitive::ShiftRight }) + for (const std::int64_t amount : { std::int64_t{ 64 }, + std::int64_t{ 65 }, + std::int64_t{ -1 }, + kLeast, + kGreatest }) + { + CAPTURE(amount); + CHECK(Stops(operation, 5, amount, false)); + CHECK(Stops(operation, 5, amount, true)); + } +} + +TEST_CASE("operations that always have a value are not checked", + "[llvm][computability]") +{ + for (const auto operation : { Core::Primitive::Add, + Core::Primitive::Subtract, + Core::Primitive::Multiply, + Core::Primitive::BitwiseAnd, + Core::Primitive::BitwiseOr, + Core::Primitive::BitwiseXor }) + { + CHECK_FALSE(Stops(operation, kLeast, 0, false)); + CHECK_FALSE(Stops(operation, kLeast, 0, true)); + } +} diff --git a/Compiler/Cli/Arguments/Options.cpp b/Compiler/Cli/Arguments/Options.cpp index cba18b07..0fc313e9 100644 --- a/Compiler/Cli/Arguments/Options.cpp +++ b/Compiler/Cli/Arguments/Options.cpp @@ -15,7 +15,7 @@ #include "TableLibrary.hpp" #ifndef VXS_PROJECT_VERSION -# define VXS_PROJECT_VERSION "0.4.1" +# define VXS_PROJECT_VERSION "0.5.0" #endif namespace diff --git a/Compiler/Cli/BUILD.bazel b/Compiler/Cli/BUILD.bazel index aecf858b..0ee1483a 100644 --- a/Compiler/Cli/BUILD.bazel +++ b/Compiler/Cli/BUILD.bazel @@ -7,5 +7,8 @@ package(default_visibility = ["//visibility:public"]) cc_binary( name = "vxs", srcs = ["Main.cpp"], - deps = ["//Compiler/Cli/Commands:commands"], + deps = [ + "//Compiler/Cli/Commands:commands", + "//Compiler/Support:compiler_stack", + ], ) diff --git a/Compiler/Cli/Commands/Commands.cpp b/Compiler/Cli/Commands/Commands.cpp index 6b78d994..cd477cdb 100644 --- a/Compiler/Cli/Commands/Commands.cpp +++ b/Compiler/Cli/Commands/Commands.cpp @@ -589,7 +589,19 @@ namespace PathText(executable)); return 1; } - const int status = RunInstalledTool(PathText(executable), arguments); + // The artifact is named relative to the working directory, and a + // name without a directory would be looked for on PATH, where the + // program just built is not. Start it by its absolute path. + const auto absolute = std::filesystem::absolute(executable, error); + if (error) + { + fmt::print(stderr, + "vxs: could not resolve native executable '{}': {}\n", + PathText(executable), + error.message()); + return 1; + } + const int status = RunInstalledTool(PathText(absolute), arguments); if (status == -1) { fmt::print( diff --git a/Compiler/Cli/Commands/Tests/BUILD.bazel b/Compiler/Cli/Commands/Tests/BUILD.bazel index ea162060..349f397a 100644 --- a/Compiler/Cli/Commands/Tests/BUILD.bazel +++ b/Compiler/Cli/Commands/Tests/BUILD.bazel @@ -24,6 +24,19 @@ cc_binary( ], ) +# Programs built into native executables and started as processes. The test +# drives the command line in its own process, so the Haskell frontend library +# has to stand beside it, where the development helper stages it. +cc_binary( + name = "executable_run_tests", + srcs = ["ExecutableRunTests.cpp"], + deps = [ + "//Compiler/Cli/Commands:commands", + "@catch3//:catch2_main", + "@catch3//src/Progmasoft:catch3", + ], +) + # Compile the ABI header with a real C11 frontend to keep C++ types outside the # boundary. This is a test-only C translation unit, never production code. cc_library( diff --git a/Compiler/Cli/Commands/Tests/ExecutableRunTests.cpp b/Compiler/Cli/Commands/Tests/ExecutableRunTests.cpp new file mode 100644 index 00000000..36367bbe --- /dev/null +++ b/Compiler/Cli/Commands/Tests/ExecutableRunTests.cpp @@ -0,0 +1,731 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Compiler/Cli/Commands/Commands.hpp" + +#ifdef _WIN32 +# ifndef WIN32_LEAN_AND_MEAN +# define WIN32_LEAN_AND_MEAN +# endif +# ifndef NOMINMAX +# define NOMINMAX +# endif +# include +#endif + +// Programs built into native executables and run as processes. +// +// Every other execution test runs generated code inside the process that +// compiled it, where the runtime library of the compiler is at hand. A native +// executable has only what was linked into it. These cases build a program +// the way a user does, start the executable, and read its exit status: zero +// for a program that ran to the end of `Main`, and the status of a stopped +// process for a program that needed a value that cannot be computed. +// +// A program checks its own results with `Check`, which divides by zero when +// two numbers differ. A wrong result therefore stops the program, and one +// case confirms that it does, so that the others cannot pass by checking +// nothing. +// +// A program that writes is run with its standard output and standard error +// sent to files, and the bytes it wrote are compared with bytes written by +// hand. That is the only place the whole of console output is tested as a +// user meets it: the runtime library an executable is linked with, the +// import of the system functions it calls, and the bytes on the stream. + +namespace +{ + constexpr std::string_view kHeader + = "namespace Demo;\n" + "public class Program {\n" + " public static int Check(_ int got, _ int expected) {\n" + " return 1 / (got == expected ? 1 : 0);\n" + " }\n" + " public static int Zero(_ int v) { return v - v; }\n" + " public static int Pick(_ int flag, _ int value) {\n" + " if (flag > 0) { return value; }\n" + " return 7;\n" + " }\n"; + + [[nodiscard]] auto + Directory() -> std::filesystem::path + { + return std::filesystem::temp_directory_path() + / "visual-xsharp-executable-run"; + } + + /// Writes the program, and gives the status of `vxs -File` on + /// it. + [[nodiscard]] auto + Drive(std::string_view command, + std::string_view name, + std::string_view members) -> int + { + std::filesystem::create_directories(Directory()); + const auto source = Directory() / (std::string(name) + ".vxs"); + { + std::ofstream stream(source, std::ios::binary | std::ios::trunc); + stream << kHeader << members << "}\n"; + REQUIRE(stream.good()); + } + std::vector storage{ "vxs", + std::string(command), + "-File", + source.string() }; + std::vector arguments; + arguments.reserve(storage.size() + 1U); + for (auto &argument : storage) + arguments.push_back(argument.data()); + arguments.push_back(nullptr); + return Visual::XSharp::Cli::Run(static_cast(storage.size()), + arguments.data()); + } + +#ifdef _WIN32 + /// The status of a process that executed an instruction the processor + /// refuses, which is how generated code stops. + constexpr int kStopped = static_cast(0xC000001DU); + + [[nodiscard]] auto + Run(std::string_view name, std::string_view members) -> int + { + return Drive("run", name, members); + } + + /// What a process did: how it ended and what it wrote. + struct Outcome final + { + int status{ -1 }; + std::string output; + std::string error; + }; + + [[nodiscard]] auto + ReadAll(const std::filesystem::path &path) -> std::string + { + std::ifstream stream(path, std::ios::binary); + return { std::istreambuf_iterator(stream), + std::istreambuf_iterator() }; + } + + /// A file a child process may write to through an inherited handle. + [[nodiscard]] auto + InheritableFile(const std::filesystem::path &path) -> HANDLE + { + SECURITY_ATTRIBUTES attributes{}; + attributes.nLength = sizeof(attributes); + attributes.bInheritHandle = TRUE; + return CreateFileW(path.c_str(), + GENERIC_WRITE, + FILE_SHARE_READ, + &attributes, + CREATE_ALWAYS, + FILE_ATTRIBUTE_NORMAL, + nullptr); + } + + /// Builds the program into an executable, runs the executable with its + /// standard streams sent to files, and returns what it wrote. + [[nodiscard]] auto + RunCapturing(std::string_view name, std::string_view members) -> Outcome + { + REQUIRE(Drive("build", name, members) == 0); + const auto executable = Directory() / (std::string(name) + ".vxse"); + const auto outputPath = Directory() / (std::string(name) + ".out"); + const auto errorPath = Directory() / (std::string(name) + ".err"); + REQUIRE(std::filesystem::is_regular_file(executable)); + + auto *output = InheritableFile(outputPath); + auto *error = InheritableFile(errorPath); + REQUIRE(output != INVALID_HANDLE_VALUE); + REQUIRE(error != INVALID_HANDLE_VALUE); + + STARTUPINFOW startup{}; + startup.cb = sizeof(startup); + startup.dwFlags = STARTF_USESTDHANDLES; + startup.hStdInput = nullptr; + startup.hStdOutput = output; + startup.hStdError = error; + PROCESS_INFORMATION process{}; + auto commandLine = L'"' + executable.wstring() + L'"'; + const auto started = CreateProcessW(executable.c_str(), + commandLine.data(), + nullptr, + nullptr, + TRUE, + 0, + nullptr, + nullptr, + &startup, + &process); + CloseHandle(output); + CloseHandle(error); + REQUIRE(started != 0); + CloseHandle(process.hThread); + REQUIRE(WaitForSingleObject(process.hProcess, 120000U) + == WAIT_OBJECT_0); + DWORD status = 0U; + REQUIRE(GetExitCodeProcess(process.hProcess, &status) != 0); + CloseHandle(process.hProcess); + return { static_cast(status), + ReadAll(outputPath), + ReadAll(errorPath) }; + } +#endif +} // namespace + +TEST_CASE("every program of this file passes the checks before code " + "generation") +{ + CHECK(Drive("check", + "Checked", + " public static void Main() { Check(Pick(0, 1), 7); }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +// The native linker is the Windows one; elsewhere `vxs` does not produce an +// executable yet. +#ifdef _WIN32 + +TEST_CASE("a program whose Main ends without a return runs to its end") +{ + CHECK(Run("Empty", " public static void Main() { }\n") == 0); + CHECK(Run("Falling", + " public static void Touch(_ int v) { int w = v + 1; }\n" + " public static void Twice(_ int v) {\n" + " if (v > 0) { Touch(v); return; }\n" + " Touch(v + 1);\n" + " }\n" + " public static void Main() {\n" + " Twice(1);\n" + " Twice(0);\n" + " for (int i = 0; i < 3; i += 1) { Touch(i); }\n" + " }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("a program that checks a wrong result stops") +{ + CHECK(Run("Right", " public static void Main() { Check(1 + 1, 2); }\n") + == 0); + CHECK(Run("Wrong", " public static void Main() { Check(1 + 1, 3); }\n") + == kStopped); + CHECK(Run("WrongLater", + " public static void Main() {\n" + " Check(Pick(1, 4), 4);\n" + " Check(Pick(0, 4), 4);\n" + " }\n") + == kStopped); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable passes arguments by need") +{ + // A division by zero in an argument the method does not read is never + // carried out; the same argument, read, is the value it should be. + CHECK(Run("ByNeed", + " public static int Run(_ int a, _ int b) {\n" + " return Pick(a, b / a);\n" + " }\n" + " public static int Pass(_ int flag, _ int value) {\n" + " return Pick(flag, value);\n" + " }\n" + " public static void Main() {\n" + " Check(Run(0, 8), 7);\n" + " Check(Run(2, 8), 4);\n" + " Check(Pass(0, 8 / Zero(3)), 7);\n" + " Check(Pass(1, 8 / (Zero(3) + 2)), 4);\n" + " int shared = 12 / (Zero(1) + 3);\n" + " Check(Pick(1, shared) + shared, 8);\n" + " }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable that needs a quotient by zero stops") +{ + CHECK(Run("Needed", + " public static void Main() {\n" + " Check(Pick(1, 8 / Zero(3)), 0);\n" + " }\n") + == kStopped); + CHECK(Run("NeededRemainder", + " public static void Main() {\n" + " int left = 8 % Zero(3);\n" + " Check(left, left);\n" + " }\n") + == kStopped); + CHECK(Run("NotNeeded", + " public static void Main() {\n" + " int never = 8 / Zero(3);\n" + " Check(Pick(0, never), 7);\n" + " }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable wraps the least integer divided by minus one") +{ + CHECK(Run("Wraps", + " public static void Main() {\n" + " int least = Zero(1) - 9223372036854775807 - 1;\n" + " int minusOne = Zero(1) - 1;\n" + " Check(least / minusOne, least);\n" + " Check(least % minusOne, 0);\n" + " }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable creates and calls closures") +{ + CHECK(Run("Closures", + " public static int Run(_ int a, _ int b) {\n" + " auto f = \\(int v) -> v + a;\n" + " auto g = \\(int w) -> f(w) * 2;\n" + " return g(b);\n" + " }\n" + " public static void Main() { Check(Run(1, 8), 18); }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable reuses the memory of the objects it releases") +{ + // Each pass creates objects and releases them. Two million passes hold + // far more than the executable's memory if nothing is reused, and no + // more than one pass does if everything is. + CHECK(Run("ReuseClosures", + " public static int Many(_ int rounds) {\n" + " int total = 0;\n" + " for (int i = 0; i < rounds; i += 1) {\n" + " auto add = \\(int v) -> v + i;\n" + " total += add(1) - i;\n" + " }\n" + " return total;\n" + " }\n" + " public static void Main() {\n" + " Check(Many(2000000), 2000000);\n" + " }\n") + == 0); + CHECK(Run("ReuseArguments", + " public static int Many(_ int rounds) {\n" + " int total = 0;\n" + " for (int i = 0; i < rounds; i += 1) {\n" + " total += Pick(i % 2, 10 / (i % 2));\n" + " }\n" + " return total;\n" + " }\n" + " public static void Main() {\n" + " Check(Many(2000000), 17000000);\n" + " }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable holds many objects at one time") +{ + // Every level of the recursion holds a closure until the levels below + // it have returned. + CHECK(Run("Held", + " public static int Deep(_ int n) {\n" + " auto here = \\(int v) -> v + n;\n" + " if (n <= 0) { return here(0); }\n" + " return Deep(n - 1) + here(1) - n;\n" + " }\n" + " public static void Main() { Check(Deep(4000), 4000); }\n") + == 0); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable writes to standard output") +{ + // The example program of the repository, as it is written there. + const auto hello + = RunCapturing("HelloWorld", + " public static void Main() {\n" + " String language = \"Visual X#\";\n" + " Console.Println(\"Hello from \" + language" + " + \"!\");\n" + " }\n"); + CHECK(hello.status == 0); + CHECK(hello.output == "Hello from Visual X#!\r\n"); + CHECK(hello.error.empty()); + + // Print ends no line, Println ends one with the line terminator of the + // platform, and a line feed written in a string is a line feed. + const auto lines = RunCapturing("Lines", + " public static void Main() {\n" + " Console.Print(\"a\");\n" + " Console.Print(\"b\");\n" + " Console.Println(\"\");\n" + " Console.Println(42);\n" + " Console.Println(true);\n" + " Console.Println('x');\n" + " }\n"); + CHECK(lines.status == 0); + CHECK(lines.output == "ab\r\n42\r\ntrue\r\nx\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable keeps standard output and standard error apart") +{ + const auto outcome + = RunCapturing("Streams", + " public static void Main() {\n" + " Console.Print(\"out one, \");\n" + " Console.Errorln(\"error one\");\n" + " Console.Println(\"out two\");\n" + " Console.Errorfn(\"error %d\", 2);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "out one, out two\r\n"); + CHECK(outcome.error == "error one\r\nerror 2\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable writes to a file as UTF-8") +{ + // One character of two bytes and one of three, written in the source. + const auto outcome + = RunCapturing("Encoded", + " public static void Main() {\n" + " Console.Println(\"\xc3\xa9\xe2\x82\xac\");\n" + " Console.Printf(\"%4s|\", \"\xc3\xa9\");\n" + " }\n"); + CHECK(outcome.status == 0); + // The width counts characters, not bytes: three spaces before one + // character of two bytes. + CHECK(outcome.output == "\xc3\xa9\xe2\x82\xac\r\n \xc3\xa9|"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable applies the conversions of a format") +{ + const auto outcome = RunCapturing( + "Formats", + " public static void Main() {\n" + " Console.Printfn(\"gcd(%d, %d) = %d\", 1071, 462, 21);\n" + " Console.Printfn(\"%5d|%-5d|%05d\", 42, 42, 0 - 42);\n" + " Console.Printfn(\"%+d %'d\", 42, 0 - 1234567);\n" + " Console.Printfn(\"%x %#x %x\", 255, 255, 0 - 255);\n" + " uint size = 4294967295;\n" + " Console.Printfn(\"%u %x\", size, size);\n" + " Console.Printfn(\"%10s|%-10s|%.2s\", \"abc\", \"abc\"," + " \"abcdef\");\n" + " Console.Printfn(\"%c%3c|%b %6b|\", 'q', 'q', true, false);\n" + " Console.Printfn(\"%d%% done%n%s\", 50, \"next\");\n" + " Console.Printfn(\"%*d|%.*f|%*.*f|\", 5, 42, 2, 3.14159, 8, 2," + " 3.14159);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output + == "gcd(1071, 462) = 21\r\n" + " 42|42 |-0042\r\n" + "+42 -1'234'567\r\n" + "ff 0xff -ff\r\n" + "4294967295 ffffffff\r\n" + " abc|abc |ab\r\n" + "q q|true false|\r\n" + "50% done\r\nnext\r\n" + " 42|3.14| 3.14|\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable writes floating-point numbers exactly") +{ + const auto outcome = RunCapturing( + "Floating", + " public static void Main() {\n" + " Console.Printfn(\"%f\", 12.5);\n" + " Console.Printfn(\"%.2f %.0f %.0f\", 12.5, 2.5, 3.5);\n" + " Console.Printfn(\"%010.3f|%-10.3f|\", 3.14159, 3.14159);\n" + " Console.Printfn(\"%'.2f\", 1234567.5);\n" + " Console.Printfn(\"%.20f\", 0.1);\n" + " float price = 19.99;\n" + " Console.Printfn(\"Price: %.2f\", price);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output + == "12.500000\r\n" + "12.50 2 4\r\n" + "000003.142|3.142 |\r\n" + "1'234'567.50\r\n" + "0.10000000000000000555\r\n" + "Price: 19.99\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable joins and compares strings") +{ + const auto outcome = RunCapturing( + "Strings", + " public static String Twice(_ String s) { return s + s; }\n" + " public static void Main() {\n" + " String line = \"x\";\n" + " line += \"y\";\n" + " line += 3;\n" + " Console.Println(Twice(line));\n" + " Console.Println(\"n=\" + 5 + \" b=\" + true + \" c=\" + " + "'z');\n" + " String first = \"ab\";\n" + " String second = \"a\" + \"b\";\n" + " Console.Println(first == second);\n" + " Console.Println(first \\= second);\n" + " Console.Println(first == \"abc\");\n" + " String made = Console.Format(\"%05d-%s\", 42, first);\n" + " Console.Println(made);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output + == "xy3xy3\r\n" + "n=5 b=true c=z\r\n" + "true\r\nfalse\r\nfalse\r\n" + "00042-ab\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable writes where the program says, in the order it " + "says") +{ + // Output is an effect: a binding that writes is carried out where it + // stands, and a value that only computes is still computed by need. + const auto outcome = RunCapturing( + "Effects", + " public static int Log(_ int v) { Console.Println(v);" + " return v; }\n" + " public static void Main() {\n" + " int unused = Log(1);\n" + " int never = 8 / Zero(3);\n" + " Console.Println(Pick(0, Log(5)));\n" + " Console.Printf(\"%*d|%n\", Log(3), Log(4));\n" + " Console.Println(Pick(0, never));\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "1\r\n5\r\n7\r\n3\r\n4\r\n 4|\r\n7\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable has written what came before it stopped") +{ + // The runtime keeps no buffer, so nothing is lost when a program stops. + const auto outcome + = RunCapturing("Stopped", + " public static void Main() {\n" + " Console.Println(\"before\");\n" + " Console.Error(\"also before\");\n" + " Check(1, 2);\n" + " Console.Println(\"after\");\n" + " }\n"); + CHECK(outcome.status == kStopped); + CHECK(outcome.output == "before\r\n"); + CHECK(outcome.error == "also before"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable releases the strings it makes") +{ + // Each pass makes and releases several strings. A million passes hold + // far more than a process may if none is released; the last line shows + // the loop ran to its end. + const auto outcome = RunCapturing( + "Many", + " public static void Main() {\n" + " int total = 0;\n" + " for (int i = 0; i < 1000000; i += 1) {\n" + " String text = \"value \" + i;\n" + " String made = Console.Format(\"%08d|%s\", i, text);\n" + " if (made == text) { total += 1; }\n" + " total += 1;\n" + " }\n" + " Console.Println(total);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "1000000\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable calls a method held as a value where the call " + "stands") +{ + // A method is a callable value. Nothing reads what the calls return, + // and both write all the same. + const auto outcome = RunCapturing( + "Held", + " public static int Log(_ int v) { Console.Println(v);" + " return v; }\n" + " public static int Run(_ (int) -> int f) { return f(3); }\n" + " public static void Main() {\n" + " auto f = Log;\n" + " int first = f(1);\n" + " int second = Run(Log);\n" + " Console.Println(9);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "1\r\n3\r\n9\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable passes strings through callables") +{ + const auto outcome = RunCapturing( + "Callables", + " public static String Name() { return \"Visual X#\"; }\n" + " public static void Main() {\n" + " String prefix = Name() + \": \";\n" + " auto label = \\(int n) -> prefix + n;\n" + " auto twice = \\(int v) -> { Console.Printf(\"%d,\", v);" + " return v * 2; };\n" + " Console.Println(label(1));\n" + " Console.Println(label(twice(twice(1))));\n" + " String all = \"\";\n" + " for (int i = 0; i < 1000; i += 1) { all += label(i % 10); }\n" + " Console.Println(all == \"\");\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "Visual X#: 1\r\n1,2,Visual X#: 4\r\nfalse\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable that appends keeps the string another name holds") +{ + const auto outcome + = RunCapturing("Append", + " public static void Main() {\n" + " String s = \"a\";\n" + " String t = s;\n" + " s += \"b\";\n" + " s += 7;\n" + " Console.Println(s + \"|\" + t);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "ab7|a\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable that shifts by the width of the type stops") +{ + const auto outcome + = RunCapturing("Shift", + " public static void Main() {\n" + " Console.Println(1 << (Zero(3) + 63) < 0);\n" + " int k = 1 << (Zero(3) + 64);\n" + " Console.Println(k);\n" + " Console.Println(\"after\");\n" + " }\n"); + CHECK(outcome.status == kStopped); + CHECK(outcome.output == "true\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable that only reports writes nothing to standard " + "output") +{ + const auto outcome + = RunCapturing("Report", + " public static void Main() {\n" + " Console.Errorfn(\"%s: %d\", \"code\", 7);\n" + " Console.Error(\"done\");\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output.empty()); + CHECK(outcome.error == "code: 7\r\ndone"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("a link leaves only the executable beside the source") +{ + // The import library and the list of names it is made from are made + // for one link and removed after it. + REQUIRE(Drive("build", + "Tidy", + " public static void Main() {" + " Console.Println(\"x\"); }\n") + == 0); + std::size_t executables = 0U; + for (const auto &entry : std::filesystem::directory_iterator(Directory())) + { + const auto extension = entry.path().extension().string(); + CHECK(extension != ".def"); + CHECK(extension != ".lib"); + CHECK(extension != ".obj"); + CHECK(extension != ".o"); + if (extension == ".vxse") + ++executables; + } + CHECK(executables == 1U); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable selects one of two strings") +{ + // Every pass selects a string and replaces the one before it; a + // million passes would exhaust a process that released none. + const auto outcome = RunCapturing( + "Select", + " public static String Name() { return \"Visual X#\"; }\n" + " public static void Main() {\n" + " String s = Zero(3) > 0 ? \"a\" : \"b\";\n" + " Console.Println(s);\n" + " Console.Println(Zero(3) == 0 ? Name() : \"nobody\");\n" + " int evens = 0;\n" + " for (int i = 0; i < 1000000; i += 1) {\n" + " String kind = i % 2 == 0 ? \"even \" + i : \"odd\";\n" + " if (kind \\= \"odd\") { evens += 1; }\n" + " }\n" + " Console.Println(evens);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "b\r\nVisual X#\r\n500000\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable names methods through the namespace and the type") +{ + const auto outcome = RunCapturing( + "Qualified", + " public static int Log(_ int v) { Console.Println(v);" + " return v; }\n" + " public static int Run(_ (int) -> int f) { return f(3); }\n" + " public static void Main() {\n" + " Check(Demo.Program.Pick(1, 4), 4);\n" + " int first = Demo.Program.Log(1);\n" + " int second = Run(Program::Log);\n" + " auto held = Demo.Program::Log;\n" + " int third = held(5);\n" + " Console.Println(9);\n" + " }\n"); + CHECK(outcome.status == 0); + CHECK(outcome.output == "1\r\n3\r\n5\r\n9\r\n"); + std::filesystem::remove_all(Directory()); +} + +TEST_CASE("an executable writes many lines") +{ + const auto outcome + = RunCapturing("Loop", + " public static void Main() {\n" + " for (int i = 0; i < 20000; i += 1) {\n" + " Console.Printfn(\"line %d\", i);\n" + " }\n" + " }\n"); + CHECK(outcome.status == 0); + std::string expected; + for (int index = 0; index < 20000; ++index) + expected += "line " + std::to_string(index) + "\r\n"; + CHECK(outcome.output == expected); + std::filesystem::remove_all(Directory()); +} + +#endif diff --git a/Compiler/Cli/Main.cpp b/Compiler/Cli/Main.cpp index e9b33bd5..0e7460f7 100644 --- a/Compiler/Cli/Main.cpp +++ b/Compiler/Cli/Main.cpp @@ -6,9 +6,15 @@ */ #include "Compiler/Cli/Commands/Commands.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" int main(int argc, char **argv) { - return Visual::XSharp::Cli::Run(argc, argv); + // Every command runs on the compiler stack: the nesting limits of the + // frontend are stated against its size, not against the stack the + // operating system gives the process. + return Visual::XSharp::Support::RunOnCompilerStack([argc, argv] { + return Visual::XSharp::Cli::Run(argc, argv); + }); } diff --git a/Compiler/Codegen/Xmm/Lower.cpp b/Compiler/Codegen/Xmm/Lower.cpp index c7164809..8192013b 100644 --- a/Compiler/Codegen/Xmm/Lower.cpp +++ b/Compiler/Codegen/Xmm/Lower.cpp @@ -116,6 +116,10 @@ namespace visual_xsharp::xmm return Opcode::BitwiseNot; case xpp::Opcode::TypeIs: return Opcode::TypeIs; + case xpp::Opcode::Memoize: + return Opcode::Memoize; + case xpp::Opcode::RuntimeCall: + return Opcode::RuntimeCall; case xpp::Opcode::MakeClosure: return Opcode::MakeClosure; case xpp::Opcode::RetainStrong: diff --git a/Compiler/Codegen/Xmm/Verifier.cpp b/Compiler/Codegen/Xmm/Verifier.cpp index 8fcc3e55..31a75375 100644 --- a/Compiler/Codegen/Xmm/Verifier.cpp +++ b/Compiler/Codegen/Xmm/Verifier.cpp @@ -16,6 +16,7 @@ #include "Visual/XSharp/Analysis/DefiniteInitialization.hpp" #include "Visual/XSharp/Core/Callable.hpp" #include "Visual/XSharp/Core/Ownership.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" #include "Visual/XSharp/Core/Scalar.hpp" #include "Visual/XSharp/Xmm/OwnershipVerifier.hpp" #include "Visual/XSharp/Xmm/Verifier.hpp" @@ -172,6 +173,7 @@ namespace Visual::XSharp::Xmm case xmm::Opcode::Negate: case xmm::Opcode::NotBool: case xmm::Opcode::BitwiseNot: + case xmm::Opcode::Memoize: case xmm::Opcode::RetainStrong: case xmm::Opcode::ReleaseStrong: case xmm::Opcode::MakeWeak: @@ -205,6 +207,7 @@ namespace Visual::XSharp::Xmm return 2; case xmm::Opcode::Call: case xmm::Opcode::MakeClosure: + case xmm::Opcode::RuntimeCall: return 0; } return 0; @@ -460,6 +463,35 @@ namespace Visual::XSharp::Xmm "producing ownership instruction must preserve " "its operand type"); } + else if (instruction.opcode == xmm::Opcode::RuntimeCall) + { + // The function is named by an immediate, never by a + // register: what is called is fixed when the program is + // compiled. + namespace runtime = core::runtime; + const runtime::Signature *signature = nullptr; + if (!instruction.operands.empty() + && instruction.operands.front().kind + == xmm::Value::Kind::Immediate) + if (const auto identity = runtime::IdentityOf( + instruction.operands.front().immediate, + instruction.operands.front().type)) + signature = runtime::Find(*identity); + std::vector arguments; + arguments.reserve(instruction.operands.size()); + for (std::size_t index = 1U; + index < instruction.operands.size(); + ++index) + arguments.push_back(instruction.operands[index].type.kind); + const auto defect + = runtime::Check(signature, + arguments, + instruction.result_type.kind); + if (defect != runtime::Defect::None) + context.add(IssueKind::OperandType, + "VXL1054", + std::string(runtime::Describe(defect))); + } else { const auto expected = ExpectedOperandCount(instruction.opcode); @@ -540,6 +572,31 @@ namespace Visual::XSharp::Xmm "bitwise instruction operands and result " "must use one integer type"); } + else if (instruction.opcode == xmm::Opcode::Memoize) + { + const auto remembers + = instruction.operands.size() == 1U + && instruction.operands.front().type.kind + == core::Type::Kind::Function + && instruction.operands.front().type.components.size() + == 1U + && (instruction.operands.front() + .type.components.front() + .kind + == core::Type::Kind::Bool + || core::is_numeric( + instruction.operands.front() + .type.components.front())); + if (!remembers + || instruction.result_type + != instruction.operands.front().type) + context.add( + IssueKind::OperandType, + "VXL1053", + "memoization requires one callable without " + "parameters whose result is Bool or numeric, and " + "yields a callable of the same type"); + } else if (instruction.opcode == xmm::Opcode::FloorDivide) { const auto hasNumericPair @@ -590,6 +647,7 @@ namespace Visual::XSharp::Xmm } else if (instruction.result_type.kind != core::Type::Kind::Unit && instruction.opcode != xmm::Opcode::Call + && instruction.opcode != xmm::Opcode::RuntimeCall && instruction.opcode != xmm::Opcode::MakeClosure) context.add( IssueKind::ResultType, diff --git a/Compiler/Codegen/Xmm/Wire.cpp b/Compiler/Codegen/Xmm/Wire.cpp index c7e5602f..18dc7ba6 100644 --- a/Compiler/Codegen/Xmm/Wire.cpp +++ b/Compiler/Codegen/Xmm/Wire.cpp @@ -161,7 +161,7 @@ namespace Visual::XSharp::Xmm::Wire IR::Instruction instruction; instruction.opcode = static_cast( ReadTag(reader, - static_cast(IR::Opcode::TypeIs), + static_cast(IR::Opcode::RuntimeCall), "Xmm opcode")); instruction.destination = reader.U32("Xmm destination register"); instruction.result_type = reader.Type("Xmm instruction result"); diff --git a/Compiler/Codegen/Xpp/BUILD.bazel b/Compiler/Codegen/Xpp/BUILD.bazel index 76fe07b4..b4063a2b 100644 --- a/Compiler/Codegen/Xpp/BUILD.bazel +++ b/Compiler/Codegen/Xpp/BUILD.bazel @@ -8,6 +8,7 @@ cc_library( "ControlFlow.cpp", "Lower.cpp", "Optimize.cpp", + "OwnershipPlacement.cpp", "OwnershipVerifier.cpp", "Verifier.cpp", "Wire.cpp", diff --git a/Compiler/Codegen/Xpp/Lower.cpp b/Compiler/Codegen/Xpp/Lower.cpp index 3233b406..e33c4712 100644 --- a/Compiler/Codegen/Xpp/Lower.cpp +++ b/Compiler/Codegen/Xpp/Lower.cpp @@ -81,6 +81,10 @@ namespace visual_xsharp::xpp return Opcode::BitwiseNot; case core::Operation::TypeIs: return Opcode::TypeIs; + case core::Operation::Memoize: + return Opcode::Memoize; + case core::Operation::RuntimeCall: + return Opcode::RuntimeCall; case core::Operation::MakeClosure: return Opcode::MakeClosure; } diff --git a/Compiler/Codegen/Xpp/OwnershipPlacement.cpp b/Compiler/Codegen/Xpp/OwnershipPlacement.cpp new file mode 100644 index 00000000..6e6d2a46 --- /dev/null +++ b/Compiler/Codegen/Xpp/OwnershipPlacement.cpp @@ -0,0 +1,660 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Analysis/Liveness.hpp" +#include "Visual/XSharp/Core/Ownership.hpp" +#include "Visual/XSharp/Xpp/OwnershipPlacement.hpp" + +namespace Visual::XSharp::Xpp +{ + namespace Core = ::visual_xsharp::core; + namespace IR = ::visual_xsharp::xpp; + namespace Live = ::Visual::XSharp::Analysis::Liveness; + + namespace + { + using SymbolSet = std::unordered_set; + using Effect = IR::Instruction::Effect; + + [[nodiscard]] auto + IsAarc(const Core::Type &type) -> bool + { + // The same classification as the ownership verifier, so every + // value it tracks has an owner here. + return Core::UsesAarc(type) || type.kind == Core::Type::Kind::Named; + } + + [[nodiscard]] auto + IsSymbol(const IR::Operand &operand) -> bool + { + return operand.kind == IR::Operand::Kind::Symbol; + } + + [[nodiscard]] auto + SymbolOperand(IR::SymbolId symbol, const Core::Type &type) + -> IR::Operand + { + IR::Operand operand; + operand.kind = IR::Operand::Kind::Symbol; + operand.type = type; + operand.symbol = symbol; + return operand; + } + + [[nodiscard]] auto + Release(IR::SymbolId symbol, const Core::Type &type) -> IR::Instruction + { + IR::Instruction instruction; + instruction.effect = Effect::Discard; + instruction.opcode = IR::Opcode::ReleaseStrong; + instruction.result_type = Core::Type::unit(); + instruction.operands.push_back(SymbolOperand(symbol, type)); + return instruction; + } + + /// `destination = opcode source`, defining a new symbol. + [[nodiscard]] auto + DefineFrom(IR::Opcode opcode, + IR::SymbolId destination, + IR::SymbolId source, + const Core::Type &type) -> IR::Instruction + { + IR::Instruction instruction; + instruction.effect = Effect::Define; + instruction.opcode = opcode; + instruction.destination = destination; + instruction.result_type = type; + instruction.operands.push_back(SymbolOperand(source, type)); + return instruction; + } + + /// A closure of a method, without captures. + [[nodiscard]] auto + MethodClosure(IR::SymbolId destination, + IR::SymbolId method, + const Core::Type &type) -> IR::Instruction + { + IR::Instruction instruction; + instruction.effect = Effect::Define; + instruction.opcode = IR::Opcode::MakeClosure; + instruction.destination = destination; + instruction.result_type = type; + instruction.closure_function = method; + return instruction; + } + + [[nodiscard]] auto + Contains(const std::vector &symbols, + IR::SymbolId symbol) -> bool + { + return std::ranges::find(symbols, symbol) != symbols.end(); + } + + /// The targets of a terminator, one for each edge it has. + [[nodiscard]] auto + Edges(IR::Terminator &terminator) -> std::vector + { + switch (terminator.kind) + { + case IR::Terminator::Kind::Branch: + return { &terminator.true_target, + &terminator.false_target }; + case IR::Terminator::Kind::Jump: + return { &terminator.true_target }; + case IR::Terminator::Kind::Return: + case IR::Terminator::Kind::Unreachable: + return {}; + } + return {}; + } + + /// Places the ownership of one function. + class Placer final + { + public: + Placer(IR::Function &function, + const SymbolSet &methods, + IR::SymbolId &nextSymbol) + : function_(function) + , methods_(methods) + , nextSymbol_(nextSymbol) + { + for (const auto &block : function_.blocks) + nextBlock_ = std::max(nextBlock_, block.id + 1U); + } + + void + Run() + { + OwnAssignedParameters(); + for (const auto ¶meter : function_.parameters) + if (IsAarc(parameter.type)) + borrowed_.insert(parameter.symbol.id); + MaterializeMethodValues(); + Normalize(); + Place(); + } + + private: + [[nodiscard]] auto + Fresh() -> IR::SymbolId + { + return nextSymbol_++; + } + + [[nodiscard]] auto + IsMethod(const IR::Operand &operand) const -> bool + { + return IsSymbol(operand) && methods_.contains(operand.symbol); + } + + /// Whether a symbol holds a reference this function must + /// release. + [[nodiscard]] auto + Owned(IR::SymbolId symbol, const Core::Type &type) const -> bool + { + return IsAarc(type) && !methods_.contains(symbol) + && !borrowed_.contains(symbol); + } + + [[nodiscard]] auto + Owned(const IR::Operand &operand) const -> bool + { + return IsSymbol(operand) && Owned(operand.symbol, operand.type); + } + + /** + * A parameter is borrowed, but one the body assigns holds an + * owned value from the assignment on. Such a parameter is + * copied into a local of the body at entry, with a reference + * of its own, and the body uses the local: a symbol is then + * borrowed or owned, never one after the other. + */ + void + OwnAssignedParameters() + { + SymbolSet assigned; + for (const auto &block : function_.blocks) + for (const auto &instruction : block.instructions) + if (instruction.effect == Effect::Store) + assigned.insert(instruction.destination); + + std::vector copies; + std::unordered_map locals; + for (const auto ¶meter : function_.parameters) + { + if (!IsAarc(parameter.type) + || !assigned.contains(parameter.symbol.id)) + continue; + const auto local = Fresh(); + locals.emplace(parameter.symbol.id, local); + copies.push_back(DefineFrom(IR::Opcode::RetainStrong, + local, + parameter.symbol.id, + parameter.type)); + } + if (copies.empty()) + return; + + const auto rename = [&locals](IR::SymbolId &symbol) { + if (const auto found = locals.find(symbol); + found != locals.end()) + symbol = found->second; + }; + for (auto &block : function_.blocks) + { + for (auto &instruction : block.instructions) + { + if (instruction.effect != Effect::Discard) + rename(instruction.destination); + for (auto &operand : instruction.operands) + if (IsSymbol(operand)) + rename(operand.symbol); + } + if (IsSymbol(block.terminator.value)) + rename(block.terminator.value.symbol); + } + + // The copies run once, also when the entry block is the + // target of a loop. + IR::Block entry; + entry.id = nextBlock_++; + entry.instructions = std::move(copies); + entry.terminator.kind = IR::Terminator::Kind::Jump; + entry.terminator.true_target = function_.entry; + function_.entry = entry.id; + function_.blocks.insert(function_.blocks.begin(), + std::move(entry)); + } + + /** + * A method named where a value is expected becomes a closure + * without captures. The callee of a direct call stays a + * method: such a call needs no closure. + * + * A string literal used as an operand becomes a local in the + * same pass, for the same reason: both are objects created at + * a use, and an object needs a symbol to be released by. + */ + void + MaterializeMethodValues() + { + for (auto &block : function_.blocks) + { + std::vector rewritten; + rewritten.reserve(block.instructions.size()); + for (auto &instruction : block.instructions) + { + if (instruction.opcode == IR::Opcode::Copy + && instruction.effect != Effect::Discard + && instruction.operands.size() == 1U + && IsMethod(instruction.operands.front())) + { + auto closure = MethodClosure( + instruction.destination, + instruction.operands.front().symbol, + instruction.result_type); + closure.effect = instruction.effect; + rewritten.push_back(std::move(closure)); + continue; + } + // A string literal that is an operand creates a + // string where it is used. Unless the instruction + // is the binding of that string, nothing names the + // new object and nothing could release it, so it + // gets a symbol of its own first. + const auto binds + = instruction.opcode == IR::Opcode::Copy + && instruction.effect != Effect::Discard + && instruction.operands.size() == 1U; + if (!binds) + for (auto &operand : instruction.operands) + { + if (operand.kind != IR::Operand::Kind::Literal + || !IsAarc(operand.type)) + continue; + const auto created = Fresh(); + IR::Instruction literal; + literal.effect = Effect::Define; + literal.opcode = IR::Opcode::Copy; + literal.destination = created; + literal.result_type = operand.type; + literal.operands.push_back(operand); + rewritten.push_back(std::move(literal)); + operand = SymbolOperand(created, operand.type); + } + const std::size_t first + = instruction.opcode == IR::Opcode::Call ? 1U : 0U; + for (std::size_t index = first; + index < instruction.operands.size(); + ++index) + { + auto &operand = instruction.operands[index]; + if (!IsMethod(operand)) + continue; + const auto closure = Fresh(); + rewritten.push_back(MethodClosure(closure, + operand.symbol, + operand.type)); + operand.symbol = closure; + } + rewritten.push_back(std::move(instruction)); + } + auto &terminator = block.terminator; + if (terminator.kind == IR::Terminator::Kind::Return + && IsMethod(terminator.value)) + { + const auto closure = Fresh(); + rewritten.push_back( + MethodClosure(closure, + terminator.value.symbol, + terminator.value.type)); + terminator.value.symbol = closure; + } + block.instructions = std::move(rewritten); + } + } + + /** + * Bring every instruction to a form in which each owned value + * has a symbol of its own: + * + * - a call whose owned result is discarded defines a symbol, + * so that the result can be released; + * - an instruction that reads the symbol it writes reads the + * old value from a symbol of its own, so that the old value + * can be released after the new one is stored; + * - a borrowed value that is returned is retained first, + * because the caller receives an owned result. + */ + void + Normalize() + { + for (auto &block : function_.blocks) + { + std::vector rewritten; + rewritten.reserve(block.instructions.size()); + for (auto &instruction : block.instructions) + { + if (instruction.effect == Effect::Discard + && (instruction.opcode == IR::Opcode::Call + || instruction.opcode + == IR::Opcode::RuntimeCall) + && IsAarc(instruction.result_type)) + { + instruction.effect = Effect::Define; + instruction.destination = Fresh(); + } + const auto rewrites + = instruction.effect != Effect::Discard + && Owned(instruction.destination, + instruction.result_type) + && !IsSelfCopy(instruction) + && std::ranges::any_of( + instruction.operands, + [&](const IR::Operand &operand) { + return IsSymbol(operand) + && operand.symbol + == instruction.destination; + }); + if (rewrites) + { + const auto previous = Fresh(); + rewritten.push_back( + DefineFrom(IR::Opcode::Copy, + previous, + instruction.destination, + instruction.result_type)); + for (auto &operand : instruction.operands) + if (IsSymbol(operand) + && operand.symbol + == instruction.destination) + operand.symbol = previous; + } + rewritten.push_back(std::move(instruction)); + } + auto &terminator = block.terminator; + if (terminator.kind == IR::Terminator::Kind::Return + && IsSymbol(terminator.value) + && IsAarc(terminator.value.type) + && borrowed_.contains(terminator.value.symbol)) + { + const auto owned = Fresh(); + rewritten.push_back(DefineFrom(IR::Opcode::RetainStrong, + owned, + terminator.value.symbol, + terminator.value.type)); + terminator.value.symbol = owned; + } + block.instructions = std::move(rewritten); + } + } + + [[nodiscard]] static auto + IsSelfCopy(const IR::Instruction &instruction) -> bool + { + return instruction.opcode == IR::Opcode::Copy + && instruction.effect != Effect::Discard + && instruction.operands.size() == 1U + && IsSymbol(instruction.operands.front()) + && instruction.operands.front().symbol + == instruction.destination; + } + + /// The liveness problem of the owned symbols alone. + [[nodiscard]] auto + LivenessModel() -> Live::Function + { + Live::Function model; + model.entry = function_.entry; + model.blocks.reserve(function_.blocks.size()); + for (auto &block : function_.blocks) + { + Live::Block live; + live.id = block.id; + for (const auto *target : Edges(block.terminator)) + live.successors.push_back(*target); + live.accesses.reserve(block.instructions.size() + 1U); + for (std::size_t index = 0U; + index < block.instructions.size(); + ++index) + { + const auto &instruction = block.instructions[index]; + Live::Access access; + access.instruction = index; + for (const auto &operand : instruction.operands) + if (Owned(operand)) + { + access.reads.push_back(operand.symbol); + types_.insert_or_assign(operand.symbol, + operand.type); + } + if (instruction.effect != Effect::Discard + && Owned(instruction.destination, + instruction.result_type)) + { + access.write = instruction.destination; + types_.insert_or_assign(instruction.destination, + instruction.result_type); + } + live.accesses.push_back(std::move(access)); + } + Live::Access terminator; + terminator.instruction = block.instructions.size(); + terminator.terminator = true; + if (block.terminator.kind == IR::Terminator::Kind::Return + && Owned(block.terminator.value)) + terminator.reads.push_back( + block.terminator.value.symbol); + live.accesses.push_back(std::move(terminator)); + model.blocks.push_back(std::move(live)); + } + return model; + } + + [[nodiscard]] auto + ReleaseOf(IR::SymbolId symbol) const -> IR::Instruction + { + return Release(symbol, types_.at(symbol)); + } + + /** + * Write the retains and the releases. A value is released + * after the instruction that uses it last, after the + * instruction that defines it when nothing uses it, and on the + * edge on which it dies when it lives on another edge of the + * same branch. A returned value leaves with its reference. + */ + void + Place() + { + const auto analysis = Live::Analyze(LivenessModel()); + // A function with a malformed graph is left as it is; the + // verifier reports the graph. + if (!analysis.valid()) + return; + + std::unordered_map facts; + facts.reserve(analysis.facts.size()); + for (const auto &block : analysis.facts) + if (block.reachable) + facts.emplace(block.block, &block); + + std::unordered_map predecessors; + predecessors[function_.entry] = 1U; + for (auto &block : function_.blocks) + if (facts.contains(block.id)) + for (const auto *target : Edges(block.terminator)) + ++predecessors[*target]; + + std::unordered_map> + prefixes; + std::vector added; + for (auto &block : function_.blocks) + { + const auto found = facts.find(block.id); + if (found == facts.end()) + continue; + const auto &blockFacts = *found->second; + PlaceInBlock(block, blockFacts); + + for (auto *target : Edges(block.terminator)) + { + const auto successor = facts.find(*target); + if (successor == facts.end()) + continue; + auto dying = blockFacts.liveOnExit; + std::erase_if(dying, [&](Live::StorageId symbol) { + return Contains(successor->second->liveOnEntry, + symbol); + }); + if (dying.empty()) + continue; + std::ranges::sort(dying); + std::vector releases; + releases.reserve(dying.size()); + for (const auto symbol : dying) + releases.push_back(ReleaseOf(symbol)); + if (predecessors[*target] == 1U) + { + prefixes[*target] = std::move(releases); + continue; + } + IR::Block edge; + edge.id = nextBlock_++; + edge.instructions = std::move(releases); + edge.terminator.kind = IR::Terminator::Kind::Jump; + edge.terminator.true_target = *target; + *target = edge.id; + added.push_back(std::move(edge)); + } + } + + for (auto &block : function_.blocks) + if (const auto prefix = prefixes.find(block.id); + prefix != prefixes.end()) + block.instructions.insert( + block.instructions.begin(), + std::make_move_iterator(prefix->second.begin()), + std::make_move_iterator(prefix->second.end())); + for (auto &block : added) + function_.blocks.push_back(std::move(block)); + } + + void + PlaceInBlock(IR::Block &block, const Live::BlockFacts &facts) + { + std::vector rewritten; + rewritten.reserve(block.instructions.size()); + for (std::size_t index = 0U; index < block.instructions.size(); + ++index) + { + auto &instruction = block.instructions[index]; + // A copy of a symbol onto itself moves nothing. + if (IsSelfCopy(instruction) + && IsAarc(instruction.result_type)) + continue; + const auto &after = facts.accesses[index].liveAfter; + const auto writes = instruction.effect != Effect::Discard + && Owned(instruction.destination, + instruction.result_type); + + // The source of a copy that takes over its reference. + bool moves = false; + IR::SymbolId moved{}; + if (instruction.opcode == IR::Opcode::Copy + && instruction.effect != Effect::Discard + && instruction.operands.size() == 1U + && IsSymbol(instruction.operands.front()) + && IsAarc(instruction.result_type)) + { + const auto source = instruction.operands.front().symbol; + if (borrowed_.contains(source) + || Contains(after, source)) + instruction.opcode = IR::Opcode::RetainStrong; + else if (Owned(instruction.operands.front())) + { + moves = true; + moved = source; + } + } + + std::vector dying; + for (const auto &operand : instruction.operands) + if (Owned(operand) && !Contains(after, operand.symbol) + && (!moves || operand.symbol != moved) + && std::ranges::find(dying, operand.symbol) + == dying.end()) + dying.push_back(operand.symbol); + const auto destination = instruction.destination; + rewritten.push_back(std::move(instruction)); + for (const auto symbol : dying) + rewritten.push_back(ReleaseOf(symbol)); + if (writes && !Contains(after, destination)) + rewritten.push_back(ReleaseOf(destination)); + } + block.instructions = std::move(rewritten); + } + + IR::Function &function_; + const SymbolSet &methods_; + IR::SymbolId &nextSymbol_; + IR::BlockId nextBlock_{}; + /// Parameters: the caller keeps them alive. + SymbolSet borrowed_; + /// The type of each owned symbol, for the operand of its + /// release. + std::unordered_map types_; + }; + + /// One more than every symbol the module mentions. + [[nodiscard]] auto + FirstFreeSymbol(const IR::Module &module) -> IR::SymbolId + { + IR::SymbolId next = 1U; + const auto see = [&next](IR::SymbolId symbol) { + next = std::max(next, symbol + 1U); + }; + for (const auto &function : module.functions) + { + see(function.symbol.id); + for (const auto ¶meter : function.parameters) + see(parameter.symbol.id); + for (const auto &block : function.blocks) + { + for (const auto &instruction : block.instructions) + { + see(instruction.destination); + see(instruction.closure_function); + for (const auto &operand : instruction.operands) + if (IsSymbol(operand)) + see(operand.symbol); + } + if (IsSymbol(block.terminator.value)) + see(block.terminator.value.symbol); + } + } + return next; + } + } // namespace + + auto + PlaceOwnership(IR::Module module) -> IR::Module + { + SymbolSet methods; + methods.reserve(module.functions.size()); + for (const auto &function : module.functions) + methods.insert(function.symbol.id); + auto nextSymbol = FirstFreeSymbol(module); + for (auto &function : module.functions) + Placer(function, methods, nextSymbol).Run(); + return module; + } +} // namespace Visual::XSharp::Xpp diff --git a/Compiler/Codegen/Xpp/Tests/BUILD.bazel b/Compiler/Codegen/Xpp/Tests/BUILD.bazel index 8b7da028..aee333cf 100644 --- a/Compiler/Codegen/Xpp/Tests/BUILD.bazel +++ b/Compiler/Codegen/Xpp/Tests/BUILD.bazel @@ -7,6 +7,7 @@ cc_binary( srcs = [ "ControlFlowTests.cpp", "OptimizerTests.cpp", + "OwnershipPlacementTests.cpp", "OwnershipVerifierTests.cpp", "VerifierTests.cpp", ], diff --git a/Compiler/Codegen/Xpp/Tests/OwnershipPlacementTests.cpp b/Compiler/Codegen/Xpp/Tests/OwnershipPlacementTests.cpp new file mode 100644 index 00000000..eecdb3b4 --- /dev/null +++ b/Compiler/Codegen/Xpp/Tests/OwnershipPlacementTests.cpp @@ -0,0 +1,802 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Xpp/OwnershipPlacement.hpp" +#include "Visual/XSharp/Xpp/OwnershipVerifier.hpp" + +// The placement pass is checked against a model that knows nothing of how it +// works: the placed function is executed on reference counts, along every +// path through its graph, and every path must end with each reference +// accounted for. A release the pass forgot, a release it wrote twice and a +// use after the last release are different failures of the model. + +namespace +{ + namespace Core = ::visual_xsharp::core; + namespace IR = ::visual_xsharp::xpp; + namespace Xpp = ::Visual::XSharp::Xpp; + using Effect = IR::Instruction::Effect; + + constexpr IR::SymbolId kFunction = 1U; + constexpr IR::SymbolId kHelper = 2U; + + [[nodiscard]] auto + Text() -> Core::Type + { + return Core::Type::string(); + } + + [[nodiscard]] auto + Callable() -> Core::Type + { + return Core::Type::function({}, Core::Type::int64()); + } + + [[nodiscard]] auto + Symbol(IR::SymbolId symbol, Core::Type type = Text()) -> IR::Operand + { + return { IR::Operand::Kind::Symbol, + std::move(type), + symbol, + std::monostate{} }; + } + + [[nodiscard]] auto + Instruction(Effect effect, + IR::Opcode opcode, + IR::SymbolId destination, + Core::Type type, + std::vector operands) -> IR::Instruction + { + return { + effect, opcode, destination, std::move(type), std::move(operands), + 0U, {} + }; + } + + /// `destination = Helper(arguments...)`: a call that returns a value. + [[nodiscard]] auto + Call(IR::SymbolId destination, + const std::vector &arguments = {}, + Effect effect = Effect::Define) -> IR::Instruction + { + std::vector operands; + operands.reserve(arguments.size() + 1U); + operands.push_back(Symbol(kHelper, Callable())); + for (const auto argument : arguments) + operands.push_back(Symbol(argument)); + return Instruction(effect, + IR::Opcode::Call, + destination, + Text(), + std::move(operands)); + } + + [[nodiscard]] auto + Copy(IR::SymbolId destination, + IR::SymbolId source, + Effect effect = Effect::Define) -> IR::Instruction + { + return Instruction(effect, + IR::Opcode::Copy, + destination, + Text(), + { Symbol(source) }); + } + + /// `destination = closure capturing the given values strongly`. + [[nodiscard]] auto + Closure(IR::SymbolId destination, const std::vector &captures) + -> IR::Instruction + { + std::vector operands; + operands.reserve(captures.size()); + for (const auto capture : captures) + operands.push_back(Symbol(capture)); + auto closure = Instruction(Effect::Define, + IR::Opcode::MakeClosure, + destination, + Callable(), + std::move(operands)); + closure.closure_function = kHelper; + closure.capture_modes.assign(captures.size(), + Core::CaptureMode::Strong); + return closure; + } + + [[nodiscard]] auto + Return(std::optional symbol = std::nullopt) -> IR::Terminator + { + IR::Terminator terminator; + terminator.kind = IR::Terminator::Kind::Return; + terminator.value = symbol ? Symbol(*symbol) + : IR::Operand{ IR::Operand::Kind::Literal, + Core::Type::unit(), + 0U, + std::monostate{} }; + return terminator; + } + + [[nodiscard]] auto + Jump(IR::BlockId target) -> IR::Terminator + { + IR::Terminator terminator; + terminator.kind = IR::Terminator::Kind::Jump; + terminator.true_target = target; + return terminator; + } + + [[nodiscard]] auto + Branch(IR::BlockId whenTrue, IR::BlockId whenFalse) -> IR::Terminator + { + IR::Terminator terminator; + terminator.kind = IR::Terminator::Kind::Branch; + terminator.value + = { IR::Operand::Kind::Literal, Core::Type::boolean(), 0U, true }; + terminator.true_target = whenTrue; + terminator.false_target = whenFalse; + return terminator; + } + + [[nodiscard]] auto + Block(IR::BlockId id, + std::vector instructions, + IR::Terminator terminator) -> IR::Block + { + return { id, std::move(instructions), std::move(terminator) }; + } + + /// A module of the function under test and a helper it may call. + [[nodiscard]] auto + Module(const std::vector ¶meters, + std::vector blocks) -> IR::Module + { + IR::Function function; + function.symbol = { kFunction, U"Subject" }; + function.parameters.reserve(parameters.size()); + for (const auto parameter : parameters) + function.parameters.push_back( + { { parameter, U"parameter" }, Text() }); + function.return_type = Text(); + function.entry = blocks.front().id; + function.blocks = std::move(blocks); + + IR::Function helper; + helper.symbol = { kHelper, U"Helper" }; + helper.return_type = Text(); + helper.entry = 0U; + helper.blocks.push_back(Block(0U, {}, Return())); + + IR::Module module; + module.name = { U"Placement" }; + module.functions.push_back(std::move(function)); + module.functions.push_back(std::move(helper)); + return module; + } + + /** + * The reference-count model. + * + * A call result and a new closure are objects with one reference. A + * closure holds a reference to each strong capture until it is + * destroyed. A parameter enters with the reference of the caller. The + * function may run a block at most twice on one path, which takes + * every loop around once more than it needs to be entered. + */ + class Model final + { + public: + explicit Model(const IR::Function &function) + : function_(function) + {} + + /// The first failure on any path, or nothing. + [[nodiscard]] auto + Check() -> std::optional + { + State state; + for (const auto ¶meter : function_.parameters) + { + state.objects.push_back({ 1, {} }); + state.symbols[parameter.symbol.id] = state.objects.size() - 1U; + } + parameters_ = state.objects.size(); + Walk(function_.entry, state, {}); + return failure_; + } + + [[nodiscard]] auto + Paths() const -> std::size_t + { + return paths_; + } + + private: + struct Object final + { + std::int64_t references{}; + std::vector captures; + }; + struct State final + { + std::vector objects; + std::map symbols; + }; + + void + Fail(std::string message) + { + if (!failure_) + failure_ = std::move(message); + } + + void + Drop(State &state, std::size_t object) + { + if (state.objects[object].references <= 0) + { + Fail("released an object that has no reference left"); + return; + } + if (--state.objects[object].references != 0) + return; + const auto captures = state.objects[object].captures; + for (const auto capture : captures) + Drop(state, capture); + } + + /// The object a symbol holds, which must still have a reference. + [[nodiscard]] auto + Read(State &state, const IR::Operand &operand) + -> std::optional + { + if (operand.kind != IR::Operand::Kind::Symbol + || operand.symbol == kHelper || operand.symbol == kFunction) + return std::nullopt; + const auto found = state.symbols.find(operand.symbol); + if (found == state.symbols.end()) + { + Fail("read a symbol that holds nothing"); + return std::nullopt; + } + if (state.objects[found->second].references <= 0) + Fail("used an object after its last reference was released"); + return found->second; + } + + void + Execute(State &state, const IR::Instruction &instruction) + { + std::vector inputs; + for (const auto &operand : instruction.operands) + if (const auto object = Read(state, operand)) + inputs.push_back(*object); + std::optional result; + switch (instruction.opcode) + { + case IR::Opcode::Call: + state.objects.push_back({ 1, {} }); + result = state.objects.size() - 1U; + break; + case IR::Opcode::MakeClosure: + for (const auto capture : inputs) + ++state.objects[capture].references; + state.objects.push_back({ 1, inputs }); + result = state.objects.size() - 1U; + break; + case IR::Opcode::Copy: + if (!inputs.empty()) + result = inputs.front(); + break; + case IR::Opcode::RetainStrong: + if (!inputs.empty()) + { + ++state.objects[inputs.front()].references; + result = inputs.front(); + } + break; + case IR::Opcode::ReleaseStrong: + if (!inputs.empty()) + Drop(state, inputs.front()); + break; + default: + Fail("the model does not know this operation"); + break; + } + if (instruction.effect != Effect::Discard && result) + state.symbols[instruction.destination] = *result; + else if (instruction.effect == Effect::Discard && result + && instruction.opcode != IR::Opcode::Copy) + // The backend releases a closure that nothing receives. + // Any other discarded result has no owner. + instruction.opcode == IR::Opcode::MakeClosure + ? Drop(state, *result) + : Fail("an owned result was discarded"); + } + + void + Finish(State state, const IR::Terminator &terminator) + { + ++paths_; + // The caller releases what it receives. + if (const auto returned = Read(state, terminator.value)) + Drop(state, *returned); + for (std::size_t object = 0U; object < state.objects.size(); + ++object) + { + const auto expected = object < parameters_ ? 1 : 0; + if (state.objects[object].references > expected) + Fail("a path ends with a reference that nobody " + "releases"); + if (state.objects[object].references < expected) + Fail("a path released a reference it did not own"); + } + } + + void + Walk(IR::BlockId id, State state, std::map visits) + { + if (failure_ || ++visits[id] > 2) + return; + const auto block + = std::ranges::find(function_.blocks, id, &IR::Block::id); + if (block == function_.blocks.end()) + { + Fail("an edge leads to a block that does not exist"); + return; + } + for (const auto &instruction : block->instructions) + Execute(state, instruction); + const auto &terminator = block->terminator; + switch (terminator.kind) + { + case IR::Terminator::Kind::Return: + Finish(std::move(state), terminator); + break; + case IR::Terminator::Kind::Jump: + Walk(terminator.true_target, std::move(state), visits); + break; + case IR::Terminator::Kind::Branch: + Walk(terminator.true_target, state, visits); + Walk(terminator.false_target, std::move(state), visits); + break; + case IR::Terminator::Kind::Unreachable: + break; + } + } + + const IR::Function &function_; + std::size_t parameters_{}; + std::size_t paths_{}; + std::optional failure_; + }; + + [[nodiscard]] auto + Count(const IR::Function &function, IR::Opcode opcode) -> std::size_t + { + std::size_t count = 0U; + for (const auto &block : function.blocks) + count += static_cast( + std::ranges::count(block.instructions, + opcode, + &IR::Instruction::opcode)); + return count; + } + + /// Place the ownership of a module and require a balanced result that + /// the ownership verifier accepts. + [[nodiscard]] auto + Placed(IR::Module module) -> IR::Function + { + auto placed = Xpp::PlaceOwnership(std::move(module)); + const auto issues = Xpp::VerifyOwnership(placed); + for (const auto &issue : issues) + UNSCOPED_INFO(issue.code << ": " << issue.message); + CHECK(issues.empty()); + Model model(placed.functions.front()); + const auto failure = model.Check(); + if (failure) + UNSCOPED_INFO(*failure); + CHECK_FALSE(failure.has_value()); + CHECK(model.Paths() > 0U); + return std::move(placed.functions.front()); + } +} // namespace + +TEST_CASE("the model rejects what the pass must not produce", + "[xpp][ownership][placement]") +{ + // Without the pass the same functions are wrong in the three ways the + // model tells apart, so a pass that did nothing could not pass below. + const auto check = [](std::vector blocks) { + const auto module = Module({}, std::move(blocks)); + return Model(module.functions.front()).Check(); + }; + const auto release = [](IR::SymbolId symbol) { + return Instruction(Effect::Discard, + IR::Opcode::ReleaseStrong, + 0U, + Core::Type::unit(), + { Symbol(symbol) }); + }; + std::vector leak; + leak.push_back(Block(0U, { Call(10U) }, Return())); + const auto leaked = check(std::move(leak)); + CHECK(leaked.has_value()); + + std::vector twice; + twice.push_back( + Block(0U, { Call(10U), release(10U), release(10U) }, Return())); + const auto releasedTwice = check(std::move(twice)); + CHECK(releasedTwice.has_value()); + + std::vector late; + late.push_back( + Block(0U, { Call(10U), release(10U), Call(11U, { 10U }) }, Return())); + const auto usedLate = check(std::move(late)); + CHECK(usedLate.has_value()); + + std::vector balanced; + balanced.push_back(Block(0U, { Call(10U), release(10U) }, Return())); + const auto sound = check(std::move(balanced)); + CHECK_FALSE(sound.has_value()); +} + +TEST_CASE("a value is released after its last use on a straight path", + "[xpp][ownership][placement]") +{ + SECTION("a value nothing uses is released where it is defined") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Return())); + const auto function = Placed(Module({}, std::move(blocks))); + REQUIRE(function.blocks.front().instructions.size() == 2U); + CHECK(function.blocks.front().instructions[1].opcode + == IR::Opcode::ReleaseStrong); + } + SECTION("a value is released after the call that uses it last") + { + std::vector blocks; + blocks.push_back( + Block(0U, + { Call(10U), Call(11U, { 10U }), Call(12U, { 10U, 11U }) }, + Return())); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 3U); + // Nothing is released before the last call has read it. + CHECK(function.blocks.front().instructions[1].opcode + == IR::Opcode::Call); + CHECK(function.blocks.front().instructions[2].opcode + == IR::Opcode::Call); + } + SECTION("a returned value leaves with its reference") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U), Call(11U) }, Return(10U))); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 1U); + } + SECTION("a discarded result is given a symbol and released") + { + std::vector blocks; + blocks.push_back( + Block(0U, { Call(0U, {}, Effect::Discard) }, Return())); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(function.blocks.front().instructions.front().effect + == Effect::Define); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 1U); + } +} + +TEST_CASE("a copy shares or takes over the reference of its source", + "[xpp][ownership][placement]") +{ + SECTION("the source is used again: the copy retains") + { + std::vector blocks; + blocks.push_back( + Block(0U, + { Call(10U), Copy(11U, 10U), Call(12U, { 10U, 11U }) }, + Return())); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(Count(function, IR::Opcode::RetainStrong) == 1U); + CHECK(Count(function, IR::Opcode::Copy) == 0U); + } + SECTION("the source is not used again: the copy moves") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U), Copy(11U, 10U) }, Return(11U))); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(Count(function, IR::Opcode::RetainStrong) == 0U); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 0U); + } + SECTION("a store over a value that is still needed elsewhere") + { + std::vector blocks; + blocks.push_back(Block(0U, + { Call(10U), + Call(11U), + Copy(10U, 11U, Effect::Store), + Call(12U, { 10U, 11U }) }, + Return())); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("an instruction that reads the symbol it writes") + { + std::vector blocks; + blocks.push_back(Block(0U, + { Call(10U), Call(10U, { 10U }, Effect::Store) }, + Return(10U))); + const auto function = Placed(Module({}, std::move(blocks))); + // The old value is read from a symbol of its own and released. + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 1U); + } +} + +TEST_CASE("a parameter is borrowed and a result is owned", + "[xpp][ownership][placement]") +{ + SECTION("a parameter that is only read is never released") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U, { 5U }) }, Return())); + const auto function = Placed(Module({ 5U }, std::move(blocks))); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 1U); + CHECK(Count(function, IR::Opcode::RetainStrong) == 0U); + } + SECTION("a returned parameter is retained for the caller") + { + std::vector blocks; + blocks.push_back(Block(0U, {}, Return(5U))); + const auto function = Placed(Module({ 5U }, std::move(blocks))); + CHECK(Count(function, IR::Opcode::RetainStrong) == 1U); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 0U); + } + SECTION("a copy of a parameter has a reference of its own") + { + std::vector blocks; + blocks.push_back(Block(0U, { Copy(10U, 5U) }, Return(10U))); + const auto function = Placed(Module({ 5U }, std::move(blocks))); + CHECK(Count(function, IR::Opcode::RetainStrong) == 1U); + } + SECTION("a parameter the body assigns is owned by the body") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U, { 5U }) }, Branch(1U, 2U))); + blocks.push_back(Block(1U, { Call(5U, {}, Effect::Store) }, Jump(2U))); + blocks.push_back(Block(2U, { Call(11U, { 5U }) }, Return(5U))); + const auto function = Placed(Module({ 5U }, std::move(blocks))); + // The entry is a block of its own, so the copy runs once. + CHECK(function.entry != 0U); + CHECK(function.blocks.front().instructions.front().opcode + == IR::Opcode::RetainStrong); + } +} + +TEST_CASE("a value that dies on one edge is released on that edge", + "[xpp][ownership][placement]") +{ + SECTION("the target has one predecessor: the release opens it") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Branch(1U, 2U))); + blocks.push_back(Block(1U, { Call(11U, { 10U }) }, Return())); + blocks.push_back(Block(2U, {}, Return())); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(function.blocks.size() == 3U); + CHECK(function.blocks[2].instructions.front().opcode + == IR::Opcode::ReleaseStrong); + } + SECTION("the target has other predecessors: the edge gets a block") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Branch(1U, 2U))); + blocks.push_back(Block(1U, { Call(11U, { 10U }) }, Jump(2U))); + blocks.push_back(Block(2U, {}, Return())); + const auto function = Placed(Module({}, std::move(blocks))); + REQUIRE(function.blocks.size() == 4U); + CHECK(function.blocks[3].instructions.front().opcode + == IR::Opcode::ReleaseStrong); + CHECK(function.blocks[3].terminator.true_target == 2U); + CHECK(function.blocks[0].terminator.false_target + == function.blocks[3].id); + // The join itself releases nothing: each way in already has. + CHECK(function.blocks[2].instructions.empty()); + } + SECTION("a value defined on one side only") + { + std::vector blocks; + blocks.push_back(Block(0U, {}, Branch(1U, 2U))); + blocks.push_back( + Block(1U, { Call(10U), Call(11U, { 10U }) }, Jump(2U))); + blocks.push_back(Block(2U, {}, Return())); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("a value returned on one side and dropped on the other") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U), Call(11U) }, Branch(1U, 2U))); + blocks.push_back(Block(1U, {}, Return(10U))); + blocks.push_back(Block(2U, {}, Return(11U))); + (void)Placed(Module({}, std::move(blocks))); + } +} + +TEST_CASE("values in loops are released once for each time they are made", + "[xpp][ownership][placement]") +{ + SECTION("a value made in every pass") + { + std::vector blocks; + blocks.push_back(Block(0U, {}, Jump(1U))); + blocks.push_back(Block(1U, {}, Branch(2U, 3U))); + blocks.push_back( + Block(2U, { Call(10U), Call(11U, { 10U }) }, Jump(1U))); + blocks.push_back(Block(3U, {}, Return())); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("a value made before the loop and used in it and after it") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Jump(1U))); + blocks.push_back(Block(1U, {}, Branch(2U, 3U))); + blocks.push_back(Block(2U, { Call(11U, { 10U }) }, Jump(1U))); + blocks.push_back(Block(3U, { Call(12U, { 10U }) }, Return())); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("a value made before the loop and used only in it") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Jump(1U))); + blocks.push_back(Block(1U, {}, Branch(2U, 3U))); + blocks.push_back(Block(2U, { Call(11U, { 10U }) }, Jump(1U))); + blocks.push_back(Block(3U, {}, Return())); + const auto function = Placed(Module({}, std::move(blocks))); + // It dies on the way out of the loop, not inside it. + CHECK(function.blocks[3].instructions.front().opcode + == IR::Opcode::ReleaseStrong); + } + SECTION("a variable reassigned in every pass and returned after") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Jump(1U))); + blocks.push_back(Block(1U, {}, Branch(2U, 3U))); + blocks.push_back( + Block(2U, { Call(10U, { 10U }, Effect::Store) }, Jump(1U))); + blocks.push_back(Block(3U, {}, Return(10U))); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("a variable overwritten in the loop without being read") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Jump(1U))); + blocks.push_back(Block(1U, {}, Branch(2U, 3U))); + blocks.push_back(Block(2U, { Call(10U, {}, Effect::Store) }, Jump(1U))); + blocks.push_back(Block(3U, {}, Return(10U))); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("a loop that leaves from its middle") + { + std::vector blocks; + blocks.push_back(Block(0U, { Call(10U) }, Jump(1U))); + blocks.push_back(Block(1U, { Call(11U) }, Branch(2U, 4U))); + blocks.push_back( + Block(2U, { Call(12U, { 10U, 11U }) }, Branch(3U, 4U))); + blocks.push_back(Block(3U, {}, Jump(1U))); + blocks.push_back(Block(4U, {}, Return(10U))); + (void)Placed(Module({}, std::move(blocks))); + } +} + +TEST_CASE("closures own their captures", "[xpp][ownership][placement]") +{ + SECTION("a capture is released by its owner and kept by the closure") + { + std::vector blocks; + blocks.push_back( + Block(0U, { Call(10U), Closure(11U, { 10U }) }, Return(11U))); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 1U); + } + SECTION("a closure that nothing uses is released with its captures") + { + std::vector blocks; + blocks.push_back(Block( + 0U, + { Call(10U), Closure(11U, { 10U }), Closure(12U, { 10U, 11U }) }, + Return())); + (void)Placed(Module({}, std::move(blocks))); + } + SECTION("a captured parameter stays the caller's") + { + std::vector blocks; + blocks.push_back(Block(0U, { Closure(11U, { 5U }) }, Return(11U))); + const auto function = Placed(Module({ 5U }, std::move(blocks))); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 0U); + } +} + +TEST_CASE("a method used as a value becomes a closure", + "[xpp][ownership][placement]") +{ + const auto method = [] { + return Symbol(kHelper, Callable()); + }; + SECTION("stored in a local") + { + std::vector blocks; + blocks.push_back(Block(0U, + { Instruction(Effect::Define, + IR::Opcode::Copy, + 10U, + Callable(), + { method() }) }, + Return())); + const auto function = Placed(Module({}, std::move(blocks))); + const auto &first = function.blocks.front().instructions.front(); + CHECK(first.opcode == IR::Opcode::MakeClosure); + CHECK(first.closure_function == kHelper); + CHECK(first.operands.empty()); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 1U); + } + SECTION("passed as an argument, while the callee stays a method") + { + std::vector blocks; + blocks.push_back(Block(0U, + { Instruction(Effect::Define, + IR::Opcode::Call, + 10U, + Text(), + { method(), method() }) }, + Return(10U))); + const auto function = Placed(Module({}, std::move(blocks))); + const auto &instructions = function.blocks.front().instructions; + REQUIRE(instructions.size() == 3U); + CHECK(instructions[0].opcode == IR::Opcode::MakeClosure); + CHECK(instructions[1].opcode == IR::Opcode::Call); + CHECK(instructions[1].operands[0].symbol == kHelper); + CHECK(instructions[1].operands[1].symbol + == instructions[0].destination); + CHECK(instructions[2].opcode == IR::Opcode::ReleaseStrong); + } + SECTION("returned") + { + std::vector blocks; + IR::Terminator terminator; + terminator.kind = IR::Terminator::Kind::Return; + terminator.value = method(); + blocks.push_back(Block(0U, {}, std::move(terminator))); + const auto function = Placed(Module({}, std::move(blocks))); + CHECK(Count(function, IR::Opcode::MakeClosure) == 1U); + CHECK(Count(function, IR::Opcode::ReleaseStrong) == 0U); + } +} + +TEST_CASE("what holds no reference is left alone", + "[xpp][ownership][placement]") +{ + std::vector blocks; + blocks.push_back(Block(0U, + { Instruction(Effect::Define, + IR::Opcode::Add, + 10U, + Core::Type::int64(), + {}) }, + Return())); + // A block nothing reaches keeps its instructions as they are. + blocks.push_back(Block(7U, { Call(11U) }, Return())); + const auto before = Module({}, std::move(blocks)); + const auto placed = Xpp::PlaceOwnership(before); + CHECK(placed == before); +} diff --git a/Compiler/Codegen/Xpp/Verifier.cpp b/Compiler/Codegen/Xpp/Verifier.cpp index 081aa5bb..9de9c249 100644 --- a/Compiler/Codegen/Xpp/Verifier.cpp +++ b/Compiler/Codegen/Xpp/Verifier.cpp @@ -15,6 +15,7 @@ #include "Visual/XSharp/Analysis/DefiniteInitialization.hpp" #include "Visual/XSharp/Core/Callable.hpp" #include "Visual/XSharp/Core/Ownership.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" #include "Visual/XSharp/Core/Scalar.hpp" #include "Visual/XSharp/Xpp/OwnershipVerifier.hpp" #include "Visual/XSharp/Xpp/Verifier.hpp" @@ -70,6 +71,7 @@ namespace Visual::XSharp::Xpp case IR::Opcode::Negate: case IR::Opcode::LogicalNot: case IR::Opcode::BitwiseNot: + case IR::Opcode::Memoize: case IR::Opcode::RetainStrong: case IR::Opcode::ReleaseStrong: case IR::Opcode::MakeWeak: @@ -342,6 +344,52 @@ namespace Visual::XSharp::Xpp "type test requires a reference subject, uint " "identity and Bool result"); } + else if (value.opcode == IR::Opcode::RuntimeCall) + { + // The function is named by a literal, never by a value that + // is computed: what is called is fixed when the program is + // compiled. + namespace runtime = Core::runtime; + const runtime::Signature *signature = nullptr; + if (!value.operands.empty() + && value.operands.front().kind + == IR::Operand::Kind::Literal) + if (const auto identity + = runtime::IdentityOf(value.operands.front().literal, + value.operands.front().type)) + signature = runtime::Find(*identity); + std::vector arguments; + arguments.reserve(value.operands.size()); + for (std::size_t index = 1U; index < value.operands.size(); + ++index) + arguments.push_back(value.operands[index].type.kind); + const auto defect = runtime::Check(signature, + arguments, + value.result_type.kind); + if (defect != runtime::Defect::None) + context.Add("VXP1048", + std::string(runtime::Describe(defect))); + } + else if (value.opcode == IR::Opcode::Memoize) + { + // The result is a new owner of the same callable type; the + // remembered value is kept in it and owns nothing. + const auto remembers + = value.operands.size() == 1U + && value.operands.front().type.kind + == Core::Type::Kind::Function + && value.operands.front().type.components.size() == 1U + && (value.operands.front().type.components.front().kind + == Core::Type::Kind::Bool + || Core::is_numeric( + value.operands.front().type.components.front())); + if (!remembers + || value.result_type != value.operands.front().type) + context.Add("VXP1047", + "memoization requires one callable without " + "parameters whose result is Bool or numeric, " + "and yields a callable of the same type"); + } else if (value.opcode == IR::Opcode::FloorDivide) { const auto hasNumericPair diff --git a/Compiler/Codegen/Xpp/Wire.cpp b/Compiler/Codegen/Xpp/Wire.cpp index d0570a5b..279740aa 100644 --- a/Compiler/Codegen/Xpp/Wire.cpp +++ b/Compiler/Codegen/Xpp/Wire.cpp @@ -145,7 +145,7 @@ namespace Visual::XSharp::Xpp::Wire ReadTag(reader, 2U, "Xpp instruction effect")); instruction.opcode = static_cast( ReadTag(reader, - static_cast(IR::Opcode::TypeIs), + static_cast(IR::Opcode::RuntimeCall), "Xpp opcode")); instruction.destination = reader.U64("Xpp destination"); instruction.result_type = reader.Type("Xpp instruction result"); diff --git a/Compiler/Core/BUILD.bazel b/Compiler/Core/BUILD.bazel index 47af4b5a..ef3f7c07 100644 --- a/Compiler/Core/BUILD.bazel +++ b/Compiler/Core/BUILD.bazel @@ -16,7 +16,9 @@ cc_library( "Scalar.cpp", "Template.cpp", "Verifier.cpp", - "Wire.cpp", + "Wire/Decode.cpp", + "Wire/Encode.cpp", + "Wire/Magic.hpp", ], deps = [ "//Compiler/Artifact:source_path", diff --git a/Compiler/Core/CorePrep/Benches/CorePrepBenches.cpp b/Compiler/Core/CorePrep/Benches/CorePrepBenches.cpp index 79145f99..5ef8bfac 100644 --- a/Compiler/Core/CorePrep/Benches/CorePrepBenches.cpp +++ b/Compiler/Core/CorePrep/Benches/CorePrepBenches.cpp @@ -76,6 +76,29 @@ namespace return { { U"Benchmark" }, { std::move(function) } }; } + // A module of the given number of functions, each of which returns a + // constant. What a function sees of the functions around it is part of + // verifying it, so the cost of that view shows as the count grows. + [[nodiscard]] auto + MakeFunctions(std::size_t functionCount) -> Core::CorePrepModule + { + std::vector functions; + functions.reserve(functionCount); + for (std::size_t index = 0; index < functionCount; ++index) + { + Core::Terminator terminator; + terminator.kind = Core::Terminator::Kind::Return; + terminator.value + = Core::Atom::constant(std::int64_t{ 0 }, Core::Type::int64()); + functions.push_back({ Symbol(index + 1U), + {}, + Core::Type::int64(), + 0U, + { { 0U, {}, std::move(terminator) } } }); + } + return { { U"Benchmark" }, std::move(functions) }; + } + void RequireValid(const Core::CorePrepModule &module) { @@ -102,6 +125,22 @@ namespace state.SetComplexityN(state.range(0)); } + void + VerifyFunctions(benchmark::State &state) + { + const auto module + = MakeFunctions(static_cast(state.range(0))); + RequireValid(module); + for (auto _ : state) + { + const auto issues = Core::verify(module); + benchmark::DoNotOptimize(issues.data()); + benchmark::DoNotOptimize(issues.size()); + } + state.SetItemsProcessed(state.iterations() * state.range(0)); + state.SetComplexityN(state.range(0)); + } + void Encode(benchmark::State &state) { @@ -144,12 +183,18 @@ namespace constexpr auto kMinimumBlocks = 4; constexpr auto kMaximumBlocks = 256; + constexpr auto kMinimumFunctions = 64; + constexpr auto kMaximumFunctions = 4096; } // namespace BENCHMARK(Verify) ->RangeMultiplier(4) ->Range(kMinimumBlocks, kMaximumBlocks) ->Complexity(); +BENCHMARK(VerifyFunctions) + ->RangeMultiplier(4) + ->Range(kMinimumFunctions, kMaximumFunctions) + ->Complexity(); BENCHMARK(Encode) ->RangeMultiplier(4) ->Range(kMinimumBlocks, kMaximumBlocks) diff --git a/Compiler/Core/CorePrep/Prepare.cpp b/Compiler/Core/CorePrep/Prepare.cpp index 5597653d..a75f034b 100644 --- a/Compiler/Core/CorePrep/Prepare.cpp +++ b/Compiler/Core/CorePrep/Prepare.cpp @@ -3,6 +3,7 @@ #include #include +#include #include #include "Visual/XSharp/Core/CorePrep/Prepare.hpp" @@ -109,6 +110,10 @@ namespace Visual::XSharp::Core::CorePrep return Prepared::Operation::BitwiseNot; case Primitive::TypeIs: return Prepared::Operation::TypeIs; + case Primitive::Memoize: + return Prepared::Operation::Memoize; + case Primitive::RuntimeCall: + return Prepared::Operation::RuntimeCall; } // Reaching this point means the Core enum and adapter diverged. // C++20 has no std::unreachable; abort explicitly instead of @@ -134,14 +139,14 @@ namespace Visual::XSharp::Core::CorePrep /// False after a return, break or continue ended the region. bool open{ true }; - void + [[gnu::noinline]] void Emit(Prepared::Instruction instruction) { instructions.push_back(std::move(instruction)); } /// End the current block; no block is open until Open is called. - void + [[gnu::noinline]] void Close(Prepared::Terminator terminator) { closed.push_back( @@ -158,13 +163,13 @@ namespace Visual::XSharp::Core::CorePrep open = true; } - void + [[gnu::noinline]] void Jump(const Prepared::BlockId target) { Close({ Prepared::Terminator::Kind::Jump, {}, target, 0U }); } - void + [[gnu::noinline]] void Branch(Prepared::Atom condition, const Prepared::BlockId whenTrue, const Prepared::BlockId whenFalse) @@ -175,7 +180,7 @@ namespace Visual::XSharp::Core::CorePrep whenFalse }); } - [[nodiscard]] auto + [[nodiscard, gnu::noinline]] auto Temporary(std::u32string prefix) -> SymbolName { const auto id = state.nextSymbol++; @@ -193,20 +198,15 @@ namespace Visual::XSharp::Core::CorePrep std::vector captures; }; + // The adapter recurses once per level of nesting, and an + // instruction, an atom and a type are large. The functions on the + // recursive paths therefore hold as few of them as they can: every + // instruction is built by a function of its own, which returns + // before the next level is entered, and the shapes that nest as + // deep as an expression is long are walked in a loop. [[nodiscard]] auto Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom; - [[nodiscard]] auto - AtomizeMany(Cursor &cursor, const std::vector &expressions) - -> std::vector - { - std::vector atoms; - atoms.reserve(expressions.size()); - for (const auto &expression : expressions) - atoms.push_back(Atomize(cursor, expression)); - return atoms; - } - [[nodiscard]] auto ZeroForBooleanContext(const Type &type) -> Prepared::Atom { @@ -220,7 +220,7 @@ namespace Visual::XSharp::Core::CorePrep /// Canonicalize a numeric truth value to `value != 0` in the open /// block; a Boolean atom is returned unchanged. - [[nodiscard]] auto + [[nodiscard, gnu::noinline]] auto Booleanize(Cursor &cursor, Prepared::Atom atom) -> Prepared::Atom { if (atom.type == Type::boolean()) @@ -252,6 +252,47 @@ namespace Visual::XSharp::Core::CorePrep && expression.operands.size() == 2U; } + /// Emit an instruction that takes the result of an operation. + [[gnu::noinline]] void + EmitOperation(Cursor &cursor, + const Prepared::Instruction::Kind kind, + const SymbolName &symbol, + const Type &type, + const bool mutableBinding, + OperationResult &operation) + { + cursor.Emit( + Prepared::Instruction{ kind, + symbol, + type, + mutableBinding, + operation.operation, + std::move(operation.operands), + std::move(operation.closureFunction), + std::move(operation.captures) }); + } + + /// Emit a binding or an assignment that copies one atom. + [[gnu::noinline]] void + EmitCopy(Cursor &cursor, + const Prepared::Instruction::Kind kind, + const SymbolName &symbol, + const Type &type, + const bool mutableBinding, + Prepared::Atom value) + { + cursor.Emit(Prepared::Instruction{ + kind, + symbol, + type, + mutableBinding, + Prepared::Operation::Copy, + { std::move(value) }, + {}, + {}, + }); + } + /** * @brief Lower `&&` and `||` as control flow, not as an eager operator. * @@ -263,27 +304,25 @@ namespace Visual::XSharp::Core::CorePrep * join therefore carries an initialized Boolean without a phi node * in CorePrep's storage-oriented form. This matches the Haskell * CorePrep lowering. + * + * The left operand has been evaluated by the caller, which walks a + * chain of operators in a loop. */ - [[nodiscard]] auto - AtomizeShortCircuit(Cursor &cursor, const Expression &expression) - -> Prepared::Atom + [[nodiscard, gnu::noinline]] auto + FinishShortCircuit(Cursor &cursor, + const Expression &expression, + Prepared::Atom left) -> Prepared::Atom { const auto isOr = expression.primitive == Primitive::LogicalOr; - auto condition - = Booleanize(cursor, - Atomize(cursor, expression.operands.front())); + auto condition = Booleanize(cursor, std::move(left)); auto result = cursor.Temporary(U"$shortcircuit"); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Bind, - result, - Type::boolean(), - true, - Prepared::Operation::Copy, - { Prepared::Atom::constant(Prepared::Literal{ isOr }, - Type::boolean()) }, - {}, - {}, - }); + EmitCopy(cursor, + Prepared::Instruction::Kind::Bind, + result, + Type::boolean(), + true, + Prepared::Atom::constant(Prepared::Literal{ isOr }, + Type::boolean())); const auto rightId = cursor.state.nextBlock; const auto joinId = rightId + 1U; cursor.state.nextBlock = joinId + 1U; @@ -293,19 +332,13 @@ namespace Visual::XSharp::Core::CorePrep cursor.Branch(std::move(condition), rightId, joinId); cursor.Open(rightId); - auto right - = Booleanize(cursor, - Atomize(cursor, expression.operands.back())); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Assign, - result, - Type::boolean(), - false, - Prepared::Operation::Copy, - { std::move(right) }, - {}, - {}, - }); + EmitCopy(cursor, + Prepared::Instruction::Kind::Assign, + result, + Type::boolean(), + false, + Booleanize(cursor, + Atomize(cursor, expression.operands.back()))); cursor.Jump(joinId); cursor.Open(joinId); return Prepared::Atom::variable(std::move(result), Type::boolean()); @@ -318,6 +351,70 @@ namespace Visual::XSharp::Core::CorePrep && expression.operands.size() == 3U; } + /// The slot and the join of a conditional expression whose false + /// arm is being evaluated. + struct OpenConditional final + { + SymbolName result; + const Type *type; + Prepared::BlockId joinId; + }; + + /// Evaluate the test and the true arm of a conditional expression + /// and leave its false block open. + [[gnu::noinline]] void + BeginConditional(Cursor &cursor, + const Expression &expression, + std::vector &open) + { + auto condition + = Booleanize(cursor, Atomize(cursor, expression.operands[0])); + auto result = cursor.Temporary(U"$conditional"); + EmitCopy(cursor, + Prepared::Instruction::Kind::Bind, + result, + expression.type, + true, + expression.type == Type::boolean() + ? Prepared::Atom::constant(Prepared::Literal{ false }, + Type::boolean()) + : ZeroForBooleanContext(expression.type)); + const auto trueId = cursor.state.nextBlock; + const auto falseId = trueId + 1U; + const auto joinId = falseId + 1U; + cursor.state.nextBlock = joinId + 1U; + cursor.Branch(std::move(condition), trueId, falseId); + + cursor.Open(trueId); + EmitCopy(cursor, + Prepared::Instruction::Kind::Assign, + result, + expression.type, + false, + Atomize(cursor, expression.operands[1])); + cursor.Jump(joinId); + cursor.Open(falseId); + open.push_back({ std::move(result), &expression.type, joinId }); + } + + /// Store the value of the false arm and continue in the join. + [[nodiscard, gnu::noinline]] auto + EndConditional(Cursor &cursor, + const OpenConditional &conditional, + Prepared::Atom value) -> Prepared::Atom + { + EmitCopy(cursor, + Prepared::Instruction::Kind::Assign, + conditional.result, + *conditional.type, + false, + std::move(value)); + cursor.Jump(conditional.joinId); + cursor.Open(conditional.joinId); + return Prepared::Atom::variable(conditional.result, + *conditional.type); + } + /** * @brief Lower a conditional expression to a branch over a slot. * @@ -328,55 +425,30 @@ namespace Visual::XSharp::Core::CorePrep * The neutral value is never observable: no path reaches the join * without passing through one of the two assignments. * - * The symbol and block allocation order matches the Haskell - * CorePrep lowering. + * A chain `a ? b : c ? d : e` nests in the false arm of each link. + * The links are lowered in a loop and their joins are completed + * from the innermost outwards, which creates the symbols and the + * blocks in the order of the recursive formulation. That order + * matches the Haskell CorePrep lowering. */ - [[nodiscard]] auto + [[nodiscard, gnu::noinline]] auto AtomizeConditional(Cursor &cursor, const Expression &expression) -> Prepared::Atom { - auto condition - = Booleanize(cursor, Atomize(cursor, expression.operands[0])); - auto result = cursor.Temporary(U"$conditional"); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Bind, - result, - expression.type, - true, - Prepared::Operation::Copy, - { expression.type == Type::boolean() - ? Prepared::Atom::constant(Prepared::Literal{ false }, - Type::boolean()) - : ZeroForBooleanContext(expression.type) }, - {}, - {}, - }); - const auto trueId = cursor.state.nextBlock; - const auto falseId = trueId + 1U; - const auto joinId = falseId + 1U; - cursor.state.nextBlock = joinId + 1U; - cursor.Branch(std::move(condition), trueId, falseId); - - const auto prepareArm - = [&](const Prepared::BlockId armId, const Expression &arm) { - cursor.Open(armId); - auto value = Atomize(cursor, arm); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Assign, - result, - expression.type, - false, - Prepared::Operation::Copy, - { std::move(value) }, - {}, - {}, - }); - cursor.Jump(joinId); - }; - prepareArm(trueId, expression.operands[1]); - prepareArm(falseId, expression.operands[2]); - cursor.Open(joinId); - return Prepared::Atom::variable(std::move(result), expression.type); + std::vector open; + const Expression *current = &expression; + do + { + BeginConditional(cursor, *current, open); + current = ¤t->operands[2]; + } while (IsConditional(*current)); + auto value = Atomize(cursor, *current); + while (!open.empty()) + { + value = EndConditional(cursor, open.back(), std::move(value)); + open.pop_back(); + } + return value; } struct PreparedFunctionResult final @@ -386,89 +458,79 @@ namespace Visual::XSharp::Core::CorePrep SymbolId nextSymbol{}; }; - [[nodiscard]] auto - HighestExpressionSymbol(const Expression &expression) -> SymbolId; - - [[nodiscard]] auto - HighestStatementSymbol(const Statement &statement) -> SymbolId - { - SymbolId highest = 0U; - if (statement.kind == Statement::Kind::Bind) - { - highest = statement.binding.symbol.id; - highest = std::max( - highest, - HighestExpressionSymbol(statement.binding.value)); - } - else - { - if (statement.kind == Statement::Kind::Assign) - highest = statement.destination.id; - highest - = std::max(highest, - HighestExpressionSymbol(statement.expression)); - } - for (const auto &nested : statement.trueBranch) - highest = std::max(highest, HighestStatementSymbol(nested)); - for (const auto &nested : statement.falseBranch) - highest = std::max(highest, HighestStatementSymbol(nested)); - for (const auto &nested : statement.loopBody) - highest = std::max(highest, HighestStatementSymbol(nested)); - for (const auto &nested : statement.loopUpdate) - highest = std::max(highest, HighestStatementSymbol(nested)); - return highest; - } - - [[nodiscard]] auto - HighestExpressionSymbol(const Expression &expression) -> SymbolId - { - SymbolId highest = expression.kind == Expression::Kind::Variable - ? expression.symbol.id - : 0U; - if (expression.callee) - highest = std::max(highest, - HighestExpressionSymbol(*expression.callee)); - for (const auto &operand : expression.operands) - highest = std::max(highest, HighestExpressionSymbol(operand)); - for (const auto &capture : expression.captures) - { - highest = std::max(highest, capture.symbol.id); - if (capture.value) - highest = std::max(highest, - HighestExpressionSymbol(*capture.value)); - } - for (const auto &[parameter, type] : expression.closureParameters) - { - static_cast(type); - highest = std::max(highest, parameter.id); - } - if (expression.closureBody) - for (const auto &statement : *expression.closureBody) - highest - = std::max(highest, HighestStatementSymbol(statement)); - if (expression.kind == Expression::Kind::Let) - { - highest = std::max(highest, expression.letSymbol.id); - if (expression.letValue) - highest = std::max( - highest, - HighestExpressionSymbol(*expression.letValue)); - if (expression.letBody) - highest = std::max( - highest, - HighestExpressionSymbol(*expression.letBody)); - } - return highest; - } - + /** + * @brief The highest symbol identity a function mentions. + * + * Every statement and expression of the function is visited once, + * from a worklist: the scan runs before anything else looks at the + * function, and a recursive one would use stack in proportion to + * the nesting of the body and to the length of every chain in it. + */ [[nodiscard]] auto HighestFunctionSymbol(const Function &function) -> SymbolId { SymbolId highest = function.symbol.id; for (const auto ¶meter : function.parameters) highest = std::max(highest, parameter.symbol.id); - for (const auto &statement : function.body) - highest = std::max(highest, HighestStatementSymbol(statement)); + + std::vector statements; + std::vector expressions; + const auto addStatements + = [&statements](const std::vector &nested) { + for (const auto &statement : nested) + statements.push_back(&statement); + }; + addStatements(function.body); + while (!statements.empty() || !expressions.empty()) + { + if (!expressions.empty()) + { + const auto &expression = *expressions.back(); + expressions.pop_back(); + if (expression.kind == Expression::Kind::Variable) + highest = std::max(highest, expression.symbol.id); + if (expression.callee) + expressions.push_back(expression.callee.get()); + for (const auto &operand : expression.operands) + expressions.push_back(&operand); + for (const auto &capture : expression.captures) + { + highest = std::max(highest, capture.symbol.id); + if (capture.value) + expressions.push_back(capture.value.get()); + } + for (const auto ¶meter : expression.closureParameters) + highest = std::max(highest, parameter.first.id); + if (expression.closureBody) + addStatements(*expression.closureBody); + if (expression.kind == Expression::Kind::Let) + { + highest = std::max(highest, expression.letSymbol.id); + if (expression.letValue) + expressions.push_back(expression.letValue.get()); + if (expression.letBody) + expressions.push_back(expression.letBody.get()); + } + continue; + } + const auto &statement = *statements.back(); + statements.pop_back(); + if (statement.kind == Statement::Kind::Bind) + { + highest = std::max(highest, statement.binding.symbol.id); + expressions.push_back(&statement.binding.value); + } + else + { + if (statement.kind == Statement::Kind::Assign) + highest = std::max(highest, statement.destination.id); + expressions.push_back(&statement.expression); + } + addStatements(statement.trueBranch); + addStatements(statement.falseBranch); + addStatements(statement.loopBody); + addStatements(statement.loopUpdate); + } return highest; } @@ -494,70 +556,18 @@ namespace Visual::XSharp::Core::CorePrep return prepared; } - [[nodiscard]] auto - AtomizeOperation(Cursor &cursor, const Expression &expression) - -> OperationResult + /// The operands of a primitive whose first operand was evaluated. + [[nodiscard, gnu::noinline]] auto + PrimitiveOperation(Cursor &cursor, + const Expression &expression, + Prepared::Atom first) -> OperationResult { - if (expression.kind == Expression::Kind::Let - || IsShortCircuit(expression) || IsConditional(expression)) - return { Prepared::Operation::Copy, - { Atomize(cursor, expression) }, - {}, - {} }; - if (expression.kind == Expression::Kind::Variable) - return { Prepared::Operation::Copy, - { Prepared::Atom::variable(expression.symbol, - expression.type) }, - {}, - {} }; - if (expression.kind == Expression::Kind::Literal) - return { Prepared::Operation::Copy, - { LowerLiteral(expression) }, - {}, - {} }; - if (expression.kind == Expression::Kind::Apply) - { - std::vector operands; - operands.reserve(expression.operands.size() + 1U); - operands.push_back(Atomize(cursor, *expression.callee)); - for (const auto &argument : expression.operands) - operands.push_back(Atomize(cursor, argument)); - return { Prepared::Operation::Call, - std::move(operands), - {}, - {} }; - } - if (expression.kind == Expression::Kind::Closure) - { - if (!expression.closureBody) - std::abort(); - const auto closureId = cursor.state.nextSymbol++; - const auto digits = std::to_string(closureId); - std::u32string spelling = U"$closure"; - spelling.append(digits.begin(), digits.end()); - SymbolName closureName{ closureId, std::move(spelling) }; - - auto captures = AtomizeCaptures(cursor, expression.captures); - std::vector parameters; - parameters.reserve(expression.captures.size() - + expression.closureParameters.size()); - for (const auto &capture : expression.captures) - parameters.push_back( - Parameter{ capture.symbol, capture.type }); - for (const auto &[symbol, type] : expression.closureParameters) - parameters.push_back(Parameter{ symbol, type }); - cursor.state.pendingFunctions.push_back(Function{ - closureName, - std::move(parameters), - expression.closureReturnType, - *expression.closureBody, - }); - return { Prepared::Operation::MakeClosure, - {}, - std::move(closureName), - std::move(captures) }; - } - auto operands = AtomizeMany(cursor, expression.operands); + std::vector operands; + operands.reserve(expression.operands.size()); + operands.push_back(std::move(first)); + for (std::size_t index = 1U; index < expression.operands.size(); + ++index) + operands.push_back(Atomize(cursor, expression.operands[index])); if (expression.primitive == Primitive::LogicalAnd || expression.primitive == Primitive::LogicalOr || expression.primitive == Primitive::LogicalNot) @@ -569,8 +579,118 @@ namespace Visual::XSharp::Core::CorePrep {} }; } - [[nodiscard]] auto - Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom + [[nodiscard, gnu::noinline]] auto + CallOperation(Cursor &cursor, const Expression &expression) + -> OperationResult + { + std::vector operands; + operands.reserve(expression.operands.size() + 1U); + operands.push_back(Atomize(cursor, *expression.callee)); + for (const auto &argument : expression.operands) + operands.push_back(Atomize(cursor, argument)); + return { Prepared::Operation::Call, std::move(operands), {}, {} }; + } + + [[nodiscard, gnu::noinline]] auto + ClosureOperation(Cursor &cursor, const Expression &expression) + -> OperationResult + { + if (!expression.closureBody) + std::abort(); + const auto closureId = cursor.state.nextSymbol++; + const auto digits = std::to_string(closureId); + std::u32string spelling = U"$closure"; + spelling.append(digits.begin(), digits.end()); + SymbolName closureName{ closureId, std::move(spelling) }; + + auto captures = AtomizeCaptures(cursor, expression.captures); + std::vector parameters; + parameters.reserve(expression.captures.size() + + expression.closureParameters.size()); + for (const auto &capture : expression.captures) + parameters.push_back(Parameter{ capture.symbol, capture.type }); + for (const auto &[symbol, type] : expression.closureParameters) + parameters.push_back(Parameter{ symbol, type }); + cursor.state.pendingFunctions.push_back(Function{ + closureName, + std::move(parameters), + expression.closureReturnType, + *expression.closureBody, + }); + return { Prepared::Operation::MakeClosure, + {}, + std::move(closureName), + std::move(captures) }; + } + + [[nodiscard, gnu::noinline]] auto + CopyOperation(Prepared::Atom atom) -> OperationResult + { + return { Prepared::Operation::Copy, { std::move(atom) }, {}, {} }; + } + + [[nodiscard, gnu::noinline]] auto + AtomizeOperation(Cursor &cursor, const Expression &expression) + -> OperationResult + { + if (expression.kind == Expression::Kind::Let + || IsShortCircuit(expression) || IsConditional(expression)) + return CopyOperation(Atomize(cursor, expression)); + if (expression.kind == Expression::Kind::Variable) + return CopyOperation(Prepared::Atom::variable(expression.symbol, + expression.type)); + if (expression.kind == Expression::Kind::Literal) + return CopyOperation(LowerLiteral(expression)); + if (expression.kind == Expression::Kind::Apply) + return CallOperation(cursor, expression); + if (expression.kind == Expression::Kind::Closure) + return ClosureOperation(cursor, expression); + if (expression.operands.empty()) + return { LowerPrimitive(expression.primitive), {}, {}, {} }; + return PrimitiveOperation( + cursor, + expression, + Atomize(cursor, expression.operands.front())); + } + + /// Bind the result of an operation to a fresh temporary. + [[nodiscard, gnu::noinline]] auto + BindTemporary(Cursor &cursor, + const Type &type, + OperationResult &operation) -> Prepared::Atom + { + auto temporary = cursor.Temporary(U"$coreprep"); + EmitOperation(cursor, + Prepared::Instruction::Kind::Bind, + temporary, + type, + false, + operation); + return Prepared::Atom::variable(std::move(temporary), type); + } + + /// Bind the value of a let expression; its body follows. + [[gnu::noinline]] void + BindLet(Cursor &cursor, const Expression &expression) + { + if (!expression.letValue || !expression.letBody) + std::abort(); + // The bound value keeps its own operation, as in the Haskell + // lowering; an intermediate temporary would make the two + // adapters disagree on every non-atomic value. + auto value = AtomizeOperation(cursor, *expression.letValue); + EmitOperation(cursor, + Prepared::Instruction::Kind::Bind, + expression.letSymbol, + expression.letType, + false, + value); + } + + /// An expression that is not walked in a loop. + [[nodiscard, gnu::noinline]] auto + AtomizeLeaf(Cursor &cursor, const Expression &expression) + -> Prepared::Atom { if (expression.kind == Expression::Kind::Variable) return Prepared::Atom::variable(expression.symbol, @@ -578,26 +698,7 @@ namespace Visual::XSharp::Core::CorePrep if (expression.kind == Expression::Kind::Literal) return LowerLiteral(expression); if (expression.kind == Expression::Kind::Let) - { - if (!expression.letValue || !expression.letBody) - std::abort(); - // The bound value keeps its own operation, as in the - // Haskell lowering; an intermediate temporary would make - // the two adapters disagree on every non-atomic value. - auto value = AtomizeOperation(cursor, *expression.letValue); - cursor.Emit( - Prepared::Instruction{ Prepared::Instruction::Kind::Bind, - expression.letSymbol, - expression.letType, - false, - value.operation, - std::move(value.operands), - std::move(value.closureFunction), - std::move(value.captures) }); - return Atomize(cursor, *expression.letBody); - } - if (IsShortCircuit(expression)) - return AtomizeShortCircuit(cursor, expression); + return Atomize(cursor, expression); if (expression.kind == Expression::Kind::Conditional) { // Core verification requires three operands, but keep this @@ -606,20 +707,59 @@ namespace Visual::XSharp::Core::CorePrep std::abort(); return AtomizeConditional(cursor, expression); } - auto operation = AtomizeOperation(cursor, expression); - auto temporary = cursor.Temporary(U"$coreprep"); - cursor.Emit( - Prepared::Instruction{ Prepared::Instruction::Kind::Bind, - temporary, - expression.type, - false, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); - return Prepared::Atom::variable(std::move(temporary), - expression.type); + return BindTemporary(cursor, expression.type, operation); + } + + /// A primitive from the atom of its first operand. + [[nodiscard, gnu::noinline]] auto + CompleteFirstOperand(Cursor &cursor, + const Expression &expression, + Prepared::Atom first) -> Prepared::Atom + { + if (IsShortCircuit(expression)) + return FinishShortCircuit(cursor, expression, std::move(first)); + auto operation + = PrimitiveOperation(cursor, expression, std::move(first)); + return BindTemporary(cursor, expression.type, operation); + } + + /** + * @brief Evaluate an expression into the open block and return the + * atom that holds its value. + * + * A sequence of bindings nests in the body of each let, and a chain + * of operators in the first operand of each primitive. Both are + * walked in a loop: the bindings are emitted in order, and the + * primitives are completed from the innermost outwards once their + * first operand has a value. The instructions, the symbols and the + * blocks are created in the order of the recursive formulation. + */ + [[nodiscard, gnu::noinline]] auto + Atomize(Cursor &cursor, const Expression &expression) -> Prepared::Atom + { + const Expression *current = &expression; + while (current->kind == Expression::Kind::Let) + { + BindLet(cursor, *current); + current = current->letBody.get(); + } + std::vector chain; + while (current->kind == Expression::Kind::Primitive + && !current->operands.empty()) + { + chain.push_back(current); + current = ¤t->operands.front(); + } + auto atom = AtomizeLeaf(cursor, *current); + while (!chain.empty()) + { + atom = CompleteFirstOperand(cursor, + *chain.back(), + std::move(atom)); + chain.pop_back(); + } + return atom; } void @@ -657,14 +797,33 @@ namespace Visual::XSharp::Core::CorePrep cursor.state.loopTargets.pop_back(); } - /// Evaluate a loop or branch condition in the open block and return - /// its Boolean atom. Short-circuit operands may leave a different + /// Evaluate a loop or branch condition in the open block and branch + /// on its Boolean atom. Short-circuit operands may leave a different /// block open than the one the condition started in. - [[nodiscard]] auto - PrepareCondition(Cursor &cursor, const Expression &condition) - -> Prepared::Atom + [[gnu::noinline]] void + BranchOnCondition(Cursor &cursor, + const Expression &condition, + const Prepared::BlockId whenTrue, + const Prepared::BlockId whenFalse) + { + cursor.Branch(Booleanize(cursor, Atomize(cursor, condition)), + whenTrue, + whenFalse); + } + + /// Evaluate the condition of an `if`, then take three consecutive + /// block identities for its true branch, its false branch and its + /// join, and branch. Returns the first of the three. The condition + /// comes first because it may create blocks of its own. + [[nodiscard, gnu::noinline]] auto + BranchToNewBlocks(Cursor &cursor, const Expression &condition) + -> Prepared::BlockId { - return Booleanize(cursor, Atomize(cursor, condition)); + auto atom = Booleanize(cursor, Atomize(cursor, condition)); + const auto trueId = cursor.state.nextBlock; + cursor.state.nextBlock = trueId + 3U; + cursor.Branch(std::move(atom), trueId, trueId + 1U); + return trueId; } void @@ -697,19 +856,16 @@ namespace Visual::XSharp::Core::CorePrep * lowerings are compared block by block, so this order is part of * the contract. */ - void + [[gnu::noinline]] void PrepareConditionalChain(Cursor &cursor, const Statement &first) { std::vector joins; const Statement *link = &first; for (;;) { - auto condition = PrepareCondition(cursor, link->expression); - const auto trueId = cursor.state.nextBlock; + const auto trueId = BranchToNewBlocks(cursor, link->expression); const auto falseId = trueId + 1U; const auto joinId = falseId + 1U; - cursor.state.nextBlock = joinId + 1U; - cursor.Branch(std::move(condition), trueId, falseId); PrepareBranchRegion(cursor, trueId, joinId, link->trueBranch); joins.push_back(joinId); const auto continues @@ -741,6 +897,152 @@ namespace Visual::XSharp::Core::CorePrep } } + [[gnu::noinline]] void + PrepareBinding(Cursor &cursor, const Statement &statement) + { + auto operation = AtomizeOperation(cursor, statement.binding.value); + EmitOperation(cursor, + Prepared::Instruction::Kind::Bind, + statement.binding.symbol, + statement.binding.type, + statement.binding.mutableBinding, + operation); + } + + [[gnu::noinline]] void + PrepareAssignment(Cursor &cursor, const Statement &statement) + { + EmitCopy(cursor, + Prepared::Instruction::Kind::Assign, + statement.destination, + statement.expression.type, + false, + Atomize(cursor, statement.expression)); + } + + [[gnu::noinline]] void + PrepareEvaluation(Cursor &cursor, const Statement &statement) + { + // Only a call may be an instruction whose result is dropped: + // the record has no result type on the wire, and a reader + // recovers it from the callee. Any other value is computed + // into an ordinary temporary, so its operands still run and + // may trap, and the unused atom is ignored. + if (statement.expression.kind != Expression::Kind::Apply) + { + static_cast(Atomize(cursor, statement.expression)); + return; + } + auto operation = AtomizeOperation(cursor, statement.expression); + EmitOperation(cursor, + Prepared::Instruction::Kind::Evaluate, + {}, + statement.expression.type, + false, + operation); + } + + [[gnu::noinline]] void + PrepareReturn(Cursor &cursor, const Statement &statement) + { + cursor.Close({ Prepared::Terminator::Kind::Return, + Atomize(cursor, statement.expression), + 0U, + 0U }); + } + + [[gnu::noinline]] void + PrepareTransfer(Cursor &cursor, const Statement &statement) + { + // Core verification rejects a transfer outside a loop; stay + // total for direct native API callers. + if (cursor.state.loopTargets.empty()) + { + cursor.Close( + { Prepared::Terminator::Kind::Unreachable, {}, 0U, 0U }); + return; + } + const auto targets = cursor.state.loopTargets.back(); + cursor.Jump(statement.kind == Statement::Kind::Break + ? targets.first + : targets.second); + } + + [[gnu::noinline]] void + PrepareWhile(Cursor &cursor, const Statement &statement) + { + // The condition owns a dedicated header block. Folding it into + // the incoming block would make the back-edge re-execute every + // straight-line statement that precedes the loop, including the + // initializers it tests. + const auto conditionId = cursor.state.nextBlock; + const auto bodyId = conditionId + 1U; + const auto exitId = bodyId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(conditionId); + cursor.Open(conditionId); + BranchOnCondition(cursor, statement.expression, bodyId, exitId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + conditionId, + conditionId, + statement.loopBody); + cursor.Open(exitId); + } + + [[gnu::noinline]] void + PrepareDoWhile(Cursor &cursor, const Statement &statement) + { + const auto bodyId = cursor.state.nextBlock; + const auto conditionId = bodyId + 1U; + const auto exitId = conditionId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(bodyId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + conditionId, + conditionId, + statement.loopBody); + cursor.Open(conditionId); + BranchOnCondition(cursor, statement.expression, bodyId, exitId); + cursor.Open(exitId); + } + + [[gnu::noinline]] void + PrepareFor(Cursor &cursor, const Statement &statement) + { + const auto conditionId = cursor.state.nextBlock; + const auto bodyId = conditionId + 1U; + const auto updateId = bodyId + 1U; + const auto exitId = updateId + 1U; + cursor.state.nextBlock = exitId + 1U; + + cursor.Jump(conditionId); + cursor.Open(conditionId); + // A numeric condition's canonicalizing comparison is part of + // the header and is re-evaluated each pass. + BranchOnCondition(cursor, statement.expression, bodyId, exitId); + PrepareLoopRegion(cursor, + bodyId, + exitId, + updateId, + updateId, + statement.loopBody); + // `continue` enters the update region; the region itself then + // returns to the condition, never to its own entry. + PrepareLoopRegion(cursor, + updateId, + exitId, + updateId, + conditionId, + statement.loopUpdate); + cursor.Open(exitId); + } + /** * @brief Append structured statements to the cursor's open block. * @@ -758,174 +1060,30 @@ namespace Visual::XSharp::Core::CorePrep switch (statement.kind) { case Statement::Kind::Bind: - { - auto operation - = AtomizeOperation(cursor, statement.binding.value); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Bind, - statement.binding.symbol, - statement.binding.type, - statement.binding.mutableBinding, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); + PrepareBinding(cursor, statement); break; - } case Statement::Kind::Assign: - { - auto value = Atomize(cursor, statement.expression); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Assign, - statement.destination, - statement.expression.type, - false, - Prepared::Operation::Copy, - { std::move(value) }, - {}, - {} }); + PrepareAssignment(cursor, statement); break; - } case Statement::Kind::Evaluate: - { - // Only a call may be an instruction whose result is - // dropped: the record has no result type on the - // wire, and a reader recovers it from the callee. - // Any other value is computed into an ordinary - // temporary, so its operands still run and may - // trap, and the unused atom is ignored. - if (statement.expression.kind - != Expression::Kind::Apply) - { - static_cast( - Atomize(cursor, statement.expression)); - break; - } - auto operation - = AtomizeOperation(cursor, statement.expression); - cursor.Emit(Prepared::Instruction{ - Prepared::Instruction::Kind::Evaluate, - {}, - statement.expression.type, - false, - operation.operation, - std::move(operation.operands), - std::move(operation.closureFunction), - std::move(operation.captures) }); + PrepareEvaluation(cursor, statement); break; - } case Statement::Kind::Return: - { - auto value = Atomize(cursor, statement.expression); - cursor.Close({ Prepared::Terminator::Kind::Return, - std::move(value), - 0U, - 0U }); + PrepareReturn(cursor, statement); return; - } case Statement::Kind::Break: case Statement::Kind::Continue: - { - // Core verification rejects a transfer outside a - // loop; stay total for direct native API callers. - if (cursor.state.loopTargets.empty()) - { - cursor.Close( - { Prepared::Terminator::Kind::Unreachable, - {}, - 0U, - 0U }); - return; - } - const auto targets = cursor.state.loopTargets.back(); - cursor.Jump(statement.kind == Statement::Kind::Break - ? targets.first - : targets.second); + PrepareTransfer(cursor, statement); return; - } case Statement::Kind::While: - { - // The condition owns a dedicated header block. - // Folding it into the incoming block would make the - // back-edge re-execute every straight-line statement - // that precedes the loop, including the initializers - // it tests. - const auto conditionId = cursor.state.nextBlock; - const auto bodyId = conditionId + 1U; - const auto exitId = bodyId + 1U; - cursor.state.nextBlock = exitId + 1U; - - cursor.Jump(conditionId); - cursor.Open(conditionId); - cursor.Branch( - PrepareCondition(cursor, statement.expression), - bodyId, - exitId); - PrepareLoopRegion(cursor, - bodyId, - exitId, - conditionId, - conditionId, - statement.loopBody); - cursor.Open(exitId); + PrepareWhile(cursor, statement); break; - } case Statement::Kind::DoWhile: - { - const auto bodyId = cursor.state.nextBlock; - const auto conditionId = bodyId + 1U; - const auto exitId = conditionId + 1U; - cursor.state.nextBlock = exitId + 1U; - - cursor.Jump(bodyId); - PrepareLoopRegion(cursor, - bodyId, - exitId, - conditionId, - conditionId, - statement.loopBody); - cursor.Open(conditionId); - cursor.Branch( - PrepareCondition(cursor, statement.expression), - bodyId, - exitId); - cursor.Open(exitId); + PrepareDoWhile(cursor, statement); break; - } case Statement::Kind::For: - { - const auto conditionId = cursor.state.nextBlock; - const auto bodyId = conditionId + 1U; - const auto updateId = bodyId + 1U; - const auto exitId = updateId + 1U; - cursor.state.nextBlock = exitId + 1U; - - cursor.Jump(conditionId); - cursor.Open(conditionId); - // A numeric condition's canonicalizing comparison is - // part of the header and is re-evaluated each pass. - cursor.Branch( - PrepareCondition(cursor, statement.expression), - bodyId, - exitId); - PrepareLoopRegion(cursor, - bodyId, - exitId, - updateId, - updateId, - statement.loopBody); - // `continue` enters the update region; the region - // itself then returns to the condition, never to its - // own entry. - PrepareLoopRegion(cursor, - updateId, - exitId, - updateId, - conditionId, - statement.loopUpdate); - cursor.Open(exitId); + PrepareFor(cursor, statement); break; - } case Statement::Kind::If: PrepareConditionalChain(cursor, statement); break; @@ -945,11 +1103,26 @@ namespace Visual::XSharp::Core::CorePrep } Cursor cursor{ State{ nextSymbol, 1U, {}, {} }, 0U, {}, {}, true }; PrepareStatements(cursor, function.body); - // Core verification proves every path returns. A body that still - // falls off its end is marked instead of given an invented value. + // A function without a result may end without a return: + // reaching the end of its body returns. For a function with a + // result Core verification proves every path returns, so a body + // that still falls off its end is marked instead of given an + // invented value. if (cursor.open) - cursor.Close( - { Prepared::Terminator::Kind::Unreachable, {}, 0U, 0U }); + { + if (function.returnType.kind == Prepared::Type::Kind::Unit) + cursor.Close( + { Prepared::Terminator::Kind::Return, + Prepared::Atom::constant(Prepared::Literal{}, + Prepared::Type::unit()), + 0U, + 0U }); + else + cursor.Close({ Prepared::Terminator::Kind::Unreachable, + {}, + 0U, + 0U }); + } return { Prepared::Function{ function.symbol, std::move(parameters), @@ -968,30 +1141,41 @@ namespace Visual::XSharp::Core::CorePrep ::visual_xsharp::core::CorePrepModule prepared{ module.name, {}, module.sourceFiles }; - std::vector> pending; + // The queue refers to the functions of the module instead of copying + // them: a copy would duplicate every statement of the program, and + // copying nested statements recurses once per level of nesting. + // Lifted closure bodies are owned by a deque, whose elements keep + // their addresses while the queue grows. + std::deque lifted; + std::vector> pending; pending.reserve(module.functions.size()); for (const auto &function : module.functions) - pending.emplace_back(function, function.sourceFile); + pending.emplace_back(&function, function.sourceFile); // Lifted closure bodies only mention identities already counted // here or generated by the shared counter, so the seed stays valid // for functions appended to the queue later. SymbolId nextSymbol = 1U; for (const auto &[function, _] : pending) nextSymbol - = std::max(nextSymbol, HighestFunctionSymbol(function) + 1U); + = std::max(nextSymbol, HighestFunctionSymbol(*function) + 1U); // Closure bodies are lifted as ordinary Core functions and fed back // through the same work queue. This naturally handles nested closures // without adding a second, subtly different lowering implementation. for (std::size_t index = 0U; index < pending.size(); ++index) { - auto result = PrepareFunction(pending[index].first, nextSymbol); + auto result = PrepareFunction(*pending[index].first, nextSymbol); result.function.sourceFile = pending[index].second; prepared.functions.push_back(std::move(result.function)); nextSymbol = result.nextSymbol; for (auto &function : result.pendingFunctions) - pending.emplace_back(std::move(function), - pending[index].second); + { + lifted.push_back(std::move(function)); + // Copy the owner's path first: growing the queue may move + // the element it is read from. + auto sourceFile = pending[index].second; + pending.emplace_back(&lifted.back(), std::move(sourceFile)); + } } return prepared; } diff --git a/Compiler/Core/CorePrep/Verifier.cpp b/Compiler/Core/CorePrep/Verifier.cpp index 31e46d15..b2954f5a 100644 --- a/Compiler/Core/CorePrep/Verifier.cpp +++ b/Compiler/Core/CorePrep/Verifier.cpp @@ -48,6 +48,7 @@ namespace visual_xsharp::core case Operation::Negate: case Operation::LogicalNot: case Operation::BitwiseNot: + case Operation::Memoize: return 1; default: return 2; @@ -69,6 +70,7 @@ namespace visual_xsharp::core block)); if (instruction.operation != Operation::Call && instruction.operation != Operation::MakeClosure + && instruction.operation != Operation::RuntimeCall && instruction.operands.size() != expected_arity(instruction.operation)) issues.push_back(issue("VXC1007", diff --git a/Compiler/Core/CorePrep/Verifier/Semantics.cpp b/Compiler/Core/CorePrep/Verifier/Semantics.cpp index 9756ae59..8d6de24b 100644 --- a/Compiler/Core/CorePrep/Verifier/Semantics.cpp +++ b/Compiler/Core/CorePrep/Verifier/Semantics.cpp @@ -8,6 +8,7 @@ #include "Visual/XSharp/Core/CorePrep/Verifier/Semantics.hpp" #include "Visual/XSharp/Core/Ownership.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" #include "Visual/XSharp/Core/Scalar.hpp" #include "Visual/XSharp/Core/Template.hpp" @@ -22,23 +23,89 @@ namespace visual_xsharp::core bool callable{}; }; - using Definitions = std::unordered_map; + using SpellingMap = std::unordered_map; - [[nodiscard]] auto - captured_parameter_symbols(const CorePrepModule &module, - SymbolId function) - -> std::unordered_set + // What every function of a module sees of the module around it: the + // functions it may call, their spellings, and the parameters that + // closures fill. It is built once for the module. Building it again + // for every function made verification quadratic in the number of + // functions: 2000 small methods took 14 seconds here alone. + struct ModuleCatalog final { - std::unordered_set symbols; - for (const auto &owner : module.functions) - for (const auto &block : owner.blocks) - for (const auto &instruction : block.instructions) - if (instruction.operation == Operation::MakeClosure - && instruction.closure_function.id == function) - for (const auto &capture : instruction.captures) - symbols.insert(capture.symbol.id); - return symbols; - } + std::unordered_map functions; + SpellingMap spellings; + // Function symbols whose spelling differs from that of an + // earlier function with the same id. + std::size_t conflicting_spellings{}; + // For a lifted function, the parameters its closures capture into. + std::unordered_map> + captured_parameters; + }; + + // The symbols a function may refer to: the functions of its module + // and what the function itself defines. A function of the module + // takes precedence over a local definition with the same id, as it + // did when both were entered into one table, functions first. + class Definitions final + { + public: + explicit Definitions(const ModuleCatalog &catalog) noexcept + : module_functions(&catalog.functions) + {} + + void + emplace(SymbolId id, Definition definition) + { + local.emplace(id, std::move(definition)); + } + + [[nodiscard]] auto + find(SymbolId id) const -> const Definition * + { + if (const auto found = module_functions->find(id); + found != module_functions->end()) + return &found->second; + if (const auto found = local.find(id); found != local.end()) + return &found->second; + return nullptr; + } + + private: + const std::unordered_map *module_functions; + std::unordered_map local; + }; + + // The spelling each symbol id carries where a function can see it: + // the function's own, then the spellings of the module's functions, + // then the first spelling the function uses for any other id. + class Spellings final + { + public: + Spellings(const ModuleCatalog &catalog, + const SymbolName &function) noexcept + : module_spellings(&catalog.spellings) + , own(&function) + {} + + // The spelling recorded for the id; the given one is recorded + // when the id had none. + [[nodiscard]] auto + record(SymbolId id, const std::u32string &spelling) + -> const std::u32string & + { + if (id == own->id && !own->spelling.empty()) + return own->spelling; + if (const auto found = module_spellings->find(id); + found != module_spellings->end()) + return found->second; + return local.emplace(id, spelling).first->second; + } + + private: + const SpellingMap *module_spellings; + const SymbolName *own; + SpellingMap local; + }; auto issue(std::string code, @@ -73,21 +140,17 @@ namespace visual_xsharp::core } void - verify_symbol_spelling( - const SymbolName &symbol, - const Function &function, - BlockId block, - std::unordered_map &spellings, - std::vector &issues) + verify_symbol_spelling(const SymbolName &symbol, + const Function &function, + BlockId block, + Spellings &spellings, + std::vector &issues) { if (symbol.id == 0) return; if (symbol.spelling.empty()) return; - const auto [found, inserted] - = spellings.emplace(symbol.id, symbol.spelling); - if (!inserted && !found->second.empty() - && found->second != symbol.spelling) + if (spellings.record(symbol.id, symbol.spelling) != symbol.spelling) issues.push_back( issue("VXC1014", "one symbol id carries conflicting spellings", @@ -100,7 +163,7 @@ namespace visual_xsharp::core const Function &function, BlockId block, std::size_t depth, - std::unordered_map &spellings, + Spellings &spellings, std::vector &issues) { // This boundary also accepts CorePrep assembled by tooling rather @@ -248,7 +311,7 @@ namespace visual_xsharp::core const Function &function, BlockId block, const Definitions &definitions, - std::unordered_map &spellings, + Spellings &spellings, std::vector &issues) { verify_type(atom.type, function, block, 0, spellings, issues); @@ -259,14 +322,14 @@ namespace visual_xsharp::core block, spellings, issues); - const auto found = definitions.find(atom.symbol.id); - if (found == definitions.end()) + const auto *const found = definitions.find(atom.symbol.id); + if (found == nullptr) issues.push_back( issue("VXC1021", "atom references an undefined symbol", function, block)); - else if (found->second.type != atom.type) + else if (found->type != atom.type) issues.push_back( issue("VXC1022", "atom type differs from its symbol definition", @@ -282,6 +345,20 @@ namespace visual_xsharp::core block)); } + /// The row of the runtime catalog a runtime call names with its + /// first operand, or null when that operand names none. + [[nodiscard]] auto + runtime_signature(const std::vector &operands) + -> const runtime::Signature * + { + if (operands.empty() + || operands.front().kind != Atom::Kind::Literal) + return nullptr; + const auto identity = runtime::IdentityOf(operands.front().literal, + operands.front().type); + return identity ? runtime::Find(*identity) : nullptr; + } + auto expected_primitive_result(Operation operation, const std::vector &operands) @@ -329,23 +406,42 @@ namespace visual_xsharp::core case Operation::BitwiseXor: case Operation::BitwiseOr: case Operation::BitwiseNot: + case Operation::Memoize: return operands.empty() ? std::nullopt : std::optional(operands.front().type); case Operation::MakeClosure: return std::nullopt; + case Operation::RuntimeCall: + if (const auto *signature = runtime_signature(operands)) + return signature->result == runtime::Result::Text + ? Type::string() + : signature->result == runtime::Result::Truth + ? Type::boolean() + : Type::unit(); + return std::nullopt; } return std::nullopt; } + // Whether a callable of this type can remember its result: it takes + // no parameters, and its result owns nothing. + [[nodiscard]] auto + remembers_result(const Type &type) -> bool + { + return type.kind == Type::Kind::Function + && type.components.size() == 1U + && (type.components.front().kind == Type::Kind::Bool + || is_numeric(type.components.front())); + } + void - verify_operation( - const Instruction &instruction, - const Function &function, - BlockId block, - const Definitions &definitions, - std::unordered_map &spellings, - std::vector &issues) + verify_operation(const Instruction &instruction, + const Function &function, + BlockId block, + const Definitions &definitions, + Spellings &spellings, + std::vector &issues) { for (const auto &operand : instruction.operands) verify_atom(operand, @@ -362,9 +458,9 @@ namespace visual_xsharp::core block, spellings, issues); - const auto target + const auto *const target = definitions.find(instruction.closure_function.id); - if (target == definitions.end() || !target->second.callable) + if (target == nullptr || !target->callable) issues.push_back(issue( "VXC1040", "closure target is not a declared lifted function", @@ -413,9 +509,9 @@ namespace visual_xsharp::core block)); } - if (target != definitions.end() && target->second.callable) + if (target != nullptr && target->callable) { - const auto &lifted = target->second.type.components; + const auto &lifted = target->type.components; if (lifted.size() <= instruction.captures.size()) issues.push_back( issue("VXC1045", @@ -562,6 +658,35 @@ namespace visual_xsharp::core function, block)); break; + case Operation::RuntimeCall: + { + std::vector arguments; + arguments.reserve(arity); + for (std::size_t index = 1U; index < arity; ++index) + arguments.push_back( + instruction.operands[index].type.kind); + const auto defect = runtime::Check( + runtime_signature(instruction.operands), + arguments, + instruction.type.kind); + if (defect != runtime::Defect::None) + issues.push_back( + issue("VXC1076", + std::string(runtime::Describe(defect)), + function, + block)); + break; + } + case Operation::Memoize: + if (arity != 1U + || !remembers_result(instruction.operands.front().type)) + issues.push_back( + issue("VXC1074", + "memoization requires one callable without " + "parameters whose result is bool or numeric", + function, + block)); + break; case Operation::ShiftLeft: case Operation::ShiftRight: case Operation::BitwiseAnd: @@ -642,27 +767,64 @@ namespace visual_xsharp::core } } - void - collect_definitions( - const CorePrepModule &module, - const Function &function, - Definitions &definitions, - std::unordered_map &spellings, - std::vector &issues) + [[nodiscard]] auto + catalog_module(const CorePrepModule &module) -> ModuleCatalog { - const auto capturedParameters - = captured_parameter_symbols(module, function.symbol.id); + ModuleCatalog catalog; + catalog.functions.reserve(module.functions.size()); + catalog.spellings.reserve(module.functions.size()); for (const auto &candidate : module.functions) { - verify_symbol_spelling(candidate.symbol, - function, - 0, - spellings, - issues); - definitions.emplace( + if (candidate.symbol.id != 0 + && !candidate.symbol.spelling.empty()) + { + const auto [found, inserted] + = catalog.spellings.emplace(candidate.symbol.id, + candidate.symbol.spelling); + if (!inserted && found->second != candidate.symbol.spelling) + ++catalog.conflicting_spellings; + } + catalog.functions.emplace( candidate.symbol.id, Definition{ function_type(candidate), false, true }); + for (const auto &block : candidate.blocks) + for (const auto &instruction : block.instructions) + if (instruction.operation == Operation::MakeClosure) + { + auto &captured + = catalog.captured_parameters + [instruction.closure_function.id]; + for (const auto &capture : instruction.captures) + captured.insert(capture.symbol.id); + } } + return catalog; + } + + void + collect_definitions(const ModuleCatalog &catalog, + const Function &function, + Definitions &definitions, + Spellings &spellings, + std::vector &issues) + { + const auto captured + = catalog.captured_parameters.find(function.symbol.id); + const auto isCaptured = [&](SymbolId parameter) { + return captured != catalog.captured_parameters.end() + && captured->second.contains(parameter); + }; + // A conflict among the function symbols of the module is a fault + // of every function that sees them, as it was when each function + // entered them into its own table. + for (std::size_t conflict = 0; + conflict < catalog.conflicting_spellings; + ++conflict) + issues.push_back( + issue("VXC1014", + "one symbol id carries conflicting spellings", + function, + 0)); for (const auto ¶meter : function.parameters) { verify_symbol_spelling(parameter.symbol, @@ -681,8 +843,7 @@ namespace visual_xsharp::core // lowering introduces storage. definitions.emplace(parameter.symbol.id, Definition{ parameter.type, - capturedParameters.contains( - parameter.symbol.id), + isCaptured(parameter.symbol.id), false }); } for (const auto &block : function.blocks) @@ -709,12 +870,12 @@ namespace visual_xsharp::core } void - verify_function(const CorePrepModule &module, + verify_function(const ModuleCatalog &catalog, const Function &function, std::vector &issues) { - Definitions definitions; - std::unordered_map spellings; + Definitions definitions{ catalog }; + Spellings spellings{ catalog, function.symbol }; verify_symbol_spelling(function.symbol, function, function.entry, @@ -726,7 +887,7 @@ namespace visual_xsharp::core 0, spellings, issues); - collect_definitions(module, + collect_definitions(catalog, function, definitions, spellings, @@ -743,23 +904,23 @@ namespace visual_xsharp::core block.id, spellings, issues); - const auto target + const auto *const target = definitions.find(instruction.destination.id); - if (target == definitions.end()) + if (target == nullptr) issues.push_back( issue("VXC1035", "assignment targets an undefined symbol", function, block.id)); - else if (!target->second.mutable_binding) + else if (!target->mutable_binding) issues.push_back( issue("VXC1036", "assignment targets an immutable symbol", function, block.id)); if (instruction.operands.size() == 1U - && target != definitions.end() - && target->second.type + && target != nullptr + && target->type != instruction.operands.front().type) issues.push_back(issue( "VXC1037", @@ -812,8 +973,9 @@ namespace visual_xsharp::core -> std::vector { std::vector issues; + const auto catalog = catalog_module(module); for (const auto &function : module.functions) - verify_function(module, function, issues); + verify_function(catalog, function, issues); return issues; } } // namespace visual_xsharp::core diff --git a/Compiler/Core/CorePrep/Wire/Decode.cpp b/Compiler/Core/CorePrep/Wire/Decode.cpp index c2a5b78f..10f766bc 100644 --- a/Compiler/Core/CorePrep/Wire/Decode.cpp +++ b/Compiler/Core/CorePrep/Wire/Decode.cpp @@ -451,7 +451,7 @@ namespace visual_xsharp::core::wire operation_tag() -> Operation { const auto tag = byte("operation tag"); - if (tag > static_cast(Operation::TypeIs)) + if (tag > static_cast(Operation::RuntimeCall)) { fail(ErrorKind::InvalidTag, "operation tag", diff --git a/Compiler/Core/IR.cpp b/Compiler/Core/IR.cpp index d38e02d8..9340b826 100644 --- a/Compiler/Core/IR.cpp +++ b/Compiler/Core/IR.cpp @@ -15,6 +15,49 @@ namespace Visual::XSharp::Core && type == other.type && equalValue; } + Expression::~Expression() + { + if (operands.empty()) + return; + std::vector pending = std::move(operands); + while (!pending.empty()) + { + // The operands of the expression taken from the list join the + // list, so that it is released without any of its own. + Expression next = std::move(pending.back()); + pending.pop_back(); + for (auto &operand : next.operands) + pending.push_back(std::move(operand)); + next.operands.clear(); + } + } + + Statement::~Statement() + { + if (trueBranch.empty() && falseBranch.empty() && loopBody.empty() + && loopUpdate.empty()) + return; + std::vector pending; + const auto take = [&pending](std::vector &statements) { + for (auto &statement : statements) + pending.push_back(std::move(statement)); + statements.clear(); + }; + take(trueBranch); + take(falseBranch); + take(loopBody); + take(loopUpdate); + while (!pending.empty()) + { + Statement next = std::move(pending.back()); + pending.pop_back(); + take(next.trueBranch); + take(next.falseBranch); + take(next.loopBody); + take(next.loopUpdate); + } + } + auto Expression::Variable(SymbolName name, Type valueType) -> Expression { diff --git a/Compiler/Core/Tests/BUILD.bazel b/Compiler/Core/Tests/BUILD.bazel index 14501319..606c8e4e 100644 --- a/Compiler/Core/Tests/BUILD.bazel +++ b/Compiler/Core/Tests/BUILD.bazel @@ -18,7 +18,11 @@ cc_binary( "ConditionalChainTests.cpp", "ConditionalLoweringTests.cpp", "CorePipelineTests.cpp", + "CorePrepVerifierTests.cpp", + "FallThroughTests.cpp", "LoopLoweringTests.cpp", + "NestingLimitTests.cpp", + "RuntimeCallVerifierTests.cpp", "ShortCircuitLoweringTests.cpp", "SymbolAllocationTests.cpp", "TemplateTests.cpp", @@ -26,6 +30,7 @@ cc_binary( deps = [ "//Compiler/Core:core", "//Compiler/Driver:core_pipeline", + "//Compiler/Support:compiler_stack", "//Compiler/Headers/Visual/XSharp:pipeline_api", "//Compiler/Headers/Visual/XSharp/Core:template_api", "//Compiler/Headers/Visual/XSharp/Core:scalar", diff --git a/Compiler/Core/Tests/ConditionalChainTests.cpp b/Compiler/Core/Tests/ConditionalChainTests.cpp index e705f821..0d689c52 100644 --- a/Compiler/Core/Tests/ConditionalChainTests.cpp +++ b/Compiler/Core/Tests/ConditionalChainTests.cpp @@ -452,3 +452,76 @@ TEST_CASE("the adapter prepares a chain of any length", CHECK(Returns(closed) == links + 1U); } } + +TEST_CASE("a loop that cannot be left never falls off the end of a body", + "[core][verifier]") +{ + const auto always = [] { + return Core::Expression::Constant(true, Core::Type::boolean()); + }; + // The function is moved into its module. A braced list would copy it, + // and copying a statement recurses once per level of nesting. + const auto ends = [](Core::Statement loop) { + Core::Function function{ { 1U, U"Pick" }, + { { { kValue, U"value" }, + Core::Type::int64() } }, + Core::Type::int64(), + {} }; + function.body.push_back(std::move(loop)); + Core::Module module{ { U"Chain" }, {} }; + module.functions.push_back(std::move(function)); + return module; + }; + + SECTION("an endless loop of each kind ends a body that returns a value") + { + CHECK_FALSE( + HasIssue(ends(Core::Statement::While(always(), {})), "VXC1005")); + CHECK_FALSE( + HasIssue(ends(Core::Statement::DoWhile({}, always())), "VXC1005")); + CHECK_FALSE( + HasIssue(ends(Core::Statement::For(always(), {}, {})), "VXC1005")); + } + + SECTION("a break leaves the loop, from its body and from its update") + { + CHECK(HasIssue(ends(Core::Statement::While( + always(), + { Core::Statement::If(Is(1), + { Core::Statement::Break() }, + {}) })), + "VXC1005")); + CHECK(HasIssue(ends(Core::Statement::For(always(), + {}, + { Core::Statement::Break() })), + "VXC1005")); + } + + SECTION("a break at the end of a long else-if chain is found") + { + // The search walks the chain from a list, not by recursion. The + // chain is moved into the loop: copying one recurses per link. + constexpr std::size_t kLinks = 600U; + const auto around = [&](std::vector last) { + std::vector body; + body.push_back(Chain(kLinks, std::move(last))); + return ends(Core::Statement::While(always(), std::move(body))); + }; + std::vector leaves; + leaves.push_back(Core::Statement::Break()); + const auto left = around(std::move(leaves)); + CHECK(HasIssue(left, "VXC1005")); + CHECK_FALSE(HasIssue(around({}), "VXC1005")); + } + + SECTION("a loop with a condition, or left only by a nested break") + { + CHECK(HasIssue(ends(Core::Statement::While(Is(1), {})), "VXC1005")); + CHECK_FALSE(HasIssue( + ends(Core::Statement::While( + always(), + { Core::Statement::While(always(), + { Core::Statement::Break() }) })), + "VXC1005")); + } +} diff --git a/Compiler/Core/Tests/ConditionalLoweringTests.cpp b/Compiler/Core/Tests/ConditionalLoweringTests.cpp index bc392970..06df9948 100644 --- a/Compiler/Core/Tests/ConditionalLoweringTests.cpp +++ b/Compiler/Core/Tests/ConditionalLoweringTests.cpp @@ -576,7 +576,7 @@ TEST_CASE("the Core verifier enforces the conditional contract", } } -TEST_CASE("Core wire v8 round-trips conditional expressions", +TEST_CASE("Core wire v10 round-trips conditional expressions", "[core][wire][conditional]") { const auto module = Returning( @@ -594,7 +594,7 @@ TEST_CASE("Core wire v8 round-trips conditional expressions", REQUIRE(Core::Verify(module).empty()); const auto encoded = Core::Wire::Encode(module); REQUIRE(encoded); - CHECK(encoded.bytes.at(4) == 8U); + CHECK(encoded.bytes.at(4) == 10U); const auto decoded = Core::Wire::Decode(encoded.bytes); REQUIRE(decoded); CHECK(*decoded.module == module); diff --git a/Compiler/Core/Tests/CorePipelineTests.cpp b/Compiler/Core/Tests/CorePipelineTests.cpp index 88ac6b11..cf0c7340 100644 --- a/Compiler/Core/Tests/CorePipelineTests.cpp +++ b/Compiler/Core/Tests/CorePipelineTests.cpp @@ -200,7 +200,7 @@ namespace } [[nodiscard]] auto - ReadGoldenHex(std::string_view filename = "wire-v8.hex") + ReadGoldenHex(std::string_view filename = "wire-v10.hex") -> std::vector { const auto path = std::filesystem::path(__FILE__).parent_path() @@ -234,7 +234,7 @@ namespace } } // namespace -TEST_CASE("native VXCR v8 codec matches the Haskell golden contract") +TEST_CASE("native VXCR v10 codec matches the Haskell golden contract") { const auto expected = ReadGoldenHex(); const auto encoded = Core::Wire::Encode(GoldenModule()); @@ -284,7 +284,7 @@ TEST_CASE("VXCR reader rejects malformed boundaries and configured limits") } } -TEST_CASE("VXCR v8 carries Haskell Core closure and source-owner fields") +TEST_CASE("VXCR v10 carries Haskell Core closure and source-owner fields") { const auto source = ClosureModule(); REQUIRE(Core::Verify(source).empty()); @@ -309,7 +309,7 @@ TEST_CASE("native pipeline consumes a closure artifact emitted by Haskell") // This golden file is emitted from closure-boundary.vxs by vxs-frontend, // rather than re-encoded by the C++ model. It therefore locks the actual // cross-language expression tag and field order that production uses. - const auto bytes = ReadGoldenHex("wire-v8-closure.hex"); + const auto bytes = ReadGoldenHex("wire-v10-closure.hex"); const auto decoded = Core::Wire::Decode(bytes); REQUIRE(decoded); REQUIRE(Core::Verify(*decoded.module).empty()); @@ -359,7 +359,7 @@ TEST_CASE("native pipeline preserves non-empty Haskell source ownership bytes") { // Both owner fields are non-empty in this golden so field order cannot be // accidentally hidden by interchangeable zero-length encodings. - const auto bytes = ReadGoldenHex("wire-v8-project-source.hex"); + const auto bytes = ReadGoldenHex("wire-v10-project-source.hex"); const auto decoded = Core::Wire::Decode(bytes); REQUIRE(decoded); REQUIRE(Core::Verify(*decoded.module).empty()); @@ -547,7 +547,7 @@ TEST_CASE("Core adapter creates explicit CorePrep CFG and temporaries") REQUIRE(visual_xsharp::core::verify(prepared).empty()); } -TEST_CASE("Core v8 loops round-trip and lower to explicit back-edges") +TEST_CASE("Core v10 loops round-trip and lower to explicit back-edges") { const auto integer = [](std::int64_t value) { return Core::Expression::Constant(value, Core::Type::int64()); @@ -599,7 +599,7 @@ TEST_CASE("Core v8 loops round-trip and lower to explicit back-edges") }; const Core::Module module{ { U"Iteration" }, { std::move(function) } }; - CHECK(Core::Wire::kCurrentVersion == 8U); + CHECK(Core::Wire::kCurrentVersion == 10U); REQUIRE(Core::Verify(module).empty()); const auto encoded = Core::Wire::Encode(module); REQUIRE(encoded); diff --git a/Compiler/Core/Tests/CorePrepVerifierTests.cpp b/Compiler/Core/Tests/CorePrepVerifierTests.cpp new file mode 100644 index 00000000..8e8f6164 --- /dev/null +++ b/Compiler/Core/Tests/CorePrepVerifierTests.cpp @@ -0,0 +1,259 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" + +// What a function of a CorePrep module sees of the module around it: the +// functions it may call, their spellings, and the parameters that closures +// fill. The verifier builds that view once for the module. These cases pin +// what each function sees through it, so that the view cannot be built +// faster by seeing something else. + +namespace +{ + namespace Prepared = visual_xsharp::core; + + [[nodiscard]] auto + Name(std::uint64_t id, std::u32string spelling) -> Prepared::SymbolName + { + return { id, std::move(spelling) }; + } + + [[nodiscard]] auto + IntegerFunction() -> Prepared::Type + { + return Prepared::Type::function({}, Prepared::Type::int64()); + } + + // `return 0;` + [[nodiscard]] auto + Constant(Prepared::SymbolName name) -> Prepared::Function + { + Prepared::Terminator terminator; + terminator.kind = Prepared::Terminator::Kind::Return; + terminator.value = Prepared::Atom::constant(std::int64_t{ 0 }, + Prepared::Type::int64()); + return { std::move(name), + {}, + Prepared::Type::int64(), + 0U, + { { 0U, {}, std::move(terminator) } } }; + } + + // `result = callee(); return result;` + [[nodiscard]] auto + Caller(Prepared::SymbolName name, + Prepared::SymbolName callee, + std::uint64_t result) -> Prepared::Function + { + Prepared::Instruction call; + call.kind = Prepared::Instruction::Kind::Bind; + call.destination = Name(result, U"result"); + call.type = Prepared::Type::int64(); + call.operation = Prepared::Operation::Call; + call.operands = { Prepared::Atom::variable(std::move(callee), + IntegerFunction()) }; + Prepared::Terminator terminator; + terminator.kind = Prepared::Terminator::Kind::Return; + terminator.value = Prepared::Atom::variable(Name(result, U"result"), + Prepared::Type::int64()); + return { std::move(name), + {}, + Prepared::Type::int64(), + 0U, + { { 0U, { std::move(call) }, std::move(terminator) } } }; + } + + [[nodiscard]] auto + Count(const std::vector &issues, + std::string_view code) -> std::ptrdiff_t + { + return std::count_if(issues.begin(), + issues.end(), + [code](const Prepared::VerificationIssue &issue) { + return issue.code == code; + }); + } +} // namespace + +TEST_CASE("a function may call a function that the module declares after it") +{ + const Prepared::CorePrepModule module{ + { U"Verifier" }, + { Caller(Name(1U, U"First"), Name(2U, U"Second"), 10U), + Constant(Name(2U, U"Second")) }, + }; + + CHECK(Prepared::verify(module).empty()); +} + +TEST_CASE("a module of several thousand functions that call one another " + "verifies") +{ + constexpr std::uint64_t kFunctions = 3000U; + Prepared::CorePrepModule module{ { U"Verifier" }, {} }; + module.functions.reserve(kFunctions); + // Every function calls the one after it; the last returns a constant. + for (std::uint64_t index = 1U; index < kFunctions; ++index) + module.functions.push_back(Caller(Name(index, U"Method"), + Name(index + 1U, U"Method"), + kFunctions + index)); + module.functions.push_back(Constant(Name(kFunctions, U"Method"))); + + const auto issues = Prepared::verify(module); + + CHECK(issues.empty()); +} + +TEST_CASE("a call of a symbol that no function and no binding defines is " + "refused") +{ + const Prepared::CorePrepModule module{ + { U"Verifier" }, + { Caller(Name(1U, U"First"), Name(7U, U"Absent"), 10U) }, + }; + + const auto issues = Prepared::verify(module); + + CHECK(Count(issues, "VXC1021") == 1); +} + +TEST_CASE("a use that spells a function of the module differently is one " + "conflict, in the function that uses it") +{ + const Prepared::CorePrepModule module{ + { U"Verifier" }, + { Caller(Name(1U, U"First"), Name(2U, U"Other"), 10U), + Constant(Name(2U, U"Second")), + Constant(Name(3U, U"Third")) }, + }; + + const auto issues = Prepared::verify(module); + + CHECK(Count(issues, "VXC1014") == 1); + const auto conflict + = std::find_if(issues.begin(), issues.end(), [](const auto &issue) { + return issue.code == "VXC1014"; + }); + REQUIRE(conflict != issues.end()); + CHECK(conflict->function == 1U); +} + +TEST_CASE("two functions with one identity and two spellings are a conflict " + "for every function that sees them") +{ + const Prepared::CorePrepModule module{ + { U"Verifier" }, + { Constant(Name(1U, U"First")), + Constant(Name(1U, U"Second")), + Constant(Name(3U, U"Third")) }, + }; + + const auto issues = Prepared::verify(module); + + // The duplicated identity is reported once, as the fault it is. + CHECK(Count(issues, "VXC1001") == 1); + // Each of the three functions sees the two spellings. + CHECK(Count(issues, "VXC1014") == 3); +} + +TEST_CASE("a function of the module takes precedence over a binding with " + "its identity") +{ + // The body binds identity 2, which is also a function of the module, + // to an integer and returns it. A read of identity 2 is a read of the + // function, whose type is not the integer the return carries. + Prepared::Instruction bind; + bind.kind = Prepared::Instruction::Kind::Bind; + bind.destination = Name(2U, U"Second"); + bind.type = Prepared::Type::int64(); + bind.operation = Prepared::Operation::Copy; + bind.operands = { Prepared::Atom::constant(std::int64_t{ 1 }, + Prepared::Type::int64()) }; + Prepared::Terminator terminator; + terminator.kind = Prepared::Terminator::Kind::Return; + terminator.value = Prepared::Atom::variable(Name(2U, U"Second"), + Prepared::Type::int64()); + const Prepared::CorePrepModule module{ + { U"Verifier" }, + { Prepared::Function{ + Name(1U, U"First"), + {}, + Prepared::Type::int64(), + 0U, + { { 0U, { std::move(bind) }, std::move(terminator) } } }, + Constant(Name(2U, U"Second")) }, + }; + + const auto issues = Prepared::verify(module); + + CHECK(Count(issues, "VXC1022") == 1); +} + +TEST_CASE("a parameter may be assigned only when a closure of the module " + "captures into it") +{ + const auto assigning = [] { + // `slot = 1; return slot;` with `slot` a parameter. + Prepared::Instruction assign; + assign.kind = Prepared::Instruction::Kind::Assign; + assign.destination = Name(20U, U"slot"); + assign.type = Prepared::Type::int64(); + assign.operation = Prepared::Operation::Copy; + assign.operands = { Prepared::Atom::constant(std::int64_t{ 1 }, + Prepared::Type::int64()) }; + Prepared::Terminator terminator; + terminator.kind = Prepared::Terminator::Kind::Return; + terminator.value = Prepared::Atom::variable(Name(20U, U"slot"), + Prepared::Type::int64()); + return Prepared::Function{ + Name(2U, U"Lifted"), + { { Name(20U, U"slot"), Prepared::Type::int64() } }, + Prepared::Type::int64(), + 0U, + { { 0U, { std::move(assign) }, std::move(terminator) } } + }; + }; + + const Prepared::CorePrepModule plain{ { U"Verifier" }, { assigning() } }; + CHECK(Count(Prepared::verify(plain), "VXC1036") == 1); + + // A closure anywhere in the module that captures into the parameter + // makes it a slot of the closure's environment, which may be assigned. + Prepared::Instruction make; + make.kind = Prepared::Instruction::Kind::Bind; + make.destination = Name(30U, U"closure"); + make.type = IntegerFunction(); + make.operation = Prepared::Operation::MakeClosure; + make.closure_function = Name(2U, U"Lifted"); + make.captures = { { Prepared::CaptureMode::Strong, + Name(20U, U"slot"), + Prepared::Type::int64(), + Prepared::Atom::constant(std::int64_t{ 5 }, + Prepared::Type::int64()) } }; + Prepared::Terminator terminator; + terminator.kind = Prepared::Terminator::Kind::Return; + terminator.value + = Prepared::Atom::constant(std::int64_t{ 0 }, Prepared::Type::int64()); + const Prepared::CorePrepModule captured{ + { U"Verifier" }, + { assigning(), + Prepared::Function{ + Name(1U, U"Owner"), + {}, + Prepared::Type::int64(), + 0U, + { { 0U, { std::move(make) }, std::move(terminator) } } } }, + }; + CHECK(Count(Prepared::verify(captured), "VXC1036") == 0); +} diff --git a/Compiler/Core/Tests/FallThroughTests.cpp b/Compiler/Core/Tests/FallThroughTests.cpp new file mode 100644 index 00000000..5de1a67b --- /dev/null +++ b/Compiler/Core/Tests/FallThroughTests.cpp @@ -0,0 +1,248 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" + +// How a function ends when its body ends. +// +// A function without a result may end without a return: reaching the end of +// its body returns. A function with a result returns on every path, so the +// end of its body is never reached and the block that stands there is marked +// unreachable instead of being given a value nobody wrote. The adapter used +// to mark both the same way, and a native program whose `Main` ended without +// a return stopped where it should have ended. The Haskell adapter has the +// same cases in `FallThroughTests.hs`. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Nothing() -> Core::Expression + { + return Core::Expression::Constant(std::monostate{}, Core::Type::unit()); + } + + [[nodiscard]] auto + Flag() -> Core::Expression + { + return Core::Expression::Variable({ 20U, U"flag" }, + Core::Type::boolean()); + } + + [[nodiscard]] auto + Bind(std::uint64_t id, std::int64_t value) -> Core::Statement + { + return Core::Statement::Bind( + { { id, U"local" }, Core::Type::int64(), false, Integer(value) }); + } + + // `if (flag) { return; }` + [[nodiscard]] auto + ReturnWhen() -> Core::Statement + { + return Core::Statement::If(Flag(), + { Core::Statement::Return(Nothing()) }, + {}); + } + + // `if (flag) { return 1; } else { return 2; }` + [[nodiscard]] auto + BothReturn(Core::Expression condition = Flag()) -> Core::Statement + { + return Core::Statement::If(std::move(condition), + { Core::Statement::Return(Integer(1)) }, + { Core::Statement::Return(Integer(2)) }); + } + + [[nodiscard]] auto + Module(Core::Type result, std::vector body) -> Core::Module + { + return { { U"FallThrough" }, + { Core::Function{ + { 1U, U"Evaluate" }, + { { { 20U, U"flag" }, Core::Type::boolean() } }, + std::move(result), + std::move(body), + } } }; + } + + [[nodiscard]] auto + ReturnsNothing(const Prepared::Terminator &terminator) -> bool + { + return terminator.kind == Prepared::Terminator::Kind::Return + && terminator.value.type.kind == Prepared::Type::Kind::Unit; + } + + struct Ends final + { + std::size_t returningNothing{}; + std::size_t unreachable{}; + std::size_t blocks{}; + }; + + [[nodiscard]] auto + EndsOf(const Prepared::Function &function) -> Ends + { + Ends ends; + for (const auto &block : function.blocks) + { + ++ends.blocks; + if (ReturnsNothing(block.terminator)) + ++ends.returningNothing; + if (block.terminator.kind + == Prepared::Terminator::Kind::Unreachable) + ++ends.unreachable; + } + return ends; + } + + [[nodiscard]] auto + Prepare(Core::Type result, std::vector body) + -> Prepared::CorePrepModule + { + const auto module = Module(std::move(result), std::move(body)); + REQUIRE(Core::Verify(module).empty()); + auto prepared = Core::CorePrep::Prepare(module); + REQUIRE(Prepared::verify(prepared).empty()); + return prepared; + } +} // namespace + +TEST_CASE("an empty body without a result returns") +{ + const auto prepared = Prepare(Core::Type::unit(), {}); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.blocks == 1U); + CHECK(ends.returningNothing == 1U); + CHECK(ends.unreachable == 0U); +} + +TEST_CASE("a body without a result that ends in a binding returns") +{ + const auto prepared = Prepare(Core::Type::unit(), { Bind(2U, 1) }); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.blocks == 1U); + CHECK(ends.returningNothing == 1U); + CHECK(ends.unreachable == 0U); +} + +TEST_CASE("the block after a conditional return returns") +{ + // One return was written; the other is the end of the body. + const auto prepared = Prepare(Core::Type::unit(), { ReturnWhen() }); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.returningNothing == 2U); + CHECK(ends.unreachable == 0U); +} + +TEST_CASE("the block after a loop returns") +{ + const auto prepared + = Prepare(Core::Type::unit(), + { Core::Statement::While(Flag(), { Bind(3U, 1) }) }); + const auto ends = EndsOf(prepared.functions.front()); + // The end of the loop body goes back to the loop; only the end of the + // function's own body returns. + CHECK(ends.returningNothing == 1U); + CHECK(ends.unreachable == 0U); +} + +TEST_CASE("the block after a loop that is left early returns") +{ + const auto prepared = Prepare( + Core::Type::unit(), + { Core::Statement::While(Flag(), { Core::Statement::Break() }) }); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.returningNothing == 1U); + CHECK(ends.unreachable == 0U); +} + +TEST_CASE("statements after a conditional return are followed by a return") +{ + const auto prepared + = Prepare(Core::Type::unit(), { ReturnWhen(), Bind(2U, 1) }); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.returningNothing == 2U); + CHECK(ends.unreachable == 0U); +} + +TEST_CASE("an explicit return without a value is kept as written") +{ + const auto prepared + = Prepare(Core::Type::unit(), { Core::Statement::Return(Nothing()) }); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.blocks == 1U); + CHECK(ends.returningNothing == 1U); +} + +TEST_CASE("a body with a result is not given a value at its end") +{ + const auto prepared = Prepare(Core::Type::int64(), { BothReturn() }); + const auto ends = EndsOf(prepared.functions.front()); + CHECK(ends.returningNothing == 0U); + // Both arms return, so the block after them is never entered. + CHECK(ends.unreachable == 1U); +} + +TEST_CASE("a closure without a result returns at the end of its body") +{ + const auto callable = Core::Type::function({}, Core::Type::unit()); + const auto prepared = Prepare( + Core::Type::unit(), + { Core::Statement::Bind({ { 6U, U"act" }, + callable, + false, + Core::Expression::Closure({}, + {}, + Core::Type::unit(), + { Bind(7U, 1) }, + callable) }) }); + REQUIRE(prepared.functions.size() == 2U); + for (const auto &function : prepared.functions) + { + const auto ends = EndsOf(function); + CHECK(ends.returningNothing == 1U); + CHECK(ends.unreachable == 0U); + } +} + +TEST_CASE("a closure with a result is not given a value at its end") +{ + const auto callable = Core::Type::function({}, Core::Type::int64()); + const auto prepared = Prepare( + Core::Type::unit(), + { Core::Statement::Bind( + { { 6U, U"pick" }, + callable, + false, + Core::Expression::Closure({}, + {}, + Core::Type::int64(), + { BothReturn(Core::Expression::Constant( + true, + Core::Type::boolean())) }, + callable) }) }); + REQUIRE(prepared.functions.size() == 2U); + const auto lifted = EndsOf(prepared.functions.back()); + CHECK(lifted.returningNothing == 0U); + CHECK(lifted.unreachable == 1U); +} diff --git a/Compiler/Core/Tests/Fixtures/Core/wire-v8-closure.hex b/Compiler/Core/Tests/Fixtures/Core/wire-v10-closure.hex similarity index 86% rename from Compiler/Core/Tests/Fixtures/Core/wire-v8-closure.hex rename to Compiler/Core/Tests/Fixtures/Core/wire-v10-closure.hex index 627dabad..86386226 100644 --- a/Compiler/Core/Tests/Fixtures/Core/wire-v8-closure.hex +++ b/Compiler/Core/Tests/Fixtures/Core/wire-v10-closure.hex @@ -1,6 +1,6 @@ -# Visual X# Core wire v8 closure document generated by the Haskell frontend. +# Visual X# Core wire v10 closure document generated by the Haskell frontend. # Source: closure-boundary.vxs -56 58 43 52 08 00 00 00 01 00 00 00 0f 00 00 00 +56 58 43 52 0a 00 00 00 01 00 00 00 0f 00 00 00 43 00 00 00 6c 00 00 00 6f 00 00 00 73 00 00 00 75 00 00 00 72 00 00 00 65 00 00 00 42 00 00 00 6f 00 00 00 75 00 00 00 6e 00 00 00 64 00 00 00 diff --git a/Compiler/Core/Tests/Fixtures/Core/wire-v8-project-source.hex b/Compiler/Core/Tests/Fixtures/Core/wire-v10-project-source.hex similarity index 86% rename from Compiler/Core/Tests/Fixtures/Core/wire-v8-project-source.hex rename to Compiler/Core/Tests/Fixtures/Core/wire-v10-project-source.hex index 0541b6db..283abc5e 100644 --- a/Compiler/Core/Tests/Fixtures/Core/wire-v8-project-source.hex +++ b/Compiler/Core/Tests/Fixtures/Core/wire-v10-project-source.hex @@ -1,6 +1,6 @@ -# Visual X# Core wire v8 document emitted by the Haskell Core writer. +# Visual X# Core wire v10 document emitted by the Haskell Core writer. # Module Demo; function Main belongs to Sources/Main.vxs. -56 58 43 52 08 00 00 00 01 00 00 00 +56 58 43 52 0a 00 00 00 01 00 00 00 04 00 00 00 44 00 00 00 65 00 00 00 6d 00 00 00 6f 00 00 00 01 00 00 00 10 00 00 00 53 00 00 00 6f 00 00 00 diff --git a/Compiler/Core/Tests/Fixtures/Core/wire-v8.hex b/Compiler/Core/Tests/Fixtures/Core/wire-v10.hex similarity index 81% rename from Compiler/Core/Tests/Fixtures/Core/wire-v8.hex rename to Compiler/Core/Tests/Fixtures/Core/wire-v10.hex index bb2b4877..dacedb1b 100644 --- a/Compiler/Core/Tests/Fixtures/Core/wire-v8.hex +++ b/Compiler/Core/Tests/Fixtures/Core/wire-v10.hex @@ -1,6 +1,6 @@ -# Visual X# Core wire v8 golden document +# Visual X# Core wire v10 golden document # module Demo; function Main() -> unit { return unit; } -56 58 43 52 08 00 00 00 +56 58 43 52 0a 00 00 00 01 00 00 00 04 00 00 00 44 00 00 00 65 00 00 00 6d 00 00 00 6f 00 00 00 00 00 00 00 diff --git a/Compiler/Core/Tests/NestingLimitTests.cpp b/Compiler/Core/Tests/NestingLimitTests.cpp new file mode 100644 index 00000000..a3e975a4 --- /dev/null +++ b/Compiler/Core/Tests/NestingLimitTests.cpp @@ -0,0 +1,498 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/Scalar.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// The native stages recurse once per level of statement nesting. The wire +// reader is the boundary at which untrusted Core enters them, so it bounds +// that level as it bounds type and expression depth, and the writer refuses +// what the reader would reject. These tests pin where the level is counted: +// every body is one level below the statement or closure that holds it, the +// links of an `else if` chain share a level, and an empty body costs none. +// The last tests nest far deeper than the stack of a test process allows and +// run on the compiler stack, the one the limits are stated against. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Prepared = visual_xsharp::core; + namespace Wire = Visual::XSharp::Core::Wire; + + constexpr std::uint64_t kValue = 2U; + constexpr std::uint64_t kTotal = 3U; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Value() -> Core::Expression + { + return Core::Expression::Variable({ kValue, U"value" }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Total() -> Core::Expression + { + return Core::Expression::Variable({ kTotal, U"total" }, + Core::Type::int64()); + } + + /// `value > literal`. + [[nodiscard]] auto + Exceeds(std::int64_t literal) -> Core::Expression + { + return Core::Expression::InvokePrimitive(Core::Primitive::GreaterThan, + { Value(), Integer(literal) }, + Core::Type::boolean()); + } + + /// `total = total + 1;` + [[nodiscard]] auto + Count() -> Core::Statement + { + return Core::Statement::Assign( + { kTotal, U"total" }, + Core::Expression::InvokePrimitive(Core::Primitive::Add, + { Total(), Integer(1) }, + Core::Type::int64())); + } + + /** + * @brief `if (value > 0) { if (value > 1) { ... total = total + 1; } }`. + * + * With `levels` conditionals the assignment is `levels` bodies below the + * function body. The nest is built from the inside out, so building it + * does not recurse. + */ + [[nodiscard]] auto + Nest(std::size_t levels) -> Core::Statement + { + auto statement = Count(); + for (std::size_t index = levels; index-- > 0U;) + { + std::vector body; + body.push_back(std::move(statement)); + statement + = Core::Statement::If(Exceeds(static_cast(index)), + std::move(body), + {}); + } + return statement; + } + + /// `long Pick(long value) { long total = 0; return total; }` + [[nodiscard]] auto + Module(Core::Statement statement) -> Core::Module + { + std::vector body; + body.push_back( + Core::Statement::Bind(Core::Binding{ { kTotal, U"total" }, + Core::Type::int64(), + true, + Integer(0) })); + body.push_back(std::move(statement)); + body.push_back(Core::Statement::Return(Total())); + // The function is moved into the module. A braced list would copy + // it, and copying nested statements recurses once per level. + Core::Module module; + module.name = { U"Nesting" }; + module.functions.push_back( + Core::Function{ { 1U, U"Pick" }, + { { { kValue, U"value" }, Core::Type::int64() } }, + Core::Type::int64(), + std::move(body) }); + return module; + } + + [[nodiscard]] auto + WithDepth(std::size_t depth) -> Wire::Limits + { + Wire::Limits limits; + limits.maximumStatementDepth = depth; + return limits; + } + + [[nodiscard]] auto + IsLimit(const std::optional &error) -> bool + { + return error && error->kind == Wire::ErrorKind::LimitExceeded; + } +} // namespace + +TEST_CASE("the default statement depth limit matches the expression limit", + "[core][wire][nesting]") +{ + const Wire::Limits limits; + CHECK(limits.maximumStatementDepth == 4096U); + CHECK(limits.maximumExpressionDepth == 4096U); +} + +TEST_CASE("the wire bounds the nesting of statement bodies", + "[core][wire][nesting]") +{ + // The function body is level 1, so `levels` conditionals put the + // innermost body at level `levels + 1`. + constexpr std::size_t kLevels = 7U; + const auto module = Module(Nest(kLevels)); + REQUIRE(Core::Verify(module).empty()); + + SECTION("a module at the limit is written and read") + { + const auto limits = WithDepth(kLevels + 1U); + const auto encoded = Wire::Encode(module, limits); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes, limits); + REQUIRE(decoded); + CHECK(*decoded.module == module); + } + SECTION("the writer refuses a module one level beyond the limit") + { + const auto encoded = Wire::Encode(module, WithDepth(kLevels)); + CHECK_FALSE(encoded); + CHECK(IsLimit(encoded.error)); + } + SECTION("the reader refuses a document one level beyond the limit") + { + const auto encoded = Wire::Encode(module); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes, WithDepth(kLevels)); + CHECK_FALSE(decoded); + CHECK(IsLimit(decoded.error)); + } + SECTION("a limit of zero refuses any body with a statement") + { + const auto encoded = Wire::Encode(module); + REQUIRE(encoded); + CHECK(IsLimit(Wire::Decode(encoded.bytes, WithDepth(0U)).error)); + CHECK(IsLimit(Wire::Encode(module, WithDepth(0U)).error)); + } +} + +TEST_CASE("the links of an else-if chain share one nesting level", + "[core][wire][nesting]") +{ + // if (value > 0) { count } else if (value > 1) { count } ... : the + // bodies of all links are at level 2, whatever the length of the chain. + auto statement = Core::Statement::If(Exceeds(99), { Count() }, { Count() }); + for (std::int64_t index = 98; index >= 0; --index) + { + std::vector next; + next.push_back(std::move(statement)); + statement + = Core::Statement::If(Exceeds(index), { Count() }, std::move(next)); + } + const auto module = Module(std::move(statement)); + REQUIRE(Core::Verify(module).empty()); + + const auto atLimit = WithDepth(2U); + const auto encoded = Wire::Encode(module, atLimit); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes, atLimit); + REQUIRE(decoded); + CHECK(*decoded.module == module); + + CHECK(IsLimit(Wire::Encode(module, WithDepth(1U)).error)); + CHECK(IsLimit(Wire::Decode(encoded.bytes, WithDepth(1U)).error)); +} + +TEST_CASE("an empty body costs no nesting level", "[core][wire][nesting]") +{ + // Two conditionals nested through their true branches, each with an + // empty false branch. The innermost body is at level 3; the empty + // branches need no level of their own, at the limit or beyond it. + std::vector inner; + inner.push_back(Core::Statement::If(Exceeds(1), { Count() }, {})); + const auto module + = Module(Core::Statement::If(Exceeds(0), std::move(inner), {})); + REQUIRE(Core::Verify(module).empty()); + + const auto limits = WithDepth(3U); + const auto encoded = Wire::Encode(module, limits); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes, limits); + REQUIRE(decoded); + CHECK(*decoded.module == module); + CHECK(IsLimit(Wire::Encode(module, WithDepth(2U)).error)); + CHECK(IsLimit(Wire::Decode(encoded.bytes, WithDepth(2U)).error)); + + // A body of only empty branches is written at a limit of one level. + const auto hollow = Module(Core::Statement::If(Exceeds(0), {}, {})); + const auto shallow = Wire::Encode(hollow, WithDepth(1U)); + REQUIRE(shallow); + CHECK(Wire::Decode(shallow.bytes, WithDepth(1U))); +} + +TEST_CASE("a loop body and a closure body are one level below their owner", + "[core][wire][nesting]") +{ + SECTION("while") + { + const auto module + = Module(Core::Statement::While(Exceeds(0), { Count() })); + CHECK(Wire::Encode(module, WithDepth(2U))); + CHECK(IsLimit(Wire::Encode(module, WithDepth(1U)).error)); + } + SECTION("closure") + { + // The closure is an operand of a statement in the function body, so + // its own body is at level 2 and the conditional's body at level 3. + auto closure = Core::Expression::Closure( + {}, + {}, + Core::Type::int64(), + { Core::Statement::If( + Core::Expression::Constant(true, Core::Type::boolean()), + { Core::Statement::Return(Integer(1)) }, + {}), + Core::Statement::Return(Integer(0)) }, + Core::Type::function({}, Core::Type::int64())); + const auto module + = Module(Core::Statement::Evaluate(std::move(closure))); + const auto encoded = Wire::Encode(module, WithDepth(3U)); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes, WithDepth(3U)); + REQUIRE(decoded); + CHECK(*decoded.module == module); + CHECK(IsLimit(Wire::Encode(module, WithDepth(2U)).error)); + CHECK(IsLimit(Wire::Decode(encoded.bytes, WithDepth(2U)).error)); + } +} + +TEST_CASE("the compiler stack carries a result back to its caller", + "[core][stack]") +{ + CHECK(Visual::XSharp::Support::RunOnCompilerStack([] { + return 42; + }) + == 42); + int touched = 0; + Visual::XSharp::Support::RunOnCompilerStack([&touched] { + touched = 7; + }); + CHECK(touched == 7); + CHECK(Visual::XSharp::Support::kCompilerStackBytes + == std::size_t{ 256U } * 1024U * 1024U); +} + +TEST_CASE("every native Core stage walks deep nesting on the compiler stack", + "[core][nesting][stack]") +{ + // 1500 levels is more than five times the frontend's statement limit + // and far beyond what the stack of this process could carry: about 80 + // levels fit in one megabyte. Building, comparing and destroying the + // nest recurse as well, so all of it happens on the compiler stack. + constexpr std::size_t kLevels = 1500U; + const auto outcome = Visual::XSharp::Support::RunOnCompilerStack([] { + const auto module = Module(Nest(kLevels)); + if (!Core::Verify(module).empty()) + return 1; + const auto encoded = Wire::Encode(module); + if (!encoded) + return 2; + const auto decoded = Wire::Decode(encoded.bytes); + if (!decoded || !(*decoded.module == module)) + return 3; + const auto prepared = Core::CorePrep::Prepare(*decoded.module); + if (!Prepared::verify(prepared).empty()) + return 4; + // One branch per conditional. + std::size_t branches = 0U; + for (const auto &block : prepared.functions.front().blocks) + if (block.terminator.kind == Prepared::Terminator::Kind::Branch) + ++branches; + return branches == kLevels ? 0 : 5; + }); + CHECK(outcome == 0); +} + +TEST_CASE("the reader refuses nesting beyond the default limit", + "[core][wire][nesting][stack]") +{ + // One level beyond the default. The writer is given a larger limit so + // that the document exists; the reader with the default must refuse it + // with a limit error and without walking it. + constexpr std::size_t kLevels = 4096U; + const auto outcome = Visual::XSharp::Support::RunOnCompilerStack([] { + const auto module = Module(Nest(kLevels)); + const auto encoded = Wire::Encode(module, WithDepth(kLevels + 1U)); + if (!encoded) + return 1; + if (!IsLimit(Wire::Encode(module).error)) + return 2; + return IsLimit(Wire::Decode(encoded.bytes).error) ? 0 : 3; + }); + CHECK(outcome == 0); +} + +TEST_CASE("the adapter prepares a long chain without copying the module", + "[core][coreprep][nesting]") +{ + // A chain of 3000 links is 3000 levels of nested statements. Copying + // such a statement recurses once per level, and the adapter used to + // copy every function before preparing it, which overflowed the stack + // near 1500 links. Verification and preparation themselves walk a chain + // in a loop; building and destroying the module still recurse, so the + // whole test runs on the compiler stack. + constexpr std::size_t kLinks = 3000U; + const auto outcome = Visual::XSharp::Support::RunOnCompilerStack([] { + auto statement = Core::Statement::If(Exceeds(0), { Count() }, {}); + for (std::size_t index = 1U; index < kLinks; ++index) + { + std::vector next; + next.push_back(std::move(statement)); + statement + = Core::Statement::If(Exceeds(static_cast(index)), + { Count() }, + std::move(next)); + } + const auto module = Module(std::move(statement)); + if (!Core::Verify(module).empty()) + return 1; + const auto prepared = Core::CorePrep::Prepare(module); + if (!Prepared::verify(prepared).empty()) + return 2; + std::size_t branches = 0U; + for (const auto &block : prepared.functions.front().blocks) + if (block.terminator.kind == Prepared::Terminator::Kind::Branch) + ++branches; + return branches == kLinks ? 0 : 3; + }); + CHECK(outcome == 0); +} + +namespace +{ + /// `value + value + ...` with the given number of additions, each sum + /// the first operand of the next. + [[nodiscard]] auto + LeftChain(std::size_t additions) -> Core::Expression + { + auto expression = Value(); + for (std::size_t index = 0U; index < additions; ++index) + { + std::vector operands; + operands.push_back(std::move(expression)); + operands.push_back(Value()); + expression = Core::Expression::InvokePrimitive(Core::Primitive::Add, + std::move(operands), + Core::Type::int64()); + } + return expression; + } + + /// `value + (value + (...))` with the given number of additions, each + /// sum the second operand of the one around it. + [[nodiscard]] auto + RightChain(std::size_t additions) -> Core::Expression + { + auto expression = Value(); + for (std::size_t index = 0U; index < additions; ++index) + { + std::vector operands; + operands.push_back(Value()); + operands.push_back(std::move(expression)); + expression = Core::Expression::InvokePrimitive(Core::Primitive::Add, + std::move(operands), + Core::Type::int64()); + } + return expression; + } + + [[nodiscard]] auto + WithExpressionDepth(std::size_t depth) -> Wire::Limits + { + Wire::Limits limits; + limits.maximumExpressionDepth = depth; + return limits; + } +} // namespace + +TEST_CASE("a chain of operators is one expression level", + "[core][wire][nesting]") +{ + // The first operand of a primitive is at the level of the primitive. + // A sum of 20001 operands therefore fits a depth limit of one, and the + // whole test runs on the stack the test process starts with: the + // writer, the reader, the verifier and the adapter walk the chain in a + // loop, and releasing the module does not recurse along it either. + constexpr std::size_t kAdditions = 20000U; + const auto module = Module( + Core::Statement::Assign({ kTotal, U"total" }, LeftChain(kAdditions))); + const auto limits = WithExpressionDepth(1U); + const auto encoded = Wire::Encode(module, limits); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes, limits); + REQUIRE(decoded); + REQUIRE(decoded.module); + const auto again = Wire::Encode(*decoded.module, limits); + REQUIRE(again); + CHECK(again.bytes == encoded.bytes); + CHECK(Core::Verify(*decoded.module).empty()); + const auto prepared = Core::CorePrep::Prepare(*decoded.module); + CHECK(Prepared::verify(prepared).empty()); + REQUIRE(prepared.functions.size() == 1U); + // One block: the binding of `total`, a temporary for every addition, + // and the assignment that copies the last one. + REQUIRE(prepared.functions.front().blocks.size() == 1U); + CHECK(prepared.functions.front().blocks.front().instructions.size() + == kAdditions + 2U); +} + +TEST_CASE("an operand after the first is one expression level deeper", + "[core][wire][nesting]") +{ + // Five additions that nest to the right put the innermost operand at + // depth five; the expression of the statement is at depth zero. + const auto module + = Module(Core::Statement::Assign({ kTotal, U"total" }, RightChain(5U))); + CHECK(Wire::Encode(module, WithExpressionDepth(5U))); + CHECK(IsLimit(Wire::Encode(module, WithExpressionDepth(4U)).error)); + const auto encoded = Wire::Encode(module, WithExpressionDepth(5U)); + REQUIRE(encoded); + CHECK(Wire::Decode(encoded.bytes, WithExpressionDepth(5U))); + CHECK(IsLimit(Wire::Decode(encoded.bytes, WithExpressionDepth(4U)).error)); +} + +TEST_CASE("a long else-if chain is handled on the default stack", + "[core][coreprep][nesting]") +{ + // 20000 links are 20000 levels of nested statements. Every native Core + // stage walks them in a loop and the statements are released from a + // list, so none of this needs the compiler stack. + constexpr std::size_t kLinks = 20000U; + auto statement = Core::Statement::If(Exceeds(0), { Count() }, {}); + for (std::size_t index = 1U; index < kLinks; ++index) + { + std::vector next; + next.push_back(std::move(statement)); + statement + = Core::Statement::If(Exceeds(static_cast(index)), + { Count() }, + std::move(next)); + } + const auto module = Module(std::move(statement)); + const auto encoded = Wire::Encode(module); + REQUIRE(encoded); + const auto decoded = Wire::Decode(encoded.bytes); + REQUIRE(decoded); + REQUIRE(decoded.module); + CHECK(Core::Verify(*decoded.module).empty()); + const auto prepared = Core::CorePrep::Prepare(*decoded.module); + CHECK(Prepared::verify(prepared).empty()); +} diff --git a/Compiler/Core/Tests/RuntimeCallVerifierTests.cpp b/Compiler/Core/Tests/RuntimeCallVerifierTests.cpp new file mode 100644 index 00000000..7830469e --- /dev/null +++ b/Compiler/Core/Tests/RuntimeCallVerifierTests.cpp @@ -0,0 +1,278 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/IR.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" + +// The runtime call in native Core: what the verifier that reads a Core +// artifact from the frontend accepts and rejects, and that an accepted call +// reaches generated code. The Haskell verifier has the same cases in +// `RuntimeCallTests.hs`; the stages after Core are in +// `RuntimeCallPipelineTests.cpp`. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Runtime = visual_xsharp::core::runtime; + namespace Pipeline = Visual::XSharp::Pipeline; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Text(std::u32string value) -> Core::Expression + { + return Core::Expression::Constant(std::move(value), + Core::Type::string()); + } + + [[nodiscard]] auto + Count() -> Core::Expression + { + return Core::Expression::Variable({ 2U, U"count" }, + Core::Type::int64()); + } + + /// A call of the function with the given arguments and type. + [[nodiscard]] auto + Call(Runtime::Function function, + std::vector arguments, + Core::Type type) -> Core::Expression + { + std::vector operands; + operands.reserve(arguments.size() + 1U); + operands.push_back(Integer(static_cast(function))); + for (auto &argument : arguments) + operands.push_back(std::move(argument)); + return Core::Expression::InvokePrimitive(Core::Primitive::RuntimeCall, + std::move(operands), + std::move(type)); + } + + /// `Run(count) { ; return count; }` + [[nodiscard]] auto + Module(Core::Statement statement) -> Core::Module + { + return { { U"Calls" }, + { Core::Function{ + { 1U, U"Run" }, + { { { 2U, U"count" }, Core::Type::int64() } }, + Core::Type::int64(), + { std::move(statement), Core::Statement::Return(Count()) }, + } } }; + } + + [[nodiscard]] auto + Evaluating(Core::Expression expression) -> Core::Module + { + return Module(Core::Statement::Evaluate(std::move(expression))); + } + + [[nodiscard]] auto + Binding(Core::Type type, Core::Expression expression) -> Core::Module + { + return Module(Core::Statement::Bind({ { 3U, U"line" }, + std::move(type), + false, + std::move(expression) })); + } + + [[nodiscard]] auto + Rejected(const Core::Module &module) -> bool + { + const auto issues = Core::Verify(module); + return std::ranges::any_of(issues, [](const auto &issue) { + return issue.code == "VXC1075"; + }); + } + + /// `Console.Printfn("Count: %5d", count)` as the frontend lowers it. + [[nodiscard]] auto + Writing() -> Core::Module + { + return Evaluating( + Call(Runtime::Function::ConsoleWrite, + { Call(Runtime::Function::TextConcat, + { Text(U"Count: "), + Call(Runtime::Function::TextFormatSigned, + { Integer(0), Integer(5), Integer(-1), Count() }, + Core::Type::string()) }, + Core::Type::string()), + Integer(1) }, + Core::Type::unit())); + } +} // namespace + +TEST_CASE("native Core accepts well-formed runtime calls") +{ + CHECK(Core::Verify(Writing()).empty()); + CHECK(Core::Verify(Binding(Core::Type::boolean(), + Call(Runtime::Function::TextEquals, + { Text(U"a"), Text(U"b") }, + Core::Type::boolean()))) + .empty()); + CHECK(Core::Verify(Binding(Core::Type::string(), + Call(Runtime::Function::TextNewline, + {}, + Core::Type::string()))) + .empty()); +} + +TEST_CASE("native Core wire carries runtime calls") +{ + const auto module = Writing(); + const auto encoded = Core::Wire::Encode(module); + REQUIRE(encoded); + const auto decoded = Core::Wire::Decode(encoded.bytes); + REQUIRE(decoded); + REQUIRE(decoded.module); + CHECK(*decoded.module == module); +} + +TEST_CASE("native Core rejects a runtime call that names no function") +{ + const auto raw = [](std::vector operands) { + return Core::Expression::InvokePrimitive(Core::Primitive::RuntimeCall, + std::move(operands), + Core::Type::unit()); + }; + // An identity the catalog does not have, zero, and a negative number. + CHECK(Rejected(Evaluating(raw({ Integer(999) })))); + CHECK(Rejected(Evaluating(raw({ Integer(0) })))); + CHECK(Rejected(Evaluating(raw({ Integer(-12) })))); + // No operand, and an identity that is computed: the function is fixed + // when the program is compiled. + CHECK(Rejected(Evaluating(raw({})))); + CHECK(Rejected(Evaluating(raw({ Count(), Text(U"a"), Integer(0) })))); + // An identity of another integer type. + CHECK(Rejected(Evaluating(raw( + { Core::Expression::Constant(std::int64_t{ 12 }, Core::Type::uint64()), + Text(U"a"), + Integer(0) })))); +} + +TEST_CASE("native Core holds a runtime call to its function's arguments") +{ + const auto text = Core::Type::string(); + // One too few, one too many, and one where none is taken. + CHECK(Rejected( + Binding(text, + Call(Runtime::Function::TextConcat, { Text(U"a") }, text)))); + CHECK(Rejected(Binding(text, + Call(Runtime::Function::TextConcat, + { Text(U"a"), Text(U"b"), Text(U"c") }, + text)))); + CHECK(Rejected( + Binding(text, + Call(Runtime::Function::TextNewline, { Count() }, text)))); + // A number where a string is taken, and a string where a number is. + CHECK(Rejected(Binding( + text, + Call(Runtime::Function::TextConcat, { Text(U"a"), Count() }, text)))); + CHECK(Rejected(Binding( + text, + Call(Runtime::Function::TextFromSigned, { Text(U"a") }, text)))); + // A signed integer where an unsigned one is taken. + CHECK(Rejected( + Binding(text, + Call(Runtime::Function::TextFromUnsigned, { Count() }, text)))); + // A width that is not an int. + CHECK( + Rejected(Binding(text, + Call(Runtime::Function::TextFormatSigned, + { Integer(0), + Core::Expression::Constant(std::int64_t{ 5 }, + Core::Type::int32()), + Integer(-1), + Count() }, + text)))); +} + +TEST_CASE("native Core holds a runtime call to its function's result") +{ + CHECK(Rejected(Binding(Core::Type::int64(), + Call(Runtime::Function::TextConcat, + { Text(U"a"), Text(U"b") }, + Core::Type::int64())))); + CHECK(Rejected(Binding(Core::Type::string(), + Call(Runtime::Function::ConsoleWrite, + { Text(U"a"), Integer(0) }, + Core::Type::string())))); + CHECK(Rejected(Binding(Core::Type::string(), + Call(Runtime::Function::TextEquals, + { Text(U"a"), Text(U"b") }, + Core::Type::string())))); +} + +TEST_CASE("the adapter names the function with a literal in CorePrep") +{ + const auto prepared = Core::CorePrep::Prepare(Writing()); + REQUIRE(visual_xsharp::core::verify(prepared).empty()); + std::size_t calls{}; + for (const auto &function : prepared.functions) + for (const auto &block : function.blocks) + for (const auto &instruction : block.instructions) + { + if (instruction.operation + != visual_xsharp::core::Operation::RuntimeCall) + continue; + ++calls; + REQUIRE_FALSE(instruction.operands.empty()); + // The identity is not bound to a temporary on the way. + CHECK(instruction.operands.front().kind + == visual_xsharp::core::Atom::Kind::Literal); + } + CHECK(calls == 3U); +} + +TEST_CASE("a runtime call in Core reaches generated code") +{ + const auto encoded = Core::Wire::Encode(Writing()); + REQUIRE(encoded); + for (const auto optimize : { false, true }) + { + Pipeline::Options options; + options.optimize_xpp = optimize; + options.optimize_xmm = optimize; + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes, options); + REQUIRE(pipeline); + REQUIRE(pipeline.llvm); + const std::string_view ir = pipeline.llvm->llvm_ir; + // A write is observed by whoever reads the output: no pipeline mode + // removes it, and it is made once. + const auto first = ir.find("call void @vxs_console_write("); + REQUIRE(first != std::string_view::npos); + CHECK(ir.find("call void @vxs_console_write(", first + 1U) + == std::string_view::npos); + CHECK(ir.find("@vxs_text_format_signed(") != std::string_view::npos); + CHECK(ir.find("@vxs_text_concat(") != std::string_view::npos); + } +} + +TEST_CASE("a runtime call that the verifier rejects is never lowered") +{ + const auto encoded = Core::Wire::Encode(Evaluating( + Core::Expression::InvokePrimitive(Core::Primitive::RuntimeCall, + { Integer(999) }, + Core::Type::unit()))); + REQUIRE(encoded); + const auto pipeline = Pipeline::ConsumeCore(encoded.bytes, {}); + CHECK_FALSE(pipeline); + CHECK_FALSE(pipeline.llvm); +} diff --git a/Compiler/Core/Tests/TemplateTests.cpp b/Compiler/Core/Tests/TemplateTests.cpp index 9ed86f06..67a77fd3 100644 --- a/Compiler/Core/Tests/TemplateTests.cpp +++ b/Compiler/Core/Tests/TemplateTests.cpp @@ -420,7 +420,7 @@ TEST_CASE("rendered identity length-prefixes qualified components") CHECK(second.find("1:") != std::string::npos); } -TEST_CASE("Core v8 preserves ordered template arguments and source ownership") +TEST_CASE("Core v10 preserves ordered template arguments and source ownership") { const auto type = Applied(U"Mix", { TypeArgument(Model::Type::string()), @@ -434,11 +434,11 @@ TEST_CASE("Core v8 preserves ordered template arguments and source ownership") const auto decoded = Core::Wire::Decode(encoded.bytes); REQUIRE(decoded); CHECK(*decoded.module == module); - CHECK(Core::Wire::kCurrentVersion == 8U); + CHECK(Core::Wire::kCurrentVersion == 10U); } TEST_CASE( - "CorePrep v6 preserves ordered template arguments and source ownership") + "CorePrep v7 preserves ordered template arguments and source ownership") { const auto type = Applied(U"Mix", { TypeArgument(Model::Type::string()), @@ -452,7 +452,7 @@ TEST_CASE( const auto decoded = CorePrepWire::decode(encoded.bytes); REQUIRE(decoded); CHECK(*decoded.module == module); - CHECK(CorePrepWire::current_version == 6U); + CHECK(CorePrepWire::current_version == 8U); } TEST_CASE("specialization table interns identical types once") diff --git a/Compiler/Core/Verifier.cpp b/Compiler/Core/Verifier.cpp index 5eed072c..21a91835 100644 --- a/Compiler/Core/Verifier.cpp +++ b/Compiler/Core/Verifier.cpp @@ -4,11 +4,14 @@ #include #include #include +#include +#include #include #include "Compiler/Artifact/SourcePath.hpp" #include "Visual/XSharp/ADTs/DenseIdMap.hpp" #include "Visual/XSharp/Core/Ownership.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" #include "Visual/XSharp/Core/Scalar.hpp" #include "Visual/XSharp/Core/Template.hpp" #include "Visual/XSharp/Core/Verifier.hpp" @@ -23,13 +26,81 @@ namespace Visual::XSharp::Core bool mutableBinding{}; std::u32string spelling; }; - using Environment = ADTs::DenseIdMap; + using Definitions = ADTs::DenseIdMap; + + /// The local definitions visible at one point of a function. + /// + /// A branch, a loop body, a let and a closure each see what is + /// defined around them and add definitions that end with them. An + /// environment therefore holds only what its own region defines and + /// refers to the environment of the region around it. Entering a + /// region costs nothing that depends on how much is already + /// defined; a copy of everything visible, which is what this + /// replaced, made a function with many branches or closures cost + /// the product of the two. + /// + /// The environment around a region must outlive the region's own + /// and must not gain definitions while the inner one is in use. + /// The verifier walks regions strictly inside one another, so both + /// hold. + class Environment final + { + public: + Environment() = default; + Environment(const Environment &) = delete; + Environment(Environment &&) = default; + auto + operator=(const Environment &) -> Environment & = delete; + auto + operator=(Environment &&) -> Environment & = delete; + ~Environment() = default; + + /// The environment of a region inside this one. + [[nodiscard]] auto + Extend() const -> Environment + { + Environment inner; + inner.outer_ = this; + return inner; + } + + /// The definition of a symbol in this region or, failing that, + /// in the nearest region around it that defines it. + [[nodiscard]] auto + Find(const SymbolId symbol) const -> const Definition * + { + for (const auto *region = this; region != nullptr; + region = region->outer_) + if (const auto *found = region->own_.Find(symbol)) + return found; + return nullptr; + } + + [[nodiscard]] auto + Contains(const SymbolId symbol) const -> bool + { + return Find(symbol) != nullptr; + } + + /// Define a symbol in this region. It hides a definition of + /// the same symbol in a region around this one until this + /// region ends. + void + InsertOrAssign(const SymbolId symbol, Definition definition) + { + own_.InsertOrAssign(symbol, std::move(definition)); + } + + private: + const Environment *outer_{}; + Definitions own_; + }; class FunctionVerifier final { public: FunctionVerifier(const Function &function, - const Environment &functions, + const Definitions &functions, std::vector &issues) : function_(function) , functions_(functions) @@ -77,7 +148,7 @@ namespace Visual::XSharp::Core private: const Function &function_; - const Environment &functions_; + const Definitions &functions_; Environment environment_; std::vector &issues_; @@ -97,40 +168,48 @@ namespace Visual::XSharp::Core return locals.Contains(symbol) || functions_.Contains(symbol); } - void - Add(std::string code, std::string message, SymbolId symbol = 0U) + // The verifier recurses once per level of nesting. Reporting + // is therefore kept out of the functions that recurse: the + // texts are passed as views and become strings only here, so a + // check costs its caller no string on the stack. + [[gnu::noinline]] void + Add(std::string_view code, + std::string_view message, + SymbolId symbol = 0U) { - issues_.push_back(VerificationIssue{ std::move(code), - std::move(message), + issues_.push_back(VerificationIssue{ std::string(code), + std::string(message), function_.symbol.id, symbol }); } void CheckSymbol(const SymbolName &symbol, - std::string code, - std::string message) + std::string_view code, + std::string_view message) { if (symbol.id == 0U) - Add(std::move(code), std::move(message), symbol.id); + Add(code, message, symbol.id); } - void - CheckType(const Type &type, std::string code, std::string message) + [[gnu::noinline]] void + CheckType(const Type &type, + std::string_view code, + std::string_view message) { if (ContainsInvalidType(type)) - Add(std::move(code), std::move(message)); + Add(code, message); for (const auto &templateIssue : Template::Validate(type)) Add("VXC1040", "invalid Core template type: " + templateIssue.message); } - void + [[gnu::noinline]] void CheckSameType(const Type &expected, const Type &actual, - std::string code, - std::string message, + std::string_view code, + std::string_view message, SymbolId symbol = 0U) { if (expected != actual) - Add(std::move(code), std::move(message), symbol); + Add(code, message, symbol); } [[nodiscard]] static auto ContainsInvalidType(const Type &type) -> bool @@ -164,7 +243,49 @@ namespace Visual::XSharp::Core Kind::Boolean; }); } - /// Whether every path through the statements returns. The false + /// Whether a statement is a loop that cannot be left: its + /// condition is the literal true and no `break` leaves it. Such + /// a loop ends only through a `return` inside it, or not at + /// all, so control never reaches the statement after it. + [[nodiscard]] static auto + CannotBeLeft(const Statement &statement) -> bool + { + if (statement.kind != Statement::Kind::While + && statement.kind != Statement::Kind::DoWhile + && statement.kind != Statement::Kind::For) + return false; + const auto *const condition + = std::get_if(&statement.expression.literal); + if (statement.expression.kind != Expression::Kind::Literal + || condition == nullptr || !*condition) + return false; + // The statements of the loop are searched from a list, so + // an `else if` chain in the body costs no stack per link. + // A break in a nested loop leaves that loop. + std::vector *> pending{ + &statement.loopBody, + &statement.loopUpdate + }; + while (!pending.empty()) + { + const auto *const statements = pending.back(); + pending.pop_back(); + for (const auto &nested : *statements) + { + if (nested.kind == Statement::Kind::Break) + return false; + if (nested.kind == Statement::Kind::If) + { + pending.push_back(&nested.trueBranch); + pending.push_back(&nested.falseBranch); + } + } + } + return true; + } + /// Whether control can never fall off the end of the statements: + /// every path returns, or runs into a loop that cannot be left. + /// The false /// branch of an `if` is followed in a loop rather than by /// recursion while it is the last statement to decide the /// answer, so an `else if` chain costs no stack per link. @@ -176,7 +297,8 @@ namespace Visual::XSharp::Core const Statement *deciding = nullptr; for (const auto &statement : statements) { - if (statement.kind == Statement::Kind::Return) + if (statement.kind == Statement::Kind::Return + || CannotBeLeft(statement)) return true; if (statement.kind != Statement::Kind::If || statement.falseBranch.empty() @@ -251,7 +373,7 @@ namespace Visual::XSharp::Core * environment for the whole chain instead of copying it per * link. */ - void + [[gnu::noinline]] void VerifyConditionalChain(const Statement &first, Environment &environment, const Type &expectedReturnType, @@ -268,7 +390,7 @@ namespace Visual::XSharp::Core if (!accepts_boolean_context(link->expression.type)) Add("VXC1017", "Core condition must be bool or numeric"); - auto trueEnvironment = *linkEnvironment; + auto trueEnvironment = linkEnvironment->Extend(); VerifyStatements(link->trueBranch, trueEnvironment, expectedReturnType, @@ -277,18 +399,22 @@ namespace Visual::XSharp::Core break; if (!chainEnvironment) { - chainEnvironment.emplace(environment); + chainEnvironment.emplace(environment.Extend()); linkEnvironment = &*chainEnvironment; } link = &link->falseBranch.front(); } - auto falseEnvironment = *linkEnvironment; + auto falseEnvironment = linkEnvironment->Extend(); VerifyStatements(link->falseBranch, falseEnvironment, expectedReturnType, scope); } + // Each kind of statement and of expression is verified by a + // function of its own, so that a level of nesting costs the + // frame of the kind that nests and not the frames of every kind + // together. void VerifyStatement(const Statement &statement, Environment &environment, @@ -298,58 +424,11 @@ namespace Visual::XSharp::Core switch (statement.kind) { case Statement::Kind::Bind: - { - const auto &binding = statement.binding; - CheckSymbol(binding.symbol, - "VXC1008", - "Core binding symbol must be positive"); - CheckType(binding.type, - "VXC1009", - "Core binding has an unresolved type"); - VerifyExpression(binding.value, environment); - CheckSameType(binding.type, - binding.value.type, - "VXC1011", - "Core binding value type does not match " - "its declaration", - binding.symbol.id); - if (ContainsDefinition(environment, binding.symbol.id)) - Add("VXC1010", - "Core binding symbol is already defined", - binding.symbol.id); - environment.InsertOrAssign( - binding.symbol.id, - Definition{ binding.type, - binding.mutableBinding, - binding.symbol.spelling }); + VerifyBinding(statement.binding, environment); return; - } case Statement::Kind::Assign: - { - CheckSymbol(statement.destination, - "VXC1015", - "Core assignment symbol must be positive"); - VerifyExpression(statement.expression, environment); - const auto *found - = FindDefinition(environment, - statement.destination.id); - if (found == nullptr) - Add("VXC1012", - "Core assignment targets an undefined symbol", - statement.destination.id); - else if (!found->mutableBinding) - Add("VXC1013", - "Core assignment targets an immutable symbol", - statement.destination.id); - else - CheckSameType( - found->type, - statement.expression.type, - "VXC1014", - "Core assignment value has the wrong type", - statement.destination.id); + VerifyAssignment(statement, environment); return; - } case Statement::Kind::Return: VerifyExpression(statement.expression, environment); CheckSameType(expectedReturnType, @@ -369,26 +448,8 @@ namespace Visual::XSharp::Core case Statement::Kind::While: case Statement::Kind::DoWhile: case Statement::Kind::For: - { - VerifyExpression(statement.expression, environment); - if (!accepts_boolean_context(statement.expression.type)) - Add("VXC1017", - "Core loop condition must be bool or numeric"); - auto loopEnvironment = environment; - VerifyStatements(statement.loopBody, - loopEnvironment, - expectedReturnType, - TransferScope::InLoopBody); - if (statement.kind == Statement::Kind::For) - { - auto updateEnvironment = environment; - VerifyStatements(statement.loopUpdate, - updateEnvironment, - expectedReturnType, - TransferScope::InForUpdate); - } + VerifyLoop(statement, environment, expectedReturnType); return; - } case Statement::Kind::Break: if (scope == TransferScope::OutsideLoop) Add("VXC1064", "Core break appears outside a loop"); @@ -404,46 +465,126 @@ namespace Visual::XSharp::Core return; } } - void + [[gnu::noinline]] void + VerifyBinding(const Binding &binding, Environment &environment) + { + CheckSymbol(binding.symbol, + "VXC1008", + "Core binding symbol must be positive"); + CheckType(binding.type, + "VXC1009", + "Core binding has an unresolved type"); + VerifyExpression(binding.value, environment); + CheckSameType(binding.type, + binding.value.type, + "VXC1011", + "Core binding value type does not match " + "its declaration", + binding.symbol.id); + if (ContainsDefinition(environment, binding.symbol.id)) + Add("VXC1010", + "Core binding symbol is already defined", + binding.symbol.id); + environment.InsertOrAssign( + binding.symbol.id, + Definition{ binding.type, + binding.mutableBinding, + binding.symbol.spelling }); + } + [[gnu::noinline]] void + VerifyAssignment(const Statement &statement, + const Environment &environment) + { + CheckSymbol(statement.destination, + "VXC1015", + "Core assignment symbol must be positive"); + VerifyExpression(statement.expression, environment); + const auto *found + = FindDefinition(environment, statement.destination.id); + if (found == nullptr) + Add("VXC1012", + "Core assignment targets an undefined symbol", + statement.destination.id); + else if (!found->mutableBinding) + Add("VXC1013", + "Core assignment targets an immutable symbol", + statement.destination.id); + else + CheckSameType(found->type, + statement.expression.type, + "VXC1014", + "Core assignment value has the wrong type", + statement.destination.id); + } + [[gnu::noinline]] void + VerifyLoop(const Statement &statement, + const Environment &environment, + const Type &expectedReturnType) + { + VerifyExpression(statement.expression, environment); + if (!accepts_boolean_context(statement.expression.type)) + Add("VXC1017", + "Core loop condition must be bool or numeric"); + auto loopEnvironment = environment.Extend(); + VerifyStatements(statement.loopBody, + loopEnvironment, + expectedReturnType, + TransferScope::InLoopBody); + if (statement.kind == Statement::Kind::For) + { + auto updateEnvironment = environment.Extend(); + VerifyStatements(statement.loopUpdate, + updateEnvironment, + expectedReturnType, + TransferScope::InForUpdate); + } + } + + /** + * @brief Verify one expression. + * + * A chain of operators nests in the first operand of each + * primitive, as deep as the chain is long. Those operands are + * walked in a loop: the type of each primitive is checked on + * the way down, and its other operands and its own rules on + * the way back, innermost first. The checks and their order + * are those of the recursive formulation. + */ + [[gnu::noinline]] void VerifyExpression(const Expression &expression, const Environment &environment) { - CheckType(expression.type, - "VXC1018", - "Core expression has an unresolved type"); + std::vector chain; + const Expression *current = &expression; + for (;;) + { + CheckType(current->type, + "VXC1018", + "Core expression has an unresolved type"); + if (current->kind != Expression::Kind::Primitive + || current->operands.empty()) + break; + chain.push_back(current); + current = ¤t->operands.front(); + } + VerifyUnchained(*current, environment); + while (!chain.empty()) + { + VerifyPrimitive(*chain.back(), environment); + chain.pop_back(); + } + } + /// An expression that is not a primitive with operands; its + /// type was checked. + void + VerifyUnchained(const Expression &expression, + const Environment &environment) + { switch (expression.kind) { case Expression::Kind::Variable: - { - CheckSymbol(expression.symbol, - "VXC1019", - "Core variable symbol must be positive"); - const auto *found - = FindDefinition(environment, expression.symbol.id); - if (found == nullptr) - Add("VXC1020", - "Core expression references an undefined " - "symbol", - expression.symbol.id); - else - { - CheckSameType(found->type, - expression.type, - "VXC1021", - "Core variable type disagrees with " - "its definition", - expression.symbol.id); - if (!expression.symbol.spelling.empty() - && !found->spelling.empty() - && expression.symbol.spelling - != found->spelling) - Add("VXC1030", - "Core symbol spelling disagrees with its " - "definition", - expression.symbol.id); - } + VerifyVariable(expression, environment); return; - } case Expression::Kind::Literal: VerifyLiteral(expression); return; @@ -457,78 +598,110 @@ namespace Visual::XSharp::Core VerifyClosure(expression, environment); return; case Expression::Kind::Let: - { - CheckSymbol(expression.letSymbol, - "VXC1045", - "Core let symbol must be positive"); - CheckType(expression.letType, - "VXC1046", - "Core let binding has an unresolved type"); - if (!expression.letValue || !expression.letBody) - { - Add("VXC1047", "Core let value or body is missing"); - return; - } - VerifyExpression(*expression.letValue, environment); - CheckSameType(expression.letType, - expression.letValue->type, - "VXC1048", - "Core let value has the wrong type"); - auto bodyEnvironment = environment; - bodyEnvironment.InsertOrAssign( - expression.letSymbol.id, - Definition{ expression.letType, - false, - expression.letSymbol.spelling }); - VerifyExpression(*expression.letBody, bodyEnvironment); - CheckSameType( - expression.type, - expression.letBody->type, - "VXC1049", - "Core let result disagrees with its body"); + VerifyLet(expression, environment); return; - } case Expression::Kind::Conditional: - { - if (expression.operands.size() != 3U) - { - Add("VXC1071", - "Core conditional must contain a test and two " - "arms"); - return; - } - const auto &test = expression.operands[0]; - const auto &whenTrue = expression.operands[1]; - const auto &whenFalse = expression.operands[2]; - VerifyExpression(test, environment); - if (!accepts_boolean_context(test.type)) - Add("VXC1067", - "Core conditional test must be bool or " - "numeric"); - VerifyExpression(whenTrue, environment); - VerifyExpression(whenFalse, environment); - CheckSameType(expression.type, - whenTrue.type, - "VXC1068", - "Core conditional result type disagrees " - "with its first arm"); - CheckSameType(expression.type, - whenFalse.type, - "VXC1069", - "Core conditional result type disagrees " - "with its second arm"); - // The result is materialized in a plain storage - // slot. Owned values would need move and release - // rules for that slot. - if (!accepts_boolean_context(expression.type)) - Add("VXC1070", - "Core conditional result must be bool or " - "numeric"); + VerifyConditional(expression, environment); return; - } } } - void + [[gnu::noinline]] void + VerifyVariable(const Expression &expression, + const Environment &environment) + { + CheckSymbol(expression.symbol, + "VXC1019", + "Core variable symbol must be positive"); + const auto *found + = FindDefinition(environment, expression.symbol.id); + if (found == nullptr) + { + Add("VXC1020", + "Core expression references an undefined symbol", + expression.symbol.id); + return; + } + CheckSameType(found->type, + expression.type, + "VXC1021", + "Core variable type disagrees with its " + "definition", + expression.symbol.id); + if (!expression.symbol.spelling.empty() + && !found->spelling.empty() + && expression.symbol.spelling != found->spelling) + Add("VXC1030", + "Core symbol spelling disagrees with its definition", + expression.symbol.id); + } + [[gnu::noinline]] void + VerifyLet(const Expression &expression, + const Environment &environment) + { + CheckSymbol(expression.letSymbol, + "VXC1045", + "Core let symbol must be positive"); + CheckType(expression.letType, + "VXC1046", + "Core let binding has an unresolved type"); + if (!expression.letValue || !expression.letBody) + { + Add("VXC1047", "Core let value or body is missing"); + return; + } + VerifyExpression(*expression.letValue, environment); + CheckSameType(expression.letType, + expression.letValue->type, + "VXC1048", + "Core let value has the wrong type"); + auto bodyEnvironment = environment.Extend(); + bodyEnvironment.InsertOrAssign( + expression.letSymbol.id, + Definition{ expression.letType, + false, + expression.letSymbol.spelling }); + VerifyExpression(*expression.letBody, bodyEnvironment); + CheckSameType(expression.type, + expression.letBody->type, + "VXC1049", + "Core let result disagrees with its body"); + } + [[gnu::noinline]] void + VerifyConditional(const Expression &expression, + const Environment &environment) + { + if (expression.operands.size() != 3U) + { + Add("VXC1071", + "Core conditional must contain a test and two arms"); + return; + } + const auto &test = expression.operands[0]; + const auto &whenTrue = expression.operands[1]; + const auto &whenFalse = expression.operands[2]; + VerifyExpression(test, environment); + if (!accepts_boolean_context(test.type)) + Add("VXC1067", + "Core conditional test must be bool or numeric"); + VerifyExpression(whenTrue, environment); + VerifyExpression(whenFalse, environment); + CheckSameType(expression.type, + whenTrue.type, + "VXC1068", + "Core conditional result type disagrees with " + "its first arm"); + CheckSameType(expression.type, + whenFalse.type, + "VXC1069", + "Core conditional result type disagrees with " + "its second arm"); + // The result is materialized in a plain storage slot. Owned + // values would need move and release rules for that slot. + if (!accepts_boolean_context(expression.type)) + Add("VXC1070", + "Core conditional result must be bool or numeric"); + } + [[gnu::noinline]] void VerifyLiteral(const Expression &expression) { if (const auto issue @@ -537,7 +710,7 @@ namespace Visual::XSharp::Core "Core literal payload does not match its type: " + *issue); } - void + [[gnu::noinline]] void VerifyCall(const Expression &expression, const Environment &environment) { @@ -572,16 +745,53 @@ namespace Visual::XSharp::Core "VXC1024", "Core call result type disagrees with the callee"); } - void + /// A call of a runtime function: the first operand is the + /// literal that names the function, and the rest are checked + /// against the row of the catalog that function has. + [[gnu::noinline]] void + VerifyRuntimeCall(const Expression &expression) + { + namespace runtime = ::visual_xsharp::core::runtime; + const runtime::Signature *signature = nullptr; + if (!expression.operands.empty() + && expression.operands.front().kind + == Expression::Kind::Literal) + if (const auto identity = runtime::IdentityOf( + expression.operands.front().literal, + expression.operands.front().type)) + signature = runtime::Find(*identity); + std::vector arguments; + arguments.reserve(expression.operands.size()); + for (std::size_t index = 1U; index < expression.operands.size(); + ++index) + arguments.push_back(expression.operands[index].type.kind); + const auto defect = runtime::Check(signature, + arguments, + expression.type.kind); + if (defect != runtime::Defect::None) + Add("VXC1075", + "Core " + std::string(runtime::Describe(defect))); + } + /// Verify the operands of a primitive after the first, which + /// the caller has verified, and then the primitive itself. + [[gnu::noinline]] void VerifyPrimitive(const Expression &expression, const Environment &environment) { - for (const auto &operand : expression.operands) - VerifyExpression(operand, environment); + for (std::size_t index = 1U; index < expression.operands.size(); + ++index) + VerifyExpression(expression.operands[index], environment); + if (expression.primitive == Primitive::RuntimeCall) + { + VerifyRuntimeCall(expression); + return; + } + const auto memoize = expression.primitive == Primitive::Memoize; const auto unary = expression.primitive == Primitive::Negate || expression.primitive == Primitive::LogicalNot - || expression.primitive == Primitive::BitwiseNot; + || expression.primitive == Primitive::BitwiseNot + || memoize; const auto logical = expression.primitive == Primitive::LogicalAnd || expression.primitive == Primitive::LogicalOr @@ -611,7 +821,22 @@ namespace Visual::XSharp::Core "VXC1027", "Core primitive operands must have matching types"); - if (typeTest) + if (memoize) + { + // The remembered result is kept in the callable, in a + // slot that owns nothing. + const auto remembers + = operandType.kind == Type::Kind::Function + && operandType.components.size() == 1U + && (operandType.components.front().kind + == Type::Kind::Bool + || is_numeric(operandType.components.front())); + if (!remembers) + Add("VXC1073", + "Core memoization requires a callable without " + "parameters whose result is bool or numeric"); + } + else if (typeTest) { if (expression.operands.size() == 2U) { @@ -663,7 +888,7 @@ namespace Visual::XSharp::Core "VXC1028", "Core primitive result has the wrong type"); } - void + [[gnu::noinline]] void VerifyClosure(const Expression &expression, const Environment &outerEnvironment) { @@ -676,7 +901,7 @@ namespace Visual::XSharp::Core CheckType(expression.closureReturnType, "VXC1032", "Core closure has an unresolved return type"); - Environment closureEnvironment = outerEnvironment; + auto closureEnvironment = outerEnvironment.Extend(); ADTs::DenseIdSet localSymbols; localSymbols.Reserve(expression.captures.size() + expression.closureParameters.size()); @@ -804,7 +1029,7 @@ namespace Visual::XSharp::Core 0U }); } - Environment functions; + Definitions functions; functions.Reserve(module.functions.size()); for (const auto &function : module.functions) { diff --git a/Compiler/Core/Wire.cpp b/Compiler/Core/Wire.cpp deleted file mode 100644 index 787c0c67..00000000 --- a/Compiler/Core/Wire.cpp +++ /dev/null @@ -1,1379 +0,0 @@ -// SPDX-FileCopyrightText: 2026 Progmasoft -// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 - -#include -#include -#include - -#include "Visual/XSharp/Core/Scalar.hpp" -#include "Visual/XSharp/Core/Wire.hpp" - -namespace Visual::XSharp::Core::Wire -{ - namespace - { - inline constexpr std::uint8_t kMagic[] = { 'V', 'X', 'C', 'R' }; - - class Reader final - { - public: - Reader(std::span bytes, const Limits &limits) - : bytes_(bytes) - , limits_(limits) - {} - - /// @brief Validate the entire envelope before publication. - /// Header checks precede allocation; the final offset check - /// rejects trailing bytes and ambiguous encodings. - [[nodiscard]] auto - Document() -> DecodeResult - { - if (bytes_.size() > limits_.maximumWireBytes) - return Failure(ErrorKind::LimitExceeded, - "wire byte length", - "input exceeds configured byte limit"); - for (const auto expected : kMagic) - if (Byte("magic") != expected) - return Failure( - ErrorKind::InvalidMagic, - "magic", - "input is not a Visual X# Core document"); - const auto version = Unsigned("version"); - if (!error_ && version != kCurrentVersion) - Fail(ErrorKind::UnsupportedVersion, - "version", - "unsupported Core wire version"); - const auto flags = Unsigned("flags"); - if (!error_ && flags != 0U) - Fail(ErrorKind::InvalidTag, - "flags", - "reserved flags must be zero"); - if (error_) - return Result(); - - Module module; - module.name = QualifiedName("module name"); - module.sourceFiles - = Vector(limits_.maximumFunctions, - "source file count", - [this] { - return Text("source file"); - }); - module.functions = Vector(limits_.maximumFunctions, - "function count", - [this] { - return ReadFunction(); - }); - if (!error_ && offset_ != bytes_.size()) - Fail(ErrorKind::TrailingInput, - "document", - "bytes remain after Core module"); - if (!error_) - module_ = std::move(module); - return Result(); - } - - private: - std::span bytes_; - const Limits &limits_; - std::size_t offset_{}; - std::optional module_; - std::optional error_; - - [[nodiscard]] auto - Result() -> DecodeResult - { - return DecodeResult{ std::move(module_), std::move(error_) }; - } - [[nodiscard]] auto - Failure(ErrorKind kind, std::string context, std::string message) - -> DecodeResult - { - Fail(kind, std::move(context), std::move(message)); - return Result(); - } - void - Fail(ErrorKind kind, std::string context, std::string message) - { - if (!error_) - error_ = Error{ kind, - offset_, - std::move(context), - std::move(message) }; - } - [[nodiscard]] auto - Byte(std::string_view context) -> std::uint8_t - { - if (offset_ >= bytes_.size()) - { - Fail(ErrorKind::TruncatedInput, - std::string(context), - "input ended before field was complete"); - return 0U; - } - return bytes_[offset_++]; - } - /** - * @brief Read an unsigned little-endian wire scalar - * of a fixed width. - * - * This byte order - * is explicit so host endianness cannot alter VXCR. - */ - template - [[nodiscard]] auto - Unsigned(std::string_view context) -> Integer - { - static_assert(std::is_unsigned_v); - Integer result{}; - for (std::size_t shift = 0; shift < sizeof(Integer) * 8U; - shift += 8U) - result |= static_cast(Byte(context)) << shift; - return result; - } - /// @brief Validate an untrusted length before narrowing it. - /// Callers use the checked count before reserving memory or - /// decoding recursively, preventing unchecked allocations. - [[nodiscard]] auto - Count(std::size_t maximum, std::string_view context) -> std::size_t - { - const auto value = Unsigned(context); - if (!error_ && value > maximum) - Fail(ErrorKind::LimitExceeded, - std::string(context), - "collection count exceeds configured limit"); - return error_ ? 0U : static_cast(value); - } - /// @brief Reserve only after Count enforces the caller's limit. - template - [[nodiscard]] auto - Vector(std::size_t maximum, std::string_view context, Decode decode) - -> std::vector - { - const auto size = Count(maximum, context); - std::vector values; - values.reserve(size); - for (std::size_t index = 0; index < size && !error_; ++index) - values.push_back(decode()); - return values; - } - /// @brief Read UTF-32 text and reject non-scalar values. - /// VXCR counts Unicode scalars, not UTF-8 bytes or UTF-16 - /// code units. Surrogates and values above U+10FFFF are invalid. - [[nodiscard]] auto - Text(std::string_view context) -> std::u32string - { - const auto size = Count(limits_.maximumTextScalars, context); - std::u32string value; - value.reserve(size); - for (std::size_t index = 0; index < size && !error_; ++index) - { - const auto scalar = Unsigned(context); - if (scalar > 0x10ffffU - || (scalar >= 0xd800U && scalar <= 0xdfffU)) - { - Fail(ErrorKind::InvalidScalar, - std::string(context), - "wire text contains a non-scalar Unicode value"); - break; - } - value.push_back(static_cast(scalar)); - } - return value; - } - [[nodiscard]] auto - QualifiedName(std::string_view context) - -> std::vector - { - return Vector(65535U, context, [this, context] { - return Text(context); - }); - } - [[nodiscard]] auto - Symbol(std::string_view context) -> SymbolName - { - const auto id = Unsigned(context); - if (!error_ && id == 0U) - Fail(ErrorKind::InvalidSymbol, - std::string(context), - "symbol id must be positive"); - return SymbolName{ id, Text(context) }; - } - [[nodiscard]] auto - ReadType(std::size_t depth = 0U) -> Type - { - if (depth > limits_.maximumTypeDepth) - { - Fail(ErrorKind::LimitExceeded, - "type", - "type nesting exceeds configured limit"); - return Type::unit(); - } - switch (Byte("type tag")) - { - case 0: - return Type::unit(); - case 1: - return Type::boolean(); - case 2: - return Type::int64(); - case 3: - return Type::string(); - case 4: - { - auto name = QualifiedName("named type"); - auto arguments = Vector< - ::visual_xsharp::core::TemplateArgument>( - limits_.maximumOperands, - "template argument count", - [this, depth] { - switch (Byte("template argument tag")) - { - case 0: - return ::visual_xsharp::core:: - TemplateArgument::type_argument( - ReadType(depth + 1U)); - case 1: - return ::visual_xsharp::core:: - TemplateArgument::value_argument( - ::visual_xsharp::core:: - TemplateValue:: - integer_value( - ReadInteger( - "template " - "integer"))); - case 2: - return ::visual_xsharp::core:: - TemplateArgument::value_argument( - ::visual_xsharp::core:: - TemplateValue:: - boolean_value(Boolean( - "template " - "boolean"))); - case 3: - return ::visual_xsharp::core:: - TemplateArgument::value_argument( - ::visual_xsharp::core:: - TemplateValue:: - character_value( - ReadInteger( - "template " - "character"))); - case 4: - return ::visual_xsharp::core:: - TemplateArgument::value_argument( - ::visual_xsharp::core:: - TemplateValue:: - parameter_value(Symbol( - "template value " - "parameter"))); - default: - Fail(ErrorKind::InvalidTag, - "template argument tag", - "unknown template argument tag"); - return ::visual_xsharp::core:: - TemplateArgument{}; - } - }); - return Type::named_template(std::move(name), - std::move(arguments)); - } - case 5: - { - auto parameters - = Vector(limits_.maximumParameters, - "function type parameter count", - [this, depth] { - return ReadType(depth + 1U); - }); - return Type::function(std::move(parameters), - ReadType(depth + 1U)); - } - case 6: - return Type::type_variable( - Symbol("type variable symbol")); - case 7: - return Type::character(); - case 8: - return Type::int8(); - case 9: - return Type::int16(); - case 10: - return Type::int32(); - case 11: - return Type::int128(); - case 12: - return Type::uint8(); - case 13: - return Type::uint16(); - case 14: - return Type::uint32(); - case 15: - return Type::uint64(); - case 16: - return Type::uint128(); - case 17: - return Type::float16(); - case 18: - return Type::float32(); - case 19: - return Type::float64(); - case 20: - return Type::float128(); - default: - Fail(ErrorKind::InvalidTag, - "type tag", - "unknown Core type tag"); - return Type::unit(); - } - } - [[nodiscard]] auto - Boolean(std::string_view context) -> bool - { - const auto value = Byte(context); - if (value > 1U) - Fail(ErrorKind::InvalidBoolean, - std::string(context), - "boolean byte must be zero or one"); - return value == 1U; - } - [[nodiscard]] auto - ReadInteger(std::string_view context) - -> ::visual_xsharp::core::IntegerLiteral - { - ::visual_xsharp::core::IntegerLiteral value; - value.negative = Boolean(std::string(context) + " sign"); - value.magnitude = Vector( - limits_.maximumNumericBytes, - std::string(context) + " magnitude", - [this, context] { - return Byte(std::string(context) + " magnitude"); - }); - if (!::visual_xsharp::core::integer_is_canonical(value)) - Fail(ErrorKind::InvalidInteger, - std::string(context), - "integer magnitude/sign is not canonical"); - return value; - } - [[nodiscard]] auto - ReadLiteral() -> Literal - { - switch (Byte("literal tag")) - { - case 0: - return std::monostate{}; - case 1: - return Boolean("boolean literal"); - case 2: - return static_cast( - Unsigned("integer literal")); - case 3: - return Text("string literal"); - case 4: - { - ::visual_xsharp::core::IntegerLiteral value; - value.negative = Boolean("integer sign"); - value.magnitude = Vector( - limits_.maximumNumericBytes, - "integer magnitude", - [this] { - return Byte("integer magnitude"); - }); - if (!::visual_xsharp::core::integer_is_canonical(value)) - Fail(ErrorKind::InvalidInteger, - "integer literal", - "integer magnitude/sign is not canonical"); - return value; - } - case 5: - { - const auto size = Count(limits_.maximumNumericBytes, - "floating literal length"); - std::string spelling; - spelling.reserve(size); - for (std::size_t index = 0; index < size && !error_; - ++index) - { - const auto value = Byte("floating literal"); - if (value > 0x7fU) - Fail(ErrorKind::InvalidInteger, - "floating literal", - "floating spelling must be ASCII"); - else - spelling.push_back(static_cast(value)); - } - if (!error_ - && !::visual_xsharp::core:: - floating_spelling_is_valid(spelling)) - Fail(ErrorKind::InvalidInteger, - "floating literal", - "floating spelling is not canonical"); - return ::visual_xsharp::core::FloatingLiteral{ - std::move(spelling) - }; - } - case 6: - return std::monostate{}; - default: - Fail(ErrorKind::InvalidTag, - "literal tag", - "unknown Core literal tag"); - return std::monostate{}; - } - } - [[nodiscard]] auto - ReadExpression(std::size_t depth = 0U) -> Expression - { - if (depth > limits_.maximumExpressionDepth) - { - Fail(ErrorKind::LimitExceeded, - "expression", - "expression nesting exceeds configured limit"); - return {}; - } - const auto tag = Byte("expression tag"); - const auto primitiveTag - = tag == 3U ? Byte("primitive tag") : 0U; - auto valueType = ReadType(); - switch (tag) - { - case 0: - return Expression::Variable(Symbol("variable symbol"), - std::move(valueType)); - case 1: - return Expression::Constant(ReadLiteral(), - std::move(valueType)); - case 2: - { - auto callee = ReadExpression(depth + 1U); - auto arguments = Vector( - limits_.maximumOperands, - "call argument count", - [this, depth] { - return ReadExpression(depth + 1U); - }); - return Expression::Apply(std::move(callee), - std::move(arguments), - std::move(valueType)); - } - case 3: - { - if (primitiveTag - > static_cast(Primitive::TypeIs)) - Fail(ErrorKind::InvalidTag, - "primitive tag", - "unknown Core primitive tag"); - auto arguments = Vector( - limits_.maximumOperands, - "primitive operand count", - [this, depth] { - return ReadExpression(depth + 1U); - }); - return Expression::InvokePrimitive( - // A rejected tag never becomes an enumerator. - primitiveTag > static_cast( - Primitive::TypeIs) - ? Primitive::Add - : static_cast(primitiveTag), - std::move(arguments), - std::move(valueType)); - } - case 4: - { - auto captures = Vector(limits_.maximumOperands, - "closure capture count", - [this, depth] { - return ReadCapture( - depth + 1U); - }); - auto parameters = Vector>( - limits_.maximumParameters, - "closure parameter count", - [this] { - auto parameter = ReadParameter(); - return std::pair{ std::move(parameter.symbol), - std::move(parameter.type) }; - }); - auto returnType = ReadType(); - auto body - = Vector(limits_.maximumStatements, - "closure statement count", - [this] { - return ReadStatement(); - }); - return Expression::Closure(std::move(captures), - std::move(parameters), - std::move(returnType), - std::move(body), - std::move(valueType)); - } - case 5: - { - auto symbol = Symbol("let symbol"); - auto bindingType = ReadType(); - auto value = ReadExpression(depth + 1U); - auto body = ReadExpression(depth + 1U); - return Expression::Let(std::move(symbol), - std::move(bindingType), - std::move(value), - std::move(body), - std::move(valueType)); - } - case 6: - { - // The three children have fixed positions, so the - // payload carries no count. - auto test = ReadExpression(depth + 1U); - auto whenTrue = ReadExpression(depth + 1U); - auto whenFalse = ReadExpression(depth + 1U); - return Expression::Conditional(std::move(test), - std::move(whenTrue), - std::move(whenFalse), - std::move(valueType)); - } - default: - Fail(ErrorKind::InvalidTag, - "expression tag", - "unknown Core expression tag"); - return {}; - } - } - [[nodiscard]] auto - ReadCapture(std::size_t depth) -> Capture - { - const auto tag = Byte("closure capture mode"); - if (tag > 2U) - Fail(ErrorKind::InvalidTag, - "closure capture mode", - "unknown Core closure capture mode"); - auto symbol = Symbol("closure capture symbol"); - auto type = ReadType(); - auto value = ReadExpression(depth); - Capture capture; - // A rejected tag never becomes an enumerator. - capture.mode = tag > 2U ? CaptureMode::Strong - : static_cast(tag); - capture.symbol = std::move(symbol); - capture.type = std::move(type); - capture.value = std::make_shared(std::move(value)); - return capture; - } - /// The tag byte of a conditional statement. - static constexpr std::uint8_t kConditionalTag = 3U; - - /** - * @brief Read a conditional statement whose tag was consumed, - * and every `else if` that continues it. - * - * An `else if` is encoded as a false branch that holds exactly - * one conditional statement. Reading such a chain by recursion - * uses stack in proportion to its length, so a long chain in - * valid source would overflow it. The links are read in a loop - * into a flat list and nested afterwards, innermost first. The - * bytes consumed and the resulting tree are those of the - * recursive formulation. - */ - [[nodiscard]] auto - ReadConditionalChain() -> Statement - { - struct Link final - { - Expression condition; - std::vector whenTrue; - }; - std::vector links; - std::vector finalBranch; - for (;;) - { - auto condition = ReadExpression(); - auto whenTrue - = Vector(limits_.maximumStatements, - "true branch statement count", - [this] { - return ReadStatement(); - }); - links.push_back( - { std::move(condition), std::move(whenTrue) }); - const auto size = Count(limits_.maximumStatements, - "false branch statement count"); - // A false branch of one conditional is the next link: - // take its tag and read it in the next iteration. - if (!error_ && size == 1U && offset_ < bytes_.size() - && bytes_[offset_] == kConditionalTag) - { - ++offset_; - continue; - } - finalBranch.reserve(size); - for (std::size_t index = 0; index < size && !error_; - ++index) - finalBranch.push_back(ReadStatement()); - break; - } - auto statement - = Statement::If(std::move(links.back().condition), - std::move(links.back().whenTrue), - std::move(finalBranch)); - links.pop_back(); - while (!links.empty()) - { - std::vector nested; - nested.push_back(std::move(statement)); - statement = Statement::If(std::move(links.back().condition), - std::move(links.back().whenTrue), - std::move(nested)); - links.pop_back(); - } - return statement; - } - - [[nodiscard]] auto - ReadStatement() -> Statement - { - switch (Byte("statement tag")) - { - case 0: - { - auto symbol = Symbol("binding symbol"); - auto type = ReadType(); - const auto mutableBinding - = Boolean("binding mutability"); - auto value = ReadExpression(); - return Statement::Bind(Binding{ std::move(symbol), - std::move(type), - mutableBinding, - std::move(value) }); - } - case 1: - { - auto symbol = Symbol("assignment symbol"); - return Statement::Assign(std::move(symbol), - ReadExpression()); - } - case 2: - return Statement::Return(ReadExpression()); - case 3: - return ReadConditionalChain(); - case 4: - return Statement::Evaluate(ReadExpression()); - case 5: - { - auto condition = ReadExpression(); - auto body - = Vector(limits_.maximumStatements, - "while body statement count", - [this] { - return ReadStatement(); - }); - return Statement::While(std::move(condition), - std::move(body)); - } - case 6: - { - auto body - = Vector(limits_.maximumStatements, - "do/while body statement count", - [this] { - return ReadStatement(); - }); - return Statement::DoWhile(std::move(body), - ReadExpression()); - } - case 7: - { - auto condition = ReadExpression(); - auto body - = Vector(limits_.maximumStatements, - "for body statement count", - [this] { - return ReadStatement(); - }); - auto update - = Vector(limits_.maximumStatements, - "for update statement count", - [this] { - return ReadStatement(); - }); - return Statement::For(std::move(condition), - std::move(body), - std::move(update)); - } - case 8: - return Statement::Break(); - case 9: - return Statement::Continue(); - default: - Fail(ErrorKind::InvalidTag, - "statement tag", - "unknown Core statement tag"); - return {}; - } - } - [[nodiscard]] auto - ReadParameter() -> Parameter - { - return Parameter{ Symbol("parameter symbol"), ReadType() }; - } - [[nodiscard]] auto - ReadFunction() -> Function - { - Function function; - function.symbol = Symbol("function symbol"); - function.sourceFile = Text("function source file"); - function.parameters - = Vector(limits_.maximumParameters, - "parameter count", - [this] { - return ReadParameter(); - }); - function.returnType = ReadType(); - function.body = Vector(limits_.maximumStatements, - "statement count", - [this] { - return ReadStatement(); - }); - return function; - } - }; - - class Writer final - { - public: - explicit Writer(const Limits &limits) - : limits_(limits) - {} - void - Document(const Module &module) - { - for (const auto value : kMagic) - Byte(value); - Unsigned(kCurrentVersion); - Unsigned(0U); - QualifiedName(module.name, "module name"); - Vector(module.sourceFiles, - limits_.maximumFunctions, - "source file count", - [this](const std::u32string &sourceFile) { - Text(sourceFile, "source file"); - }); - Vector(module.functions, - limits_.maximumFunctions, - "function count", - [this](const Function &function) { - WriteFunction(function); - }); - } - [[nodiscard]] auto - Finish() && -> EncodeResult - { - if (!error_ && bytes_.size() > limits_.maximumWireBytes) - Fail(ErrorKind::LimitExceeded, - "wire byte length", - "encoded document exceeds configured limit"); - return EncodeResult{ std::move(bytes_), std::move(error_) }; - } - - private: - const Limits &limits_; - std::vector bytes_; - std::optional error_; - - void - Fail(ErrorKind kind, std::string context, std::string message) - { - if (!error_) - error_ = Error{ kind, - bytes_.size(), - std::move(context), - std::move(message) }; - } - void - Byte(std::uint8_t value) - { - if (!error_) - bytes_.push_back(value); - } - template - void - Unsigned(Integer value) - { - static_assert(std::is_unsigned_v); - for (std::size_t shift = 0; shift < sizeof(Integer) * 8U; - shift += 8U) - Byte(static_cast( - (value >> shift) & static_cast(0xffU))); - } - void - Count(std::size_t value, - std::size_t maximum, - std::string_view context) - { - if (value > maximum - || value > std::numeric_limits::max()) - Fail(ErrorKind::LimitExceeded, - std::string(context), - "collection count exceeds wire limit"); - else - Unsigned(static_cast(value)); - } - template - void - Vector(const std::vector &values, - std::size_t maximum, - std::string_view context, - Encode encode) - { - Count(values.size(), maximum, context); - for (const auto &value : values) - { - if (error_) - return; - encode(value); - } - } - void - Text(const std::u32string &value, std::string_view context) - { - Count(value.size(), limits_.maximumTextScalars, context); - for (const auto scalar : value) - { - const auto numeric = static_cast(scalar); - if (numeric > 0x10ffffU - || (numeric >= 0xd800U && numeric <= 0xdfffU)) - { - Fail(ErrorKind::InvalidScalar, - std::string(context), - "text contains a non-scalar Unicode value"); - return; - } - Unsigned(numeric); - } - } - void - QualifiedName(const std::vector &parts, - std::string_view context) - { - Vector(parts, - 65535U, - context, - [this, context](const auto &part) { - Text(part, context); - }); - } - void - Symbol(const SymbolName &symbol, std::string_view context) - { - if (symbol.id == 0U) - { - Fail(ErrorKind::InvalidSymbol, - std::string(context), - "symbol id must be positive"); - return; - } - Unsigned(symbol.id); - Text(symbol.spelling, context); - } - void - WriteType(const Type &type, std::size_t depth = 0U) - { - if (depth > limits_.maximumTypeDepth) - { - Fail(ErrorKind::LimitExceeded, - "type", - "type nesting exceeds configured limit"); - return; - } - switch (type.kind) - { - case Type::Kind::Unit: - Byte(0); - return; - case Type::Kind::Bool: - Byte(1); - return; - case Type::Kind::Int64: - Byte(2); - return; - case Type::Kind::String: - Byte(3); - return; - case Type::Kind::Named: - Byte(4); - QualifiedName(type.name, "named type"); - Vector(type.templateArguments, - limits_.maximumOperands, - "template argument count", - [this, depth](const auto &argument) { - if (argument.kind - == ::visual_xsharp::core:: - TemplateArgument::Kind::Type) - { - Byte(0); - if (!argument.type) - { - Fail(ErrorKind::UnsupportedType, - "template argument", - "type argument has no payload"); - return; - } - WriteType(*argument.type, depth + 1U); - return; - } - switch (argument.value.kind) - { - case ::visual_xsharp::core:: - TemplateValue::Kind::Integer: - Byte(1); - WriteInteger(argument.value.integer, - "template integer"); - return; - case ::visual_xsharp::core:: - TemplateValue::Kind::Boolean: - Byte(2); - Byte(argument.value.boolean ? 1U - : 0U); - return; - case ::visual_xsharp::core:: - TemplateValue::Kind::Character: - Byte(3); - WriteInteger(argument.value.integer, - "template character"); - return; - case ::visual_xsharp::core:: - TemplateValue::Kind::Parameter: - Byte(4); - Symbol(argument.value.parameter, - "template value parameter"); - return; - } - }); - return; - case Type::Kind::Function: - Byte(5); - if (type.components.empty()) - { - Fail(ErrorKind::UnsupportedType, - "function type", - "function type has no result component"); - return; - } - Count(type.components.size() - 1U, - limits_.maximumParameters, - "function type parameter count"); - for (std::size_t index = 0; - index + 1U < type.components.size(); - ++index) - WriteType(type.components[index], depth + 1U); - WriteType(type.components.back(), depth + 1U); - return; - case Type::Kind::TypeVariable: - Byte(6); - Symbol(type.variable, "type variable symbol"); - return; - case Type::Kind::Character: - Byte(7); - return; - case Type::Kind::Int8: - Byte(8); - return; - case Type::Kind::Int16: - Byte(9); - return; - case Type::Kind::Int32: - Byte(10); - return; - case Type::Kind::Int128: - Byte(11); - return; - case Type::Kind::UInt8: - Byte(12); - return; - case Type::Kind::UInt16: - Byte(13); - return; - case Type::Kind::UInt32: - Byte(14); - return; - case Type::Kind::UInt64: - Byte(15); - return; - case Type::Kind::UInt128: - Byte(16); - return; - case Type::Kind::Float16: - Byte(17); - return; - case Type::Kind::Float32: - Byte(18); - return; - case Type::Kind::Float64: - Byte(19); - return; - case Type::Kind::Float128: - Byte(20); - return; - } - } - void - WriteInteger(const ::visual_xsharp::core::IntegerLiteral &integer, - std::string_view context) - { - if (!::visual_xsharp::core::integer_is_canonical(integer)) - { - Fail(ErrorKind::InvalidInteger, - std::string(context), - "integer magnitude/sign is not canonical"); - return; - } - Byte(integer.negative ? 1U : 0U); - Vector(integer.magnitude, - limits_.maximumNumericBytes, - std::string(context) + " magnitude", - [this](const auto octet) { - Byte(octet); - }); - } - void - WriteLiteral(const Literal &literal, const Type &valueType) - { - if (std::holds_alternative(literal)) - Byte(valueType.kind == Type::Kind::Unit ? 0U : 6U); - else if (const auto *boolean = std::get_if(&literal)) - { - Byte(1); - Byte(*boolean ? 1U : 0U); - } - else if (const auto *integer - = std::get_if(&literal)) - { - Byte(2); - Unsigned(static_cast(*integer)); - } - else if (const auto *string - = std::get_if(&literal)) - { - Byte(3); - Text(*string, "string literal"); - } - else if (const auto *wideInteger - = std::get_if<::visual_xsharp::core::IntegerLiteral>( - &literal)) - { - if (!::visual_xsharp::core::integer_is_canonical( - *wideInteger)) - { - Fail(ErrorKind::InvalidInteger, - "integer literal", - "integer magnitude/sign is not canonical"); - return; - } - Byte(4); - Byte(wideInteger->negative ? 1U : 0U); - Vector(wideInteger->magnitude, - limits_.maximumNumericBytes, - "integer magnitude", - [this](const std::uint8_t octet) { - Byte(octet); - }); - } - else if (const auto *floating - = std::get_if<::visual_xsharp::core::FloatingLiteral>( - &literal)) - { - if (!::visual_xsharp::core::floating_spelling_is_valid( - floating->spelling)) - { - Fail(ErrorKind::InvalidInteger, - "floating literal", - "floating spelling is not canonical"); - return; - } - Byte(5); - Count(floating->spelling.size(), - limits_.maximumNumericBytes, - "floating literal length"); - for (const auto character : floating->spelling) - Byte(static_cast(character)); - } - else if (const auto *narrowInteger - = std::get_if(&literal)) - { - Byte(4); - const auto normalized - = ::visual_xsharp::core::integer_from_signed( - *narrowInteger); - Byte(normalized.negative ? 1U : 0U); - Vector(normalized.magnitude, - limits_.maximumNumericBytes, - "integer magnitude", - [this](const std::uint8_t octet) { - Byte(octet); - }); - } - else - Fail(ErrorKind::UnsupportedType, - "literal", - "literal cannot cross the Core v5 boundary"); - } - void - WriteExpression(const Expression &expression, - std::size_t depth = 0U) - { - if (depth > limits_.maximumExpressionDepth) - { - Fail(ErrorKind::LimitExceeded, - "expression", - "expression nesting exceeds configured limit"); - return; - } - Byte(static_cast(expression.kind)); - if (expression.kind == Expression::Kind::Primitive) - Byte(static_cast(expression.primitive)); - WriteType(expression.type); - switch (expression.kind) - { - case Expression::Kind::Variable: - Symbol(expression.symbol, "variable symbol"); - return; - case Expression::Kind::Literal: - WriteLiteral(expression.literal, expression.type); - return; - case Expression::Kind::Apply: - if (!expression.callee) - { - Fail(ErrorKind::InvalidCount, - "callee", - "Core call must contain a callee"); - return; - } - WriteExpression(*expression.callee, depth + 1U); - Vector(expression.operands, - limits_.maximumOperands, - "call argument count", - [this, depth](const Expression &value) { - WriteExpression(value, depth + 1U); - }); - return; - case Expression::Kind::Primitive: - Vector(expression.operands, - limits_.maximumOperands, - "primitive operand count", - [this, depth](const Expression &value) { - WriteExpression(value, depth + 1U); - }); - return; - case Expression::Kind::Closure: - Vector( - expression.captures, - limits_.maximumOperands, - "closure capture count", - [this, depth](const Capture &capture) { - if (capture.mode != CaptureMode::Strong - && capture.mode != CaptureMode::Weak - && capture.mode != CaptureMode::Unowned) - { - Fail(ErrorKind::InvalidTag, - "closure capture mode", - "unknown Core closure capture mode"); - return; - } - Byte(static_cast(capture.mode)); - Symbol(capture.symbol, - "closure capture symbol"); - WriteType(capture.type); - if (!capture.value) - { - Fail(ErrorKind::InvalidCount, - "closure capture value", - "Core closure capture must contain a " - "value"); - return; - } - WriteExpression(*capture.value, depth + 1U); - }); - Vector(expression.closureParameters, - limits_.maximumParameters, - "closure parameter count", - [this](const auto ¶meter) { - Symbol(parameter.first, "parameter symbol"); - WriteType(parameter.second); - }); - WriteType(expression.closureReturnType); - if (!expression.closureBody) - { - Fail(ErrorKind::InvalidCount, - "closure body", - "Core closure must contain a body"); - return; - } - Vector(*expression.closureBody, - limits_.maximumStatements, - "closure statement count", - [this](const Statement &statement) { - WriteStatement(statement); - }); - return; - case Expression::Kind::Let: - Symbol(expression.letSymbol, "let symbol"); - WriteType(expression.letType); - if (!expression.letValue || !expression.letBody) - { - Fail(ErrorKind::InvalidCount, - "let expression", - "Core let must contain a value and body"); - return; - } - WriteExpression(*expression.letValue, depth + 1U); - WriteExpression(*expression.letBody, depth + 1U); - return; - case Expression::Kind::Conditional: - if (expression.operands.size() != 3U) - { - Fail(ErrorKind::InvalidCount, - "conditional expression", - "Core conditional must contain a test and " - "two arms"); - return; - } - for (const auto &operand : expression.operands) - WriteExpression(operand, depth + 1U); - return; - } - } - void - WriteStatement(const Statement &statement) - { - Byte(static_cast(statement.kind)); - switch (statement.kind) - { - case Statement::Kind::Bind: - Symbol(statement.binding.symbol, "binding symbol"); - WriteType(statement.binding.type); - Byte(statement.binding.mutableBinding ? 1U : 0U); - WriteExpression(statement.binding.value); - return; - case Statement::Kind::Assign: - Symbol(statement.destination, "assignment symbol"); - WriteExpression(statement.expression); - return; - case Statement::Kind::Return: - WriteExpression(statement.expression); - return; - case Statement::Kind::If: - { - // An `else if` chain is written in a loop for the - // reason ReadConditionalChain reads it in one: a - // false branch of exactly one conditional is the - // next link, and its bytes are its count, its tag - // and then the link itself. - const Statement *link = &statement; - while (!error_) - { - WriteExpression(link->expression); - Vector(link->trueBranch, - limits_.maximumStatements, - "true branch statement count", - [this](const Statement &value) { - WriteStatement(value); - }); - const auto continues - = link->falseBranch.size() == 1U - && link->falseBranch.front().kind - == Statement::Kind::If; - if (!continues) - { - Vector(link->falseBranch, - limits_.maximumStatements, - "false branch statement count", - [this](const Statement &value) { - WriteStatement(value); - }); - break; - } - Count(1U, - limits_.maximumStatements, - "false branch statement count"); - link = &link->falseBranch.front(); - Byte(static_cast(link->kind)); - } - return; - } - case Statement::Kind::Evaluate: - WriteExpression(statement.expression); - return; - case Statement::Kind::While: - WriteExpression(statement.expression); - Vector(statement.loopBody, - limits_.maximumStatements, - "while body statement count", - [this](const Statement &value) { - WriteStatement(value); - }); - return; - case Statement::Kind::DoWhile: - Vector(statement.loopBody, - limits_.maximumStatements, - "do/while body statement count", - [this](const Statement &value) { - WriteStatement(value); - }); - WriteExpression(statement.expression); - return; - case Statement::Kind::For: - WriteExpression(statement.expression); - Vector(statement.loopBody, - limits_.maximumStatements, - "for body statement count", - [this](const Statement &value) { - WriteStatement(value); - }); - Vector(statement.loopUpdate, - limits_.maximumStatements, - "for update statement count", - [this](const Statement &value) { - WriteStatement(value); - }); - return; - case Statement::Kind::Break: - case Statement::Kind::Continue: - return; - } - } - void - WriteFunction(const Function &function) - { - Symbol(function.symbol, "function symbol"); - Text(function.sourceFile, "function source file"); - Vector(function.parameters, - limits_.maximumParameters, - "parameter count", - [this](const Parameter ¶meter) { - Symbol(parameter.symbol, "parameter symbol"); - WriteType(parameter.type); - }); - WriteType(function.returnType); - Vector(function.body, - limits_.maximumStatements, - "statement count", - [this](const Statement &statement) { - WriteStatement(statement); - }); - } - }; - } // namespace - - auto - Encode(const Module &module, const Limits &limits) -> EncodeResult - { - Writer writer(limits); - writer.Document(module); - return std::move(writer).Finish(); - } - - auto - Decode(std::span bytes, const Limits &limits) - -> DecodeResult - { - return Reader(bytes, limits).Document(); - } -} // namespace Visual::XSharp::Core::Wire diff --git a/Compiler/Core/Wire/Decode.cpp b/Compiler/Core/Wire/Decode.cpp new file mode 100644 index 00000000..210947ff --- /dev/null +++ b/Compiler/Core/Wire/Decode.cpp @@ -0,0 +1,1065 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include + +#include "Compiler/Core/Wire/Magic.hpp" +#include "Visual/XSharp/Core/Scalar.hpp" +#include "Visual/XSharp/Core/Wire.hpp" + +namespace Visual::XSharp::Core::Wire +{ + namespace + { + class Reader final + { + public: + Reader(std::span bytes, const Limits &limits) + : bytes_(bytes) + , limits_(limits) + {} + + /// @brief Validate the entire envelope before publication. + /// Header checks precede allocation; the final offset check + /// rejects trailing bytes and ambiguous encodings. + [[nodiscard]] auto + Document() -> DecodeResult + { + if (bytes_.size() > limits_.maximumWireBytes) + return Failure(ErrorKind::LimitExceeded, + "wire byte length", + "input exceeds configured byte limit"); + for (const auto expected : kMagic) + if (Byte("magic") != expected) + return Failure( + ErrorKind::InvalidMagic, + "magic", + "input is not a Visual X# Core document"); + const auto version = Unsigned("version"); + if (!error_ && version != kCurrentVersion) + Fail(ErrorKind::UnsupportedVersion, + "version", + "unsupported Core wire version"); + const auto flags = Unsigned("flags"); + if (!error_ && flags != 0U) + Fail(ErrorKind::InvalidTag, + "flags", + "reserved flags must be zero"); + if (error_) + return Result(); + + Module module; + module.name = QualifiedName("module name"); + module.sourceFiles + = Vector(limits_.maximumFunctions, + "source file count", + [this] { + return Text("source file"); + }); + module.functions = Vector(limits_.maximumFunctions, + "function count", + [this] { + return ReadFunction(); + }); + if (!error_ && offset_ != bytes_.size()) + Fail(ErrorKind::TrailingInput, + "document", + "bytes remain after Core module"); + if (!error_) + module_ = std::move(module); + return Result(); + } + + private: + std::span bytes_; + const Limits &limits_; + std::size_t offset_{}; + std::size_t statementDepth_{}; + std::optional module_; + std::optional error_; + + [[nodiscard]] auto + Result() -> DecodeResult + { + return DecodeResult{ std::move(module_), std::move(error_) }; + } + [[nodiscard]] auto + Failure(ErrorKind kind, std::string context, std::string message) + -> DecodeResult + { + Fail(kind, std::move(context), std::move(message)); + return Result(); + } + [[gnu::noinline]] void + Fail(ErrorKind kind, std::string context, std::string message) + { + if (!error_) + error_ = Error{ kind, + offset_, + std::move(context), + std::move(message) }; + } + [[nodiscard]] auto + Byte(std::string_view context) -> std::uint8_t + { + if (offset_ >= bytes_.size()) + { + Fail(ErrorKind::TruncatedInput, + std::string(context), + "input ended before field was complete"); + return 0U; + } + return bytes_[offset_++]; + } + /** + * @brief Read an unsigned little-endian wire scalar + * of a fixed width. + * + * This byte order + * is explicit so host endianness cannot alter VXCR. + */ + template + [[nodiscard]] auto + Unsigned(std::string_view context) -> Integer + { + static_assert(std::is_unsigned_v); + Integer result{}; + for (std::size_t shift = 0; shift < sizeof(Integer) * 8U; + shift += 8U) + result |= static_cast(Byte(context)) << shift; + return result; + } + /// @brief Validate an untrusted length before narrowing it. + /// Callers use the checked count before reserving memory or + /// decoding recursively, preventing unchecked allocations. + [[nodiscard]] auto + Count(std::size_t maximum, std::string_view context) -> std::size_t + { + const auto value = Unsigned(context); + if (!error_ && value > maximum) + Fail(ErrorKind::LimitExceeded, + std::string(context), + "collection count exceeds configured limit"); + return error_ ? 0U : static_cast(value); + } + /// @brief Reserve only after Count enforces the caller's limit. + template + [[nodiscard]] auto + Vector(std::size_t maximum, std::string_view context, Decode decode) + -> std::vector + { + const auto size = Count(maximum, context); + std::vector values; + values.reserve(size); + for (std::size_t index = 0; index < size && !error_; ++index) + values.push_back(decode()); + return values; + } + /// @brief Read UTF-32 text and reject non-scalar values. + /// VXCR counts Unicode scalars, not UTF-8 bytes or UTF-16 + /// code units. Surrogates and values above U+10FFFF are invalid. + [[nodiscard, gnu::noinline]] auto + Text(std::string_view context) -> std::u32string + { + const auto size = Count(limits_.maximumTextScalars, context); + std::u32string value; + value.reserve(size); + for (std::size_t index = 0; index < size && !error_; ++index) + { + const auto scalar = Unsigned(context); + if (scalar > 0x10ffffU + || (scalar >= 0xd800U && scalar <= 0xdfffU)) + { + Fail(ErrorKind::InvalidScalar, + std::string(context), + "wire text contains a non-scalar Unicode value"); + break; + } + value.push_back(static_cast(scalar)); + } + return value; + } + [[nodiscard, gnu::noinline]] auto + QualifiedName(std::string_view context) + -> std::vector + { + return Vector(65535U, context, [this, context] { + return Text(context); + }); + } + [[nodiscard, gnu::noinline]] auto + Symbol(std::string_view context) -> SymbolName + { + const auto id = Unsigned(context); + if (!error_ && id == 0U) + Fail(ErrorKind::InvalidSymbol, + std::string(context), + "symbol id must be positive"); + return SymbolName{ id, Text(context) }; + } + [[nodiscard, gnu::noinline]] auto + ReadType(std::size_t depth = 0U) -> Type + { + if (depth > limits_.maximumTypeDepth) + { + Fail(ErrorKind::LimitExceeded, + "type", + "type nesting exceeds configured limit"); + return Type::unit(); + } + switch (Byte("type tag")) + { + case 0: + return Type::unit(); + case 1: + return Type::boolean(); + case 2: + return Type::int64(); + case 3: + return Type::string(); + case 4: + { + auto name = QualifiedName("named type"); + auto arguments = Vector< + ::visual_xsharp::core::TemplateArgument>( + limits_.maximumOperands, + "template argument count", + [this, depth] { + switch (Byte("template argument tag")) + { + case 0: + return ::visual_xsharp::core:: + TemplateArgument::type_argument( + ReadType(depth + 1U)); + case 1: + return ::visual_xsharp::core:: + TemplateArgument::value_argument( + ::visual_xsharp::core:: + TemplateValue:: + integer_value( + ReadInteger( + "template " + "integer"))); + case 2: + return ::visual_xsharp::core:: + TemplateArgument::value_argument( + ::visual_xsharp::core:: + TemplateValue:: + boolean_value(Boolean( + "template " + "boolean"))); + case 3: + return ::visual_xsharp::core:: + TemplateArgument::value_argument( + ::visual_xsharp::core:: + TemplateValue:: + character_value( + ReadInteger( + "template " + "character"))); + case 4: + return ::visual_xsharp::core:: + TemplateArgument::value_argument( + ::visual_xsharp::core:: + TemplateValue:: + parameter_value(Symbol( + "template value " + "parameter"))); + default: + Fail(ErrorKind::InvalidTag, + "template argument tag", + "unknown template argument tag"); + return ::visual_xsharp::core:: + TemplateArgument{}; + } + }); + return Type::named_template(std::move(name), + std::move(arguments)); + } + case 5: + { + auto parameters + = Vector(limits_.maximumParameters, + "function type parameter count", + [this, depth] { + return ReadType(depth + 1U); + }); + return Type::function(std::move(parameters), + ReadType(depth + 1U)); + } + case 6: + return Type::type_variable( + Symbol("type variable symbol")); + case 7: + return Type::character(); + case 8: + return Type::int8(); + case 9: + return Type::int16(); + case 10: + return Type::int32(); + case 11: + return Type::int128(); + case 12: + return Type::uint8(); + case 13: + return Type::uint16(); + case 14: + return Type::uint32(); + case 15: + return Type::uint64(); + case 16: + return Type::uint128(); + case 17: + return Type::float16(); + case 18: + return Type::float32(); + case 19: + return Type::float64(); + case 20: + return Type::float128(); + default: + Fail(ErrorKind::InvalidTag, + "type tag", + "unknown Core type tag"); + return Type::unit(); + } + } + [[nodiscard]] auto + Boolean(std::string_view context) -> bool + { + const auto value = Byte(context); + if (value > 1U) + Fail(ErrorKind::InvalidBoolean, + std::string(context), + "boolean byte must be zero or one"); + return value == 1U; + } + [[nodiscard, gnu::noinline]] auto + ReadInteger(std::string_view context) + -> ::visual_xsharp::core::IntegerLiteral + { + ::visual_xsharp::core::IntegerLiteral value; + value.negative = Boolean(std::string(context) + " sign"); + value.magnitude = Vector( + limits_.maximumNumericBytes, + std::string(context) + " magnitude", + [this, context] { + return Byte(std::string(context) + " magnitude"); + }); + if (!::visual_xsharp::core::integer_is_canonical(value)) + Fail(ErrorKind::InvalidInteger, + std::string(context), + "integer magnitude/sign is not canonical"); + return value; + } + [[nodiscard, gnu::noinline]] auto + ReadLiteral() -> Literal + { + switch (Byte("literal tag")) + { + case 0: + return std::monostate{}; + case 1: + return Boolean("boolean literal"); + case 2: + return static_cast( + Unsigned("integer literal")); + case 3: + return Text("string literal"); + case 4: + { + ::visual_xsharp::core::IntegerLiteral value; + value.negative = Boolean("integer sign"); + value.magnitude = Vector( + limits_.maximumNumericBytes, + "integer magnitude", + [this] { + return Byte("integer magnitude"); + }); + if (!::visual_xsharp::core::integer_is_canonical(value)) + Fail(ErrorKind::InvalidInteger, + "integer literal", + "integer magnitude/sign is not canonical"); + return value; + } + case 5: + { + const auto size = Count(limits_.maximumNumericBytes, + "floating literal length"); + std::string spelling; + spelling.reserve(size); + for (std::size_t index = 0; index < size && !error_; + ++index) + { + const auto value = Byte("floating literal"); + if (value > 0x7fU) + Fail(ErrorKind::InvalidInteger, + "floating literal", + "floating spelling must be ASCII"); + else + spelling.push_back(static_cast(value)); + } + if (!error_ + && !::visual_xsharp::core:: + floating_spelling_is_valid(spelling)) + Fail(ErrorKind::InvalidInteger, + "floating literal", + "floating spelling is not canonical"); + return ::visual_xsharp::core::FloatingLiteral{ + std::move(spelling) + }; + } + case 6: + return std::monostate{}; + default: + Fail(ErrorKind::InvalidTag, + "literal tag", + "unknown Core literal tag"); + return std::monostate{}; + } + } + /// The tag bytes of the expressions that the reader walks in a + /// loop. + static constexpr std::uint8_t kPrimitiveTag = 3U; + static constexpr std::uint8_t kLetTag = 5U; + static constexpr std::uint8_t kConditionalExpressionTag = 6U; + + /// An expression whose header and leading children were read + /// and that waits for the one child the loop reads next. + struct Pending final + { + std::uint8_t tag{}; + /// The level of this expression. + std::size_t level{}; + Primitive primitive{ Primitive::Add }; + Type valueType; + /// Primitive: the number of operands, the first of which is + /// awaited. + std::size_t operands{}; + /// Let: the bound symbol, its type and its value; the body + /// is awaited. + SymbolName symbol; + Type bindingType; + /// Let: the value. Conditional: the test. + Expression first; + /// Conditional: the true branch; the false one is awaited. + Expression second; + }; + + /** + * @brief Read one expression into its place. + * + * Three shapes nest as deep as an expression is long: a chain + * of operators nests in the first operand of each primitive, a + * chain of conditional expressions in each false branch, and a + * sequence of bindings in each let body. Reading them by + * recursion would use stack in proportion to the length of the + * chain, so those children are read in a loop: the enclosing + * expression is kept in a list while its awaited child is read, + * and the list is folded innermost first. Every other child is + * read by recursion, which the depth limit bounds. The bytes + * consumed and the resulting tree are those of the recursive + * formulation. The first operand of a primitive is at the level + * of the primitive, so a chain of operators does not count + * against the depth limit; every other child is one level + * deeper than its parent. + * + * An expression is large. The functions on the path that + * recurses once per level therefore hold none: an expression is + * read into its place, and the functions that build one from + * its parts are separate and return before the next level is + * entered. + */ + [[gnu::noinline]] void + ReadExpressionInto(Expression &expression, std::size_t depth = 0U) + { + std::vector spine; + auto level = depth; + for (;;) + { + if (level > limits_.maximumExpressionDepth) + { + Fail(ErrorKind::LimitExceeded, + "expression", + "expression nesting exceeds configured limit"); + break; + } + const auto tag = Byte("expression tag"); + if (tag != kPrimitiveTag && tag != kLetTag + && tag != kConditionalExpressionTag) + { + ReadUnchained(expression, tag, level); + break; + } + if (!ReadPending(tag, level, spine, expression)) + break; + if (tag != kPrimitiveTag) + ++level; + } + while (!spine.empty() && !error_) + { + Complete(spine.back(), expression); + spine.pop_back(); + } + if (error_) + Reset(expression); + } + /// Read one expression and return it. For callers that are not + /// on the path that recurses once per level. + [[nodiscard, gnu::noinline]] auto + ReadExpression() -> Expression + { + Expression expression; + ReadExpressionInto(expression); + return expression; + } + [[gnu::noinline]] static void + Reset(Expression &expression) + { + expression = Expression{}; + } + + /// Read the header and the leading children of a primitive, a + /// let or a conditional whose tag was consumed, into a new last + /// entry of the list, where it waits for its next child. + /// Returns false when no child is awaited: `expression` then + /// holds the result, or the input was rejected. + [[nodiscard, gnu::noinline]] auto + ReadPending(std::uint8_t tag, + std::size_t level, + std::vector &spine, + Expression &expression) -> bool + { + // The entry is filled in place. Nothing appends to this list + // while its children are read, so the reference stays valid. + spine.emplace_back(); + auto &pending = spine.back(); + pending.tag = tag; + pending.level = level; + const auto primitiveTag + = tag == kPrimitiveTag ? Byte("primitive tag") : 0U; + ReadTypeInto(pending.valueType); + if (tag == kPrimitiveTag) + { + if (primitiveTag + > static_cast(Primitive::RuntimeCall)) + Fail(ErrorKind::InvalidTag, + "primitive tag", + "unknown Core primitive tag"); + else + pending.primitive + = static_cast(primitiveTag); + pending.operands = Count(limits_.maximumOperands, + "primitive operand count"); + if (!error_ && pending.operands == 0U) + { + BuildPrimitive(pending, {}, expression); + spine.pop_back(); + return false; + } + } + else if (tag == kLetTag) + { + ReadLetHeader(pending); + ReadExpressionInto(pending.first, level + 1U); + } + else + { + // The three children have fixed positions, so the + // payload carries no count. + ReadExpressionInto(pending.first, level + 1U); + ReadExpressionInto(pending.second, level + 1U); + } + if (error_) + { + spine.pop_back(); + return false; + } + return true; + } + [[gnu::noinline]] void + ReadTypeInto(Type &type) + { + type = ReadType(); + } + [[gnu::noinline]] void + ReadLetHeader(Pending &pending) + { + pending.symbol = Symbol("let symbol"); + pending.bindingType = ReadType(); + } + + /// Build a pending expression from the child it waited for, + /// which `expression` holds, reading the operands of a + /// primitive that follow the first. The result replaces the + /// child in `expression`. + [[gnu::noinline]] void + Complete(Pending &pending, Expression &expression) + { + if (pending.tag == kLetTag) + { + BuildLet(pending, expression); + return; + } + if (pending.tag == kConditionalExpressionTag) + { + BuildConditional(pending, expression); + return; + } + std::vector operands; + operands.reserve(pending.operands); + operands.push_back(std::move(expression)); + for (std::size_t index = 1U; + index < pending.operands && !error_; + ++index) + { + operands.emplace_back(); + ReadExpressionInto(operands.back(), pending.level + 1U); + } + BuildPrimitive(pending, std::move(operands), expression); + } + [[gnu::noinline]] static void + BuildPrimitive(Pending &pending, + std::vector operands, + Expression &expression) + { + expression + = Expression::InvokePrimitive(pending.primitive, + std::move(operands), + std::move(pending.valueType)); + } + [[gnu::noinline]] static void + BuildLet(Pending &pending, Expression &expression) + { + expression = Expression::Let(std::move(pending.symbol), + std::move(pending.bindingType), + std::move(pending.first), + std::move(expression), + std::move(pending.valueType)); + } + [[gnu::noinline]] static void + BuildConditional(Pending &pending, Expression &expression) + { + expression + = Expression::Conditional(std::move(pending.first), + std::move(pending.second), + std::move(expression), + std::move(pending.valueType)); + } + + /// Read an expression that is not walked in a loop; its tag was + /// consumed. + [[gnu::noinline]] void + ReadUnchained(Expression &expression, + std::uint8_t tag, + std::size_t depth) + { + switch (tag) + { + case 0: + case 1: + ReadLeaf(expression, tag); + return; + case 2: + ReadApply(expression, depth); + return; + case 4: + ReadClosure(expression, depth); + return; + default: + // The type precedes the payload of every tag and is + // consumed before the tag is rejected, as the + // recursive formulation did. + ReadLeaf(expression, tag); + return; + } + } + [[gnu::noinline]] void + ReadLeaf(Expression &expression, std::uint8_t tag) + { + auto valueType = ReadType(); + if (tag == 0U) + expression = Expression::Variable(Symbol("variable symbol"), + std::move(valueType)); + else if (tag == 1U) + expression = Expression::Constant(ReadLiteral(), + std::move(valueType)); + else + Fail(ErrorKind::InvalidTag, + "expression tag", + "unknown Core expression tag"); + } + /// Read the operands of a call or the like into a list, each + /// into its place. + void + ReadOperands(std::vector &operands, + std::string_view context, + std::size_t depth) + { + const auto size = Count(limits_.maximumOperands, context); + operands.reserve(size); + for (std::size_t index = 0U; index < size && !error_; ++index) + { + operands.emplace_back(); + ReadExpressionInto(operands.back(), depth); + } + } + [[gnu::noinline]] void + ReadApply(Expression &expression, std::size_t depth) + { + auto valueType = ReadType(); + // The callee is read into the place of the result and moved + // out of it when the call is built. + ReadExpressionInto(expression, depth + 1U); + std::vector arguments; + ReadOperands(arguments, "call argument count", depth + 1U); + BuildApply(expression, arguments, valueType); + } + [[gnu::noinline]] static void + BuildApply(Expression &expression, + std::vector &arguments, + Type &valueType) + { + expression = Expression::Apply(std::move(expression), + std::move(arguments), + std::move(valueType)); + } + [[gnu::noinline]] void + ReadClosure(Expression &expression, std::size_t depth) + { + auto valueType = ReadType(); + auto captures + = Vector(limits_.maximumOperands, + "closure capture count", + [this, depth] { + return ReadCapture(depth + 1U); + }); + auto parameters = Vector>( + limits_.maximumParameters, + "closure parameter count", + [this] { + auto parameter = ReadParameter(); + return std::pair{ std::move(parameter.symbol), + std::move(parameter.type) }; + }); + auto returnType = ReadType(); + auto body = ReadBody("closure statement count"); + expression = Expression::Closure(std::move(captures), + std::move(parameters), + std::move(returnType), + std::move(body), + std::move(valueType)); + } + [[nodiscard, gnu::noinline]] auto + ReadCapture(std::size_t depth) -> Capture + { + const auto tag = Byte("closure capture mode"); + if (tag > 2U) + Fail(ErrorKind::InvalidTag, + "closure capture mode", + "unknown Core closure capture mode"); + Capture capture; + // A rejected tag never becomes an enumerator. + capture.mode = tag > 2U ? CaptureMode::Strong + : static_cast(tag); + capture.symbol = Symbol("closure capture symbol"); + capture.type = ReadType(); + capture.value = std::make_shared(); + ReadExpressionInto(*capture.value, depth); + return capture; + } + /// Enter one level of statement nesting, or fail when the level + /// is beyond the limit. Every successful call is paired with + /// LeaveBody. + [[nodiscard]] auto + EnterBody(std::string_view context) -> bool + { + if (statementDepth_ >= limits_.maximumStatementDepth) + { + Fail(ErrorKind::LimitExceeded, + std::string(context), + "statement nesting exceeds configured limit"); + return false; + } + ++statementDepth_; + return true; + } + void + LeaveBody() + { + --statementDepth_; + } + /** + * @brief Read the statements of a function, branch, loop or + * closure body, one nesting level below the current one. + * + * The reader recurses once per level, so the level is bounded + * like type and expression depth are: input that nests deeper + * than the limit is rejected before it can exhaust the stack. + * The level is reader state rather than a parameter because a + * closure body is reached through an expression. + * + * A statement holds two expressions by value and is large. The + * functions on the path that recurses once per level therefore + * hold none: a statement is read into its place in the list, + * and the functions that build one from its parts are separate + * and return before the next level is entered. + */ + [[nodiscard, gnu::noinline]] auto + ReadBody(std::string_view context) -> std::vector + { + // An empty body holds nothing to recurse into, so it is + // accepted at any level, as the writer writes it. + const auto size = Count(limits_.maximumStatements, context); + std::vector statements; + if (size == 0U || !EnterBody(context)) + return statements; + ReadStatements(size, statements); + LeaveBody(); + return statements; + } + void + ReadStatements(std::size_t size, std::vector &statements) + { + statements.reserve(size); + for (std::size_t index = 0; index < size && !error_; ++index) + { + // Nothing else appends to this list while the statement + // is read, so the reference stays valid. + statements.emplace_back(); + ReadStatement(statements.back()); + } + } + + /// The tag byte of a conditional statement. + static constexpr std::uint8_t kConditionalTag = 3U; + + /// One link of an `else if` chain whose false branch is the + /// next link. + struct Link final + { + Expression condition; + std::vector whenTrue; + }; + + /** + * @brief Read a conditional statement whose tag was consumed, + * and every `else if` that continues it. + * + * An `else if` is encoded as a false branch that holds exactly + * one conditional statement. Reading such a chain by recursion + * uses stack in proportion to its length, so a long chain in + * valid source would overflow it. The links are read in a loop + * into a flat list and nested afterwards, innermost first. The + * bytes consumed and the resulting tree are those of the + * recursive formulation. + */ + [[gnu::noinline]] void + ReadConditionalChain(Statement &statement) + { + std::vector links; + std::vector finalBranch; + for (;;) + { + links.emplace_back(); + ReadExpressionInto(links.back().condition); + links.back().whenTrue + = ReadBody("true branch statement count"); + const auto size = Count(limits_.maximumStatements, + "false branch statement count"); + // A false branch of one conditional is the next link: + // take its tag and read it in the next iteration. + if (!error_ && size == 1U && offset_ < bytes_.size() + && bytes_[offset_] == kConditionalTag) + { + ++offset_; + continue; + } + // The links of a chain share one level; the last false + // branch is a body one level below it. + if (size != 0U && EnterBody("false branch statement count")) + { + ReadStatements(size, finalBranch); + LeaveBody(); + } + break; + } + NestLink(statement, links.back(), finalBranch); + links.pop_back(); + while (!links.empty()) + { + std::vector nested; + nested.push_back(std::move(statement)); + NestLink(statement, links.back(), nested); + links.pop_back(); + } + } + [[gnu::noinline]] static void + NestLink(Statement &statement, + Link &link, + std::vector &whenFalse) + { + statement = Statement::If(std::move(link.condition), + std::move(link.whenTrue), + std::move(whenFalse)); + } + + // Each kind of statement is read by a function of its own, so + // that a level of nesting costs the frame of the kind that + // nests and not the frames of every kind together. + void + ReadStatement(Statement &statement) + { + switch (Byte("statement tag")) + { + case 0: + ReadBinding(statement); + return; + case 1: + ReadAssignment(statement); + return; + case 2: + ReadValueStatement(statement, Statement::Kind::Return); + return; + case 3: + ReadConditionalChain(statement); + return; + case 4: + ReadValueStatement(statement, + Statement::Kind::Evaluate); + return; + case 5: + ReadWhile(statement); + return; + case 6: + ReadDoWhile(statement); + return; + case 7: + ReadFor(statement); + return; + case 8: + SetJump(statement, Statement::Kind::Break); + return; + case 9: + SetJump(statement, Statement::Kind::Continue); + return; + default: + Fail(ErrorKind::InvalidTag, + "statement tag", + "unknown Core statement tag"); + return; + } + } + [[gnu::noinline]] void + ReadBinding(Statement &statement) + { + auto symbol = Symbol("binding symbol"); + auto type = ReadType(); + const auto mutableBinding = Boolean("binding mutability"); + auto value = ReadExpression(); + statement = Statement::Bind(Binding{ std::move(symbol), + std::move(type), + mutableBinding, + std::move(value) }); + } + [[gnu::noinline]] void + ReadAssignment(Statement &statement) + { + auto symbol = Symbol("assignment symbol"); + statement + = Statement::Assign(std::move(symbol), ReadExpression()); + } + [[gnu::noinline]] void + ReadValueStatement(Statement &statement, Statement::Kind kind) + { + statement = kind == Statement::Kind::Return + ? Statement::Return(ReadExpression()) + : Statement::Evaluate(ReadExpression()); + } + [[gnu::noinline]] static void + SetJump(Statement &statement, Statement::Kind kind) + { + statement = kind == Statement::Kind::Break + ? Statement::Break() + : Statement::Continue(); + } + // The loops read their parts into locals that are lists, which + // are small, and into the statement itself; the builders below + // put the parts in their final places. + [[gnu::noinline]] void + ReadWhile(Statement &statement) + { + ReadExpressionInto(statement.expression); + auto body = ReadBody("while body statement count"); + BuildWhile(statement, body); + } + [[gnu::noinline]] void + ReadDoWhile(Statement &statement) + { + auto body = ReadBody("do/while body statement count"); + ReadExpressionInto(statement.expression); + BuildDoWhile(statement, body); + } + [[gnu::noinline]] void + ReadFor(Statement &statement) + { + ReadExpressionInto(statement.expression); + auto body = ReadBody("for body statement count"); + auto update = ReadBody("for update statement count"); + BuildFor(statement, body, update); + } + [[gnu::noinline]] static void + BuildWhile(Statement &statement, std::vector &body) + { + statement = Statement::While(std::move(statement.expression), + std::move(body)); + } + [[gnu::noinline]] static void + BuildDoWhile(Statement &statement, std::vector &body) + { + statement = Statement::DoWhile(std::move(body), + std::move(statement.expression)); + } + [[gnu::noinline]] static void + BuildFor(Statement &statement, + std::vector &body, + std::vector &update) + { + statement = Statement::For(std::move(statement.expression), + std::move(body), + std::move(update)); + } + [[nodiscard, gnu::noinline]] auto + ReadParameter() -> Parameter + { + return Parameter{ Symbol("parameter symbol"), ReadType() }; + } + [[nodiscard]] auto + ReadFunction() -> Function + { + Function function; + function.symbol = Symbol("function symbol"); + function.sourceFile = Text("function source file"); + function.parameters + = Vector(limits_.maximumParameters, + "parameter count", + [this] { + return ReadParameter(); + }); + function.returnType = ReadType(); + function.body = ReadBody("statement count"); + return function; + } + }; + } // namespace + + auto + Decode(std::span bytes, const Limits &limits) + -> DecodeResult + { + return Reader(bytes, limits).Document(); + } +} // namespace Visual::XSharp::Core::Wire diff --git a/Compiler/Core/Wire/Encode.cpp b/Compiler/Core/Wire/Encode.cpp new file mode 100644 index 00000000..91b7e38c --- /dev/null +++ b/Compiler/Core/Wire/Encode.cpp @@ -0,0 +1,689 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include + +#include "Compiler/Core/Wire/Magic.hpp" +#include "Visual/XSharp/Core/Scalar.hpp" +#include "Visual/XSharp/Core/Wire.hpp" + +namespace Visual::XSharp::Core::Wire +{ + namespace + { + class Writer final + { + public: + explicit Writer(const Limits &limits) + : limits_(limits) + {} + void + Document(const Module &module) + { + for (const auto value : kMagic) + Byte(value); + Unsigned(kCurrentVersion); + Unsigned(0U); + QualifiedName(module.name, "module name"); + Vector(module.sourceFiles, + limits_.maximumFunctions, + "source file count", + [this](const std::u32string &sourceFile) { + Text(sourceFile, "source file"); + }); + Vector(module.functions, + limits_.maximumFunctions, + "function count", + [this](const Function &function) { + WriteFunction(function); + }); + } + [[nodiscard]] auto + Finish() && -> EncodeResult + { + if (!error_ && bytes_.size() > limits_.maximumWireBytes) + Fail(ErrorKind::LimitExceeded, + "wire byte length", + "encoded document exceeds configured limit"); + return EncodeResult{ std::move(bytes_), std::move(error_) }; + } + + private: + const Limits &limits_; + std::vector bytes_; + std::size_t statementDepth_{}; + std::optional error_; + + void + Fail(ErrorKind kind, std::string context, std::string message) + { + if (!error_) + error_ = Error{ kind, + bytes_.size(), + std::move(context), + std::move(message) }; + } + void + Byte(std::uint8_t value) + { + if (!error_) + bytes_.push_back(value); + } + template + void + Unsigned(Integer value) + { + static_assert(std::is_unsigned_v); + for (std::size_t shift = 0; shift < sizeof(Integer) * 8U; + shift += 8U) + Byte(static_cast( + (value >> shift) & static_cast(0xffU))); + } + void + Count(std::size_t value, + std::size_t maximum, + std::string_view context) + { + if (value > maximum + || value > std::numeric_limits::max()) + Fail(ErrorKind::LimitExceeded, + std::string(context), + "collection count exceeds wire limit"); + else + Unsigned(static_cast(value)); + } + template + void + Vector(const std::vector &values, + std::size_t maximum, + std::string_view context, + Encode encode) + { + Count(values.size(), maximum, context); + for (const auto &value : values) + { + if (error_) + return; + encode(value); + } + } + void + Text(const std::u32string &value, std::string_view context) + { + Count(value.size(), limits_.maximumTextScalars, context); + for (const auto scalar : value) + { + const auto numeric = static_cast(scalar); + if (numeric > 0x10ffffU + || (numeric >= 0xd800U && numeric <= 0xdfffU)) + { + Fail(ErrorKind::InvalidScalar, + std::string(context), + "text contains a non-scalar Unicode value"); + return; + } + Unsigned(numeric); + } + } + void + QualifiedName(const std::vector &parts, + std::string_view context) + { + Vector(parts, + 65535U, + context, + [this, context](const auto &part) { + Text(part, context); + }); + } + void + Symbol(const SymbolName &symbol, std::string_view context) + { + if (symbol.id == 0U) + { + Fail(ErrorKind::InvalidSymbol, + std::string(context), + "symbol id must be positive"); + return; + } + Unsigned(symbol.id); + Text(symbol.spelling, context); + } + void + WriteType(const Type &type, std::size_t depth = 0U) + { + if (depth > limits_.maximumTypeDepth) + { + Fail(ErrorKind::LimitExceeded, + "type", + "type nesting exceeds configured limit"); + return; + } + switch (type.kind) + { + case Type::Kind::Unit: + Byte(0); + return; + case Type::Kind::Bool: + Byte(1); + return; + case Type::Kind::Int64: + Byte(2); + return; + case Type::Kind::String: + Byte(3); + return; + case Type::Kind::Named: + Byte(4); + QualifiedName(type.name, "named type"); + Vector(type.templateArguments, + limits_.maximumOperands, + "template argument count", + [this, depth](const auto &argument) { + if (argument.kind + == ::visual_xsharp::core:: + TemplateArgument::Kind::Type) + { + Byte(0); + if (!argument.type) + { + Fail(ErrorKind::UnsupportedType, + "template argument", + "type argument has no payload"); + return; + } + WriteType(*argument.type, depth + 1U); + return; + } + switch (argument.value.kind) + { + case ::visual_xsharp::core:: + TemplateValue::Kind::Integer: + Byte(1); + WriteInteger(argument.value.integer, + "template integer"); + return; + case ::visual_xsharp::core:: + TemplateValue::Kind::Boolean: + Byte(2); + Byte(argument.value.boolean ? 1U + : 0U); + return; + case ::visual_xsharp::core:: + TemplateValue::Kind::Character: + Byte(3); + WriteInteger(argument.value.integer, + "template character"); + return; + case ::visual_xsharp::core:: + TemplateValue::Kind::Parameter: + Byte(4); + Symbol(argument.value.parameter, + "template value parameter"); + return; + } + }); + return; + case Type::Kind::Function: + Byte(5); + if (type.components.empty()) + { + Fail(ErrorKind::UnsupportedType, + "function type", + "function type has no result component"); + return; + } + Count(type.components.size() - 1U, + limits_.maximumParameters, + "function type parameter count"); + for (std::size_t index = 0; + index + 1U < type.components.size(); + ++index) + WriteType(type.components[index], depth + 1U); + WriteType(type.components.back(), depth + 1U); + return; + case Type::Kind::TypeVariable: + Byte(6); + Symbol(type.variable, "type variable symbol"); + return; + case Type::Kind::Character: + Byte(7); + return; + case Type::Kind::Int8: + Byte(8); + return; + case Type::Kind::Int16: + Byte(9); + return; + case Type::Kind::Int32: + Byte(10); + return; + case Type::Kind::Int128: + Byte(11); + return; + case Type::Kind::UInt8: + Byte(12); + return; + case Type::Kind::UInt16: + Byte(13); + return; + case Type::Kind::UInt32: + Byte(14); + return; + case Type::Kind::UInt64: + Byte(15); + return; + case Type::Kind::UInt128: + Byte(16); + return; + case Type::Kind::Float16: + Byte(17); + return; + case Type::Kind::Float32: + Byte(18); + return; + case Type::Kind::Float64: + Byte(19); + return; + case Type::Kind::Float128: + Byte(20); + return; + } + } + void + WriteInteger(const ::visual_xsharp::core::IntegerLiteral &integer, + std::string_view context) + { + if (!::visual_xsharp::core::integer_is_canonical(integer)) + { + Fail(ErrorKind::InvalidInteger, + std::string(context), + "integer magnitude/sign is not canonical"); + return; + } + Byte(integer.negative ? 1U : 0U); + Vector(integer.magnitude, + limits_.maximumNumericBytes, + std::string(context) + " magnitude", + [this](const auto octet) { + Byte(octet); + }); + } + void + WriteLiteral(const Literal &literal, const Type &valueType) + { + if (std::holds_alternative(literal)) + Byte(valueType.kind == Type::Kind::Unit ? 0U : 6U); + else if (const auto *boolean = std::get_if(&literal)) + { + Byte(1); + Byte(*boolean ? 1U : 0U); + } + else if (const auto *integer + = std::get_if(&literal)) + { + Byte(2); + Unsigned(static_cast(*integer)); + } + else if (const auto *string + = std::get_if(&literal)) + { + Byte(3); + Text(*string, "string literal"); + } + else if (const auto *wideInteger + = std::get_if<::visual_xsharp::core::IntegerLiteral>( + &literal)) + { + if (!::visual_xsharp::core::integer_is_canonical( + *wideInteger)) + { + Fail(ErrorKind::InvalidInteger, + "integer literal", + "integer magnitude/sign is not canonical"); + return; + } + Byte(4); + Byte(wideInteger->negative ? 1U : 0U); + Vector(wideInteger->magnitude, + limits_.maximumNumericBytes, + "integer magnitude", + [this](const std::uint8_t octet) { + Byte(octet); + }); + } + else if (const auto *floating + = std::get_if<::visual_xsharp::core::FloatingLiteral>( + &literal)) + { + if (!::visual_xsharp::core::floating_spelling_is_valid( + floating->spelling)) + { + Fail(ErrorKind::InvalidInteger, + "floating literal", + "floating spelling is not canonical"); + return; + } + Byte(5); + Count(floating->spelling.size(), + limits_.maximumNumericBytes, + "floating literal length"); + for (const auto character : floating->spelling) + Byte(static_cast(character)); + } + else if (const auto *narrowInteger + = std::get_if(&literal)) + { + Byte(4); + const auto normalized + = ::visual_xsharp::core::integer_from_signed( + *narrowInteger); + Byte(normalized.negative ? 1U : 0U); + Vector(normalized.magnitude, + limits_.maximumNumericBytes, + "integer magnitude", + [this](const std::uint8_t octet) { + Byte(octet); + }); + } + else + Fail(ErrorKind::UnsupportedType, + "literal", + "literal cannot cross the Core v5 boundary"); + } + /** + * @brief Write one expression. + * + * A chain of operators nests in the first operand of each + * primitive, as deep as the chain is long. The headers of those + * primitives are written in a loop, then the innermost first + * operand, then the remaining operands of each primitive from + * the innermost outwards, which is the order of the nested + * formulation. The first operand of a primitive is at the depth + * of the primitive, so a chain does not count against the + * expression depth limit; every other child is one level deeper. + */ + void + WriteExpression(const Expression &root, std::size_t depth = 0U) + { + std::vector chain; + const Expression *current = &root; + while (!error_ && current->kind == Expression::Kind::Primitive + && !current->operands.empty()) + { + if (depth > limits_.maximumExpressionDepth) + break; + Byte(static_cast(current->kind)); + Byte(static_cast(current->primitive)); + WriteType(current->type); + Count(current->operands.size(), + limits_.maximumOperands, + "primitive operand count"); + chain.push_back(current); + current = ¤t->operands.front(); + } + WriteUnchained(*current, depth); + while (!chain.empty() && !error_) + { + const auto &operands = chain.back()->operands; + for (std::size_t index = 1U; + index < operands.size() && !error_; + ++index) + WriteExpression(operands[index], depth + 1U); + chain.pop_back(); + } + } + /// An expression that is not a primitive with operands. + void + WriteUnchained(const Expression &expression, std::size_t depth) + { + if (depth > limits_.maximumExpressionDepth) + { + Fail(ErrorKind::LimitExceeded, + "expression", + "expression nesting exceeds configured limit"); + return; + } + Byte(static_cast(expression.kind)); + if (expression.kind == Expression::Kind::Primitive) + Byte(static_cast(expression.primitive)); + WriteType(expression.type); + switch (expression.kind) + { + case Expression::Kind::Variable: + Symbol(expression.symbol, "variable symbol"); + return; + case Expression::Kind::Literal: + WriteLiteral(expression.literal, expression.type); + return; + case Expression::Kind::Apply: + if (!expression.callee) + { + Fail(ErrorKind::InvalidCount, + "callee", + "Core call must contain a callee"); + return; + } + WriteExpression(*expression.callee, depth + 1U); + Vector(expression.operands, + limits_.maximumOperands, + "call argument count", + [this, depth](const Expression &value) { + WriteExpression(value, depth + 1U); + }); + return; + case Expression::Kind::Primitive: + Vector(expression.operands, + limits_.maximumOperands, + "primitive operand count", + [this, depth](const Expression &value) { + WriteExpression(value, depth + 1U); + }); + return; + case Expression::Kind::Closure: + Vector( + expression.captures, + limits_.maximumOperands, + "closure capture count", + [this, depth](const Capture &capture) { + if (capture.mode != CaptureMode::Strong + && capture.mode != CaptureMode::Weak + && capture.mode != CaptureMode::Unowned) + { + Fail(ErrorKind::InvalidTag, + "closure capture mode", + "unknown Core closure capture mode"); + return; + } + Byte(static_cast(capture.mode)); + Symbol(capture.symbol, + "closure capture symbol"); + WriteType(capture.type); + if (!capture.value) + { + Fail(ErrorKind::InvalidCount, + "closure capture value", + "Core closure capture must contain a " + "value"); + return; + } + WriteExpression(*capture.value, depth + 1U); + }); + Vector(expression.closureParameters, + limits_.maximumParameters, + "closure parameter count", + [this](const auto ¶meter) { + Symbol(parameter.first, "parameter symbol"); + WriteType(parameter.second); + }); + WriteType(expression.closureReturnType); + if (!expression.closureBody) + { + Fail(ErrorKind::InvalidCount, + "closure body", + "Core closure must contain a body"); + return; + } + WriteBody(*expression.closureBody, + "closure statement count"); + return; + case Expression::Kind::Let: + Symbol(expression.letSymbol, "let symbol"); + WriteType(expression.letType); + if (!expression.letValue || !expression.letBody) + { + Fail(ErrorKind::InvalidCount, + "let expression", + "Core let must contain a value and body"); + return; + } + WriteExpression(*expression.letValue, depth + 1U); + WriteExpression(*expression.letBody, depth + 1U); + return; + case Expression::Kind::Conditional: + if (expression.operands.size() != 3U) + { + Fail(ErrorKind::InvalidCount, + "conditional expression", + "Core conditional must contain a test and " + "two arms"); + return; + } + for (const auto &operand : expression.operands) + WriteExpression(operand, depth + 1U); + return; + } + } + /** + * @brief Write the statements of a body, one nesting level below + * the current one, with the limit the reader enforces. A module + * the reader would reject is not written. + */ + void + WriteBody(const std::vector &statements, + std::string_view context) + { + if (statements.empty()) + { + Count(0U, limits_.maximumStatements, context); + return; + } + if (statementDepth_ >= limits_.maximumStatementDepth) + { + Fail(ErrorKind::LimitExceeded, + std::string(context), + "statement nesting exceeds wire limit"); + return; + } + ++statementDepth_; + Vector(statements, + limits_.maximumStatements, + context, + [this](const Statement &value) { + WriteStatement(value); + }); + --statementDepth_; + } + void + WriteStatement(const Statement &statement) + { + Byte(static_cast(statement.kind)); + switch (statement.kind) + { + case Statement::Kind::Bind: + Symbol(statement.binding.symbol, "binding symbol"); + WriteType(statement.binding.type); + Byte(statement.binding.mutableBinding ? 1U : 0U); + WriteExpression(statement.binding.value); + return; + case Statement::Kind::Assign: + Symbol(statement.destination, "assignment symbol"); + WriteExpression(statement.expression); + return; + case Statement::Kind::Return: + WriteExpression(statement.expression); + return; + case Statement::Kind::If: + { + // An `else if` chain is written in a loop for the + // reason ReadConditionalChain reads it in one: a + // false branch of exactly one conditional is the + // next link, and its bytes are its count, its tag + // and then the link itself. + const Statement *link = &statement; + while (!error_) + { + WriteExpression(link->expression); + WriteBody(link->trueBranch, + "true branch statement count"); + const auto continues + = link->falseBranch.size() == 1U + && link->falseBranch.front().kind + == Statement::Kind::If; + if (!continues) + { + WriteBody(link->falseBranch, + "false branch statement count"); + break; + } + Count(1U, + limits_.maximumStatements, + "false branch statement count"); + link = &link->falseBranch.front(); + Byte(static_cast(link->kind)); + } + return; + } + case Statement::Kind::Evaluate: + WriteExpression(statement.expression); + return; + case Statement::Kind::While: + WriteExpression(statement.expression); + WriteBody(statement.loopBody, + "while body statement count"); + return; + case Statement::Kind::DoWhile: + WriteBody(statement.loopBody, + "do/while body statement count"); + WriteExpression(statement.expression); + return; + case Statement::Kind::For: + WriteExpression(statement.expression); + WriteBody(statement.loopBody, + "for body statement count"); + WriteBody(statement.loopUpdate, + "for update statement count"); + return; + case Statement::Kind::Break: + case Statement::Kind::Continue: + return; + } + } + void + WriteFunction(const Function &function) + { + Symbol(function.symbol, "function symbol"); + Text(function.sourceFile, "function source file"); + Vector(function.parameters, + limits_.maximumParameters, + "parameter count", + [this](const Parameter ¶meter) { + Symbol(parameter.symbol, "parameter symbol"); + WriteType(parameter.type); + }); + WriteType(function.returnType); + WriteBody(function.body, "statement count"); + } + }; + } // namespace + + auto + Encode(const Module &module, const Limits &limits) -> EncodeResult + { + Writer writer(limits); + writer.Document(module); + return std::move(writer).Finish(); + } +} // namespace Visual::XSharp::Core::Wire diff --git a/Compiler/Core/Wire/Magic.hpp b/Compiler/Core/Wire/Magic.hpp new file mode 100644 index 00000000..f5f3f242 --- /dev/null +++ b/Compiler/Core/Wire/Magic.hpp @@ -0,0 +1,13 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include + +namespace Visual::XSharp::Core::Wire +{ + /// The four bytes every Core wire document starts with. The reader and + /// the writer are separate translation units and share this one + /// definition. + inline constexpr std::uint8_t kMagic[] = { 'V', 'X', 'C', 'R' }; +} // namespace Visual::XSharp::Core::Wire diff --git a/Compiler/Driver/Pipeline.cpp b/Compiler/Driver/Pipeline.cpp index 44ab1c16..eb812ec0 100644 --- a/Compiler/Driver/Pipeline.cpp +++ b/Compiler/Driver/Pipeline.cpp @@ -2,6 +2,7 @@ // SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 #include "Visual/XSharp/Pipeline.hpp" +#include "Visual/XSharp/Xpp/OwnershipPlacement.hpp" namespace { @@ -85,10 +86,11 @@ namespace visual_xsharp auto lowered_xpp = xpp::lower(*result.core_prep); // Optimization toggles select identity-vs-optimized forms; they never // skip a stage. Xmm therefore receives the same typed Xpp contract in - // debug and release modes. - result.xpp = options.optimize_xpp - ? xpp::optimize(std::move(lowered_xpp)) - : std::move(lowered_xpp); + // debug and release modes. Ownership is placed last, on the form + // that is kept: it is part of that contract in either mode. + result.xpp = ::Visual::XSharp::Xpp::PlaceOwnership( + options.optimize_xpp ? xpp::optimize(std::move(lowered_xpp)) + : std::move(lowered_xpp)); ContinueFromXpp(result, options); return result; } @@ -127,9 +129,12 @@ namespace Visual::XSharp::Pipeline return result; auto loweredXpp = ::visual_xsharp::xpp::lower(*result.core_prep); - result.xpp = options.optimize_xpp - ? ::visual_xsharp::xpp::optimize(std::move(loweredXpp)) - : std::move(loweredXpp); + // Ownership is placed once, here: an Xpp artifact read from disk + // already carries its retains and releases. + result.xpp = Xpp::PlaceOwnership( + options.optimize_xpp + ? ::visual_xsharp::xpp::optimize(std::move(loweredXpp)) + : std::move(loweredXpp)); ContinueFromXpp(result, options); return result; } diff --git a/Compiler/Driver/Tests/ArtifactWireTests.cpp b/Compiler/Driver/Tests/ArtifactWireTests.cpp index 04067445..2d433624 100644 --- a/Compiler/Driver/Tests/ArtifactWireTests.cpp +++ b/Compiler/Driver/Tests/ArtifactWireTests.cpp @@ -421,7 +421,7 @@ TEST_CASE("Xmm wire preserves virtual-register ABI and typed immediates") } TEST_CASE( - "Xpp and Xmm v5 wire preserve explicit ownership and type-test operations") + "Xpp and Xmm v6 wire preserve explicit ownership and type-test operations") { const auto xpp = OwnershipXppModule(); const auto encodedXpp = XppWire::Encode(xpp); @@ -436,16 +436,16 @@ TEST_CASE( const auto decodedXmm = XmmWire::Decode(encodedXmm.bytes); REQUIRE(decodedXmm); REQUIRE(*decodedXmm.module == xmm); - CHECK(XppWire::kCurrentVersion == 5U); - CHECK(XmmWire::kCurrentVersion == 5U); + CHECK(XppWire::kCurrentVersion == 7U); + CHECK(XmmWire::kCurrentVersion == 7U); } -TEST_CASE("Xpp v5 wire preserves ordered type and value template arguments") +TEST_CASE("Xpp v6 wire preserves ordered type and value template arguments") { const auto original = XppModule(TemplateModule()); const auto encoded = XppWire::Encode(original); REQUIRE(encoded); - CHECK(encoded.bytes[4] == 5U); + CHECK(encoded.bytes[4] == 7U); const auto decoded = XppWire::Decode(encoded.bytes); REQUIRE(decoded); CHECK(*decoded.module == original); @@ -464,12 +464,12 @@ TEST_CASE("Xpp v5 wire preserves ordered type and value template arguments") == Core::TemplateValue::Kind::Character); } -TEST_CASE("Xmm v5 wire preserves ordered type and value template arguments") +TEST_CASE("Xmm v6 wire preserves ordered type and value template arguments") { const auto original = XmmModule(XppModule(TemplateModule())); const auto encoded = XmmWire::Encode(original); REQUIRE(encoded); - CHECK(encoded.bytes[4] == 5U); + CHECK(encoded.bytes[4] == 7U); const auto decoded = XmmWire::Decode(encoded.bytes); REQUIRE(decoded); CHECK(*decoded.module == original); diff --git a/Compiler/Driver/Tests/BUILD.bazel b/Compiler/Driver/Tests/BUILD.bazel index ee3c2154..c6cdc81d 100644 --- a/Compiler/Driver/Tests/BUILD.bazel +++ b/Compiler/Driver/Tests/BUILD.bazel @@ -4,8 +4,13 @@ package(default_visibility = ["//visibility:private"]) cc_binary( name = "closure_pipeline_tests", - srcs = ["ClosurePipelineTests.cpp"], + srcs = [ + "ClosurePipelineTests.cpp", + "MemoizePipelineTests.cpp", + "RuntimeCallPipelineTests.cpp", + ], deps = [ + "//Compiler/Backend/LLVM:llvm_backend", "//Compiler/Codegen/Xmm:xmm", "//Compiler/Codegen/Xpp:xpp", "//Compiler/Core:core", diff --git a/Compiler/Driver/Tests/ClosurePipelineTests.cpp b/Compiler/Driver/Tests/ClosurePipelineTests.cpp index 33dc9a32..21ee9f9a 100644 --- a/Compiler/Driver/Tests/ClosurePipelineTests.cpp +++ b/Compiler/Driver/Tests/ClosurePipelineTests.cpp @@ -220,13 +220,13 @@ namespace } } // namespace -TEST_CASE("Core v8 round-trips closure targets captures and modes") +TEST_CASE("CorePrep v8 round-trips closure targets captures and modes") { const auto source = ClosureModule(); const auto encoded = Core::wire::encode(source); REQUIRE_FALSE(encoded.error); REQUIRE(encoded.bytes.size() > 8U); - CHECK(encoded.bytes[4] == 6U); + CHECK(encoded.bytes[4] == 8U); CHECK(encoded.bytes[5] == 0U); const auto decoded = Core::wire::decode(encoded.bytes); diff --git a/Compiler/Driver/Tests/MemoizePipelineTests.cpp b/Compiler/Driver/Tests/MemoizePipelineTests.cpp new file mode 100644 index 00000000..14be03e4 --- /dev/null +++ b/Compiler/Driver/Tests/MemoizePipelineTests.cpp @@ -0,0 +1,521 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/CorePrep.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/CorePrep/Wire.hpp" +#include "Visual/XSharp/Core/Scalar.hpp" +#include "Visual/XSharp/Xmm/IR.hpp" +#include "Visual/XSharp/Xmm/Verifier.hpp" +#include "Visual/XSharp/Xmm/Wire.hpp" +#include "Visual/XSharp/Xpp/IR.hpp" +#include "Visual/XSharp/Xpp/Verifier.hpp" +#include "Visual/XSharp/Xpp/Wire.hpp" + +// A callable that remembers its result, followed through every native stage. +// +// The operation takes a callable without parameters whose result is a Bool +// or a number, and yields a callable of the same type that calls the first +// one at most once. Each stage carries the operation under its own name, +// checks that rule for itself, and writes it to its artifact. These cases +// take one module through the stages and then break the rule at each of +// them, so that no stage relies on the one before it having checked. + +namespace +{ + namespace Core = visual_xsharp::core; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Xmm = visual_xsharp::xmm; + namespace Xpp = visual_xsharp::xpp; + + [[nodiscard]] auto + Name(Core::SymbolId id, std::u32string spelling) -> Core::SymbolName + { + return { id, std::move(spelling) }; + } + + [[nodiscard]] auto + Variable(Core::SymbolId id, Core::Type type) -> Core::Atom + { + return Core::Atom::variable(Name(id, {}), std::move(type)); + } + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Atom + { + return Core::Atom::constant(Core::integer_from_signed(value), + Core::Type::int64()); + } + + [[nodiscard]] auto + Computation(Core::Type result = Core::Type::int64()) -> Core::Type + { + return Core::Type::function({}, std::move(result)); + } + + // Position of the remembering instruction in the entry block of `Main`. + constexpr std::size_t kMemoize = 2U; + + /// ``` + /// Main(): + /// count = 0 + /// next = closure $closure10 capturing count + /// once = memoize next + /// first = once() + /// second = once() + /// return + /// $closure10(count): return count + /// ``` + [[nodiscard]] auto + MemoizeModule() -> Core::CorePrepModule + { + Core::Instruction seed; + seed.kind = Core::Instruction::Kind::Bind; + seed.destination = Name(2U, U"count"); + seed.type = Core::Type::int64(); + seed.mutable_binding = true; + seed.operation = Core::Operation::Copy; + seed.operands = { Integer(0) }; + + Core::Instruction closure; + closure.kind = Core::Instruction::Kind::Bind; + closure.destination = Name(3U, U"next"); + closure.type = Computation(); + closure.operation = Core::Operation::MakeClosure; + closure.closure_function = Name(10U, U"$closure10"); + closure.captures.push_back(Core::Capture{ + Core::CaptureMode::Strong, + Name(4U, U"count"), + Core::Type::int64(), + Integer(0), + }); + + Core::Instruction memoize; + memoize.kind = Core::Instruction::Kind::Bind; + memoize.destination = Name(5U, U"once"); + memoize.type = Computation(); + memoize.operation = Core::Operation::Memoize; + memoize.operands = { Variable(3U, Computation()) }; + + const auto call = [](Core::SymbolId id, std::u32string spelling) { + Core::Instruction instruction; + instruction.kind = Core::Instruction::Kind::Bind; + instruction.destination = Name(id, std::move(spelling)); + instruction.type = Core::Type::int64(); + instruction.operation = Core::Operation::Call; + instruction.operands = { Variable(5U, Computation()) }; + return instruction; + }; + + Core::Function main; + main.symbol = Name(1U, U"Main"); + main.return_type = Core::Type::unit(); + main.entry = 0U; + main.blocks = { + Core::Block{ + 0U, + { std::move(seed), + std::move(closure), + std::move(memoize), + call(6U, U"first"), + call(7U, U"second") }, + Core::Terminator{ + Core::Terminator::Kind::Return, + Core::Atom::constant(std::monostate{}, Core::Type::unit()), + 0U, + 0U }, + }, + }; + + Core::Function lifted; + lifted.symbol = Name(10U, U"$closure10"); + lifted.parameters + = { Core::Parameter{ Name(4U, U"count"), Core::Type::int64() } }; + lifted.return_type = Core::Type::int64(); + lifted.entry = 0U; + lifted.blocks = { + Core::Block{ + 0U, + {}, + Core::Terminator{ Core::Terminator::Kind::Return, + Variable(4U, Core::Type::int64()), + 0U, + 0U }, + }, + }; + + return Core::CorePrepModule{ { U"MemoizeTests" }, + { std::move(main), std::move(lifted) } }; + } + + [[nodiscard]] auto + MemoizeOf(Core::CorePrepModule &module) -> Core::Instruction & + { + return module.functions.front().blocks.front().instructions[kMemoize]; + } + + template + [[nodiscard]] auto + HasIssue(const Issues &issues, std::string_view code) -> bool + { + return std::ranges::any_of(issues, [code](const auto &issue) { + return issue.code == code; + }); + } + + template + [[nodiscard]] auto + Find(Module &module, Opcode opcode) -> decltype(&module.functions.front() + .blocks.front() + .instructions.front()) + { + for (auto &function : module.functions) + for (auto &block : function.blocks) + for (auto &instruction : block.instructions) + if (instruction.opcode == opcode) + return &instruction; + return nullptr; + } + + template + [[nodiscard]] auto + Count(const Module &module, Opcode opcode) -> std::size_t + { + std::size_t count{}; + for (const auto &function : module.functions) + for (const auto &block : function.blocks) + for (const auto &instruction : block.instructions) + if (instruction.opcode == opcode) + ++count; + return count; + } + + [[nodiscard]] auto + Occurrences(std::string_view text, std::string_view part) -> std::size_t + { + std::size_t count{}; + for (auto found = text.find(part); found != std::string_view::npos; + found = text.find(part, found + part.size())) + ++count; + return count; + } +} // namespace + +TEST_CASE("CorePrep accepts a callable that remembers its result") +{ + CHECK(Core::verify(MemoizeModule()).empty()); +} + +TEST_CASE("CorePrep writes and reads the remembering operation unchanged") +{ + const auto source = MemoizeModule(); + const auto encoded = Core::wire::encode(source); + REQUIRE_FALSE(encoded.error); + const auto decoded = Core::wire::decode(encoded.bytes); + REQUIRE_FALSE(decoded.error); + REQUIRE(decoded.module); + CHECK(*decoded.module == source); + auto read = *decoded.module; + CHECK(MemoizeOf(read).operation == Core::Operation::Memoize); +} + +TEST_CASE("CorePrep remembers only a computation without parameters") +{ + auto module = MemoizeModule(); + const auto taking + = Core::Type::function({ Core::Type::int64() }, Core::Type::int64()); + MemoizeOf(module).operands = { Variable(3U, taking) }; + MemoizeOf(module).type = taking; + CHECK(HasIssue(Core::verify(module), "VXC1074")); +} + +TEST_CASE("CorePrep remembers only a Bool or a number") +{ + SECTION("a Bool is remembered") + { + auto module = MemoizeModule(); + auto &instructions = module.functions.front().blocks.front(); + instructions.instructions[1].type = Computation(Core::Type::boolean()); + MemoizeOf(module).operands + = { Variable(3U, Computation(Core::Type::boolean())) }; + MemoizeOf(module).type = Computation(Core::Type::boolean()); + CHECK_FALSE(HasIssue(Core::verify(module), "VXC1074")); + } + SECTION("no result is not") + { + auto module = MemoizeModule(); + MemoizeOf(module).operands + = { Variable(3U, Computation(Core::Type::unit())) }; + MemoizeOf(module).type = Computation(Core::Type::unit()); + CHECK(HasIssue(Core::verify(module), "VXC1074")); + } + SECTION("a string is not") + { + auto module = MemoizeModule(); + MemoizeOf(module).operands + = { Variable(3U, Computation(Core::Type::string())) }; + MemoizeOf(module).type = Computation(Core::Type::string()); + CHECK(HasIssue(Core::verify(module), "VXC1074")); + } + SECTION("a callable is not") + { + auto module = MemoizeModule(); + MemoizeOf(module).operands + = { Variable(3U, Computation(Computation())) }; + MemoizeOf(module).type = Computation(Computation()); + CHECK(HasIssue(Core::verify(module), "VXC1074")); + } +} + +TEST_CASE("CorePrep does not remember a value that is not a callable") +{ + auto module = MemoizeModule(); + MemoizeOf(module).operands = { Variable(2U, Core::Type::int64()) }; + MemoizeOf(module).type = Core::Type::int64(); + CHECK(HasIssue(Core::verify(module), "VXC1074")); +} + +TEST_CASE("CorePrep requires the remembering callable to keep the type") +{ + auto module = MemoizeModule(); + MemoizeOf(module).type = Computation(Core::Type::boolean()); + CHECK_FALSE(Core::verify(module).empty()); +} + +TEST_CASE("CorePrep requires exactly one computation to remember") +{ + SECTION("none") + { + auto module = MemoizeModule(); + MemoizeOf(module).operands.clear(); + CHECK_FALSE(Core::verify(module).empty()); + } + SECTION("two") + { + auto module = MemoizeModule(); + MemoizeOf(module).operands.push_back(Variable(3U, Computation())); + CHECK_FALSE(Core::verify(module).empty()); + } +} + +TEST_CASE("Xpp carries the remembering operation and its operand") +{ + auto xpp = Xpp::lower(MemoizeModule()); + CHECK(Visual::XSharp::Xpp::Verify(xpp).empty()); + REQUIRE(Count(xpp, Xpp::Opcode::Memoize) == 1U); + const auto *memoize = Find(xpp, Xpp::Opcode::Memoize); + REQUIRE(memoize != nullptr); + CHECK(memoize->destination == 5U); + CHECK(memoize->result_type == Computation()); + REQUIRE(memoize->operands.size() == 1U); + CHECK(memoize->operands.front().kind == Xpp::Operand::Kind::Symbol); + CHECK(memoize->operands.front().symbol == 3U); + CHECK(memoize->operands.front().type == Computation()); +} + +TEST_CASE("Xpp calls the remembering callable, not the computation") +{ + const auto xpp = Xpp::lower(MemoizeModule()); + std::size_t calls{}; + for (const auto &instruction : + xpp.functions.front().blocks.front().instructions) + { + if (instruction.opcode != Xpp::Opcode::Call) + continue; + ++calls; + REQUIRE(instruction.operands.size() == 1U); + CHECK(instruction.operands.front().symbol == 5U); + } + CHECK(calls == 2U); +} + +TEST_CASE("Xpp writes and reads the remembering operation unchanged") +{ + const auto xpp = Xpp::lower(MemoizeModule()); + const auto encoded = Visual::XSharp::Xpp::Wire::Encode(xpp); + REQUIRE(encoded); + const auto decoded = Visual::XSharp::Xpp::Wire::Decode(encoded.bytes); + REQUIRE(decoded); + REQUIRE(decoded.module); + CHECK(*decoded.module == xpp); + CHECK(Count(*decoded.module, Xpp::Opcode::Memoize) == 1U); +} + +TEST_CASE("Xpp checks the remembering rule for itself") +{ + SECTION("a computation with a parameter") + { + auto xpp = Xpp::lower(MemoizeModule()); + auto *memoize = Find(xpp, Xpp::Opcode::Memoize); + REQUIRE(memoize != nullptr); + const auto taking = Core::Type::function({ Core::Type::int64() }, + Core::Type::int64()); + memoize->operands.front().type = taking; + memoize->result_type = taking; + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1047")); + } + SECTION("a result that owns something") + { + auto xpp = Xpp::lower(MemoizeModule()); + auto *memoize = Find(xpp, Xpp::Opcode::Memoize); + REQUIRE(memoize != nullptr); + memoize->operands.front().type = Computation(Core::Type::string()); + memoize->result_type = Computation(Core::Type::string()); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1047")); + } + SECTION("a callable of another type") + { + auto xpp = Xpp::lower(MemoizeModule()); + auto *memoize = Find(xpp, Xpp::Opcode::Memoize); + REQUIRE(memoize != nullptr); + memoize->result_type = Computation(Core::Type::boolean()); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1047")); + } + SECTION("a value that is not a callable") + { + auto xpp = Xpp::lower(MemoizeModule()); + auto *memoize = Find(xpp, Xpp::Opcode::Memoize); + REQUIRE(memoize != nullptr); + memoize->operands.front().type = Core::Type::int64(); + memoize->result_type = Core::Type::int64(); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1047")); + } +} + +TEST_CASE("Xmm carries the remembering operation in a register") +{ + auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + CHECK(Visual::XSharp::Xmm::Verify(xmm).empty()); + REQUIRE(Count(xmm, Xmm::Opcode::Memoize) == 1U); + const auto *memoize = Find(xmm, Xmm::Opcode::Memoize); + REQUIRE(memoize != nullptr); + CHECK(memoize->has_result); + CHECK(memoize->result_type == Computation()); + REQUIRE(memoize->operands.size() == 1U); + CHECK(memoize->operands.front().kind == Xmm::Value::Kind::Register); + CHECK(memoize->operands.front().type == Computation()); +} + +TEST_CASE("Xmm writes and reads the remembering operation unchanged") +{ + const auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + const auto encoded = Visual::XSharp::Xmm::Wire::Encode(xmm); + REQUIRE(encoded); + const auto decoded = Visual::XSharp::Xmm::Wire::Decode(encoded.bytes); + REQUIRE(decoded); + REQUIRE(decoded.module); + CHECK(*decoded.module == xmm); + CHECK(Count(*decoded.module, Xmm::Opcode::Memoize) == 1U); +} + +TEST_CASE("Xmm checks the remembering rule for itself") +{ + SECTION("a computation with a parameter") + { + auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + auto *memoize = Find(xmm, Xmm::Opcode::Memoize); + REQUIRE(memoize != nullptr); + const auto taking = Core::Type::function({ Core::Type::int64() }, + Core::Type::int64()); + memoize->operands.front().type = taking; + memoize->result_type = taking; + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1053")); + } + SECTION("a result that owns something") + { + auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + auto *memoize = Find(xmm, Xmm::Opcode::Memoize); + REQUIRE(memoize != nullptr); + memoize->operands.front().type = Computation(Core::Type::string()); + memoize->result_type = Computation(Core::Type::string()); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1053")); + } + SECTION("a callable of another type") + { + auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + auto *memoize = Find(xmm, Xmm::Opcode::Memoize); + REQUIRE(memoize != nullptr); + memoize->result_type = Computation(Core::Type::boolean()); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1053")); + } + SECTION("a value that is not a callable") + { + auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + auto *memoize = Find(xmm, Xmm::Opcode::Memoize); + REQUIRE(memoize != nullptr); + memoize->operands.front().type = Core::Type::int64(); + memoize->result_type = Core::Type::int64(); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1053")); + } +} + +TEST_CASE("LLVM gives the remembering callable an object of its own") +{ + const auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + Llvm::Options options; + options.optimization = Llvm::OptimizationLevel::Debug; + const auto result = Llvm::Lower(xmm, options); + REQUIRE(result); + const std::string_view ir = result.artifact->llvm_ir; + + // The object: how to call it, whether the result is known, the result, + // and the computation it owns. + CHECK(ir.find("%.vxs.aarc.memo.payload.") != std::string_view::npos); + CHECK(ir.find("{ ptr, i8, i64, ptr }") != std::string_view::npos); + // It is called the way a closure is called, and released the way a + // closure is released: through the functions its object names. + CHECK(ir.find("define internal i64 @.vxs.aarc.memo.invoke.") + != std::string_view::npos); + CHECK(ir.find("define internal void @.vxs.aarc.memo.destroy.") + != std::string_view::npos); + CHECK(ir.find("@.vxs.aarc.memo.metadata.") != std::string_view::npos); + // One object for the closure and one for the callable that remembers. + CHECK(Occurrences(ir, "call ptr @vxs_aarc_allocate(") == 2U); +} + +TEST_CASE("LLVM keeps the computation alive for as long as it is remembered") +{ + const auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + Llvm::Options options; + options.optimization = Llvm::OptimizationLevel::Debug; + const auto result = Llvm::Lower(xmm, options); + REQUIRE(result); + const std::string_view ir = result.artifact->llvm_ir; + + // The remembering object takes a reference of its own to the + // computation, and its destructor gives that reference back. + CHECK(ir.find("%memo.owned = call ptr @vxs_aarc_retain_strong(") + != std::string_view::npos); + const auto destructor + = ir.find("define internal void @.vxs.aarc.memo.destroy."); + REQUIRE(destructor != std::string_view::npos); + const auto body = ir.substr(destructor, ir.find("\n}\n", destructor)); + CHECK(body.find("call void @vxs_aarc_release_strong(") + != std::string_view::npos); +} + +TEST_CASE("LLVM lowers the remembering callable at every optimization level") +{ + const auto xmm = Xmm::lower(Xpp::lower(MemoizeModule())); + for (const auto level : { Llvm::OptimizationLevel::Debug, + Llvm::OptimizationLevel::Less, + Llvm::OptimizationLevel::Default, + Llvm::OptimizationLevel::Aggressive }) + { + Llvm::Options options; + options.optimization = level; + const auto result = Llvm::Lower(xmm, options); + REQUIRE(result); + CHECK_FALSE(result.artifact->bitcode.empty()); + } +} diff --git a/Compiler/Driver/Tests/RuntimeCallPipelineTests.cpp b/Compiler/Driver/Tests/RuntimeCallPipelineTests.cpp new file mode 100644 index 00000000..d444a0d4 --- /dev/null +++ b/Compiler/Driver/Tests/RuntimeCallPipelineTests.cpp @@ -0,0 +1,723 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Backend/LLVM.hpp" +#include "Visual/XSharp/Core/CorePrep.hpp" +#include "Visual/XSharp/Core/CorePrep/Verifier.hpp" +#include "Visual/XSharp/Core/CorePrep/Wire.hpp" +#include "Visual/XSharp/Core/RuntimeCall.hpp" +#include "Visual/XSharp/Core/Scalar.hpp" +#include "Visual/XSharp/Xmm/IR.hpp" +#include "Visual/XSharp/Xmm/Verifier.hpp" +#include "Visual/XSharp/Xmm/Wire.hpp" +#include "Visual/XSharp/Xpp/IR.hpp" +#include "Visual/XSharp/Xpp/OwnershipPlacement.hpp" +#include "Visual/XSharp/Xpp/Verifier.hpp" +#include "Visual/XSharp/Xpp/Wire.hpp" + +// A call of a runtime function, followed through every native stage. +// +// The operation names its function with a literal first operand and gives +// the arguments after it. The catalog says what each function takes and +// returns, and every stage checks a call against it for itself: an artifact +// may enter the pipeline at any stage, so no stage relies on the one before +// it having checked. These cases take one module through the stages, pin +// the catalog row by row, and then break a call in each way it can be +// broken, at each stage. + +namespace +{ + namespace Core = visual_xsharp::core; + namespace Runtime = visual_xsharp::core::runtime; + namespace Llvm = Visual::XSharp::Backend::LLVM; + namespace Xmm = visual_xsharp::xmm; + namespace Xpp = visual_xsharp::xpp; + + [[nodiscard]] auto + Name(Core::SymbolId id, std::u32string spelling) -> Core::SymbolName + { + return { id, std::move(spelling) }; + } + + [[nodiscard]] auto + Variable(Core::SymbolId id, Core::Type type) -> Core::Atom + { + return Core::Atom::variable(Name(id, {}), std::move(type)); + } + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Atom + { + return Core::Atom::constant(Core::integer_from_signed(value), + Core::Type::int64()); + } + + [[nodiscard]] auto + Identity(Runtime::Function function) -> Core::Atom + { + return Integer(static_cast(function)); + } + + [[nodiscard]] auto + Text(std::u32string value) -> Core::Atom + { + return Core::Atom::constant(std::move(value), Core::Type::string()); + } + + [[nodiscard]] auto + Bind(Core::SymbolId id, + std::u32string spelling, + Core::Type type, + Core::Operation operation, + std::vector operands) -> Core::Instruction + { + Core::Instruction instruction; + instruction.kind = Core::Instruction::Kind::Bind; + instruction.destination = Name(id, std::move(spelling)); + instruction.type = std::move(type); + instruction.operation = operation; + instruction.operands = std::move(operands); + return instruction; + } + + // Positions of the calls in the entry block of `Main`. + constexpr std::size_t kFormat = 1U; + constexpr std::size_t kConcat = 2U; + constexpr std::size_t kWrite = 3U; + + /// ``` + /// Main(count): + /// label = "Count: " + /// digits = runtime TextFormatSigned(0, 5, -1, count) + /// line = runtime TextConcat(label, digits) + /// runtime ConsoleWrite(line, 1) + /// return + /// ``` + [[nodiscard]] auto + WritingModule() -> Core::CorePrepModule + { + Core::Instruction write; + write.kind = Core::Instruction::Kind::Evaluate; + write.type = Core::Type::unit(); + write.operation = Core::Operation::RuntimeCall; + write.operands = { Identity(Runtime::Function::ConsoleWrite), + Variable(5U, Core::Type::string()), + Integer(1) }; + + Core::Function main; + main.symbol = Name(1U, U"Main"); + main.parameters + = { Core::Parameter{ Name(2U, U"count"), Core::Type::int64() } }; + main.return_type = Core::Type::unit(); + main.entry = 0U; + main.blocks = { + Core::Block{ + 0U, + { Bind(3U, + U"label", + Core::Type::string(), + Core::Operation::Copy, + { Text(U"Count: ") }), + Bind(4U, + U"digits", + Core::Type::string(), + Core::Operation::RuntimeCall, + { Identity(Runtime::Function::TextFormatSigned), + Integer(0), + Integer(5), + Integer(-1), + Variable(2U, Core::Type::int64()) }), + Bind(5U, + U"line", + Core::Type::string(), + Core::Operation::RuntimeCall, + { Identity(Runtime::Function::TextConcat), + Variable(3U, Core::Type::string()), + Variable(4U, Core::Type::string()) }), + std::move(write) }, + Core::Terminator{ + Core::Terminator::Kind::Return, + Core::Atom::constant(std::monostate{}, Core::Type::unit()), + 0U, + 0U }, + }, + }; + return Core::CorePrepModule{ { U"RuntimeCallTests" }, + { std::move(main) } }; + } + + [[nodiscard]] auto + Call(Core::CorePrepModule &module, std::size_t position) + -> Core::Instruction & + { + return module.functions.front().blocks.front().instructions[position]; + } + + template + [[nodiscard]] auto + HasIssue(const Issues &issues, std::string_view code) -> bool + { + return std::ranges::any_of(issues, [code](const auto &issue) { + return issue.code == code; + }); + } + + template + [[nodiscard]] auto + Calls(Module &module, Opcode opcode) -> std::vector< + decltype(&module.functions.front().blocks.front().instructions.front())> + { + std::vector + found; + for (auto &function : module.functions) + for (auto &block : function.blocks) + for (auto &instruction : block.instructions) + if (instruction.opcode == opcode) + found.push_back(&instruction); + return found; + } + + [[nodiscard]] auto + Occurrences(std::string_view text, std::string_view part) -> std::size_t + { + std::size_t count{}; + for (auto found = text.find(part); found != std::string_view::npos; + found = text.find(part, found + part.size())) + ++count; + return count; + } + + /// One row of the catalog as the tests state it. + struct Row final + { + Runtime::Function function; + std::uint64_t identity; + std::string_view symbol; + std::vector parameters; + Runtime::Result result; + bool observable; + }; + + [[nodiscard]] auto + Rows() -> std::vector + { + using enum Runtime::Parameter; + using Runtime::Function; + using Runtime::Result; + return { + { Function::TextConcat, + 1U, + "vxs_text_concat", + { Text, Text }, + Result::Text, + false }, + { Function::TextFromSigned, + 2U, + "vxs_text_from_signed", + { Signed }, + Result::Text, + false }, + { Function::TextFromUnsigned, + 3U, + "vxs_text_from_unsigned", + { Unsigned }, + Result::Text, + false }, + { Function::TextFromBool, + 4U, + "vxs_text_from_bool", + { Bool }, + Result::Text, + false }, + { Function::TextFromChar, + 5U, + "vxs_text_from_char", + { Char }, + Result::Text, + false }, + { Function::TextFormatSigned, + 6U, + "vxs_text_format_signed", + { Count, Count, Count, Signed }, + Result::Text, + false }, + { Function::TextFormatUnsigned, + 7U, + "vxs_text_format_unsigned", + { Count, Count, Count, Unsigned }, + Result::Text, + false }, + { Function::TextFormatFloating, + 8U, + "vxs_text_format_floating", + { Count, Count, Count, Floating }, + Result::Text, + false }, + { Function::TextFormatString, + 9U, + "vxs_text_format_string", + { Count, Count, Count, Text }, + Result::Text, + false }, + { Function::TextFormatChar, + 10U, + "vxs_text_format_char", + { Count, Count, Count, Char }, + Result::Text, + false }, + { Function::TextNewline, + 11U, + "vxs_text_newline", + {}, + Result::Text, + false }, + { Function::ConsoleWrite, + 12U, + "vxs_console_write", + { Text, Count }, + Result::Nothing, + true }, + { Function::TextEquals, + 13U, + "vxs_text_equals", + { Text, Text }, + Result::Truth, + false }, + }; + } +} // namespace + +TEST_CASE("the runtime catalog has the rows the artifact formats rely on") +{ + // An identity is written into artifacts and a symbol is linked against: + // a row that changed would silently change what compiled programs call. + const auto rows = Rows(); + REQUIRE(Runtime::Catalog().size() == rows.size()); + for (const auto &row : rows) + { + CAPTURE(row.symbol); + CHECK(static_cast(row.function) == row.identity); + const auto *signature = Runtime::Find(row.identity); + REQUIRE(signature != nullptr); + CHECK(signature->function == row.function); + CHECK(signature->symbol == row.symbol); + CHECK(std::ranges::equal(signature->parameters, row.parameters)); + CHECK(signature->result == row.result); + CHECK(signature->observable == row.observable); + } +} + +TEST_CASE("every runtime function has an identity and a symbol of its own") +{ + std::set identities; + std::set symbols; + for (const auto &signature : Runtime::Catalog()) + { + CHECK(identities.insert(static_cast(signature.function)) + .second); + CHECK(symbols.insert(signature.symbol).second); + CHECK(signature.symbol.starts_with("vxs_")); + } + // Zero is no function, and neither is the number after the last. + CHECK(Runtime::Find(0U) == nullptr); + CHECK(Runtime::Find(Runtime::Catalog().size() + 1U) == nullptr); + CHECK(Runtime::Find(~std::uint64_t{ 0U }) == nullptr); +} + +TEST_CASE("a parameter takes its family of types and no other") +{ + using Kind = Core::Type::Kind; + using enum Runtime::Parameter; + const std::array kinds{ Kind::Unit, Kind::Bool, Kind::Character, + Kind::Int8, Kind::Int16, Kind::Int32, + Kind::Int64, Kind::Int128, Kind::UInt8, + Kind::UInt16, Kind::UInt32, Kind::UInt64, + Kind::UInt128, Kind::Float16, Kind::Float32, + Kind::Float64, Kind::Float128, Kind::String, + Kind::Function, Kind::Named }; + const auto accepted = [&kinds](Runtime::Parameter parameter) { + std::vector taken; + for (const auto kind : kinds) + if (Runtime::Accepts(parameter, kind)) + taken.push_back(kind); + return taken; + }; + // A signed and an unsigned integer of at most sixty-four bits, a + // floating-point number of at most sixty-four: what the runtime + // functions are written for after widening. + CHECK(accepted(Signed) + == std::vector{ Kind::Int8, Kind::Int16, Kind::Int32, Kind::Int64 }); + CHECK(accepted(Unsigned) + == std::vector{ Kind::UInt8, + Kind::UInt16, + Kind::UInt32, + Kind::UInt64 }); + CHECK(accepted(Floating) + == std::vector{ Kind::Float16, Kind::Float32, Kind::Float64 }); + CHECK(accepted(Bool) == std::vector{ Kind::Bool }); + CHECK(accepted(Char) == std::vector{ Kind::Character }); + CHECK(accepted(Text) == std::vector{ Kind::String }); + CHECK(accepted(Count) == std::vector{ Kind::Int64 }); +} + +TEST_CASE("an identity is a literal int that is not negative") +{ + const auto literal = [](std::int64_t value) { + return Core::Literal{ Core::integer_from_signed(value) }; + }; + CHECK(Runtime::IdentityOf(literal(12), Core::Type::int64()) == 12U); + CHECK(Runtime::IdentityOf(literal(0), Core::Type::int64()) == 0U); + CHECK_FALSE(Runtime::IdentityOf(literal(-1), Core::Type::int64())); + // The type is that of the language's int; another integer type, or a + // literal that is not an integer, names nothing. + CHECK_FALSE(Runtime::IdentityOf(literal(12), Core::Type::int32())); + CHECK_FALSE(Runtime::IdentityOf(literal(12), Core::Type::uint64())); + CHECK_FALSE( + Runtime::IdentityOf(Core::Literal{ true }, Core::Type::int64())); + CHECK_FALSE(Runtime::IdentityOf(Core::Literal{ std::u32string(U"12") }, + Core::Type::int64())); + // A magnitude beyond sixty-four bits is not an identity. + Core::IntegerLiteral wide; + wide.magnitude.assign(9U, 0xffU); + CHECK_FALSE( + Runtime::IdentityOf(Core::Literal{ wide }, Core::Type::int64())); +} + +TEST_CASE("CorePrep accepts well-formed runtime calls") +{ + CHECK(Core::verify(WritingModule()).empty()); +} + +TEST_CASE("CorePrep writes and reads runtime calls unchanged") +{ + const auto source = WritingModule(); + const auto encoded = Core::wire::encode(source); + REQUIRE_FALSE(encoded.error); + const auto decoded = Core::wire::decode(encoded.bytes); + REQUIRE_FALSE(decoded.error); + REQUIRE(decoded.module); + CHECK(*decoded.module == source); +} + +TEST_CASE("CorePrep rejects a runtime call that names no function") +{ + SECTION("an identity the catalog does not have") + { + auto module = WritingModule(); + Call(module, kConcat).operands.front() = Integer(999); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("zero") + { + auto module = WritingModule(); + Call(module, kConcat).operands.front() = Integer(0); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("a negative number") + { + auto module = WritingModule(); + Call(module, kConcat).operands.front() = Integer(-1); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("a value that is computed") + { + // The function is fixed when the program is compiled. + auto module = WritingModule(); + Call(module, kConcat).operands.front() + = Variable(2U, Core::Type::int64()); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("no operand at all") + { + auto module = WritingModule(); + Call(module, kConcat).operands.clear(); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } +} + +TEST_CASE("CorePrep holds a runtime call to its function's arguments") +{ + SECTION("one too few") + { + auto module = WritingModule(); + Call(module, kConcat).operands.pop_back(); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("one too many") + { + auto module = WritingModule(); + Call(module, kConcat) + .operands.push_back(Variable(3U, Core::Type::string())); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("a number where a string is taken") + { + auto module = WritingModule(); + Call(module, kConcat).operands.back() + = Variable(2U, Core::Type::int64()); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("a string where a signed integer is taken") + { + auto module = WritingModule(); + Call(module, kFormat).operands.back() + = Variable(3U, Core::Type::string()); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("a width that is not an int") + { + auto module = WritingModule(); + Call(module, kFormat).operands[2] + = Core::Atom::constant(Core::integer_from_signed(5), + Core::Type::int32()); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } +} + +TEST_CASE("CorePrep holds a runtime call to its function's result") +{ + SECTION("a string result bound as a number") + { + auto module = WritingModule(); + Call(module, kConcat).type = Core::Type::int64(); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } + SECTION("a write given a result") + { + auto module = WritingModule(); + Call(module, kWrite).type = Core::Type::string(); + CHECK(HasIssue(Core::verify(module), "VXC1076")); + } +} + +TEST_CASE("Xpp carries runtime calls with their identity and arguments") +{ + auto xpp = Xpp::lower(WritingModule()); + CHECK(Visual::XSharp::Xpp::Verify(xpp).empty()); + const auto calls = Calls(xpp, Xpp::Opcode::RuntimeCall); + REQUIRE(calls.size() == 3U); + // The function is a literal operand, not a symbol that is read. + for (const auto *call : calls) + { + REQUIRE_FALSE(call->operands.empty()); + CHECK(call->operands.front().kind == Xpp::Operand::Kind::Literal); + CHECK(Runtime::IdentityOf(call->operands.front().literal, + call->operands.front().type)); + } + CHECK(Runtime::IdentityOf(calls[0]->operands.front().literal, + calls[0]->operands.front().type) + == 6U); + CHECK(calls[0]->operands.size() == 5U); + CHECK(calls[0]->result_type == Core::Type::string()); + CHECK(calls[1]->operands.size() == 3U); + // The write yields nothing and is kept for what it does. + CHECK(calls[2]->result_type == Core::Type::unit()); + CHECK(calls[2]->effect == Xpp::Instruction::Effect::Discard); +} + +TEST_CASE("Xpp writes and reads runtime calls unchanged") +{ + const auto xpp = Xpp::lower(WritingModule()); + const auto encoded = Visual::XSharp::Xpp::Wire::Encode(xpp); + REQUIRE(encoded); + const auto decoded = Visual::XSharp::Xpp::Wire::Decode(encoded.bytes); + REQUIRE(decoded); + REQUIRE(decoded.module); + CHECK(*decoded.module == xpp); +} + +TEST_CASE("Xpp checks a runtime call for itself") +{ + SECTION("an unknown function") + { + auto xpp = Xpp::lower(WritingModule()); + auto *call = Calls(xpp, Xpp::Opcode::RuntimeCall).front(); + call->operands.front().literal = Core::integer_from_signed(999); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1048")); + } + SECTION("a function named by a symbol") + { + auto xpp = Xpp::lower(WritingModule()); + auto *call = Calls(xpp, Xpp::Opcode::RuntimeCall).front(); + call->operands.front().kind = Xpp::Operand::Kind::Symbol; + call->operands.front().symbol = 2U; + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1048")); + } + SECTION("a missing argument") + { + auto xpp = Xpp::lower(WritingModule()); + auto *call = Calls(xpp, Xpp::Opcode::RuntimeCall).front(); + call->operands.pop_back(); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1048")); + } + SECTION("an argument of another type") + { + auto xpp = Xpp::lower(WritingModule()); + auto *call = Calls(xpp, Xpp::Opcode::RuntimeCall).front(); + call->operands.back().type = Core::Type::uint64(); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1048")); + } + SECTION("a result of another type") + { + auto xpp = Xpp::lower(WritingModule()); + auto *call = Calls(xpp, Xpp::Opcode::RuntimeCall).front(); + call->result_type = Core::Type::boolean(); + CHECK(HasIssue(Visual::XSharp::Xpp::Verify(xpp), "VXP1048")); + } +} + +TEST_CASE("Xmm carries runtime calls with an immediate identity") +{ + auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + CHECK(Visual::XSharp::Xmm::Verify(xmm).empty()); + const auto calls = Calls(xmm, Xmm::Opcode::RuntimeCall); + REQUIRE(calls.size() == 3U); + for (const auto *call : calls) + { + REQUIRE_FALSE(call->operands.empty()); + CHECK(call->operands.front().kind == Xmm::Value::Kind::Immediate); + } + CHECK(calls[0]->has_result); + CHECK(calls[1]->has_result); + CHECK_FALSE(calls[2]->has_result); +} + +TEST_CASE("Xmm writes and reads runtime calls unchanged") +{ + const auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + const auto encoded = Visual::XSharp::Xmm::Wire::Encode(xmm); + REQUIRE(encoded); + const auto decoded = Visual::XSharp::Xmm::Wire::Decode(encoded.bytes); + REQUIRE(decoded); + REQUIRE(decoded.module); + CHECK(*decoded.module == xmm); +} + +TEST_CASE("Xmm checks a runtime call for itself") +{ + SECTION("an unknown function") + { + auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + auto *call = Calls(xmm, Xmm::Opcode::RuntimeCall).front(); + call->operands.front().immediate = Core::integer_from_signed(0); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1054")); + } + SECTION("a missing argument") + { + auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + auto *call = Calls(xmm, Xmm::Opcode::RuntimeCall).front(); + call->operands.pop_back(); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1054")); + } + SECTION("an argument of another type") + { + auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + auto *call = Calls(xmm, Xmm::Opcode::RuntimeCall).front(); + call->operands.back().type = Core::Type::float64(); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1054")); + } + SECTION("a result of another type") + { + auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + auto *call = Calls(xmm, Xmm::Opcode::RuntimeCall).front(); + call->result_type = Core::Type::int64(); + CHECK(HasIssue(Visual::XSharp::Xmm::Verify(xmm), "VXL1054")); + } +} + +TEST_CASE("ownership placement releases the strings a runtime call returns") +{ + // Three strings are created: the literal and the two results. Each is + // released once, after its last use; the write borrows its argument. + const auto xpp + = Visual::XSharp::Xpp::PlaceOwnership(Xpp::lower(WritingModule())); + CHECK(Visual::XSharp::Xpp::Verify(xpp).empty()); + std::size_t releases{}; + for (const auto &instruction : + xpp.functions.front().blocks.front().instructions) + if (instruction.opcode == Xpp::Opcode::ReleaseStrong) + ++releases; + CHECK(releases == 3U); +} + +TEST_CASE("LLVM calls the runtime function by the symbol of its row") +{ + const auto xmm = Xmm::lower( + Visual::XSharp::Xpp::PlaceOwnership(Xpp::lower(WritingModule()))); + CHECK(Visual::XSharp::Xmm::Verify(xmm).empty()); + Llvm::Options options; + options.optimization = Llvm::OptimizationLevel::Debug; + const auto result = Llvm::Lower(xmm, options); + REQUIRE(result); + const std::string_view ir = result.artifact->llvm_ir; + + // Flags, width and precision before the value, each sixty-four bits. + CHECK(ir.find("declare ptr @vxs_text_format_signed(i64, i64, i64, i64)") + != std::string_view::npos); + CHECK(ir.find("declare ptr @vxs_text_concat(ptr, ptr)") + != std::string_view::npos); + CHECK(ir.find("declare void @vxs_console_write(ptr, i64)") + != std::string_view::npos); + CHECK(Occurrences(ir, "call ptr @vxs_text_format_signed(") == 1U); + CHECK(Occurrences(ir, "call ptr @vxs_text_concat(") == 1U); + CHECK(Occurrences(ir, "call void @vxs_console_write(") == 1U); + // The literal, the digits and the line are each released. + CHECK(Occurrences(ir, "call void @vxs_aarc_release_strong(") == 3U); +} + +TEST_CASE("LLVM lowers runtime calls at every optimization level") +{ + const auto xmm = Xmm::lower(Xpp::lower(WritingModule())); + for (const auto level : { Llvm::OptimizationLevel::Debug, + Llvm::OptimizationLevel::Less, + Llvm::OptimizationLevel::Default, + Llvm::OptimizationLevel::Aggressive }) + { + Llvm::Options options; + options.optimization = level; + const auto result = Llvm::Lower(xmm, options); + REQUIRE(result); + // A write is observed by whoever reads the output: no level of + // optimization removes it. + CHECK(result.artifact->llvm_ir.find("@vxs_console_write(") + != std::string::npos); + } +} + +TEST_CASE("a string literal given to a runtime call is released") +{ + // The literal is the operand itself here, not a binding: it creates a + // string where it is used, and nothing names that string. Ownership + // placement gives it a symbol, so that it can be released after the + // call that borrowed it. + auto module = WritingModule(); + Call(module, kConcat).operands[1] = Text(U"Count: "); + auto placed = Visual::XSharp::Xpp::PlaceOwnership(Xpp::lower(module)); + CHECK(Visual::XSharp::Xpp::Verify(placed).empty()); + for (const auto *call : Calls(placed, Xpp::Opcode::RuntimeCall)) + for (std::size_t index = 1U; index < call->operands.size(); ++index) + if (call->operands[index].type == Core::Type::string()) + CHECK(call->operands[index].kind == Xpp::Operand::Kind::Symbol); + // The label nothing reads any more, the literal, the digits and the + // line: four strings, four releases. + std::size_t releases{}; + for (const auto &instruction : + placed.functions.front().blocks.front().instructions) + if (instruction.opcode == Xpp::Opcode::ReleaseStrong) + ++releases; + CHECK(releases == 4U); +} diff --git a/Compiler/Driver/Tests/ScalarPipelineTests.cpp b/Compiler/Driver/Tests/ScalarPipelineTests.cpp index 1239c6ec..53e87864 100644 --- a/Compiler/Driver/Tests/ScalarPipelineTests.cpp +++ b/Compiler/Driver/Tests/ScalarPipelineTests.cpp @@ -413,10 +413,10 @@ TEST_CASE("literal validation combines payload kind and scalar range", core::validate_literal(std::u32string(U"text"), core::Type::boolean())); } -TEST_CASE("CorePrep wire v6 round-trips every scalar family", +TEST_CASE("CorePrep wire v7 round-trips every scalar family", "[scalar][wire][coreprep]") { - CHECK(core_wire::current_version == 6U); + CHECK(core_wire::current_version == 8U); CHECK(wire_round_trips(core::Type::unit())); CHECK(wire_round_trips(core::Type::string())); for (const auto &entry : kScalarCases) @@ -426,10 +426,10 @@ TEST_CASE("CorePrep wire v6 round-trips every scalar family", } } -TEST_CASE("Core wire v8 round-trips every scalar family", +TEST_CASE("Core wire v10 round-trips every scalar family", "[scalar][wire][core]") { - CHECK(native_wire::kCurrentVersion == 8U); + CHECK(native_wire::kCurrentVersion == 10U); CHECK(native_wire_round_trips(core::Type::unit())); CHECK(native_wire_round_trips(core::Type::string())); for (const auto &entry : kScalarCases) diff --git a/Compiler/Fuzzing/BUILD.bazel b/Compiler/Fuzzing/BUILD.bazel index 72878080..87dbf169 100644 --- a/Compiler/Fuzzing/BUILD.bazel +++ b/Compiler/Fuzzing/BUILD.bazel @@ -1,3 +1,4 @@ +load("//Compiler/Runtime:host.bzl", "RUNTIME_HOST_DEPS", "RUNTIME_HOST_LINKOPTS") load("@rules_cc//cc:cc_binary.bzl", "cc_binary") load("@rules_cc//cc:cc_library.bzl", "cc_library") @@ -43,10 +44,14 @@ cc_library( deps = ["//Compiler/Headers/Visual/XSharp/Core:core"], ) +# A compiled program calls the runtime, and the JIT resolves a runtime symbol +# in the process that hosts it. The options and libraries are those of every +# such host; see //Compiler/Runtime:host.bzl. cc_library( name = "source_fuzz_harness", srcs = ["SourceFuzz.cpp"], hdrs = ["SourceFuzz.hpp"], + linkopts = RUNTIME_HOST_LINKOPTS, deps = [ ":coreprep_parity", "//Compiler/Backend/LLVM:llvm_backend", @@ -55,21 +60,83 @@ cc_library( "//Compiler/Driver:pipeline", "//Compiler/Headers/Visual/XSharp/Core:core", "//Compiler/Headers/Visual/XSharp/Core:coreprep_wire", - ], + "//Compiler/Support:compiler_stack", + ] + RUNTIME_HOST_DEPS, ) cc_binary( name = "source_fuzz_smoke", + srcs = [ + "SourceFuzzSmoke.cpp", + ], + deps = [":source_fuzz_harness"], +) + +# The large tables of programs with hand-written results: the branching +# table here, the expression and leaving tables in source_expression_smoke. +# Programs of their own, so that each smoke program runs under one process +# watchdog. +cc_binary( + name = "source_execution_smoke", srcs = [ "BranchingExecutionCases.cpp", "BranchingExecutionCases.hpp", + "ExecutionCases.cpp", + "ExecutionCases.hpp", + "Generated/SelectionCases.inc", + "SourceExecutionSmoke.cpp", + ], + deps = [":source_fuzz_harness"], +) + +cc_binary( + name = "source_expression_smoke", + srcs = [ + "ExecutionCases.cpp", + "ExecutionCases.hpp", "ExpressionExecutionCases.cpp", "ExpressionExecutionCases.hpp", - "SourceFuzzSmoke.cpp", + "Generated/LeavingCases.inc", + "LeavingExecutionCases.cpp", + "LeavingExecutionCases.hpp", + "SourceExpressionSmoke.cpp", ], deps = [":source_fuzz_harness"], ) +# The tables written by hand for single features: inferred return types, +# evaluation by need, enums and closures. +cc_binary( + name = "source_feature_smoke", + srcs = [ + "ExecutionCases.cpp", + "ExecutionCases.hpp", + "SourceFeatureSmoke.cpp", + ], + deps = [ + ":source_fuzz_harness", + "//Compiler/Headers/Visual/XSharp/Runtime:aarc_api", + "//Compiler/Runtime/AARC:aarc", + ], +) + +# Programs that write to the console, with the text each must write. The +# harness links the runtime the generated code calls; its output is read +# through a sink. +cc_binary( + name = "source_console_smoke", + srcs = ["SourceConsoleSmoke.cpp"], + deps = [":source_fuzz_harness"] + RUNTIME_HOST_DEPS, +) + +# A measuring instrument, not a test: it compiles one source file on a stack +# of a chosen size and prints the stack the compilation committed. +cc_binary( + name = "source_stack_probe", + srcs = ["SourceStackProbe.cpp"], + deps = [":source_fuzz_harness"], +) + cc_binary( name = "lexer_fuzzer", srcs = ["SourceLibFuzzer.cpp"], diff --git a/Compiler/Fuzzing/BranchingExecutionCases.cpp b/Compiler/Fuzzing/BranchingExecutionCases.cpp index 7e2b03a8..1317eb91 100644 --- a/Compiler/Fuzzing/BranchingExecutionCases.cpp +++ b/Compiler/Fuzzing/BranchingExecutionCases.cpp @@ -2,13 +2,14 @@ // SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 #include +#include #include -#include #include #include +#include #include "BranchingExecutionCases.hpp" -#include "SourceFuzz.hpp" +#include "ExecutionCases.hpp" // Executable regressions for `match`, for `if` used as an expression and for // `guard`. Each case is a method body, the arguments it runs on and the value @@ -16,618 +17,29 @@ // rules: the subjects of a match are evaluated once and left to right, the // arms are tested in order, a guard runs only when the patterns of its arm // accept, exactly one body runs, and the block of a guard runs only when its -// condition is false. The same table is checked against a reference -// evaluator in `BranchingTests.hs` of the frontend test suite; here every -// case runs through CorePrep, Xpp, Xmm, LLVM and the ORC JIT, unoptimized -// and optimized. +// condition is false. The cases are written in `Cases/Selection.cases`; the +// rows here and the list `selectionCases` that the frontend test suite runs +// in a reference evaluator are generated from that file by +// `go -C helpers run ./cmd/execution-cases generate`. Here every case runs +// through CorePrep, Xpp, Xmm, LLVM and the ORC JIT, unoptimized and +// optimized. The cases that leave instead of yielding a value are in +// `LeavingExecutionCases.cpp`. namespace Visual::XSharp::Fuzzing { namespace { - struct Case final - { - bool flag; - bool other; - int left; - int right; - std::int64_t expected; - std::string_view body; - }; - - constexpr std::array kCases{ { - { false, - false, - 1, - 0, - 10, - "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" }, - { false, - false, - 2, - 0, - 20, - "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" }, - { false, - false, - 5, - 0, - 30, - "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" }, - { false, - false, - 1, - 0, - 10, - "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } " - "return r;" }, - { false, - false, - 2, - 0, - 20, - "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } " - "return r;" }, - { false, - false, - 7, - 0, - 0, - "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } " - "return r;" }, - { false, - false, - 20, - 0, - 40, - "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 " - "-> n + 1, _ -> 0 };" }, - { false, - false, - 7, - 0, - 8, - "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 " - "-> n + 1, _ -> 0 };" }, - { false, - false, - 3, - 0, - 0, - "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 " - "-> n + 1, _ -> 0 };" }, - { false, - false, - 1, - 1, - 11, - "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, " - "(_), (1) -> 1, (_), (_) -> 0 };" }, - { false, - false, - 1, - 5, - 10, - "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, " - "(_), (1) -> 1, (_), (_) -> 0 };" }, - { false, - false, - 4, - 1, - 1, - "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, " - "(_), (1) -> 1, (_), (_) -> 0 };" }, - { false, - false, - 4, - 4, - 0, - "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, " - "(_), (1) -> 1, (_), (_) -> 0 };" }, - { true, - false, - 0, - 0, - 1, - "return match (flag) { true -> 1, false -> 2 };" }, - { false, - false, - 0, - 0, - 2, - "return match (flag) { true -> 1, false -> 2 };" }, - { true, - true, - 0, - 0, - 1, - "return match (flag), (other) { (true), (_) -> 1, (false), " - "(true) -> 2, (false), (false) -> 3 };" }, - { true, - false, - 0, - 0, - 1, - "return match (flag), (other) { (true), (_) -> 1, (false), " - "(true) -> 2, (false), (false) -> 3 };" }, - { false, - true, - 0, - 0, - 2, - "return match (flag), (other) { (true), (_) -> 1, (false), " - "(true) -> 2, (false), (false) -> 3 };" }, - { false, - false, - 0, - 0, - 3, - "return match (flag), (other) { (true), (_) -> 1, (false), " - "(true) -> 2, (false), (false) -> 3 };" }, - { true, - true, - 0, - 0, - 3, - "return match (flag), (other) { (true), (true) -> 3, (true), (_) " - "-> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, - { true, - false, - 0, - 0, - 2, - "return match (flag), (other) { (true), (true) -> 3, (true), (_) " - "-> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, - { false, - true, - 0, - 0, - 1, - "return match (flag), (other) { (true), (true) -> 3, (true), (_) " - "-> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, - { false, - false, - 0, - 0, - 0, - "return match (flag), (other) { (true), (true) -> 3, (true), (_) " - "-> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, - { false, - false, - 0, - 0, - 101, - "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> " - "300 }; return r + n;" }, - { false, - false, - 1, - 0, - 202, - "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> " - "300 }; return r + n;" }, - { false, - false, - 5, - 0, - 306, - "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> " - "300 }; return r + n;" }, - { false, - false, - 1, - 0, - 1, - "int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, " - "(_), (_) -> 0 };" }, - { false, - false, - 2, - 0, - 0, - "int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, " - "(_), (_) -> 0 };" }, - { false, - false, - 1, - 0, - 1001, - "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> " - "10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + " - "calls;" }, - { false, - false, - 2, - 0, - 2010, - "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> " - "10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + " - "calls;" }, - { false, - false, - 3, - 0, - 3000, - "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> " - "10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + " - "calls;" }, - { true, - false, - 1, - 0, - 1, - "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" }, - { false, - false, - 1, - 0, - 2, - "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" }, - { true, - false, - 9, - 0, - 3, - "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" }, - { false, - false, - 1, - 5, - 1, - "return match (left) { 1 if right -> 1, _ -> 0 };" }, - { false, - false, - 1, - 0, - 0, - "return match (left) { 1 if right -> 1, _ -> 0 };" }, - { false, - false, - 2, - 5, - 0, - "return match (left) { 1 if right -> 1, _ -> 0 };" }, - { false, - false, - 1, - 0, - 1001, - "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += " - "10), _ -> (n += 100) }; return r * 1000 + n;" }, - { false, - false, - 2, - 0, - 10010, - "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += " - "10), _ -> (n += 100) }; return r * 1000 + n;" }, - { false, - false, - 3, - 0, - 100100, - "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += " - "10), _ -> (n += 100) }; return r * 1000 + n;" }, - { false, - false, - 2, - 0, - 1, - "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" }, - { false, - false, - 3, - 0, - 2, - "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" }, - { false, - false, - 4, - 0, - 0, - "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" }, - { false, - false, - 0, - 0, - 1, - "long wide = 5; return match (wide) { 5 -> 1, _ -> 0 };" }, - { false, - false, - 1, - 0, - 0, - "long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 " - "? 1 : 0;" }, - { false, - false, - 2, - 0, - 1, - "long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 " - "? 1 : 0;" }, - { false, - false, - 1, - 0, - 5, - "int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } " - "return r;" }, - { false, - false, - 4, - 0, - 8, - "int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } " - "return r;" }, - { true, - false, - 1, - 0, - 5, - "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> " - "match (right) { 0 -> 7, _ -> 8 } };" }, - { false, - false, - 1, - 0, - 6, - "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> " - "match (right) { 0 -> 7, _ -> 8 } };" }, - { false, - false, - 2, - 0, - 7, - "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> " - "match (right) { 0 -> 7, _ -> 8 } };" }, - { false, - false, - 2, - 3, - 8, - "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> " - "match (right) { 0 -> 7, _ -> 8 } };" }, - { false, - false, - 1, - 4, - 9, - "return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { " - "int t = right * 3; t - 1 } };" }, - { false, - false, - 2, - 4, - 11, - "return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { " - "int t = right * 3; t - 1 } };" }, - { false, - false, - 10, - 0, - 8, - "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 " - "-> { continue; }, 5 -> { break; }, _ -> { total += i; } } } " - "return total;" }, - { false, - false, - 3, - 0, - 1, - "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 " - "-> { continue; }, 5 -> { break; }, _ -> { total += i; } } } " - "return total;" }, - { false, - false, - 0, - 0, - 0, - "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 " - "-> { continue; }, 5 -> { break; }, _ -> { total += i; } } } " - "return total;" }, - { false, - false, - 0, - 0, - 40, - "int n = 0; int r = while (true) { n += 1; match (n) { 4 -> { " - "break n * 10; }, _ -> { } } }; return r;" }, - { false, - false, - 1, - 0, - 100, - "match (left) { 1 -> { return 100; }, _ -> { } } return 5;" }, - { false, - false, - 2, - 0, - 5, - "match (left) { 1 -> { return 100; }, _ -> { } } return 5;" }, - { false, false, 1, 0, 5, "match (left) { } return 5;" }, - { false, - false, - 3, - 5, - 5, - "int r = if (left > right) { left } else { right }; return r;" }, - { false, - false, - 9, - 2, - 9, - "int r = if (left > right) { left } else { right }; return r;" }, - { true, - false, - 4, - 0, - 9, - "int r = if (flag) { int t = left * 2; t + 1 } else { int t = " - "right * 3; t - 1 }; return r;" }, - { false, - false, - 0, - 5, - 14, - "int r = if (flag) { int t = left * 2; t + 1 } else { int t = " - "right * 3; t - 1 }; return r;" }, - { true, - false, - 0, - 0, - 10001, - "int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 " - "}; return r * 1000 + n;" }, - { false, - false, - 0, - 0, - 20100, - "int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 " - "}; return r * 1000 + n;" }, - { true, - true, - 1, - 2, - 101, - "return (if (flag) { left } else { right }) + (if (other) { 100 " - "} else { 200 });" }, - { false, - false, - 1, - 2, - 202, - "return (if (flag) { left } else { right }) + (if (other) { 100 " - "} else { 200 });" }, - { false, - false, - 3, - 0, - 6, - "guard (left > 0) else { return 0 - 1; } return left * 2;" }, - { false, - false, - 0, - 0, - -1, - "guard (left > 0) else { return 0 - 1; } return left * 2;" }, - { false, - false, - 6, - 0, - 6, - "int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == " - "0) else { continue; } total += i; } return total;" }, - { false, - false, - 1, - 0, - 0, - "int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == " - "0) else { continue; } total += i; } return total;" }, - { false, - false, - 4, - 0, - 4, - "int n = 0; while (true) { guard (n < left) else { break; } n += " - "1; } return n;" }, - { false, - false, - 0, - 0, - 0, - "int n = 0; while (true) { guard (n < left) else { break; } n += " - "1; } return n;" }, - { false, - false, - 2, - 3, - 13, - "int total = 0; { int part = left * 2; total += part; } { int " - "part = right * 3; total += part; } return total;" }, - { false, - false, - 0, - 0, - 0, - "int total = 0; { int part = left * 2; total += part; } { int " - "part = right * 3; total += part; } return total;" }, - { false, - false, - 3, - 0, - 40, - "int n = 0; while (true) { { n += 1; if (n > left) { break; } } " - "} { { return n * 10; } }" }, - { false, - false, - 0, - 0, - 10, - "int n = 0; while (true) { { n += 1; if (n > left) { break; } } " - "} { { return n * 10; } }" }, - { false, - false, - 4, - 0, - 10, - "int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { " - "continue; } } { int step = i * 2; total += step; } } return " - "total;" }, - { false, - false, - 1, - 0, - 0, - "int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { " - "continue; } } { int step = i * 2; total += step; } } return " - "total;" }, - { false, - false, - 5, - 0, - 6, - "int n = left; guard ((n += 1) > 3) else { return n * 10; } " - "return n;" }, - { false, - false, - 1, - 0, - 20, - "int n = left; guard ((n += 1) > 3) else { return n * 10; } " - "return n;" }, - } }; + constexpr auto kCases = std::to_array({ +#include "Generated/SelectionCases.inc" + }); - [[nodiscard]] auto - Truth(bool value) -> std::string - { - // `Id` is recursive, so the optimizer cannot fold the arguments - // away and the body really executes on run-time values. - return value ? "Id(1) > 0" : "Id(0) > 0"; - } + // The methods a body may call besides `Run` itself. + constexpr std::string_view kHelpers + = " public static int Twice(_ int value) { return value + " + "value; }\n"; - [[nodiscard]] auto - Program(const Case &entry) -> std::string - { - return "namespace Fuzz;\n" - "class Program {\n" - " public static int Id(_ int n) { return n > 0 ? 1 + " - "Id(n - 1) : 0; }\n" - " public static int Twice(_ int value) { return value + " - "value; }\n" - " public static int Run(_ bool flag, _ bool other, _ int " - "left, _ int right) {\n " - + std::string(entry.body) - + "\n }\n" - " public static int Evaluate() { return Run(" - + Truth(entry.flag) + ", " + Truth(entry.other) + ", Id(" - + std::to_string(entry.left) + "), Id(" - + std::to_string(entry.right) + ")); }\n}\n"; - } - // A match with far more arms than the lowering nests in one group. - // Arm `index` yields `index * 3 + 1` and the catch-all yields 0. + // A match with many arms. Arm `index` yields `index * 3 + 1` and the + // catch-all yields 0. [[nodiscard]] auto WideMatchBody(int arms) -> std::string { @@ -638,8 +50,7 @@ namespace Visual::XSharp::Fuzzing return body + "_ -> 0 };"; } - // The same table written as an `else if` chain, which reaches Core - // as one level of nesting per link. + // The same table written as an `else if` chain. [[nodiscard]] auto ElseIfChainBody(int links) -> std::string { @@ -650,53 +61,110 @@ namespace Visual::XSharp::Fuzzing + std::to_string(index * 3 + 1) + "; }"; return body + " return 0;"; } + + // `if (left > 0) { if (left > 1) { ... total += 1; } }` + [[nodiscard]] auto + NestedIfBody(int levels) -> std::string + { + std::string body = "int total = 0; "; + for (int index = 0; index < levels; ++index) + body += "if (left > " + std::to_string(index) + ") { "; + body += "total += 1; "; + for (int index = 0; index < levels; ++index) + body += "} "; + return body + "return total;"; + } + + // `Id(Id(...Id(left)...))`: calls nested in arguments, which is + // the shape that costs most compiler stack for each level. + [[nodiscard]] auto + NestedCallBody(int levels) -> std::string + { + std::string body = "return "; + for (int index = 0; index < levels; ++index) + body += "Id("; + body += "left"; + body.append(static_cast(levels), ')'); + return body + ";"; + } + + // `left + left + ... + left` + [[nodiscard]] auto + SumBody(int operands) -> std::string + { + std::string body = "return left"; + for (int index = 1; index < operands; ++index) + body += " + left"; + return body + ";"; + } + + // The value of arm or link `index` in the tables above. + [[nodiscard]] constexpr auto + TableValue(int index) -> std::int64_t + { + return std::int64_t{ index } * 3 + 1; + } } // namespace void ExerciseBranchingCases() { - for (const auto &entry : kCases) - { - llvm::errs() << "Branching execution: " << entry.body << '\n'; - ExerciseExpectedValue(Program(entry), entry.expected); - } - // The subjects select the first arm, the arms on both sides of a - // group boundary, arms deep in later groups, and the catch-all. A + ExerciseExecutionCases("Branching execution", kCases, kHelpers); + + // Programs whose size is the point. Every body is compiled once and + // all of its runs are checked by that one program. + std::vector large; + + // The subjects select the first arm, arms around a multiple of + // sixteen, an arm in the middle, the last arm and the catch-all. A // lowering that nested one level per arm would overflow the stack // of the stages after Core long before this many arms. constexpr int kWideArms = 200; - constexpr std::array subjects{ 0, 15, 16, 17, 150, 199, 200 }; const auto wide = WideMatchBody(kWideArms); - for (const auto subject : subjects) - { - const Case entry{ false, + for (const auto subject : { 0, 15, 16, 17, 150, 199, 200 }) + large.push_back({ false, false, subject, 0, - subject < kWideArms ? subject * 3 + 1 : 0, - wide }; - llvm::errs() << "Branching execution: match of " << kWideArms - << " arms on " << subject << '\n'; - ExerciseExpectedValue(Program(entry), entry.expected); - } + subject < kWideArms ? TableValue(subject) : 0, + wide }); + // An `else if` chain of 300 links. The native wire reader, the Core // verifier and the CorePrep adapter walk a chain in a loop; when // they recursed, 150 links overflowed the stack. Both CorePrep // lowerings are compared on it as on every other program here. constexpr int kChainLinks = 300; - constexpr std::array selected{ 0, 149, 150, 299, 300 }; const auto chain = ElseIfChainBody(kChainLinks); - for (const auto subject : selected) - { - const Case entry{ false, + for (const auto subject : { 0, 149, 150, 299, 300 }) + large.push_back({ false, false, subject, 0, - subject < kChainLinks ? subject * 3 + 1 : 0, - chain }; - llvm::errs() << "Branching execution: else-if chain of " - << kChainLinks << " links on " << subject << '\n'; - ExerciseExpectedValue(Program(entry), entry.expected); - } + subject < kChainLinks ? TableValue(subject) : 0, + chain }); + + // Programs at the nesting limits of the frontend: a statement at + // level 256 and an expression at level 1024. They compile only on + // the compiler stack, which the smoke program runs on like `vxs`. + constexpr int kLevels = 255; + const auto nested = NestedIfBody(kLevels); + for (const auto argument : { kLevels, kLevels - 1 }) + large.push_back({ false, + false, + argument, + 0, + argument == kLevels ? 1 : 0, + nested }); + // 1023 calls nested in each other's arguments put the innermost + // operand at expression level 1024. + constexpr int kNestedCalls = 1023; + const auto calls = NestedCallBody(kNestedCalls); + large.push_back({ false, false, 3, 0, 3, calls }); + constexpr int kOperands = 1024; + const auto sum = SumBody(kOperands); + large.push_back( + { false, false, 3, 0, std::int64_t{ 3 } * kOperands, sum }); + + ExerciseExecutionCases("Branching execution", large, kHelpers); } } // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/Cases/Leaving.cases b/Compiler/Fuzzing/Cases/Leaving.cases new file mode 100644 index 00000000..3c27132d --- /dev/null +++ b/Compiler/Fuzzing/Cases/Leaving.cases @@ -0,0 +1,317 @@ +# SPDX-FileCopyrightText: 2026 Progmasoft +# SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +# +# Expressions that leave instead of yielding a value: `return`, `break` and +# `continue` out of blocks used as values, in loop bodies, loop headers and +# loop expressions. +# +# This file is the source of the cases. `go -C helpers run ./cmd/execution-cases +# generate` writes them into the Haskell and the C++ test tables, and +# `check` fails when a table is not what this file generates. +# +# A case is the body of `int Run(bool flag, bool other, int left, int right)` +# and its runs. The expected value of every run is written by hand from the +# language rules, never taken from a compiler. +# +# body: +# run: plain -> +# run: flags -> +# +# Lines that start with `#` before a body are its comment. + +# An expression every branch of which leaves never yields a value. +body: int r = if (flag) { return 1; } else { return 2; }; return r + 50; +run: flags true false 0 0 -> 1 +run: plain 0 0 -> 2 + +body: int r = match (left) { 0 -> { return 10; }, _ -> { return 20; } }; return r + 1; +run: plain 0 0 -> 10 +run: plain 5 0 -> 20 + +body: return Twice(if (flag) { return 7; } else { return 9; }); +run: flags true false 0 0 -> 7 +run: plain 0 0 -> 9 + +body: int r = 1 + (if (flag) { return 1; } else { return 2; }); return r; +run: flags true false 0 0 -> 1 +run: plain 0 0 -> 2 + +body: int t = 0; int i = 0; while (i < 5) { i += 1; int q = if (i > left) { break; } else { continue; }; t += q; } return t * 10 + i; +run: plain 2 0 -> 3 +run: plain 9 0 -> 5 + +body: bool b = flag && (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4; +run: flags true false 5 0 -> 1 +run: flags true false 0 0 -> 2 +run: plain 5 0 -> 4 + +body: bool b = flag || (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4; +run: flags true false 5 0 -> 3 +run: plain 5 0 -> 1 +run: plain 0 0 -> 2 + +body: int n = left ?: (if (flag) { return 100; } else { return 200; }); return n; +run: plain 5 0 -> 5 +run: flags true false 0 0 -> 100 +run: plain 0 0 -> 200 + +body: int r = if (left > 0) { 5 } else { if (flag) { return 1; } else { return 2; } }; return r; +run: plain 3 0 -> 5 +run: flags true false 0 0 -> 1 +run: plain 0 0 -> 2 + +body: int n = left; do { n += 1; } while (if (n > 3) { return n; } else { return 0 - n; }); return 99; +run: plain 5 0 -> 6 +run: plain 1 0 -> -2 + +# The condition of a do/while that never completes still runs after +# the body and after a continue, with its effects, and is skipped by +# a break. +body: int t = 0; int n = 0; do { n += 1; if (n == left) { continue; } t += 10; } while (if ((t += 1) > 100) { return 1; } else { return t * 100 + n; }); return 99; +run: plain 1 0 -> 101 +run: plain 5 0 -> 1101 + +body: int t = 0; do { if (left > 0) { break; } t += 5; } while (if ((t += 1) > 0) { return t; } else { return 0; }); return t + 1000; +run: plain 1 0 -> 1000 +run: plain 0 0 -> 6 + +body: int t = 0; for (int i = 0; i < 3; i += 1) { do { t += 1; } while (if (t > left) { return t * 10 + i; } else { return t; }); } return 77; +run: plain 0 0 -> 10 +run: plain 5 0 -> 1 + +body: int t = 0; int i = 0; while (i < 3) { i += 1; int j = 0; do { j += 1; if (j < 3) { continue; } t += j; } while (if (i > left) { return t * 10 + i; } else { j < 4 }); } return t; +run: plain 0 0 -> 1 +run: plain 9 0 -> 21 + +# A transfer in a value block targets the innermost loop only. +body: int t = 0; int i = 0; while (i < 5) { i += 1; int j = 0; while (j < 3) { j += 1; t += if (j == 2) { break; } else { 1 }; } } return t * 10 + i; +run: plain 0 0 -> 55 + +body: int t = 0; int i = 0; while (i < 5) { i += 1; int j = 0; while (j < 3) { j += 1; t += if (j == 2) { continue; } else { 1 }; } } return t * 10 + i; +run: plain 0 0 -> 105 + +body: int t = 0; while (t < 10) { t += 1; int q = while (true) { break t; }; if (q > left) { break; } } return t; +run: plain 3 0 -> 4 +run: plain 20 0 -> 10 + +# A short-circuit operator evaluates its left operand, with its +# effects, and reaches a right operand that never completes only +# when the left one does not decide. +body: int t = 0; bool b = (t += 1) > 0 && (if (left > 0) { return t * 10; } else { return t * 100; }); return 7; +run: plain 1 0 -> 10 +run: plain 0 0 -> 100 + +body: int t = 0; bool b = (t += 1) > 5 && (if (left > 0) { return t * 10; } else { return t * 100; }); return t + (b ? 1000 : 0); +run: plain 1 0 -> 1 +run: plain 0 0 -> 1 + +body: int t = 0; bool b = (t += left) > 0 || (if (flag) { return t * 10; } else { return 50 + t; }); return b ? t + 1 : 7; +run: plain 3 0 -> 4 +run: flags true false 0 0 -> 0 +run: plain 0 0 -> 50 + +body: int t = 0; int q = (t += left) ? t * 2 : (if (flag) { return 0 - 1; } else { return 0 - 2; }); return q + t; +run: plain 4 0 -> 12 +run: flags true false 0 0 -> -1 +run: plain 0 0 -> -2 + +# Operands are evaluated left to right up to the one that never +# completes; the ones after it are not evaluated. +body: int t = 0; int r = (t += 1) + (if (flag) { return t * 10; } else { return t * 100; }) + (t += 50); return r; +run: flags true false 0 0 -> 10 +run: plain 0 0 -> 100 + +body: int t = 0; return Twice(Twice(t += 3) + (if (flag) { return t; } else { return t * 2; })); +run: flags true false 0 0 -> 3 +run: plain 0 0 -> 6 + +body: int t = left; t = Twice(t += 1) + (if (flag) { return t; } else { return t + 100; }); return 0 - 1; +run: flags true false 4 0 -> 5 +run: plain 4 0 -> 105 + +# A guard runs only when its patterns accept, and a guarded arm that +# leaves does not make the match leave. +body: int t = 0; int r = match (left) { 0 if (t += 1) > 5 -> { return 1; }, 0 -> { return 10 + t; }, _ -> { return 20 + t; } }; return r; +run: plain 0 0 -> 11 +run: plain 3 0 -> 20 + +body: int r = match (left) { _ if flag -> { return 1; }, _ -> 5 }; return r + 1; +run: flags true false 0 0 -> 1 +run: plain 0 0 -> 6 + +# A break in the condition of a loop leaves that loop, once, with the +# effects of the condition up to it. +body: int n = 0; while (if (n >= left) { break; } else { true }) { n += 1; } return n * 10 + 1; +run: plain 3 0 -> 31 +run: plain 0 0 -> 1 + +body: int c = 0; int n = 0; while (if ((c += 1) > left) { break; } else { true }) { n += 1; } return c * 100 + n; +run: plain 2 0 -> 302 +run: plain 0 0 -> 100 + +body: int n = 0; do { n += 1; } while (if (n >= left) { break; } else { true }); return n; +run: plain 3 0 -> 3 +run: plain 0 0 -> 1 + +body: int t = 0; for (int i = 0; if (i > left) { break; } else { i < 10 }; i += 1) { t += i; } return t; +run: plain 2 0 -> 3 +run: plain 20 0 -> 45 + +body: int t = 0; for (int i = 0; i < 3; i += 1) { int j = 0; while (if (j == 2) { break; } else { true }) { j += 1; t += 1; } t += 10; } return t; +run: plain 0 0 -> 36 + +body: int t = 0; int i = 0; while (i < 4) { i += 1; int j = 0; while (if (j >= i) { break; } else { true }) { j += 1; if (j == 2) { continue; } t += 1; } } return t * 10 + i; +run: plain 0 0 -> 74 + +# A continue in the update of a loop ends the update; the condition +# is tested next, and the update is not run again for that pass. +body: int t = 0; int skips = 0; for (int i = 0; i < 6; i += if (i == left && skips == 0) { skips += 1; continue; } else { 1 }) { t += 1; } return t * 10 + skips; +run: plain 2 0 -> 71 +run: plain 9 0 -> 60 + +body: int t = 0; int u = 0; for (int i = 0; i < 4; i += 1, u += if (i == left) { continue; } else { 10 }) { t += 1; } return u * 10 + t; +run: plain 2 0 -> 304 +run: plain 9 0 -> 404 + +body: int t = 0; for (int i = 0; i < 3; i += 1) { int c = 0; for (int j = 0; j < 4; j += if (c == 0 && j == left) { c += 1; continue; } else { 1 }) { t += 1; } t += c * 100; } return t; +run: plain 1 0 -> 315 +run: plain 7 0 -> 12 + +# A continue in the condition of a loop abandons the rest of the +# condition and evaluates the condition of that loop again. The body +# does not run in between, and neither does the update of a for. +body: int c = 0; int n = 0; while (if ((c += 1) < left) { continue; } else { n < 2 }) { n += 1; } return c * 100 + n; +run: plain 3 0 -> 502 +run: plain 0 0 -> 302 + +body: int u = 0; int c = 0; int t = 0; for (int i = 0; if ((c += 1) == left) { continue; } else { i < 3 }; i += 1, u += 1) { t += 10; } return c * 1000 + u * 100 + t; +run: plain 2 0 -> 5330 +run: plain 9 0 -> 4330 + +body: int c = 0; int n = 0; do { n += 1; } while (if ((c += 1) < left) { continue; } else { n < 3 }); return c * 100 + n; +run: plain 3 0 -> 503 +run: plain 0 0 -> 303 + +# It targets the loop the condition belongs to, not one around it. +body: int t = 0; int c = 0; for (int i = 0; i < 3; i += 1) { int j = 0; int once = 0; while (if (once == 0) { once = 1; c += 1; continue; } else { j < 2 }) { j += 1; t += 1; } t += 10; } return t * 100 + c; +run: plain 0 0 -> 3603 + +# A condition every path of which leaves: it repeats until it breaks. +body: int c = 0; while (if ((c += 1) < left) { continue; } else { break; }) { c += 100; } return c; +run: plain 4 0 -> 4 +run: plain 0 0 -> 1 + +body: int c = 0; int u = 0; for (int i = 0; if ((c += 1) < left) { continue; } else { break; }; i += 1, u += 1) { c += 100; } return c * 10 + u; +run: plain 4 0 -> 40 +run: plain 0 0 -> 10 + +# A break in the update of a loop leaves that loop, with the effects +# of the update up to it and without the rest of the update. +body: int t = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }) { t += 1; } return t; +run: plain 2 0 -> 3 +run: plain 20 0 -> 10 + +body: int t = 0; int u = 0; for (int i = 0; i < 10; u += 1, i += if (i == left) { break; } else { 1 }, u += 100) { t += 1; } return u * 10 + t; +run: plain 1 0 -> 1022 +run: plain 20 0 -> 10110 + +# A break and a continue in one update clause. +body: int t = 0; int s = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }, s += if (i == 2) { continue; } else { 1 }, s += 10) { t += 1; } return s * 100 + t; +run: plain 4 0 -> 3305 +run: plain 20 0 -> 9910 + +# It leaves the loop the update belongs to, not one around it. +body: int t = 0; for (int a = 0; a < 3; a += 1) { for (int b = 0; b < 5; b += if (b == 1) { break; } else { 1 }) { t += 1; } t += 10; } return t; +run: plain 0 0 -> 36 + +# In the update of a loop used as an expression it carries the value. +body: int r = for (int i = 0; ; i += if (i == left) { break i * 10; } else { 1 }) { if (i > 5) { break 99; } }; return r; +run: plain 3 0 -> 30 +run: plain 9 0 -> 99 + +# A value block may carry a value out of the loop expression around +# it; the value goes to that loop, not to one around it. +body: int r = while (true) { int q = if (flag) { break left + 1; } else { 2 }; break q; }; return r; +run: flags true false 4 0 -> 5 +run: plain 4 0 -> 2 + +body: int r = while (true) { int q = match (left) { 0 -> { break 100; }, int n -> n * 2 }; break q; }; return r; +run: plain 0 0 -> 100 +run: plain 4 0 -> 8 + +body: int t = 0; int r = for (int i = 0; ; i += 1) { int inner = while (true) { t += 1; int q = if (t > left) { break t * 2; } else { 0 }; t += q; }; if (inner > 0) { break inner + i; } }; return r * 10 + t; +run: plain 2 0 -> 63 +run: plain 0 0 -> 21 + +body: int t = 0; int r = Twice(while (true) { t += 1; int q = (t += 10) + (if (t > left) { break t; } else { 1 }); t += q; }); return r * 100 + t; +run: plain 5 0 -> 2211 +run: plain 30 0 -> 6834 + +# A return leaves the method from a loop expression, directly and +# through a value block. +body: int r = while (true) { int q = if (flag) { return 77; } else { 2 }; break q + left; }; return r; +run: flags true false 0 0 -> 77 +run: plain 3 0 -> 5 + +body: int r = while (true) { if (flag) { return 1; } break 2; }; return r + 10; +run: flags true false 0 0 -> 1 +run: plain 0 0 -> 12 + +body: int r = while (true) { return left; }; return r + 1; +run: plain 6 0 -> 6 + +body: int t = 0; int r = while (true) { t += 1; int q = for (int i = 0; ; i += 1) { if (i + t > left) { return i * 10 + t; } if (i == 2) { break i; } }; if (t == 3) { break q; } }; return 0 - r; +run: plain 1 0 -> 11 +run: plain 2 0 -> 21 +run: plain 99 0 -> -2 + +# A block used as a value may leave instead of yielding one. +body: int r = if (left > 5) { return 100; } else { left * 2 }; return r + 1; +run: plain 9 0 -> 100 +run: plain 3 0 -> 7 + +body: int r = match (left) { 0 -> { return 50; }, int n -> n + 1 }; return r * 2; +run: plain 0 0 -> 50 +run: plain 4 0 -> 10 + +body: return Twice(if (flag) { return 7; } else { left }); +run: flags true false 4 0 -> 7 +run: plain 4 0 -> 8 + +body: int t = 0; int i = 0; while (i < 10) { i += 1; t += if (i > left) { break; } else { i }; } return t * 100 + i; +run: plain 3 0 -> 604 +run: plain 0 0 -> 1 + +body: int t = 0; for (int i = 0; i < 6; i += 1) { t += match (i) { 2 -> { continue; }, int n -> { if (n == left) { continue; } else { n } } }; } return t; +run: plain 4 0 -> 9 +run: plain 9 0 -> 13 + +# A guard block that leaves through a match, and through continue. +body: guard (left > 0) else { match (flag) { true -> { return 1; }, false -> { return 2; } } } return left + 10; +run: flags true false 0 0 -> 1 +run: plain 0 0 -> 2 +run: plain 5 0 -> 15 + +body: int t = 0; int i = 0; while (i < left) { i += 1; guard (i \= 2) else { continue; } t += i; } return t; +run: plain 4 0 -> 8 +run: plain 1 0 -> 1 + +body: int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } return r; +run: plain 1 0 -> 10 +run: plain 2 0 -> 20 +run: plain 7 0 -> 0 + +body: return match (left) { int n if n > 10 -> n * 2, int n if n > 5 -> n + 1, _ -> 0 }; +run: plain 20 0 -> 40 +run: plain 7 0 -> 8 +run: plain 3 0 -> 0 + +body: return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 }; +run: plain 1 1 -> 11 +run: plain 1 5 -> 10 +run: plain 4 1 -> 1 +run: plain 4 4 -> 0 + +body: return match (flag) { true -> 1, false -> 2 }; +run: flags true false 0 0 -> 1 +run: flags false false 0 0 -> 2 diff --git a/Compiler/Fuzzing/Cases/Selection.cases b/Compiler/Fuzzing/Cases/Selection.cases new file mode 100644 index 00000000..6e961f54 --- /dev/null +++ b/Compiler/Fuzzing/Cases/Selection.cases @@ -0,0 +1,207 @@ +# SPDX-FileCopyrightText: 2026 Progmasoft +# SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +# +# Which arm, block or body is selected, and how often each part runs. +# +# This file is the source of the cases. `go -C helpers run ./cmd/execution-cases +# generate` writes them into the Haskell and the C++ test tables, and +# `check` fails when a table is not what this file generates. +# +# A case is the body of `int Run(bool flag, bool other, int left, int right)` +# and its runs. The expected value of every run is written by hand from the +# language rules, never taken from a compiler. +# +# body: +# run: plain -> +# run: flags -> +# +# Lines that start with `#` before a body are its comment. + +body: return match (left) { 1 -> 10, 2 -> 20, _ -> 30 }; +run: plain 1 0 -> 10 +run: plain 2 0 -> 20 +run: plain 5 0 -> 30 + +# Negative constants. +body: int v = 0 - left; return match (v) { -1 -> 10, -2 -> 20, 0 -> 5, _ -> 7 }; +run: plain 1 0 -> 10 +run: plain 2 0 -> 20 +run: plain 0 0 -> 5 +run: plain 3 0 -> 7 + +body: int v = 0 - left - 9223372036854775807; return match (v) { -9223372036854775808 -> 1, -9223372036854775807 -> 2, _ -> 0 }; +run: plain 1 0 -> 1 +run: plain 0 0 -> 2 + +body: int v = right - left; return match (v), (flag) { (-1), (true) -> 1, (-1), (_) -> 2, (_), (_) -> 3 }; +run: flags true false 3 2 -> 1 +run: flags false false 3 2 -> 2 +run: flags true false 2 2 -> 3 + +# Two bool subjects covered without a catch-all arm. +body: return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 }; +run: flags true true 0 0 -> 1 +run: flags true false 0 0 -> 1 +run: flags false true 0 0 -> 2 +run: flags false false 0 0 -> 3 + +body: return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 }; +run: flags true true 0 0 -> 3 +run: flags true false 0 0 -> 2 +run: flags false true 0 0 -> 1 +run: flags false false 0 0 -> 0 + +# The subject is evaluated once, whichever arm accepts. +body: int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> 300 }; return r + n; +run: plain 0 0 -> 101 +run: plain 1 0 -> 202 +run: plain 5 0 -> 306 + +# Subjects are evaluated left to right, before any arm is tested. +body: int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, (_), (_) -> 0 }; +run: plain 1 0 -> 1 +run: plain 2 0 -> 0 + +# A guard runs only when the patterns of its arm accept. +body: int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> 10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + calls; +run: plain 1 0 -> 1001 +run: plain 2 0 -> 2010 +run: plain 3 0 -> 3000 + +# A guard that is false passes the value on to the arms after it. +body: return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 }; +run: flags true false 1 0 -> 1 +run: flags false false 1 0 -> 2 +run: flags true false 9 0 -> 3 + +# A numeric guard is tested in Boolean context. +body: return match (left) { 1 if right -> 1, _ -> 0 }; +run: plain 1 5 -> 1 +run: plain 1 0 -> 0 +run: plain 2 5 -> 0 + +# Only the body of the accepting arm runs. +body: int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += 10), _ -> (n += 100) }; return r * 1000 + n; +run: plain 1 0 -> 1001 +run: plain 2 0 -> 10010 +run: plain 3 0 -> 100100 + +body: return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 }; +run: plain 2 0 -> 1 +run: plain 3 0 -> 2 +run: plain 4 0 -> 0 + +body: long wide = 5; return match (wide) { 5 -> 1, _ -> 0 }; +run: plain 0 0 -> 1 + +body: long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 ? 1 : 0; +run: plain 1 0 -> 0 +run: plain 2 0 -> 1 + +body: int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } return r; +run: plain 1 0 -> 5 +run: plain 4 0 -> 8 + +body: return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } }; +run: flags true false 1 0 -> 5 +run: flags false false 1 0 -> 6 +run: plain 2 0 -> 7 +run: plain 2 3 -> 8 + +body: return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { int t = right * 3; t - 1 } }; +run: plain 1 4 -> 9 +run: plain 2 4 -> 11 + +# A statement arm may continue or leave the enclosing loop. +body: int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 -> { continue; }, 5 -> { break; }, _ -> { total += i; } } } return total; +run: plain 10 0 -> 8 +run: plain 3 0 -> 1 +run: plain 0 0 -> 0 + +# A statement arm may supply the value of an enclosing loop expression. +body: int n = 0; int r = while (true) { n += 1; match (n) { 4 -> { break n * 10; }, _ -> { } } }; return r; +run: plain 0 0 -> 40 + +body: match (left) { 1 -> { return 100; }, _ -> { } } return 5; +run: plain 1 0 -> 100 +run: plain 2 0 -> 5 + +body: match (left) { } return 5; +run: plain 1 0 -> 5 + +body: int r = if (left > right) { left } else { right }; return r; +run: plain 3 5 -> 5 +run: plain 9 2 -> 9 + +body: int r = if (flag) { int t = left * 2; t + 1 } else { int t = right * 3; t - 1 }; return r; +run: flags true false 4 0 -> 9 +run: flags false false 0 5 -> 14 + +# Only the selected block of an if expression runs. +body: int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 }; return r * 1000 + n; +run: flags true false 0 0 -> 10001 +run: flags false false 0 0 -> 20100 + +body: return (if (flag) { left } else { right }) + (if (other) { 100 } else { 200 }); +run: flags true true 1 2 -> 101 +run: flags false false 1 2 -> 202 + +body: guard (left > 0) else { return 0 - 1; } return left * 2; +run: plain 3 0 -> 6 +run: plain 0 0 -> -1 + +body: int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == 0) else { continue; } total += i; } return total; +run: plain 6 0 -> 6 +run: plain 1 0 -> 0 + +body: int n = 0; while (true) { guard (n < left) else { break; } n += 1; } return n; +run: plain 4 0 -> 4 +run: plain 0 0 -> 0 + +body: int total = 0; { int part = left * 2; total += part; } { int part = right * 3; total += part; } return total; +run: plain 2 3 -> 13 +run: plain 0 0 -> 0 + +body: int n = 0; while (true) { { n += 1; if (n > left) { break; } } } { { return n * 10; } } +run: plain 3 0 -> 40 +run: plain 0 0 -> 10 + +body: int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { continue; } } { int step = i * 2; total += step; } } return total; +run: plain 4 0 -> 10 +run: plain 1 0 -> 0 + +body: return match (left) { 1 -> 10 2 -> 20 _ -> 30 }; +run: plain 1 0 -> 10 +run: plain 2 0 -> 20 +run: plain 9 0 -> 30 + +# A pattern binding is a local of its arm and may be assigned. +body: return match (left) { int value if value > 2 -> { value = value * 2; value }, int value -> { value += 1; value } }; +run: plain 5 0 -> 10 +run: plain 1 0 -> 2 + +# Assigning a binding does not change the subject the later arms test. +body: int r = 0; match (left) { int value if (value = 7) > 9 -> { r = 1; }, 3 -> { r = 2; }, _ -> { r = 3; } } return r; +run: plain 3 0 -> 2 +run: plain 4 0 -> 3 + +body: int r = if (flag) { if (other) { 1 } else { 2 } } else { match (left) { 1 -> 10, _ -> 20 } }; return r; +run: flags true true 0 0 -> 1 +run: flags true false 0 0 -> 2 +run: flags false false 1 0 -> 10 +run: flags false false 2 0 -> 20 + +body: int n = 0; int r = if (flag) { if (other) { n += 1; } else { n += 2; } match (left) { 1 -> { n += 10; } } n } else { 0 }; return r; +run: flags true true 1 0 -> 11 +run: flags true false 2 0 -> 2 +run: flags false false 1 0 -> 0 + +body: match (left) { 1 -> { match (right) { 2 -> { return 12; } } }, _ -> { } } return 5; +run: plain 1 2 -> 12 +run: plain 1 3 -> 5 +run: plain 2 2 -> 5 + +# A guard condition with a store runs once, before the block decision. +body: int n = left; guard ((n += 1) > 3) else { return n * 10; } return n; +run: plain 5 0 -> 6 +run: plain 1 0 -> 20 diff --git a/Compiler/Fuzzing/Corpus/parser/unbalanced-nesting.seed b/Compiler/Fuzzing/Corpus/parser/unbalanced-nesting.seed new file mode 100644 index 00000000..12cb1968 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/parser/unbalanced-nesting.seed @@ -0,0 +1 @@ +class Program { public static int Evaluate() { int value = 1; if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { if (value > 0) { return ((((((((value + 1) + 1) + 1) + 1) + 1); } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } \ No newline at end of file diff --git a/Compiler/Fuzzing/Corpus/source/deep-nesting.seed b/Compiler/Fuzzing/Corpus/source/deep-nesting.seed new file mode 100644 index 00000000..9f4f248b --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/deep-nesting.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int total = 0; int value = 300; if (value > 0) { if (value > 1) { if (value > 2) { if (value > 3) { if (value > 4) { if (value > 5) { if (value > 6) { if (value > 7) { if (value > 8) { if (value > 9) { if (value > 10) { if (value > 11) { if (value > 12) { if (value > 13) { if (value > 14) { if (value > 15) { if (value > 16) { if (value > 17) { if (value > 18) { if (value > 19) { if (value > 20) { if (value > 21) { if (value > 22) { if (value > 23) { if (value > 24) { if (value > 25) { if (value > 26) { if (value > 27) { if (value > 28) { if (value > 29) { if (value > 30) { if (value > 31) { if (value > 32) { if (value > 33) { if (value > 34) { if (value > 35) { if (value > 36) { if (value > 37) { if (value > 38) { if (value > 39) { if (value > 40) { if (value > 41) { if (value > 42) { if (value > 43) { if (value > 44) { if (value > 45) { if (value > 46) { if (value > 47) { total += 1; } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } return total; } } diff --git a/Compiler/Fuzzing/Corpus/source/leaving-value-blocks.seed b/Compiler/Fuzzing/Corpus/source/leaving-value-blocks.seed new file mode 100644 index 00000000..1aee8332 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/leaving-value-blocks.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; int total = 0; for (int i = 0; i < 8; i += 1) { total += match (i) { 2 -> { continue; }, 6 -> { break; }, int n -> { if (n == value) { continue; } else { n } } }; } int r = if (total > 100) { return 0 - 1; } else { total }; guard (r > 0) else { match (r == 0) { true -> { return 1; }, false -> { return 2; } } } guard (r < 100) else { while (true) { } } return r; } } diff --git a/Compiler/Fuzzing/Corpus/source/long-else-if-chain.seed b/Compiler/Fuzzing/Corpus/source/long-else-if-chain.seed new file mode 100644 index 00000000..755ff1dc --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/long-else-if-chain.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 399; if (value == 0) { return 1; } else if (value == 1) { return 4; } else if (value == 2) { return 7; } else if (value == 3) { return 10; } else if (value == 4) { return 13; } else if (value == 5) { return 16; } else if (value == 6) { return 19; } else if (value == 7) { return 22; } else if (value == 8) { return 25; } else if (value == 9) { return 28; } else if (value == 10) { return 31; } else if (value == 11) { return 34; } else if (value == 12) { return 37; } else if (value == 13) { return 40; } else if (value == 14) { return 43; } else if (value == 15) { return 46; } else if (value == 16) { return 49; } else if (value == 17) { return 52; } else if (value == 18) { return 55; } else if (value == 19) { return 58; } else if (value == 20) { return 61; } else if (value == 21) { return 64; } else if (value == 22) { return 67; } else if (value == 23) { return 70; } else if (value == 24) { return 73; } else if (value == 25) { return 76; } else if (value == 26) { return 79; } else if (value == 27) { return 82; } else if (value == 28) { return 85; } else if (value == 29) { return 88; } else if (value == 30) { return 91; } else if (value == 31) { return 94; } else if (value == 32) { return 97; } else if (value == 33) { return 100; } else if (value == 34) { return 103; } else if (value == 35) { return 106; } else if (value == 36) { return 109; } else if (value == 37) { return 112; } else if (value == 38) { return 115; } else if (value == 39) { return 118; } else if (value == 40) { return 121; } else if (value == 41) { return 124; } else if (value == 42) { return 127; } else if (value == 43) { return 130; } else if (value == 44) { return 133; } else if (value == 45) { return 136; } else if (value == 46) { return 139; } else if (value == 47) { return 142; } else if (value == 48) { return 145; } else if (value == 49) { return 148; } else if (value == 50) { return 151; } else if (value == 51) { return 154; } else if (value == 52) { return 157; } else if (value == 53) { return 160; } else if (value == 54) { return 163; } else if (value == 55) { return 166; } else if (value == 56) { return 169; } else if (value == 57) { return 172; } else if (value == 58) { return 175; } else if (value == 59) { return 178; } else if (value == 60) { return 181; } else if (value == 61) { return 184; } else if (value == 62) { return 187; } else if (value == 63) { return 190; } else if (value == 64) { return 193; } else if (value == 65) { return 196; } else if (value == 66) { return 199; } else if (value == 67) { return 202; } else if (value == 68) { return 205; } else if (value == 69) { return 208; } else if (value == 70) { return 211; } else if (value == 71) { return 214; } else if (value == 72) { return 217; } else if (value == 73) { return 220; } else if (value == 74) { return 223; } else if (value == 75) { return 226; } else if (value == 76) { return 229; } else if (value == 77) { return 232; } else if (value == 78) { return 235; } else if (value == 79) { return 238; } else if (value == 80) { return 241; } else if (value == 81) { return 244; } else if (value == 82) { return 247; } else if (value == 83) { return 250; } else if (value == 84) { return 253; } else if (value == 85) { return 256; } else if (value == 86) { return 259; } else if (value == 87) { return 262; } else if (value == 88) { return 265; } else if (value == 89) { return 268; } else if (value == 90) { return 271; } else if (value == 91) { return 274; } else if (value == 92) { return 277; } else if (value == 93) { return 280; } else if (value == 94) { return 283; } else if (value == 95) { return 286; } else if (value == 96) { return 289; } else if (value == 97) { return 292; } else if (value == 98) { return 295; } else if (value == 99) { return 298; } else if (value == 100) { return 301; } else if (value == 101) { return 304; } else if (value == 102) { return 307; } else if (value == 103) { return 310; } else if (value == 104) { return 313; } else if (value == 105) { return 316; } else if (value == 106) { return 319; } else if (value == 107) { return 322; } else if (value == 108) { return 325; } else if (value == 109) { return 328; } else if (value == 110) { return 331; } else if (value == 111) { return 334; } else if (value == 112) { return 337; } else if (value == 113) { return 340; } else if (value == 114) { return 343; } else if (value == 115) { return 346; } else if (value == 116) { return 349; } else if (value == 117) { return 352; } else if (value == 118) { return 355; } else if (value == 119) { return 358; } else if (value == 120) { return 361; } else if (value == 121) { return 364; } else if (value == 122) { return 367; } else if (value == 123) { return 370; } else if (value == 124) { return 373; } else if (value == 125) { return 376; } else if (value == 126) { return 379; } else if (value == 127) { return 382; } else if (value == 128) { return 385; } else if (value == 129) { return 388; } else if (value == 130) { return 391; } else if (value == 131) { return 394; } else if (value == 132) { return 397; } else if (value == 133) { return 400; } else if (value == 134) { return 403; } else if (value == 135) { return 406; } else if (value == 136) { return 409; } else if (value == 137) { return 412; } else if (value == 138) { return 415; } else if (value == 139) { return 418; } else if (value == 140) { return 421; } else if (value == 141) { return 424; } else if (value == 142) { return 427; } else if (value == 143) { return 430; } else if (value == 144) { return 433; } else if (value == 145) { return 436; } else if (value == 146) { return 439; } else if (value == 147) { return 442; } else if (value == 148) { return 445; } else if (value == 149) { return 448; } else if (value == 150) { return 451; } else if (value == 151) { return 454; } else if (value == 152) { return 457; } else if (value == 153) { return 460; } else if (value == 154) { return 463; } else if (value == 155) { return 466; } else if (value == 156) { return 469; } else if (value == 157) { return 472; } else if (value == 158) { return 475; } else if (value == 159) { return 478; } else if (value == 160) { return 481; } else if (value == 161) { return 484; } else if (value == 162) { return 487; } else if (value == 163) { return 490; } else if (value == 164) { return 493; } else if (value == 165) { return 496; } else if (value == 166) { return 499; } else if (value == 167) { return 502; } else if (value == 168) { return 505; } else if (value == 169) { return 508; } else if (value == 170) { return 511; } else if (value == 171) { return 514; } else if (value == 172) { return 517; } else if (value == 173) { return 520; } else if (value == 174) { return 523; } else if (value == 175) { return 526; } else if (value == 176) { return 529; } else if (value == 177) { return 532; } else if (value == 178) { return 535; } else if (value == 179) { return 538; } else if (value == 180) { return 541; } else if (value == 181) { return 544; } else if (value == 182) { return 547; } else if (value == 183) { return 550; } else if (value == 184) { return 553; } else if (value == 185) { return 556; } else if (value == 186) { return 559; } else if (value == 187) { return 562; } else if (value == 188) { return 565; } else if (value == 189) { return 568; } else if (value == 190) { return 571; } else if (value == 191) { return 574; } else if (value == 192) { return 577; } else if (value == 193) { return 580; } else if (value == 194) { return 583; } else if (value == 195) { return 586; } else if (value == 196) { return 589; } else if (value == 197) { return 592; } else if (value == 198) { return 595; } else if (value == 199) { return 598; } else if (value == 200) { return 601; } else if (value == 201) { return 604; } else if (value == 202) { return 607; } else if (value == 203) { return 610; } else if (value == 204) { return 613; } else if (value == 205) { return 616; } else if (value == 206) { return 619; } else if (value == 207) { return 622; } else if (value == 208) { return 625; } else if (value == 209) { return 628; } else if (value == 210) { return 631; } else if (value == 211) { return 634; } else if (value == 212) { return 637; } else if (value == 213) { return 640; } else if (value == 214) { return 643; } else if (value == 215) { return 646; } else if (value == 216) { return 649; } else if (value == 217) { return 652; } else if (value == 218) { return 655; } else if (value == 219) { return 658; } else if (value == 220) { return 661; } else if (value == 221) { return 664; } else if (value == 222) { return 667; } else if (value == 223) { return 670; } else if (value == 224) { return 673; } else if (value == 225) { return 676; } else if (value == 226) { return 679; } else if (value == 227) { return 682; } else if (value == 228) { return 685; } else if (value == 229) { return 688; } else if (value == 230) { return 691; } else if (value == 231) { return 694; } else if (value == 232) { return 697; } else if (value == 233) { return 700; } else if (value == 234) { return 703; } else if (value == 235) { return 706; } else if (value == 236) { return 709; } else if (value == 237) { return 712; } else if (value == 238) { return 715; } else if (value == 239) { return 718; } else if (value == 240) { return 721; } else if (value == 241) { return 724; } else if (value == 242) { return 727; } else if (value == 243) { return 730; } else if (value == 244) { return 733; } else if (value == 245) { return 736; } else if (value == 246) { return 739; } else if (value == 247) { return 742; } else if (value == 248) { return 745; } else if (value == 249) { return 748; } else if (value == 250) { return 751; } else if (value == 251) { return 754; } else if (value == 252) { return 757; } else if (value == 253) { return 760; } else if (value == 254) { return 763; } else if (value == 255) { return 766; } else if (value == 256) { return 769; } else if (value == 257) { return 772; } else if (value == 258) { return 775; } else if (value == 259) { return 778; } else if (value == 260) { return 781; } else if (value == 261) { return 784; } else if (value == 262) { return 787; } else if (value == 263) { return 790; } else if (value == 264) { return 793; } else if (value == 265) { return 796; } else if (value == 266) { return 799; } else if (value == 267) { return 802; } else if (value == 268) { return 805; } else if (value == 269) { return 808; } else if (value == 270) { return 811; } else if (value == 271) { return 814; } else if (value == 272) { return 817; } else if (value == 273) { return 820; } else if (value == 274) { return 823; } else if (value == 275) { return 826; } else if (value == 276) { return 829; } else if (value == 277) { return 832; } else if (value == 278) { return 835; } else if (value == 279) { return 838; } else if (value == 280) { return 841; } else if (value == 281) { return 844; } else if (value == 282) { return 847; } else if (value == 283) { return 850; } else if (value == 284) { return 853; } else if (value == 285) { return 856; } else if (value == 286) { return 859; } else if (value == 287) { return 862; } else if (value == 288) { return 865; } else if (value == 289) { return 868; } else if (value == 290) { return 871; } else if (value == 291) { return 874; } else if (value == 292) { return 877; } else if (value == 293) { return 880; } else if (value == 294) { return 883; } else if (value == 295) { return 886; } else if (value == 296) { return 889; } else if (value == 297) { return 892; } else if (value == 298) { return 895; } else if (value == 299) { return 898; } else if (value == 300) { return 901; } else if (value == 301) { return 904; } else if (value == 302) { return 907; } else if (value == 303) { return 910; } else if (value == 304) { return 913; } else if (value == 305) { return 916; } else if (value == 306) { return 919; } else if (value == 307) { return 922; } else if (value == 308) { return 925; } else if (value == 309) { return 928; } else if (value == 310) { return 931; } else if (value == 311) { return 934; } else if (value == 312) { return 937; } else if (value == 313) { return 940; } else if (value == 314) { return 943; } else if (value == 315) { return 946; } else if (value == 316) { return 949; } else if (value == 317) { return 952; } else if (value == 318) { return 955; } else if (value == 319) { return 958; } else if (value == 320) { return 961; } else if (value == 321) { return 964; } else if (value == 322) { return 967; } else if (value == 323) { return 970; } else if (value == 324) { return 973; } else if (value == 325) { return 976; } else if (value == 326) { return 979; } else if (value == 327) { return 982; } else if (value == 328) { return 985; } else if (value == 329) { return 988; } else if (value == 330) { return 991; } else if (value == 331) { return 994; } else if (value == 332) { return 997; } else if (value == 333) { return 1000; } else if (value == 334) { return 1003; } else if (value == 335) { return 1006; } else if (value == 336) { return 1009; } else if (value == 337) { return 1012; } else if (value == 338) { return 1015; } else if (value == 339) { return 1018; } else if (value == 340) { return 1021; } else if (value == 341) { return 1024; } else if (value == 342) { return 1027; } else if (value == 343) { return 1030; } else if (value == 344) { return 1033; } else if (value == 345) { return 1036; } else if (value == 346) { return 1039; } else if (value == 347) { return 1042; } else if (value == 348) { return 1045; } else if (value == 349) { return 1048; } else if (value == 350) { return 1051; } else if (value == 351) { return 1054; } else if (value == 352) { return 1057; } else if (value == 353) { return 1060; } else if (value == 354) { return 1063; } else if (value == 355) { return 1066; } else if (value == 356) { return 1069; } else if (value == 357) { return 1072; } else if (value == 358) { return 1075; } else if (value == 359) { return 1078; } else if (value == 360) { return 1081; } else if (value == 361) { return 1084; } else if (value == 362) { return 1087; } else if (value == 363) { return 1090; } else if (value == 364) { return 1093; } else if (value == 365) { return 1096; } else if (value == 366) { return 1099; } else if (value == 367) { return 1102; } else if (value == 368) { return 1105; } else if (value == 369) { return 1108; } else if (value == 370) { return 1111; } else if (value == 371) { return 1114; } else if (value == 372) { return 1117; } else if (value == 373) { return 1120; } else if (value == 374) { return 1123; } else if (value == 375) { return 1126; } else if (value == 376) { return 1129; } else if (value == 377) { return 1132; } else if (value == 378) { return 1135; } else if (value == 379) { return 1138; } else if (value == 380) { return 1141; } else if (value == 381) { return 1144; } else if (value == 382) { return 1147; } else if (value == 383) { return 1150; } else if (value == 384) { return 1153; } else if (value == 385) { return 1156; } else if (value == 386) { return 1159; } else if (value == 387) { return 1162; } else if (value == 388) { return 1165; } else if (value == 389) { return 1168; } else if (value == 390) { return 1171; } else if (value == 391) { return 1174; } else if (value == 392) { return 1177; } else if (value == 393) { return 1180; } else if (value == 394) { return 1183; } else if (value == 395) { return 1186; } else if (value == 396) { return 1189; } else if (value == 397) { return 1192; } else if (value == 398) { return 1195; } else if (value == 399) { return 1198; } return 0; } } diff --git a/Compiler/Fuzzing/Corpus/source/long-operator-chain.seed b/Compiler/Fuzzing/Corpus/source/long-operator-chain.seed new file mode 100644 index 00000000..118b2847 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/long-operator-chain.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; return value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value; } } diff --git a/Compiler/Fuzzing/Corpus/source/negative-patterns.seed b/Compiler/Fuzzing/Corpus/source/negative-patterns.seed new file mode 100644 index 00000000..bfb34818 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/negative-patterns.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; int low = 0 - value; int a = match (low) { -3 -> 10, -2 -> 20, - 1 -> 30, 0 -> 40, _ -> 50 }; byte small = 1; int b = match (small) { -128 -> 1, 127 -> 2, _ -> 3 }; int c = match (low), (value) { (-3), (3) -> 1, (_), (_) -> 0 }; return a + b + c; } } diff --git a/Compiler/Fuzzing/Corpus/source/nested-match-and-blocks.seed b/Compiler/Fuzzing/Corpus/source/nested-match-and-blocks.seed new file mode 100644 index 00000000..5cb58808 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/nested-match-and-blocks.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 5; int total = 0; { match (value) { 0 -> { total += 0; }, _ -> { { match (value) { 1 -> { total += 1; }, _ -> { { match (value) { 2 -> { total += 2; }, _ -> { { match (value) { 3 -> { total += 3; }, _ -> { { match (value) { 4 -> { total += 4; }, _ -> { { match (value) { 5 -> { total += 5; }, _ -> { { match (value) { 6 -> { total += 6; }, _ -> { { match (value) { 7 -> { total += 7; }, _ -> { { match (value) { 8 -> { total += 8; }, _ -> { { match (value) { 9 -> { total += 9; }, _ -> { { match (value) { 10 -> { total += 10; }, _ -> { { match (value) { 11 -> { total += 11; }, _ -> { total += 1; } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } return total; } } diff --git a/Compiler/Fuzzing/Corpus/source/nesting-at-limit.seed b/Compiler/Fuzzing/Corpus/source/nesting-at-limit.seed new file mode 100644 index 00000000..49baf767 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/nesting-at-limit.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int total = 0; int value = 300; if (value > 0) { if (value > 1) { if (value > 2) { if (value > 3) { if (value > 4) { if (value > 5) { if (value > 6) { if (value > 7) { if (value > 8) { if (value > 9) { if (value > 10) { if (value > 11) { if (value > 12) { if (value > 13) { if (value > 14) { if (value > 15) { if (value > 16) { if (value > 17) { if (value > 18) { if (value > 19) { if (value > 20) { if (value > 21) { if (value > 22) { if (value > 23) { if (value > 24) { if (value > 25) { if (value > 26) { if (value > 27) { if (value > 28) { if (value > 29) { if (value > 30) { if (value > 31) { if (value > 32) { if (value > 33) { if (value > 34) { if (value > 35) { if (value > 36) { if (value > 37) { if (value > 38) { if (value > 39) { if (value > 40) { if (value > 41) { if (value > 42) { if (value > 43) { if (value > 44) { if (value > 45) { if (value > 46) { if (value > 47) { if (value > 48) { if (value > 49) { if (value > 50) { if (value > 51) { if (value > 52) { if (value > 53) { if (value > 54) { if (value > 55) { if (value > 56) { if (value > 57) { if (value > 58) { if (value > 59) { if (value > 60) { if (value > 61) { if (value > 62) { if (value > 63) { if (value > 64) { if (value > 65) { if (value > 66) { if (value > 67) { if (value > 68) { if (value > 69) { if (value > 70) { if (value > 71) { if (value > 72) { if (value > 73) { if (value > 74) { if (value > 75) { if (value > 76) { if (value > 77) { if (value > 78) { if (value > 79) { if (value > 80) { if (value > 81) { if (value > 82) { if (value > 83) { if (value > 84) { if (value > 85) { if (value > 86) { if (value > 87) { if (value > 88) { if (value > 89) { if (value > 90) { if (value > 91) { if (value > 92) { if (value > 93) { if (value > 94) { if (value > 95) { if (value > 96) { if (value > 97) { if (value > 98) { if (value > 99) { if (value > 100) { if (value > 101) { if (value > 102) { if (value > 103) { if (value > 104) { if (value > 105) { if (value > 106) { if (value > 107) { if (value > 108) { if (value > 109) { if (value > 110) { if (value > 111) { if (value > 112) { if (value > 113) { if (value > 114) { if (value > 115) { if (value > 116) { if (value > 117) { if (value > 118) { if (value > 119) { if (value > 120) { if (value > 121) { if (value > 122) { if (value > 123) { if (value > 124) { if (value > 125) { if (value > 126) { if (value > 127) { if (value > 128) { if (value > 129) { if (value > 130) { if (value > 131) { if (value > 132) { if (value > 133) { if (value > 134) { if (value > 135) { if (value > 136) { if (value > 137) { if (value > 138) { if (value > 139) { if (value > 140) { if (value > 141) { if (value > 142) { if (value > 143) { if (value > 144) { if (value > 145) { if (value > 146) { if (value > 147) { if (value > 148) { if (value > 149) { if (value > 150) { if (value > 151) { if (value > 152) { if (value > 153) { if (value > 154) { if (value > 155) { if (value > 156) { if (value > 157) { if (value > 158) { if (value > 159) { if (value > 160) { if (value > 161) { if (value > 162) { if (value > 163) { if (value > 164) { if (value > 165) { if (value > 166) { if (value > 167) { if (value > 168) { if (value > 169) { if (value > 170) { if (value > 171) { if (value > 172) { if (value > 173) { if (value > 174) { if (value > 175) { if (value > 176) { if (value > 177) { if (value > 178) { if (value > 179) { if (value > 180) { if (value > 181) { if (value > 182) { if (value > 183) { if (value > 184) { if (value > 185) { if (value > 186) { if (value > 187) { if (value > 188) { if (value > 189) { if (value > 190) { if (value > 191) { if (value > 192) { if (value > 193) { if (value > 194) { if (value > 195) { if (value > 196) { if (value > 197) { if (value > 198) { if (value > 199) { if (value > 200) { if (value > 201) { if (value > 202) { if (value > 203) { if (value > 204) { if (value > 205) { if (value > 206) { if (value > 207) { if (value > 208) { if (value > 209) { if (value > 210) { if (value > 211) { if (value > 212) { if (value > 213) { if (value > 214) { if (value > 215) { if (value > 216) { if (value > 217) { if (value > 218) { if (value > 219) { if (value > 220) { if (value > 221) { if (value > 222) { if (value > 223) { if (value > 224) { if (value > 225) { if (value > 226) { if (value > 227) { if (value > 228) { if (value > 229) { if (value > 230) { if (value > 231) { if (value > 232) { if (value > 233) { if (value > 234) { if (value > 235) { if (value > 236) { if (value > 237) { if (value > 238) { if (value > 239) { if (value > 240) { if (value > 241) { if (value > 242) { if (value > 243) { if (value > 244) { if (value > 245) { if (value > 246) { if (value > 247) { if (value > 248) { if (value > 249) { if (value > 250) { if (value > 251) { if (value > 252) { if (value > 253) { if (value > 254) { total += 1; } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } return total; } } diff --git a/Compiler/Fuzzing/Corpus/source/nesting-beyond-limit.seed b/Compiler/Fuzzing/Corpus/source/nesting-beyond-limit.seed new file mode 100644 index 00000000..7eaafd44 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/nesting-beyond-limit.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int total = 0; int value = 300; if (value > 0) { if (value > 1) { if (value > 2) { if (value > 3) { if (value > 4) { if (value > 5) { if (value > 6) { if (value > 7) { if (value > 8) { if (value > 9) { if (value > 10) { if (value > 11) { if (value > 12) { if (value > 13) { if (value > 14) { if (value > 15) { if (value > 16) { if (value > 17) { if (value > 18) { if (value > 19) { if (value > 20) { if (value > 21) { if (value > 22) { if (value > 23) { if (value > 24) { if (value > 25) { if (value > 26) { if (value > 27) { if (value > 28) { if (value > 29) { if (value > 30) { if (value > 31) { if (value > 32) { if (value > 33) { if (value > 34) { if (value > 35) { if (value > 36) { if (value > 37) { if (value > 38) { if (value > 39) { if (value > 40) { if (value > 41) { if (value > 42) { if (value > 43) { if (value > 44) { if (value > 45) { if (value > 46) { if (value > 47) { if (value > 48) { if (value > 49) { if (value > 50) { if (value > 51) { if (value > 52) { if (value > 53) { if (value > 54) { if (value > 55) { if (value > 56) { if (value > 57) { if (value > 58) { if (value > 59) { if (value > 60) { if (value > 61) { if (value > 62) { if (value > 63) { if (value > 64) { if (value > 65) { if (value > 66) { if (value > 67) { if (value > 68) { if (value > 69) { if (value > 70) { if (value > 71) { if (value > 72) { if (value > 73) { if (value > 74) { if (value > 75) { if (value > 76) { if (value > 77) { if (value > 78) { if (value > 79) { if (value > 80) { if (value > 81) { if (value > 82) { if (value > 83) { if (value > 84) { if (value > 85) { if (value > 86) { if (value > 87) { if (value > 88) { if (value > 89) { if (value > 90) { if (value > 91) { if (value > 92) { if (value > 93) { if (value > 94) { if (value > 95) { if (value > 96) { if (value > 97) { if (value > 98) { if (value > 99) { if (value > 100) { if (value > 101) { if (value > 102) { if (value > 103) { if (value > 104) { if (value > 105) { if (value > 106) { if (value > 107) { if (value > 108) { if (value > 109) { if (value > 110) { if (value > 111) { if (value > 112) { if (value > 113) { if (value > 114) { if (value > 115) { if (value > 116) { if (value > 117) { if (value > 118) { if (value > 119) { if (value > 120) { if (value > 121) { if (value > 122) { if (value > 123) { if (value > 124) { if (value > 125) { if (value > 126) { if (value > 127) { if (value > 128) { if (value > 129) { if (value > 130) { if (value > 131) { if (value > 132) { if (value > 133) { if (value > 134) { if (value > 135) { if (value > 136) { if (value > 137) { if (value > 138) { if (value > 139) { if (value > 140) { if (value > 141) { if (value > 142) { if (value > 143) { if (value > 144) { if (value > 145) { if (value > 146) { if (value > 147) { if (value > 148) { if (value > 149) { if (value > 150) { if (value > 151) { if (value > 152) { if (value > 153) { if (value > 154) { if (value > 155) { if (value > 156) { if (value > 157) { if (value > 158) { if (value > 159) { if (value > 160) { if (value > 161) { if (value > 162) { if (value > 163) { if (value > 164) { if (value > 165) { if (value > 166) { if (value > 167) { if (value > 168) { if (value > 169) { if (value > 170) { if (value > 171) { if (value > 172) { if (value > 173) { if (value > 174) { if (value > 175) { if (value > 176) { if (value > 177) { if (value > 178) { if (value > 179) { if (value > 180) { if (value > 181) { if (value > 182) { if (value > 183) { if (value > 184) { if (value > 185) { if (value > 186) { if (value > 187) { if (value > 188) { if (value > 189) { if (value > 190) { if (value > 191) { if (value > 192) { if (value > 193) { if (value > 194) { if (value > 195) { if (value > 196) { if (value > 197) { if (value > 198) { if (value > 199) { if (value > 200) { if (value > 201) { if (value > 202) { if (value > 203) { if (value > 204) { if (value > 205) { if (value > 206) { if (value > 207) { if (value > 208) { if (value > 209) { if (value > 210) { if (value > 211) { if (value > 212) { if (value > 213) { if (value > 214) { if (value > 215) { if (value > 216) { if (value > 217) { if (value > 218) { if (value > 219) { if (value > 220) { if (value > 221) { if (value > 222) { if (value > 223) { if (value > 224) { if (value > 225) { if (value > 226) { if (value > 227) { if (value > 228) { if (value > 229) { if (value > 230) { if (value > 231) { if (value > 232) { if (value > 233) { if (value > 234) { if (value > 235) { if (value > 236) { if (value > 237) { if (value > 238) { if (value > 239) { if (value > 240) { if (value > 241) { if (value > 242) { if (value > 243) { if (value > 244) { if (value > 245) { if (value > 246) { if (value > 247) { if (value > 248) { if (value > 249) { if (value > 250) { if (value > 251) { if (value > 252) { if (value > 253) { if (value > 254) { if (value > 255) { total += 1; } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } } return total; } } diff --git a/Compiler/Fuzzing/Corpus/source/operator-chain-of-1025.seed b/Compiler/Fuzzing/Corpus/source/operator-chain-of-1025.seed new file mode 100644 index 00000000..2b46b6e0 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/operator-chain-of-1025.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; return value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value; } } diff --git a/Compiler/Fuzzing/Corpus/source/right-nesting-at-limit.seed b/Compiler/Fuzzing/Corpus/source/right-nesting-at-limit.seed new file mode 100644 index 00000000..cb0150e8 --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/right-nesting-at-limit.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; return value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))); } } diff --git a/Compiler/Fuzzing/Corpus/source/right-nesting-beyond-limit.seed b/Compiler/Fuzzing/Corpus/source/right-nesting-beyond-limit.seed new file mode 100644 index 00000000..7cbc83bc --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/right-nesting-beyond-limit.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; return value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value + (value)))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))))); } } diff --git a/Compiler/Fuzzing/Corpus/source/very-long-operator-chain.seed b/Compiler/Fuzzing/Corpus/source/very-long-operator-chain.seed new file mode 100644 index 00000000..56af743d --- /dev/null +++ b/Compiler/Fuzzing/Corpus/source/very-long-operator-chain.seed @@ -0,0 +1 @@ +namespace Fuzz; class Program { public static int Evaluate() { int value = 3; return value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value + value; } } diff --git a/Compiler/Fuzzing/ExecutionCases.cpp b/Compiler/Fuzzing/ExecutionCases.cpp new file mode 100644 index 00000000..f60036a9 --- /dev/null +++ b/Compiler/Fuzzing/ExecutionCases.cpp @@ -0,0 +1,160 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "ExecutionCases.hpp" +#include "SourceFuzz.hpp" + +namespace Visual::XSharp::Fuzzing +{ + namespace + { + // Each run of a program owns one bit of its result, which is an + // `int`. Twenty bits keep every sum well inside its range. + constexpr std::size_t kMaximumRunsPerProgram = 20U; + // Small bodies share a program; the cost of a program under + // sanitizers is mostly the fixed cost of compiling one at all. + constexpr std::size_t kMaximumBodiesPerProgram = 8U; + // A body larger than this is a program of its own: its size is what + // it tests, and its cost should be visible on its own line. + constexpr std::size_t kMaximumSharedBodyBytes = 1024U; + + /// One body and the runs that call it. + struct Method final + { + std::string_view body; + std::vector runs; + }; + + [[nodiscard]] auto + Truth(bool value) -> std::string + { + // `Id` is recursive, so the optimizer cannot fold the arguments + // away and the body really executes on run-time values. + return value ? "Id(1) > 0" : "Id(0) > 0"; + } + + /// A literal has no sign, so a negative value is spelled as a + /// subtraction. + [[nodiscard]] auto + Literal(std::int64_t value) -> std::string + { + return value < 0 ? "(0 - " + std::to_string(-value) + ")" + : std::to_string(value); + } + + [[nodiscard]] auto + Call(std::size_t method, const ExecutionCase &run) -> std::string + { + return "Run" + std::to_string(method) + "(" + Truth(run.flag) + ", " + + Truth(run.other) + ", Id(" + std::to_string(run.left) + + "), Id(" + std::to_string(run.right) + "))"; + } + + [[nodiscard]] auto + Program(const std::vector &methods, + std::string_view helpers, + std::string_view declarations) -> std::string + { + std::string program = "namespace Fuzz;\n"; + program += declarations; + program += "class Program {\n" + " public static int Id(_ int n) { return n > 0 ? 1 + " + "Id(n - 1) : 0; }\n"; + program += helpers; + for (std::size_t method = 0U; method < methods.size(); ++method) + { + program += " public static int Run" + std::to_string(method) + + "(_ bool flag, _ bool other, _ int left, _ int " + "right) {\n "; + program += methods[method].body; + program += "\n }\n"; + } + program += " public static int Evaluate() {\n" + " int failed = 0;\n"; + std::size_t bit = 0U; + for (std::size_t method = 0U; method < methods.size(); ++method) + for (const auto *run : methods[method].runs) + program += " if (" + Call(method, *run) + " \\= " + + Literal(run->expected) + ") { failed += " + + std::to_string(std::size_t{ 1U } << bit++) + + "; }\n"; + program += " return failed;\n }\n}\n"; + return program; + } + + void + Exercise(std::string_view label, + const std::vector &methods, + std::string_view helpers, + std::string_view declarations) + { + std::size_t bit = 0U; + for (const auto &method : methods) + { + llvm::errs() << label << ": " << method.body << '\n'; + for (const auto *run : method.runs) + llvm::errs() + << " run " << (std::size_t{ 1U } << bit++) + << ": flag " << (run->flag ? "true" : "false") + << ", other " << (run->other ? "true" : "false") + << ", left " << run->left << ", right " << run->right + << " -> " << run->expected << '\n'; + } + // A result other than 0 is the sum of the run numbers printed + // above whose value differed. + ExerciseExpectedValue(Program(methods, helpers, declarations), 0); + } + } // namespace + + void + ExerciseExecutionCases(std::string_view label, + std::span cases, + std::string_view helpers, + std::string_view declarations) + { + std::vector taken(cases.size(), false); + std::vector methods; + std::size_t runs = 0U; + const auto flush = [&] { + if (!methods.empty()) + Exercise(label, methods, helpers, declarations); + methods.clear(); + runs = 0U; + }; + for (std::size_t first = 0U; first < cases.size(); ++first) + { + if (taken[first]) + continue; + // The runs of one body, in table order; a body with more runs + // than a program holds continues in a later program. + Method method{ cases[first].body, {} }; + for (std::size_t index = first; + index < cases.size() + && method.runs.size() < kMaximumRunsPerProgram; + ++index) + { + if (taken[index] || cases[index].body != method.body) + continue; + taken[index] = true; + method.runs.push_back(&cases[index]); + } + const auto large = method.body.size() > kMaximumSharedBodyBytes; + if (large || methods.size() == kMaximumBodiesPerProgram + || runs + method.runs.size() > kMaximumRunsPerProgram) + flush(); + runs += method.runs.size(); + methods.push_back(std::move(method)); + if (large) + flush(); + } + flush(); + } +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/ExecutionCases.hpp b/Compiler/Fuzzing/ExecutionCases.hpp new file mode 100644 index 00000000..a0d4dc61 --- /dev/null +++ b/Compiler/Fuzzing/ExecutionCases.hpp @@ -0,0 +1,53 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include +#include + +namespace Visual::XSharp::Fuzzing +{ + /// One run of an executable regression: the body of + /// `int Run(bool flag, bool other, int left, int right)`, the arguments + /// it is called with, and the value it must return. + struct ExecutionCase final + { + bool flag; + bool other; + int left; + int right; + std::int64_t expected; + std::string_view body; + }; + + /** + * @brief Compile every distinct body once and check all of its runs. + * + * A program holds up to eight bodies, each in a method of its own, and + * up to twenty runs. Its `Evaluate` calls the method of each run, + * compares the result with the expected value, and returns the sum of a + * distinct power of two for every run that differed. The program must + * return 0 from both native pipeline modes, so every run is checked + * exactly as when each had a program of its own, and a nonzero result + * names the runs that failed. What is saved is the fixed cost of + * compiling a program, which dominates these cases under sanitizers. A + * body of more than a kilobyte is not shared: its size is what it + * tests. + * + * The arguments are passed through a recursive identity method, so the + * optimizer cannot fold them and the bodies execute on run-time values. + * + * @param label Printed before each run, to name the table. + * @param cases Runs in table order; equal bodies need not be adjacent. + * @param helpers Source of additional methods of the class the body may + * call. + * @param declarations Source of declarations that stand before the + * class, such as the enums the bodies use. + */ + void + ExerciseExecutionCases(std::string_view label, + std::span cases, + std::string_view helpers, + std::string_view declarations = {}); +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/ExpressionExecutionCases.cpp b/Compiler/Fuzzing/ExpressionExecutionCases.cpp index 814dabd9..1f7e36ed 100644 --- a/Compiler/Fuzzing/ExpressionExecutionCases.cpp +++ b/Compiler/Fuzzing/ExpressionExecutionCases.cpp @@ -2,13 +2,10 @@ // SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 #include -#include -#include -#include #include +#include "ExecutionCases.hpp" #include "ExpressionExecutionCases.hpp" -#include "SourceFuzz.hpp" // Executable regressions for expressions that store into locals: // assignments and increments used as values, and loops used as expressions. @@ -25,17 +22,7 @@ namespace Visual::XSharp::Fuzzing { namespace { - struct Case final - { - bool flag; - bool other; - int left; - int right; - std::int64_t expected; - std::string_view body; - }; - - constexpr std::array kCases{ { + constexpr std::array kCases{ { { false, false, 3, @@ -777,48 +764,21 @@ namespace Visual::XSharp::Fuzzing "0;" }, } }; - [[nodiscard]] auto - Truth(bool value) -> std::string - { - // `Id` is recursive, so the optimizer cannot fold the arguments - // away and the body really executes on run-time values. - return value ? "Id(1) > 0" : "Id(0) > 0"; - } - - [[nodiscard]] auto - Program(const Case &entry) -> std::string - { - return "namespace Fuzz;\n" - "class Program {\n" - " public static int Id(_ int n) { return n > 0 ? 1 + " - "Id(n - 1) : 0; }\n" - " public static int Next() { return 7; }\n" - " public static int Twice(_ int value) { return value + " - "value; }\n" - " public static int Minus(_ int first, _ int second) { " - "return first - second; }\n" - " public static int Pick(_ int first, _ int second, _ " - "int " - "third) { return first * 100 + second * 10 + third; }\n" - " public static void Touch() { }\n" - " public static int Run(_ bool flag, _ bool other, _ int " - "left, _ int right) {\n " - + std::string(entry.body) - + "\n }\n" - " public static int Evaluate() { return Run(" - + Truth(entry.flag) + ", " + Truth(entry.other) + ", Id(" - + std::to_string(entry.left) + "), Id(" - + std::to_string(entry.right) + ")); }\n}\n"; - } + // The methods a body may call besides `Run` itself. + constexpr std::string_view kHelpers + = " public static int Next() { return 7; }\n" + " public static int Twice(_ int value) { return value + " + "value; }\n" + " public static int Minus(_ int first, _ int second) { " + "return first - second; }\n" + " public static int Pick(_ int first, _ int second, _ int " + "third) { return first * 100 + second * 10 + third; }\n" + " public static void Touch() { }\n"; } // namespace void ExerciseExpressionCases() { - for (const auto &entry : kCases) - { - llvm::errs() << "Expression execution: " << entry.body << '\n'; - ExerciseExpectedValue(Program(entry), entry.expected); - } + ExerciseExecutionCases("Expression execution", kCases, kHelpers); } } // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/Generated/LeavingCases.inc b/Compiler/Fuzzing/Generated/LeavingCases.inc new file mode 100644 index 00000000..c3870b0d --- /dev/null +++ b/Compiler/Fuzzing/Generated/LeavingCases.inc @@ -0,0 +1,310 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +// +// Generated from Compiler/Fuzzing/Cases/Leaving.cases by +// `go -C helpers run ./cmd/execution-cases generate`. Do not edit. +// +// clang-format off +// An expression every branch of which leaves never yields a value. +{ true, false, 0, 0, 1, + "int r = if (flag) { return 1; } else { return 2; }; return r + 50;" }, +{ false, false, 0, 0, 2, + "int r = if (flag) { return 1; } else { return 2; }; return r + 50;" }, +{ false, false, 0, 0, 10, + "int r = match (left) { 0 -> { return 10; }, _ -> { return 20; } }; return r + 1;" }, +{ false, false, 5, 0, 20, + "int r = match (left) { 0 -> { return 10; }, _ -> { return 20; } }; return r + 1;" }, +{ true, false, 0, 0, 7, + "return Twice(if (flag) { return 7; } else { return 9; });" }, +{ false, false, 0, 0, 9, + "return Twice(if (flag) { return 7; } else { return 9; });" }, +{ true, false, 0, 0, 1, + "int r = 1 + (if (flag) { return 1; } else { return 2; }); return r;" }, +{ false, false, 0, 0, 2, + "int r = 1 + (if (flag) { return 1; } else { return 2; }); return r;" }, +{ false, false, 2, 0, 3, + "int t = 0; int i = 0; while (i < 5) { i += 1; int q = if (i > left) { break; } else { continue; }; t += q; } return t * 10 + i;" }, +{ false, false, 9, 0, 5, + "int t = 0; int i = 0; while (i < 5) { i += 1; int q = if (i > left) { break; } else { continue; }; t += q; } return t * 10 + i;" }, +{ true, false, 5, 0, 1, + "bool b = flag && (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" }, +{ true, false, 0, 0, 2, + "bool b = flag && (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" }, +{ false, false, 5, 0, 4, + "bool b = flag && (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" }, +{ true, false, 5, 0, 3, + "bool b = flag || (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" }, +{ false, false, 5, 0, 1, + "bool b = flag || (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" }, +{ false, false, 0, 0, 2, + "bool b = flag || (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" }, +{ false, false, 5, 0, 5, + "int n = left ?: (if (flag) { return 100; } else { return 200; }); return n;" }, +{ true, false, 0, 0, 100, + "int n = left ?: (if (flag) { return 100; } else { return 200; }); return n;" }, +{ false, false, 0, 0, 200, + "int n = left ?: (if (flag) { return 100; } else { return 200; }); return n;" }, +{ false, false, 3, 0, 5, + "int r = if (left > 0) { 5 } else { if (flag) { return 1; } else { return 2; } }; return r;" }, +{ true, false, 0, 0, 1, + "int r = if (left > 0) { 5 } else { if (flag) { return 1; } else { return 2; } }; return r;" }, +{ false, false, 0, 0, 2, + "int r = if (left > 0) { 5 } else { if (flag) { return 1; } else { return 2; } }; return r;" }, +{ false, false, 5, 0, 6, + "int n = left; do { n += 1; } while (if (n > 3) { return n; } else { return 0 - n; }); return 99;" }, +{ false, false, 1, 0, -2, + "int n = left; do { n += 1; } while (if (n > 3) { return n; } else { return 0 - n; }); return 99;" }, +// The condition of a do/while that never completes still runs after +// the body and after a continue, with its effects, and is skipped by +// a break. +{ false, false, 1, 0, 101, + "int t = 0; int n = 0; do { n += 1; if (n == left) { continue; } t += 10; } while (if ((t += 1) > 100) { return 1; } else { return t * 100 + n; }); return 99;" }, +{ false, false, 5, 0, 1101, + "int t = 0; int n = 0; do { n += 1; if (n == left) { continue; } t += 10; } while (if ((t += 1) > 100) { return 1; } else { return t * 100 + n; }); return 99;" }, +{ false, false, 1, 0, 1000, + "int t = 0; do { if (left > 0) { break; } t += 5; } while (if ((t += 1) > 0) { return t; } else { return 0; }); return t + 1000;" }, +{ false, false, 0, 0, 6, + "int t = 0; do { if (left > 0) { break; } t += 5; } while (if ((t += 1) > 0) { return t; } else { return 0; }); return t + 1000;" }, +{ false, false, 0, 0, 10, + "int t = 0; for (int i = 0; i < 3; i += 1) { do { t += 1; } while (if (t > left) { return t * 10 + i; } else { return t; }); } return 77;" }, +{ false, false, 5, 0, 1, + "int t = 0; for (int i = 0; i < 3; i += 1) { do { t += 1; } while (if (t > left) { return t * 10 + i; } else { return t; }); } return 77;" }, +{ false, false, 0, 0, 1, + "int t = 0; int i = 0; while (i < 3) { i += 1; int j = 0; do { j += 1; if (j < 3) { continue; } t += j; } while (if (i > left) { return t * 10 + i; } else { j < 4 }); } return t;" }, +{ false, false, 9, 0, 21, + "int t = 0; int i = 0; while (i < 3) { i += 1; int j = 0; do { j += 1; if (j < 3) { continue; } t += j; } while (if (i > left) { return t * 10 + i; } else { j < 4 }); } return t;" }, +// A transfer in a value block targets the innermost loop only. +{ false, false, 0, 0, 55, + "int t = 0; int i = 0; while (i < 5) { i += 1; int j = 0; while (j < 3) { j += 1; t += if (j == 2) { break; } else { 1 }; } } return t * 10 + i;" }, +{ false, false, 0, 0, 105, + "int t = 0; int i = 0; while (i < 5) { i += 1; int j = 0; while (j < 3) { j += 1; t += if (j == 2) { continue; } else { 1 }; } } return t * 10 + i;" }, +{ false, false, 3, 0, 4, + "int t = 0; while (t < 10) { t += 1; int q = while (true) { break t; }; if (q > left) { break; } } return t;" }, +{ false, false, 20, 0, 10, + "int t = 0; while (t < 10) { t += 1; int q = while (true) { break t; }; if (q > left) { break; } } return t;" }, +// A short-circuit operator evaluates its left operand, with its +// effects, and reaches a right operand that never completes only +// when the left one does not decide. +{ false, false, 1, 0, 10, + "int t = 0; bool b = (t += 1) > 0 && (if (left > 0) { return t * 10; } else { return t * 100; }); return 7;" }, +{ false, false, 0, 0, 100, + "int t = 0; bool b = (t += 1) > 0 && (if (left > 0) { return t * 10; } else { return t * 100; }); return 7;" }, +{ false, false, 1, 0, 1, + "int t = 0; bool b = (t += 1) > 5 && (if (left > 0) { return t * 10; } else { return t * 100; }); return t + (b ? 1000 : 0);" }, +{ false, false, 0, 0, 1, + "int t = 0; bool b = (t += 1) > 5 && (if (left > 0) { return t * 10; } else { return t * 100; }); return t + (b ? 1000 : 0);" }, +{ false, false, 3, 0, 4, + "int t = 0; bool b = (t += left) > 0 || (if (flag) { return t * 10; } else { return 50 + t; }); return b ? t + 1 : 7;" }, +{ true, false, 0, 0, 0, + "int t = 0; bool b = (t += left) > 0 || (if (flag) { return t * 10; } else { return 50 + t; }); return b ? t + 1 : 7;" }, +{ false, false, 0, 0, 50, + "int t = 0; bool b = (t += left) > 0 || (if (flag) { return t * 10; } else { return 50 + t; }); return b ? t + 1 : 7;" }, +{ false, false, 4, 0, 12, + "int t = 0; int q = (t += left) ? t * 2 : (if (flag) { return 0 - 1; } else { return 0 - 2; }); return q + t;" }, +{ true, false, 0, 0, -1, + "int t = 0; int q = (t += left) ? t * 2 : (if (flag) { return 0 - 1; } else { return 0 - 2; }); return q + t;" }, +{ false, false, 0, 0, -2, + "int t = 0; int q = (t += left) ? t * 2 : (if (flag) { return 0 - 1; } else { return 0 - 2; }); return q + t;" }, +// Operands are evaluated left to right up to the one that never +// completes; the ones after it are not evaluated. +{ true, false, 0, 0, 10, + "int t = 0; int r = (t += 1) + (if (flag) { return t * 10; } else { return t * 100; }) + (t += 50); return r;" }, +{ false, false, 0, 0, 100, + "int t = 0; int r = (t += 1) + (if (flag) { return t * 10; } else { return t * 100; }) + (t += 50); return r;" }, +{ true, false, 0, 0, 3, + "int t = 0; return Twice(Twice(t += 3) + (if (flag) { return t; } else { return t * 2; }));" }, +{ false, false, 0, 0, 6, + "int t = 0; return Twice(Twice(t += 3) + (if (flag) { return t; } else { return t * 2; }));" }, +{ true, false, 4, 0, 5, + "int t = left; t = Twice(t += 1) + (if (flag) { return t; } else { return t + 100; }); return 0 - 1;" }, +{ false, false, 4, 0, 105, + "int t = left; t = Twice(t += 1) + (if (flag) { return t; } else { return t + 100; }); return 0 - 1;" }, +// A guard runs only when its patterns accept, and a guarded arm that +// leaves does not make the match leave. +{ false, false, 0, 0, 11, + "int t = 0; int r = match (left) { 0 if (t += 1) > 5 -> { return 1; }, 0 -> { return 10 + t; }, _ -> { return 20 + t; } }; return r;" }, +{ false, false, 3, 0, 20, + "int t = 0; int r = match (left) { 0 if (t += 1) > 5 -> { return 1; }, 0 -> { return 10 + t; }, _ -> { return 20 + t; } }; return r;" }, +{ true, false, 0, 0, 1, + "int r = match (left) { _ if flag -> { return 1; }, _ -> 5 }; return r + 1;" }, +{ false, false, 0, 0, 6, + "int r = match (left) { _ if flag -> { return 1; }, _ -> 5 }; return r + 1;" }, +// A break in the condition of a loop leaves that loop, once, with the +// effects of the condition up to it. +{ false, false, 3, 0, 31, + "int n = 0; while (if (n >= left) { break; } else { true }) { n += 1; } return n * 10 + 1;" }, +{ false, false, 0, 0, 1, + "int n = 0; while (if (n >= left) { break; } else { true }) { n += 1; } return n * 10 + 1;" }, +{ false, false, 2, 0, 302, + "int c = 0; int n = 0; while (if ((c += 1) > left) { break; } else { true }) { n += 1; } return c * 100 + n;" }, +{ false, false, 0, 0, 100, + "int c = 0; int n = 0; while (if ((c += 1) > left) { break; } else { true }) { n += 1; } return c * 100 + n;" }, +{ false, false, 3, 0, 3, + "int n = 0; do { n += 1; } while (if (n >= left) { break; } else { true }); return n;" }, +{ false, false, 0, 0, 1, + "int n = 0; do { n += 1; } while (if (n >= left) { break; } else { true }); return n;" }, +{ false, false, 2, 0, 3, + "int t = 0; for (int i = 0; if (i > left) { break; } else { i < 10 }; i += 1) { t += i; } return t;" }, +{ false, false, 20, 0, 45, + "int t = 0; for (int i = 0; if (i > left) { break; } else { i < 10 }; i += 1) { t += i; } return t;" }, +{ false, false, 0, 0, 36, + "int t = 0; for (int i = 0; i < 3; i += 1) { int j = 0; while (if (j == 2) { break; } else { true }) { j += 1; t += 1; } t += 10; } return t;" }, +{ false, false, 0, 0, 74, + "int t = 0; int i = 0; while (i < 4) { i += 1; int j = 0; while (if (j >= i) { break; } else { true }) { j += 1; if (j == 2) { continue; } t += 1; } } return t * 10 + i;" }, +// A continue in the update of a loop ends the update; the condition +// is tested next, and the update is not run again for that pass. +{ false, false, 2, 0, 71, + "int t = 0; int skips = 0; for (int i = 0; i < 6; i += if (i == left && skips == 0) { skips += 1; continue; } else { 1 }) { t += 1; } return t * 10 + skips;" }, +{ false, false, 9, 0, 60, + "int t = 0; int skips = 0; for (int i = 0; i < 6; i += if (i == left && skips == 0) { skips += 1; continue; } else { 1 }) { t += 1; } return t * 10 + skips;" }, +{ false, false, 2, 0, 304, + "int t = 0; int u = 0; for (int i = 0; i < 4; i += 1, u += if (i == left) { continue; } else { 10 }) { t += 1; } return u * 10 + t;" }, +{ false, false, 9, 0, 404, + "int t = 0; int u = 0; for (int i = 0; i < 4; i += 1, u += if (i == left) { continue; } else { 10 }) { t += 1; } return u * 10 + t;" }, +{ false, false, 1, 0, 315, + "int t = 0; for (int i = 0; i < 3; i += 1) { int c = 0; for (int j = 0; j < 4; j += if (c == 0 && j == left) { c += 1; continue; } else { 1 }) { t += 1; } t += c * 100; } return t;" }, +{ false, false, 7, 0, 12, + "int t = 0; for (int i = 0; i < 3; i += 1) { int c = 0; for (int j = 0; j < 4; j += if (c == 0 && j == left) { c += 1; continue; } else { 1 }) { t += 1; } t += c * 100; } return t;" }, +// A continue in the condition of a loop abandons the rest of the +// condition and evaluates the condition of that loop again. The body +// does not run in between, and neither does the update of a for. +{ false, false, 3, 0, 502, + "int c = 0; int n = 0; while (if ((c += 1) < left) { continue; } else { n < 2 }) { n += 1; } return c * 100 + n;" }, +{ false, false, 0, 0, 302, + "int c = 0; int n = 0; while (if ((c += 1) < left) { continue; } else { n < 2 }) { n += 1; } return c * 100 + n;" }, +{ false, false, 2, 0, 5330, + "int u = 0; int c = 0; int t = 0; for (int i = 0; if ((c += 1) == left) { continue; } else { i < 3 }; i += 1, u += 1) { t += 10; } return c * 1000 + u * 100 + t;" }, +{ false, false, 9, 0, 4330, + "int u = 0; int c = 0; int t = 0; for (int i = 0; if ((c += 1) == left) { continue; } else { i < 3 }; i += 1, u += 1) { t += 10; } return c * 1000 + u * 100 + t;" }, +{ false, false, 3, 0, 503, + "int c = 0; int n = 0; do { n += 1; } while (if ((c += 1) < left) { continue; } else { n < 3 }); return c * 100 + n;" }, +{ false, false, 0, 0, 303, + "int c = 0; int n = 0; do { n += 1; } while (if ((c += 1) < left) { continue; } else { n < 3 }); return c * 100 + n;" }, +// It targets the loop the condition belongs to, not one around it. +{ false, false, 0, 0, 3603, + "int t = 0; int c = 0; for (int i = 0; i < 3; i += 1) { int j = 0; int once = 0; while (if (once == 0) { once = 1; c += 1; continue; } else { j < 2 }) { j += 1; t += 1; } t += 10; } return t * 100 + c;" }, +// A condition every path of which leaves: it repeats until it breaks. +{ false, false, 4, 0, 4, + "int c = 0; while (if ((c += 1) < left) { continue; } else { break; }) { c += 100; } return c;" }, +{ false, false, 0, 0, 1, + "int c = 0; while (if ((c += 1) < left) { continue; } else { break; }) { c += 100; } return c;" }, +{ false, false, 4, 0, 40, + "int c = 0; int u = 0; for (int i = 0; if ((c += 1) < left) { continue; } else { break; }; i += 1, u += 1) { c += 100; } return c * 10 + u;" }, +{ false, false, 0, 0, 10, + "int c = 0; int u = 0; for (int i = 0; if ((c += 1) < left) { continue; } else { break; }; i += 1, u += 1) { c += 100; } return c * 10 + u;" }, +// A break in the update of a loop leaves that loop, with the effects +// of the update up to it and without the rest of the update. +{ false, false, 2, 0, 3, + "int t = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }) { t += 1; } return t;" }, +{ false, false, 20, 0, 10, + "int t = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }) { t += 1; } return t;" }, +{ false, false, 1, 0, 1022, + "int t = 0; int u = 0; for (int i = 0; i < 10; u += 1, i += if (i == left) { break; } else { 1 }, u += 100) { t += 1; } return u * 10 + t;" }, +{ false, false, 20, 0, 10110, + "int t = 0; int u = 0; for (int i = 0; i < 10; u += 1, i += if (i == left) { break; } else { 1 }, u += 100) { t += 1; } return u * 10 + t;" }, +// A break and a continue in one update clause. +{ false, false, 4, 0, 3305, + "int t = 0; int s = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }, s += if (i == 2) { continue; } else { 1 }, s += 10) { t += 1; } return s * 100 + t;" }, +{ false, false, 20, 0, 9910, + "int t = 0; int s = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }, s += if (i == 2) { continue; } else { 1 }, s += 10) { t += 1; } return s * 100 + t;" }, +// It leaves the loop the update belongs to, not one around it. +{ false, false, 0, 0, 36, + "int t = 0; for (int a = 0; a < 3; a += 1) { for (int b = 0; b < 5; b += if (b == 1) { break; } else { 1 }) { t += 1; } t += 10; } return t;" }, +// In the update of a loop used as an expression it carries the value. +{ false, false, 3, 0, 30, + "int r = for (int i = 0; ; i += if (i == left) { break i * 10; } else { 1 }) { if (i > 5) { break 99; } }; return r;" }, +{ false, false, 9, 0, 99, + "int r = for (int i = 0; ; i += if (i == left) { break i * 10; } else { 1 }) { if (i > 5) { break 99; } }; return r;" }, +// A value block may carry a value out of the loop expression around +// it; the value goes to that loop, not to one around it. +{ true, false, 4, 0, 5, + "int r = while (true) { int q = if (flag) { break left + 1; } else { 2 }; break q; }; return r;" }, +{ false, false, 4, 0, 2, + "int r = while (true) { int q = if (flag) { break left + 1; } else { 2 }; break q; }; return r;" }, +{ false, false, 0, 0, 100, + "int r = while (true) { int q = match (left) { 0 -> { break 100; }, int n -> n * 2 }; break q; }; return r;" }, +{ false, false, 4, 0, 8, + "int r = while (true) { int q = match (left) { 0 -> { break 100; }, int n -> n * 2 }; break q; }; return r;" }, +{ false, false, 2, 0, 63, + "int t = 0; int r = for (int i = 0; ; i += 1) { int inner = while (true) { t += 1; int q = if (t > left) { break t * 2; } else { 0 }; t += q; }; if (inner > 0) { break inner + i; } }; return r * 10 + t;" }, +{ false, false, 0, 0, 21, + "int t = 0; int r = for (int i = 0; ; i += 1) { int inner = while (true) { t += 1; int q = if (t > left) { break t * 2; } else { 0 }; t += q; }; if (inner > 0) { break inner + i; } }; return r * 10 + t;" }, +{ false, false, 5, 0, 2211, + "int t = 0; int r = Twice(while (true) { t += 1; int q = (t += 10) + (if (t > left) { break t; } else { 1 }); t += q; }); return r * 100 + t;" }, +{ false, false, 30, 0, 6834, + "int t = 0; int r = Twice(while (true) { t += 1; int q = (t += 10) + (if (t > left) { break t; } else { 1 }); t += q; }); return r * 100 + t;" }, +// A return leaves the method from a loop expression, directly and +// through a value block. +{ true, false, 0, 0, 77, + "int r = while (true) { int q = if (flag) { return 77; } else { 2 }; break q + left; }; return r;" }, +{ false, false, 3, 0, 5, + "int r = while (true) { int q = if (flag) { return 77; } else { 2 }; break q + left; }; return r;" }, +{ true, false, 0, 0, 1, + "int r = while (true) { if (flag) { return 1; } break 2; }; return r + 10;" }, +{ false, false, 0, 0, 12, + "int r = while (true) { if (flag) { return 1; } break 2; }; return r + 10;" }, +{ false, false, 6, 0, 6, + "int r = while (true) { return left; }; return r + 1;" }, +{ false, false, 1, 0, 11, + "int t = 0; int r = while (true) { t += 1; int q = for (int i = 0; ; i += 1) { if (i + t > left) { return i * 10 + t; } if (i == 2) { break i; } }; if (t == 3) { break q; } }; return 0 - r;" }, +{ false, false, 2, 0, 21, + "int t = 0; int r = while (true) { t += 1; int q = for (int i = 0; ; i += 1) { if (i + t > left) { return i * 10 + t; } if (i == 2) { break i; } }; if (t == 3) { break q; } }; return 0 - r;" }, +{ false, false, 99, 0, -2, + "int t = 0; int r = while (true) { t += 1; int q = for (int i = 0; ; i += 1) { if (i + t > left) { return i * 10 + t; } if (i == 2) { break i; } }; if (t == 3) { break q; } }; return 0 - r;" }, +// A block used as a value may leave instead of yielding one. +{ false, false, 9, 0, 100, + "int r = if (left > 5) { return 100; } else { left * 2 }; return r + 1;" }, +{ false, false, 3, 0, 7, + "int r = if (left > 5) { return 100; } else { left * 2 }; return r + 1;" }, +{ false, false, 0, 0, 50, + "int r = match (left) { 0 -> { return 50; }, int n -> n + 1 }; return r * 2;" }, +{ false, false, 4, 0, 10, + "int r = match (left) { 0 -> { return 50; }, int n -> n + 1 }; return r * 2;" }, +{ true, false, 4, 0, 7, + "return Twice(if (flag) { return 7; } else { left });" }, +{ false, false, 4, 0, 8, + "return Twice(if (flag) { return 7; } else { left });" }, +{ false, false, 3, 0, 604, + "int t = 0; int i = 0; while (i < 10) { i += 1; t += if (i > left) { break; } else { i }; } return t * 100 + i;" }, +{ false, false, 0, 0, 1, + "int t = 0; int i = 0; while (i < 10) { i += 1; t += if (i > left) { break; } else { i }; } return t * 100 + i;" }, +{ false, false, 4, 0, 9, + "int t = 0; for (int i = 0; i < 6; i += 1) { t += match (i) { 2 -> { continue; }, int n -> { if (n == left) { continue; } else { n } } }; } return t;" }, +{ false, false, 9, 0, 13, + "int t = 0; for (int i = 0; i < 6; i += 1) { t += match (i) { 2 -> { continue; }, int n -> { if (n == left) { continue; } else { n } } }; } return t;" }, +// A guard block that leaves through a match, and through continue. +{ true, false, 0, 0, 1, + "guard (left > 0) else { match (flag) { true -> { return 1; }, false -> { return 2; } } } return left + 10;" }, +{ false, false, 0, 0, 2, + "guard (left > 0) else { match (flag) { true -> { return 1; }, false -> { return 2; } } } return left + 10;" }, +{ false, false, 5, 0, 15, + "guard (left > 0) else { match (flag) { true -> { return 1; }, false -> { return 2; } } } return left + 10;" }, +{ false, false, 4, 0, 8, + "int t = 0; int i = 0; while (i < left) { i += 1; guard (i \\= 2) else { continue; } t += i; } return t;" }, +{ false, false, 1, 0, 1, + "int t = 0; int i = 0; while (i < left) { i += 1; guard (i \\= 2) else { continue; } t += i; } return t;" }, +{ false, false, 1, 0, 10, + "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } return r;" }, +{ false, false, 2, 0, 20, + "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } return r;" }, +{ false, false, 7, 0, 0, + "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } return r;" }, +{ false, false, 20, 0, 40, + "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 -> n + 1, _ -> 0 };" }, +{ false, false, 7, 0, 8, + "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 -> n + 1, _ -> 0 };" }, +{ false, false, 3, 0, 0, + "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 -> n + 1, _ -> 0 };" }, +{ false, false, 1, 1, 11, + "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 };" }, +{ false, false, 1, 5, 10, + "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 };" }, +{ false, false, 4, 1, 1, + "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 };" }, +{ false, false, 4, 4, 0, + "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 };" }, +{ true, false, 0, 0, 1, + "return match (flag) { true -> 1, false -> 2 };" }, +{ false, false, 0, 0, 2, + "return match (flag) { true -> 1, false -> 2 };" }, +// clang-format on diff --git a/Compiler/Fuzzing/Generated/SelectionCases.inc b/Compiler/Fuzzing/Generated/SelectionCases.inc new file mode 100644 index 00000000..02b7a180 --- /dev/null +++ b/Compiler/Fuzzing/Generated/SelectionCases.inc @@ -0,0 +1,216 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +// +// Generated from Compiler/Fuzzing/Cases/Selection.cases by +// `go -C helpers run ./cmd/execution-cases generate`. Do not edit. +// +// clang-format off +{ false, false, 1, 0, 10, + "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" }, +{ false, false, 2, 0, 20, + "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" }, +{ false, false, 5, 0, 30, + "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" }, +// Negative constants. +{ false, false, 1, 0, 10, + "int v = 0 - left; return match (v) { -1 -> 10, -2 -> 20, 0 -> 5, _ -> 7 };" }, +{ false, false, 2, 0, 20, + "int v = 0 - left; return match (v) { -1 -> 10, -2 -> 20, 0 -> 5, _ -> 7 };" }, +{ false, false, 0, 0, 5, + "int v = 0 - left; return match (v) { -1 -> 10, -2 -> 20, 0 -> 5, _ -> 7 };" }, +{ false, false, 3, 0, 7, + "int v = 0 - left; return match (v) { -1 -> 10, -2 -> 20, 0 -> 5, _ -> 7 };" }, +{ false, false, 1, 0, 1, + "int v = 0 - left - 9223372036854775807; return match (v) { -9223372036854775808 -> 1, -9223372036854775807 -> 2, _ -> 0 };" }, +{ false, false, 0, 0, 2, + "int v = 0 - left - 9223372036854775807; return match (v) { -9223372036854775808 -> 1, -9223372036854775807 -> 2, _ -> 0 };" }, +{ true, false, 3, 2, 1, + "int v = right - left; return match (v), (flag) { (-1), (true) -> 1, (-1), (_) -> 2, (_), (_) -> 3 };" }, +{ false, false, 3, 2, 2, + "int v = right - left; return match (v), (flag) { (-1), (true) -> 1, (-1), (_) -> 2, (_), (_) -> 3 };" }, +{ true, false, 2, 2, 3, + "int v = right - left; return match (v), (flag) { (-1), (true) -> 1, (-1), (_) -> 2, (_), (_) -> 3 };" }, +// Two bool subjects covered without a catch-all arm. +{ true, true, 0, 0, 1, + "return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 };" }, +{ true, false, 0, 0, 1, + "return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 };" }, +{ false, true, 0, 0, 2, + "return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 };" }, +{ false, false, 0, 0, 3, + "return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 };" }, +{ true, true, 0, 0, 3, + "return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, +{ true, false, 0, 0, 2, + "return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, +{ false, true, 0, 0, 1, + "return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, +{ false, false, 0, 0, 0, + "return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 };" }, +// The subject is evaluated once, whichever arm accepts. +{ false, false, 0, 0, 101, + "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> 300 }; return r + n;" }, +{ false, false, 1, 0, 202, + "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> 300 }; return r + n;" }, +{ false, false, 5, 0, 306, + "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> 300 }; return r + n;" }, +// Subjects are evaluated left to right, before any arm is tested. +{ false, false, 1, 0, 1, + "int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, (_), (_) -> 0 };" }, +{ false, false, 2, 0, 0, + "int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, (_), (_) -> 0 };" }, +// A guard runs only when the patterns of its arm accept. +{ false, false, 1, 0, 1001, + "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> 10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + calls;" }, +{ false, false, 2, 0, 2010, + "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> 10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + calls;" }, +{ false, false, 3, 0, 3000, + "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> 10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + calls;" }, +// A guard that is false passes the value on to the arms after it. +{ true, false, 1, 0, 1, + "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" }, +{ false, false, 1, 0, 2, + "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" }, +{ true, false, 9, 0, 3, + "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" }, +// A numeric guard is tested in Boolean context. +{ false, false, 1, 5, 1, + "return match (left) { 1 if right -> 1, _ -> 0 };" }, +{ false, false, 1, 0, 0, + "return match (left) { 1 if right -> 1, _ -> 0 };" }, +{ false, false, 2, 5, 0, + "return match (left) { 1 if right -> 1, _ -> 0 };" }, +// Only the body of the accepting arm runs. +{ false, false, 1, 0, 1001, + "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += 10), _ -> (n += 100) }; return r * 1000 + n;" }, +{ false, false, 2, 0, 10010, + "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += 10), _ -> (n += 100) }; return r * 1000 + n;" }, +{ false, false, 3, 0, 100100, + "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += 10), _ -> (n += 100) }; return r * 1000 + n;" }, +{ false, false, 2, 0, 1, + "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" }, +{ false, false, 3, 0, 2, + "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" }, +{ false, false, 4, 0, 0, + "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" }, +{ false, false, 0, 0, 1, + "long wide = 5; return match (wide) { 5 -> 1, _ -> 0 };" }, +{ false, false, 1, 0, 0, + "long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 ? 1 : 0;" }, +{ false, false, 2, 0, 1, + "long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 ? 1 : 0;" }, +{ false, false, 1, 0, 5, + "int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } return r;" }, +{ false, false, 4, 0, 8, + "int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } return r;" }, +{ true, false, 1, 0, 5, + "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } };" }, +{ false, false, 1, 0, 6, + "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } };" }, +{ false, false, 2, 0, 7, + "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } };" }, +{ false, false, 2, 3, 8, + "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } };" }, +{ false, false, 1, 4, 9, + "return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { int t = right * 3; t - 1 } };" }, +{ false, false, 2, 4, 11, + "return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { int t = right * 3; t - 1 } };" }, +// A statement arm may continue or leave the enclosing loop. +{ false, false, 10, 0, 8, + "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 -> { continue; }, 5 -> { break; }, _ -> { total += i; } } } return total;" }, +{ false, false, 3, 0, 1, + "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 -> { continue; }, 5 -> { break; }, _ -> { total += i; } } } return total;" }, +{ false, false, 0, 0, 0, + "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 -> { continue; }, 5 -> { break; }, _ -> { total += i; } } } return total;" }, +// A statement arm may supply the value of an enclosing loop expression. +{ false, false, 0, 0, 40, + "int n = 0; int r = while (true) { n += 1; match (n) { 4 -> { break n * 10; }, _ -> { } } }; return r;" }, +{ false, false, 1, 0, 100, + "match (left) { 1 -> { return 100; }, _ -> { } } return 5;" }, +{ false, false, 2, 0, 5, + "match (left) { 1 -> { return 100; }, _ -> { } } return 5;" }, +{ false, false, 1, 0, 5, + "match (left) { } return 5;" }, +{ false, false, 3, 5, 5, + "int r = if (left > right) { left } else { right }; return r;" }, +{ false, false, 9, 2, 9, + "int r = if (left > right) { left } else { right }; return r;" }, +{ true, false, 4, 0, 9, + "int r = if (flag) { int t = left * 2; t + 1 } else { int t = right * 3; t - 1 }; return r;" }, +{ false, false, 0, 5, 14, + "int r = if (flag) { int t = left * 2; t + 1 } else { int t = right * 3; t - 1 }; return r;" }, +// Only the selected block of an if expression runs. +{ true, false, 0, 0, 10001, + "int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 }; return r * 1000 + n;" }, +{ false, false, 0, 0, 20100, + "int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 }; return r * 1000 + n;" }, +{ true, true, 1, 2, 101, + "return (if (flag) { left } else { right }) + (if (other) { 100 } else { 200 });" }, +{ false, false, 1, 2, 202, + "return (if (flag) { left } else { right }) + (if (other) { 100 } else { 200 });" }, +{ false, false, 3, 0, 6, + "guard (left > 0) else { return 0 - 1; } return left * 2;" }, +{ false, false, 0, 0, -1, + "guard (left > 0) else { return 0 - 1; } return left * 2;" }, +{ false, false, 6, 0, 6, + "int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == 0) else { continue; } total += i; } return total;" }, +{ false, false, 1, 0, 0, + "int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == 0) else { continue; } total += i; } return total;" }, +{ false, false, 4, 0, 4, + "int n = 0; while (true) { guard (n < left) else { break; } n += 1; } return n;" }, +{ false, false, 0, 0, 0, + "int n = 0; while (true) { guard (n < left) else { break; } n += 1; } return n;" }, +{ false, false, 2, 3, 13, + "int total = 0; { int part = left * 2; total += part; } { int part = right * 3; total += part; } return total;" }, +{ false, false, 0, 0, 0, + "int total = 0; { int part = left * 2; total += part; } { int part = right * 3; total += part; } return total;" }, +{ false, false, 3, 0, 40, + "int n = 0; while (true) { { n += 1; if (n > left) { break; } } } { { return n * 10; } }" }, +{ false, false, 0, 0, 10, + "int n = 0; while (true) { { n += 1; if (n > left) { break; } } } { { return n * 10; } }" }, +{ false, false, 4, 0, 10, + "int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { continue; } } { int step = i * 2; total += step; } } return total;" }, +{ false, false, 1, 0, 0, + "int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { continue; } } { int step = i * 2; total += step; } } return total;" }, +{ false, false, 1, 0, 10, + "return match (left) { 1 -> 10 2 -> 20 _ -> 30 };" }, +{ false, false, 2, 0, 20, + "return match (left) { 1 -> 10 2 -> 20 _ -> 30 };" }, +{ false, false, 9, 0, 30, + "return match (left) { 1 -> 10 2 -> 20 _ -> 30 };" }, +// A pattern binding is a local of its arm and may be assigned. +{ false, false, 5, 0, 10, + "return match (left) { int value if value > 2 -> { value = value * 2; value }, int value -> { value += 1; value } };" }, +{ false, false, 1, 0, 2, + "return match (left) { int value if value > 2 -> { value = value * 2; value }, int value -> { value += 1; value } };" }, +// Assigning a binding does not change the subject the later arms test. +{ false, false, 3, 0, 2, + "int r = 0; match (left) { int value if (value = 7) > 9 -> { r = 1; }, 3 -> { r = 2; }, _ -> { r = 3; } } return r;" }, +{ false, false, 4, 0, 3, + "int r = 0; match (left) { int value if (value = 7) > 9 -> { r = 1; }, 3 -> { r = 2; }, _ -> { r = 3; } } return r;" }, +{ true, true, 0, 0, 1, + "int r = if (flag) { if (other) { 1 } else { 2 } } else { match (left) { 1 -> 10, _ -> 20 } }; return r;" }, +{ true, false, 0, 0, 2, + "int r = if (flag) { if (other) { 1 } else { 2 } } else { match (left) { 1 -> 10, _ -> 20 } }; return r;" }, +{ false, false, 1, 0, 10, + "int r = if (flag) { if (other) { 1 } else { 2 } } else { match (left) { 1 -> 10, _ -> 20 } }; return r;" }, +{ false, false, 2, 0, 20, + "int r = if (flag) { if (other) { 1 } else { 2 } } else { match (left) { 1 -> 10, _ -> 20 } }; return r;" }, +{ true, true, 1, 0, 11, + "int n = 0; int r = if (flag) { if (other) { n += 1; } else { n += 2; } match (left) { 1 -> { n += 10; } } n } else { 0 }; return r;" }, +{ true, false, 2, 0, 2, + "int n = 0; int r = if (flag) { if (other) { n += 1; } else { n += 2; } match (left) { 1 -> { n += 10; } } n } else { 0 }; return r;" }, +{ false, false, 1, 0, 0, + "int n = 0; int r = if (flag) { if (other) { n += 1; } else { n += 2; } match (left) { 1 -> { n += 10; } } n } else { 0 }; return r;" }, +{ false, false, 1, 2, 12, + "match (left) { 1 -> { match (right) { 2 -> { return 12; } } }, _ -> { } } return 5;" }, +{ false, false, 1, 3, 5, + "match (left) { 1 -> { match (right) { 2 -> { return 12; } } }, _ -> { } } return 5;" }, +{ false, false, 2, 2, 5, + "match (left) { 1 -> { match (right) { 2 -> { return 12; } } }, _ -> { } } return 5;" }, +// A guard condition with a store runs once, before the block decision. +{ false, false, 5, 0, 6, + "int n = left; guard ((n += 1) > 3) else { return n * 10; } return n;" }, +{ false, false, 1, 0, 20, + "int n = left; guard ((n += 1) > 3) else { return n * 10; } return n;" }, +// clang-format on diff --git a/Compiler/Fuzzing/LeavingExecutionCases.cpp b/Compiler/Fuzzing/LeavingExecutionCases.cpp new file mode 100644 index 00000000..b752bd5a --- /dev/null +++ b/Compiler/Fuzzing/LeavingExecutionCases.cpp @@ -0,0 +1,46 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include + +#include "ExecutionCases.hpp" +#include "LeavingExecutionCases.hpp" + +// Executable regressions for expressions that leave instead of yielding a +// value: `return`, `break` and `continue` out of blocks used as values, in +// loop bodies, in the condition and the update clause of a loop, and in +// loops used as expressions. Each case is a method body, the arguments it +// runs on and the value it must return. The expected values are written by +// hand from the language rules: what precedes the transfer is evaluated +// once and in order, what follows it is not evaluated, `break` and +// `continue` target the nearest loop, a `break` in the condition or the +// update clause of a loop leaves that loop, a `continue` in the condition +// evaluates the condition again, and a `continue` in an update clause ends +// the update. The cases are written in `Cases/Leaving.cases`; the rows here +// and the list `leavingCases` that the frontend test suite runs in a +// reference evaluator are generated from that file by +// `go -C helpers run ./cmd/execution-cases generate`. Here every case runs +// through CorePrep, Xpp, Xmm, LLVM and the ORC JIT, unoptimized and +// optimized. + +namespace Visual::XSharp::Fuzzing +{ + namespace + { + constexpr auto kCases = std::to_array({ +#include "Generated/LeavingCases.inc" + }); + + // The methods a body may call besides `Run` itself. + constexpr std::string_view kHelpers + = " public static int Twice(_ int value) { return value + " + "value; }\n"; + } // namespace + + void + ExerciseLeavingCases() + { + ExerciseExecutionCases("Leaving execution", kCases, kHelpers); + } +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/LeavingExecutionCases.hpp b/Compiler/Fuzzing/LeavingExecutionCases.hpp new file mode 100644 index 00000000..17ea2fc8 --- /dev/null +++ b/Compiler/Fuzzing/LeavingExecutionCases.hpp @@ -0,0 +1,13 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +namespace Visual::XSharp::Fuzzing +{ + /// Compile and run every regression program for expressions that leave + /// instead of yielding a value, and require the hand-written result + /// from both native pipeline modes. A wrong result or a rejected + /// program ends the process with a report. + void + ExerciseLeavingCases(); +} // namespace Visual::XSharp::Fuzzing diff --git a/Compiler/Fuzzing/SourceConsoleSmoke.cpp b/Compiler/Fuzzing/SourceConsoleSmoke.cpp new file mode 100644 index 00000000..3e7870c3 --- /dev/null +++ b/Compiler/Fuzzing/SourceConsoleSmoke.cpp @@ -0,0 +1,402 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "SourceFuzz.hpp" +#include "Visual/XSharp/Runtime/AARC.hpp" +#include "Visual/XSharp/Runtime/Text.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// Programs that write, run through the whole compiler. +// +// Each program is compiled from source through CorePrep, Xpp, Xmm, LLVM and +// the ORC JIT, unoptimized and optimized, and what it wrote to standard +// output and to standard error is compared with text written by hand. The +// runtime the generated code calls is the library this program links; its +// output goes to a sink instead of the streams of the process. +// +// The frontend tests run the same kind of program in a reference evaluator +// whose text functions are written a second time, in Haskell. This program +// is the other half: the generated code, the calling convention of each +// runtime function, and the runtime library itself. +// +// A string is an object of the runtime, so every program runs alone and +// must leave no allocation behind. + +namespace +{ + namespace Aarc = Visual::XSharp::Runtime::Aarc; + namespace Console = Visual::XSharp::Runtime::Console; + + struct Captured final + { + std::string output; + std::string error; + }; + + void + Receive(std::int64_t stream, + const char *bytes, + std::size_t count, + void *context) noexcept + { + auto *captured = static_cast(context); + (stream == VXS_CONSOLE_ERROR ? captured->error : captured->output) + .append(bytes, count); + } + + /// One program: the body of `Evaluate`, and what it writes. In the + /// expected text a line feed stands for the line terminator of the + /// platform; no program here writes a line feed of its own. + struct ConsoleCase final + { + std::string_view body; + std::string_view output; + std::string_view error; + }; + + constexpr std::string_view kHelpers + = " public static int Zero() { return 0; }\n" + " public static int Half(_ int v) { return v / 2; }\n" + " public static int Pick(_ int flag, _ int value) {" + " if (flag > 0) { return value; } return 7; }\n" + " public static int Log(_ int v) {" + " Console.Println(v); return v; }\n" + " public static String Name() { return \"Visual X#\"; }\n" + " public static String Twice(_ String s) { return s + s; }\n" + " public static void Greet(_ String who) {" + " Console.Println(\"Hello, \" + who + \"!\"); }\n" + " public static String Fizz(_ int n) {\n" + " if (n % 15 == 0) { return \"FizzBuzz\"; }\n" + " if (n % 3 == 0) { return \"Fizz\"; }\n" + " if (n % 5 == 0) { return \"Buzz\"; }\n" + " return \"\" + n;\n" + " }\n"; + + constexpr ConsoleCase kCases[] = { + // Plain output. + { "Console.Print(\"a\"); Console.Print(\"b\");", "ab", "" }, + { "Console.Println(\"Hello\");", "Hello\n", "" }, + { "Console.Println(\"\");", "\n", "" }, + { "System.Console.Println(\"qualified\");", "qualified\n", "" }, + { "Console.Println(\"%d\");", "%d\n", "" }, + { "Console.Println(42);", "42\n", "" }, + { "Console.Println(0 - 7);", "-7\n", "" }, + { "Console.Println(true);", "true\n", "" }, + { "Console.Println('x');", "x\n", "" }, + { "int n = 9223372036854775807; Console.Println(n);", + "9223372036854775807\n", + "" }, + { "uint u = 18446744073709551615; Console.Println(u);", + "18446744073709551615\n", + "" }, + { "byte small = 100; Console.Println(small);", "100\n", "" }, + { "ushort medium = 65535; Console.Println(medium);", "65535\n", "" }, + // Characters outside ASCII leave as UTF-8. + { "Console.Println(\"\xc3\xa9\xe2\x82\xac\");", + "\xc3\xa9\xe2\x82\xac\n", + "" }, + // Standard error. + { "Console.Error(\"e\"); Console.Errorln(\"f\");", "", "ef\n" }, + { "Console.Print(\"a\"); Console.Errorln(\"b\"); Console.Print(\"c\");", + "ac", + "b\n" }, + { "Console.Errorfn(\"Code: %d\", 7);", "", "Code: 7\n" }, + // Strings as values. + { "String s = \"kept\"; Console.Println(s);", "kept\n", "" }, + { "Console.Println(Name());", "Visual X#\n", "" }, + { "Console.Println(Twice(\"ab\"));", "abab\n", "" }, + // + joins. + { "String language = \"Visual X#\";" + " Console.Println(\"Hello from \" + language + \"!\");", + "Hello from Visual X#!\n", + "" }, + { "Console.Println(\"n=\" + 5);", "n=5\n", "" }, + { "Console.Println(5 + \"n\");", "5n\n", "" }, + { "Console.Println(\"b=\" + true);", "b=true\n", "" }, + { "Console.Println(\"c=\" + 'z');", "c=z\n", "" }, + { "Console.Println(\"a\" + 1 + 2);", "a12\n", "" }, + { "uint u = 7; Console.Println(\"u=\" + u);", "u=7\n", "" }, + { "String t = \"x\"; t += \"y\"; t += 3; Console.Println(t);", + "xy3\n", + "" }, + { "String t = \"\";" + " for (int i = 0; i < 4; i += 1) { t += i; } Console.Println(t);", + "0123\n", + "" }, + // Strings are compared by what they hold. + { "Console.Println(\"a\" == \"a\");", "true\n", "" }, + { "Console.Println(\"a\" \\= \"a\");", "false\n", "" }, + { "String a = \"ab\"; String b = \"a\" + \"b\";" + " Console.Println(a == b);", + "true\n", + "" }, + { "String a = \"ab\"; String b = \"a\" + \"c\";" + " Console.Println(a == b);", + "false\n", + "" }, + { "if (Name() == \"Visual X#\") { Console.Println(\"same\"); }" + " else { Console.Println(\"other\"); }", + "same\n", + "" }, + // Formats. + { "Console.Printf(\"plain\");", "plain", "" }, + { "Console.Printfn(\"%d\", 42);", "42\n", "" }, + { "Console.Printfn(\"gcd(%d, %d) = %d\", 1071, 462, 21);", + "gcd(1071, 462) = 21\n", + "" }, + { "Console.Printf(\"%5d|\", 42);", " 42|", "" }, + { "Console.Printf(\"%-5d|\", 42);", "42 |", "" }, + { "Console.Printf(\"%05d\", 0 - 42);", "-0042", "" }, + { "Console.Printf(\"%+d\", 42);", "+42", "" }, + { "Console.Printf(\"%'d\", 0 - 1234567);", "-1'234'567", "" }, + { "byte small = 0 - 100; Console.Printf(\"%d\", small);", "-100", "" }, + { "Console.Printf(\"%x\", 0 - 255);", "-ff", "" }, + { "Console.Printf(\"%#08x\", 255);", "0x0000ff", "" }, + { "uint u = 4294967295; Console.Printf(\"%x\", u);", "ffffffff", "" }, + { "ubyte tiny = 255; Console.Printf(\"%u\", tiny);", "255", "" }, + { "uint u = 1234567; Console.Printf(\"%'u\", u);", "1'234'567", "" }, + { "Console.Printf(\"%10s|\", \"abc\");", " abc|", "" }, + { "Console.Printf(\"%-10s|\", \"abc\");", "abc |", "" }, + { "Console.Printf(\"%5.1s|\", \"abc\");", " a|", "" }, + { "Console.Printf(\"%s: %d\", \"total\", 7);", "total: 7", "" }, + { "Console.Printf(\"%3c|\", 'q');", " q|", "" }, + { "Console.Printf(\"%6b|\", false);", " false|", "" }, + { "Console.Printf(\"%b\", 1 > 2);", "false", "" }, + { "Console.Printf(\"%d%%\", 50);", "50%", "" }, + { "Console.Printf(\"A%nB\");", "A\nB", "" }, + { "Console.Printf(\"%f\", 12.5);", "12.500000", "" }, + { "Console.Printf(\"%.0f\", 2.5);", "2", "" }, + { "Console.Printf(\"%.0f\", 3.5);", "4", "" }, + { "Console.Printf(\"%010.3f\", 3.14159);", "000003.142", "" }, + { "Console.Printf(\"%'.2f\", 1234567.5);", "1'234'567.50", "" }, + { "Console.Printf(\"%.20f\", 0.1);", "0.10000000000000000555", "" }, + { "float price = 19.99; Console.Printfn(\"Price: %.2f\", price);", + "Price: 19.99\n", + "" }, + { "lfloat single = 0.5; Console.Printf(\"%.3f\", single);", + "0.500", + "" }, + { "Console.Printf(\"%*d|\", 5, 42);", " 42|", "" }, + { "Console.Printf(\"%*.*f|\", 8, 2, 3.14159);", " 3.14|", "" }, + { "String t = Console.Format(\"Name: %s, Age: %d\", \"Ada\", 36);" + " Console.Println(t);", + "Name: Ada, Age: 36\n", + "" }, + { "Console.Print(Console.Format(\"%05d\", 42)" + " + Console.Format(\"%x\", 255));", + "00042ff", + "" }, + // Strings that callables take, capture, make and return. + { "auto greet = \\(String who) -> \"Hi \" + who;" + " Console.Println(greet(\"Ada\"));", + "Hi Ada\n", + "" }, + { "String prefix = Name() + \": \";" + " auto label = \\(int n) -> prefix + n;" + " Console.Println(label(1)); Console.Println(label(2));", + "Visual X#: 1\nVisual X#: 2\n", + "" }, + { "auto twice = \\(int v) -> { Console.Printf(\"%d,\", v);" + " return v * 2; }; Console.Println(twice(twice(1)));", + "1,2,4\n", + "" }, + { "auto make = \\(int n) -> Console.Format(\"<%03d>\", n);" + " String all = \"\"; for (int i = 0; i < 3; i += 1)" + " { all += make(i); } Console.Println(all);", + "<000><001><002>\n", + "" }, + // A callable that is made and never called leaves nothing behind. + { "String kept = Twice(\"ab\");" + " auto never = \\(int n) -> kept + n; Console.Println(\"x\");", + "x\n", + "" }, + // A method is a callable value: it writes where its call stands. + { "auto f = Log; int x = f(1); Console.Println(9);", "1\n9\n", "" }, + // Appending makes a new string; another name keeps the old one. + { "String s = \"a\"; String t = s; s += \"b\";" + " Console.Println(s + t);", + "aba\n", + "" }, + { "String s = \"\"; for (int i = 0; i < 5; i += 1)" + " { String old = s; s += i; if (old == s) { s += \"!\"; } }" + " Console.Println(s);", + "01234\n", + "" }, + // String operations in arguments that are and are not needed. + { "Console.Println(Pick(1, Half(8)) + Name());", "4Visual X#\n", "" }, + { "String made = Twice(Name() + \"!\");" + " Console.Println(Pick(0, 8 / Zero()));", + "7\n", + "" }, + { "Console.Println(Console.Format(\"[%s]\"," + " Console.Format(\"%5s\", Console.Format(\"%d\", 42))));", + "[ 42]\n", + "" }, + { "Console.Println(Twice(Twice(\"ab\")) == \"abababab\");", + "true\n", + "" }, + // A conditional selects one of two strings; the other is never + // made, and the one that is replaced is released. + { "String s = Zero() > 0 ? \"a\" : \"b\"; Console.Println(s);", + "b\n", + "" }, + { "Console.Println(Zero() == 0 ? Name() : Twice(\"x\"));", + "Visual X#\n", + "" }, + { "String a = \"x\"; String b = \"y\";" + " String c = Zero() > 0 ? a : b; Console.Println(c + a + b);", + "yxy\n", + "" }, + { "Console.Println(Zero() > 0 ? \"a\"" + " : Zero() == 0 ? \"b\" : \"c\");", + "b\n", + "" }, + { "String s = Zero() > 0 ? \"\" + Log(1) : \"\" + Log(2);" + " Console.Println(s);", + "2\n2\n", + "" }, + { "String s = \"\"; for (int i = 0; i < 4; i += 1)" + " { s += i % 2 == 0 ? \"e\" : \"o\"; } Console.Println(s);", + "eoeo\n", + "" }, + { "Greet(Zero() == 0 ? Name() : \"nobody\");", + "Hello, Visual X#!\n", + "" }, + { "String s = \"keep\"; s = Zero() > 0 ? \"lost\" : s;" + " Console.Println(s);", + "keep\n", + "" }, + // A selected string nothing reads is released all the same. + { "String s = Zero() == 0 ? Twice(\"ab\") : Name();" + " Console.Println(\"x\");", + "x\n", + "" }, + // Names written through the namespace and the type. + { "Console.Println(Fuzz.Program.Half(8));", "4\n", "" }, + { "auto f = Program::Log; int x = f(1); Console.Println(9);", + "1\n9\n", + "" }, + { "auto f = Fuzz.Program::Half; Console.Println(f(10));", "5\n", "" }, + // A string that is made and never written is released as well. + { "String t = Console.Format(\"%d\", 1); Console.Println(\"x\");", + "x\n", + "" }, + // Loops, branches and methods. + { "for (int i = 0; i < 3; i += 1) { Console.Print(i); }", "012", "" }, + { "Greet(\"Ada\"); Greet(\"Alan\");", + "Hello, Ada!\nHello, Alan!\n", + "" }, + { "Console.Println(Fizz(3)); Console.Println(Fizz(5));" + " Console.Println(Fizz(15)); Console.Println(Fizz(7));", + "Fizz\nBuzz\nFizzBuzz\n7\n", + "" }, + // Output is an effect: it happens where it is written. + { "int x = Log(1); Console.Println(2);", "1\n2\n", "" }, + { "int x = Log(1); int y = Log(2); Console.Println(y + x);", + "1\n2\n3\n", + "" }, + { "Console.Println(Pick(0, Log(5)));", "5\n7\n", "" }, + { "Console.Printf(\"%d %d\", Log(1), Log(2));", "1\n2\n1 2", "" }, + { "Console.Printf(\"%*d\", Log(3), Log(4));", "3\n4\n 4", "" }, + { "Console.Println(\"a\" + Log(1) + Log(2));", "1\n2\na12\n", "" }, + { "int v = Zero() > 0 ? Log(1) : Log(2);", "2\n", "" }, + { "auto say = \\(int v) -> Log(v); int x = say(1);" + " Console.Println(9);", + "1\n9\n", + "" }, + // A value that only computes is still computed by need. + { "int z = 1 / Zero(); Console.Println(\"ok\");", "ok\n", "" }, + { "int z = 8 / Zero(); Console.Println(Pick(0, z));", "7\n", "" }, + }; + + /// The expected text with each line feed as the line terminator of the + /// platform. + [[nodiscard]] auto + OnPlatform(std::string_view expected) -> std::string + { + std::string text; + for (const auto character : expected) + if (character == '\n') + { +#ifdef _WIN32 + text += "\r\n"; +#else + text += '\n'; +#endif + } + else + { + text += character; + } + return text; + } + + [[nodiscard]] auto + Program(std::string_view body) -> std::string + { + std::string program = "namespace Fuzz;\nclass Program {\n"; + program += kHelpers; + program += " public static int Evaluate() {\n "; + program += body; + program += "\n return 0;\n }\n}\n"; + return program; + } + + int + Smoke() + { + std::size_t index = 0U; + for (const auto &consoleCase : kCases) + { + llvm::errs() << "Console execution " << ++index << " of " + << std::size(kCases) << ": " << consoleCase.body + << '\n'; + const auto before = Aarc::LiveAllocations(); + Captured captured; + Console::SetSink(Receive, &captured); + // Both pipeline modes run the program, so everything is written + // twice: once by the unoptimized code and once by the optimized. + Visual::XSharp::Fuzzing::ExerciseExpectedValue( + Program(consoleCase.body), + 0); + Console::SetSink(nullptr, nullptr); + const auto output = OnPlatform(consoleCase.output); + const auto error = OnPlatform(consoleCase.error); + if (captured.output != output + output + || captured.error != error + error) + { + llvm::errs() + << "console case wrote the wrong text\n" + << " expected output, twice: " << output << '\n' + << " written output: " << captured.output << '\n' + << " expected error, twice: " << error << '\n' + << " written error: " << captured.error << '\n'; + return 1; + } + const auto after = Aarc::LiveAllocations(); + if (after != before) + { + llvm::errs() << "console case left " << (after - before) + << " AARC allocation(s) behind\n"; + return 1; + } + } + return 0; + } +} // namespace + +int +main() +{ + return Visual::XSharp::Support::RunOnCompilerStack([] { + return Smoke(); + }); +} diff --git a/Compiler/Fuzzing/SourceExecutionSmoke.cpp b/Compiler/Fuzzing/SourceExecutionSmoke.cpp new file mode 100644 index 00000000..3272f832 --- /dev/null +++ b/Compiler/Fuzzing/SourceExecutionSmoke.cpp @@ -0,0 +1,46 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include + +#include "BranchingExecutionCases.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// The executable regressions of the branching table: every program of it +// runs through CorePrep, Xpp, Xmm, LLVM and the ORC JIT, unoptimized and +// optimized, and must return its expected value from both. The expression +// and leaving tables are `source_expression_smoke`. They are programs of +// their own, beside `source_fuzz_smoke` and `source_feature_smoke`, because +// each is a deterministic check under one process watchdog, and a watchdog +// is meant to end a run that never finishes, not to bound the size of a +// test table. + +namespace +{ + int + Smoke() + { + // Hand-written results for match, if expressions and guard, and + // the programs at the nesting limits of the frontend. + Visual::XSharp::Fuzzing::ExerciseBranchingCases(); + return 0; + } +} // namespace + +int +main() +{ + // The programs include ones nested up to the frontend's limits, which + // only compile on the stack the compiler runs on in `vxs`. + return Visual::XSharp::Support::RunOnCompilerStack([] { + const auto result = Smoke(); + // What the compiler stack held for all of the above, the programs + // at the nesting limits among them. The line is how the figure is + // read on platforms where nobody measures by hand; zero means the + // platform does not report it. + llvm::errs() << "compiler stack committed: " + << Visual::XSharp::Support::CommittedStackBytes() / 1024U + << " KiB\n"; + return result; + }); +} diff --git a/Compiler/Fuzzing/SourceExpressionSmoke.cpp b/Compiler/Fuzzing/SourceExpressionSmoke.cpp new file mode 100644 index 00000000..7785d9bd --- /dev/null +++ b/Compiler/Fuzzing/SourceExpressionSmoke.cpp @@ -0,0 +1,43 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include "ExpressionExecutionCases.hpp" +#include "LeavingExecutionCases.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// The executable regressions of two of the three large tables: every program +// of the expression and leaving tables runs through CorePrep, Xpp, Xmm, LLVM +// and the ORC JIT, unoptimized and optimized, and must return its expected +// value from both. The branching table is `source_execution_smoke`. +// +// The three tables were one program. Every run of a body passes its inputs +// as calls, so that no stage can fold them, and since arguments are passed by +// need each such argument is a suspended computation with code of its own. +// The programs grew with that, and under the sanitizers the one program ran +// past its process watchdog with every case passing. A watchdog is meant to +// end a run that never finishes, not to bound the size of a test table, so +// the tables are two programs and the watchdog is what it was. + +namespace +{ + int + Smoke() + { + // Hand-written results for assignments and increments used as values + // and for loops used as expressions. + Visual::XSharp::Fuzzing::ExerciseExpressionCases(); + // Hand-written results for expressions that leave instead of + // yielding a value. + Visual::XSharp::Fuzzing::ExerciseLeavingCases(); + return 0; + } +} // namespace + +int +main() +{ + // The compiler runs on the stack it runs on in `vxs`. + return Visual::XSharp::Support::RunOnCompilerStack([] { + return Smoke(); + }); +} diff --git a/Compiler/Fuzzing/SourceFeatureSmoke.cpp b/Compiler/Fuzzing/SourceFeatureSmoke.cpp new file mode 100644 index 00000000..77963c20 --- /dev/null +++ b/Compiler/Fuzzing/SourceFeatureSmoke.cpp @@ -0,0 +1,1083 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include + +#include "ExecutionCases.hpp" +#include "SourceFuzz.hpp" +#include "Visual/XSharp/Runtime/AARC.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// Executable regressions for the features whose tables are written by hand in +// this file: methods with inferred return types, evaluation by need, +// arguments passed by need, classic enums, closures with the runtime that +// owns them, and programs that own +// closures while control leaves through a block used as a value. Every +// program runs through CorePrep, Xpp, Xmm, LLVM and the ORC JIT, unoptimized +// and optimized, and must return its expected value from both. +// +// They are a program of their own, beside `source_execution_smoke` and +// `source_fuzz_smoke`, because each smoke program is a deterministic check +// under one process watchdog, and a watchdog is meant to end a run that +// never finishes, not to bound the size of a test table. + +namespace +{ + // Programs that own closures while control leaves through a block used + // as a value. The ownership-flow verifiers of Xpp and Xmm run on every + // program compiled here, so a path that left without releasing what it + // owns, or released it twice, is rejected. The two CorePrep lowerings + // are compared on them as well. These are compiled and verified; the + // table of closure cases below runs closures. + constexpr std::array kOwnershipCases{ { + // An initializer that never completes, after a closure was created. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); " + "auto held = [kept = a] \\ -> kept; " + "int r = if (held() > 3) { return held(); } else { return 0; }; " + "return r; } }", + // A do/while whose condition never completes, reached after the + // body and after a continue, with a closure created in the body. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); " + "auto held = [kept = a] \\ -> kept; int n = 0; " + "do { n += held(); auto inner = [seen = n] \\ -> seen + 1; " + "if (inner() < 3) { continue; } } " + "while (if (n > 6) { return n; } else { return 0 - n; }); " + "return 0; } }", + // Break and continue out of value blocks in nested loops, with + // closures created in both loops. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); int t = 0; " + "for (int i = 0; i < a; i += 1) { " + "auto step = [by = i] \\ -> by + 1; int j = 0; " + "while (j < 3) { j += 1; auto pick = [at = j] \\ -> at; " + "t += if (pick() == 2) { break; } else { step() }; } " + "t += match (i) { 1 -> { continue; }, int n -> step() + n }; } " + "return t; } }", + // A value carried out of a value block to a loop expression, a + // break in a loop condition and a continue in a loop update, each + // with a closure alive at the transfer. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); int t = 0; " + "int found = while (true) { t += 1; " + "auto seen = [at = t] \\ -> at * 2; " + "int q = if (seen() > a) { break seen(); } else { 0 }; t += q; }; " + "auto keep = [of = found] \\ -> of; int n = 0; " + "while (if (n > keep()) { break; } else { true }) { n += 1; } " + "for (int i = 0; i < 4; i += if (i == 1) { continue; } else { 1 }) " + "{ auto tick = [by = i] \\ -> by; n += tick(); if (i == 1) { i += 2; " + "} } return found + n; } }", + // A return out of a value block inside a loop expression, with + // closures created in the loop. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); int t = 0; " + "int r = while (true) { t += 1; auto seen = [at = t] \\ -> at; " + "int q = if (seen() > a) { return seen() * 10; } else { seen() }; " + "if (q == 3) { break q; } }; return r; } }", + // Callables whose result type is inferred from returns that stand + // in a value block and in a loop expression. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); " + "auto pick = \\(int v) -> { int q = if (v > 0) { return 1; } " + "else { 2 }; return q; }; " + "auto scan = [limit = a] \\(int v) -> { int q = while (true) { " + "if (v > limit) { return 7; } break 2; }; return q + v; }; " + "return pick(a) + scan(a); } }", + // A callable every path of which returns from a value block, and + // one whose returns are a different type from its creator's. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); " + "auto twice = \\(int v) -> { int q = if (v > 0) { return v * 2; } " + "else { return 0 - v; }; return q; }; " + "auto positive = \\(int w) -> { bool b = if (w > 0) { return true; " + "} else { false }; return b; }; " + "int q = if (positive(a)) { return twice(a); } else { 20 }; " + "return q; } }", + // A continue in a loop condition and a break in a loop update, + // each with a closure alive at the transfer. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int a = Step(4); int c = 0; " + "auto limit = [of = a] \\ -> of; int n = 0; " + "while (if ((c += 1) < limit()) { continue; } else { n < 2 }) { " + "auto tick = [by = n] \\ -> by + 1; n = tick(); } " + "for (int i = 0; i < 9; i += if (i == limit()) { break; } else { 1 " + "}) { auto add = [by = i] \\(int w) -> w + by; n = add(n); } " + "return c * 100 + n; } }", + // A callable created inside a callable: the inner one reads a + // parameter of the outer one and a local of the method, which the + // outer one must capture for it and nothing else. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int k = Step(4); " + "auto outer = \\(int v) -> { auto inner = \\(int w) -> w + k + v; " + "return inner(v) * 2; }; return outer(k); } }", + // Three levels, with explicit and implicit captures mixed. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int k = Step(4); " + "auto a = [k] \\(int v) -> { auto b = \\(int w) -> { " + "auto c = [k, v, w] \\(int x) -> x + w * 10 + v * 100 + k * 1000; " + "return c(1); }; return b(2); }; return a(3); } }", + // An inner callable that outlives the call that created it. + "namespace Fuzz; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": 0; } " + "public static int Evaluate() { int k = Step(4); " + "auto make = \\(int v) -> { auto inner = \\(int w) -> w + v; " + "return inner; }; auto f = make(k); auto g = make(k + 2); " + "return f(2) * 100 + g(2); } }", + } }; + + // Methods whose return type is inferred, called from the bodies below. + // A chain is declared against the order of its inference, and two + // methods return each other. + constexpr std::string_view kInferredHelpers + = " public static auto Twice(_ int value) { return value + " + "value; }\n" + " public static auto Factorial(_ int value) { if (value <= 1) " + "{ return 1; } return value * Factorial(value - 1); }\n" + " public static auto First(_ int v) { return Second(v) + 1; " + "}\n" + " public static auto Second(_ int v) { return Third(v) + 10; " + "}\n" + " public static auto Third(_ int v) { return v * 2; }\n" + " public static auto Even(_ int v) { if (v == 0) { return " + "true; } return Odd(v - 1); }\n" + " public static auto Odd(_ int v) { if (v == 0) { return " + "false; } return Even(v - 1); }\n" + " public static auto Pick(_ int v) { int q = if (v > 0) { " + "return v * 3; } else { 5 }; return q + 1; }\n" + " public static auto Scan(_ int v) { int q = while (true) { " + "if (v > 3) { return 7; } break 2; }; return q + v; }\n"; + + // Closures that are created and called. Each expected value is worked + // out by hand from the capture rules: a capture initializer is + // evaluated once, where the closure is created; a closure created + // inside another reads the names around both through the outer one; + // and a closure keeps what it captured after the call that created it + // has returned. + constexpr auto kClosureCases = std::to_array< + Visual::XSharp::Fuzzing::ExecutionCase>({ + { false, + false, + 3, + 0, + 8, + "auto outer = \\(int v) -> { auto inner = \\(int w) -> w + 1; " + "return inner(v) * 2; }; return outer(left);" }, + { false, + false, + 3, + 3, + 18, + "int k = left; auto outer = \\(int v) -> { auto inner = " + "\\(int w) -> w + k + v; return inner(v) * 2; }; " + "return outer(right);" }, + { false, + false, + 5, + 1, + 14, + "int k = left; auto outer = \\(int v) -> { auto inner = " + "\\(int w) -> w + k + v; return inner(v) * 2; }; " + "return outer(right);" }, + { false, + false, + 3, + 3, + 9, + "int k = left; auto outer = [k] \\(int v) -> { auto inner = " + "[k, v] \\(int w) -> w + k + v; return inner(v); }; " + "return outer(right);" }, + { false, + false, + 4, + 0, + 4321, + "int k = left; auto a = \\(int v) -> { auto b = \\(int w) -> { " + "auto c = \\(int x) -> x + w * 10 + v * 100 + k * 1000; " + "return c(1); }; return b(2); }; return a(3);" }, + { false, + false, + 5, + 7, + 709, + "auto make = \\(int v) -> { auto inner = \\(int w) -> w + v; " + "return inner; }; auto f = make(left); auto g = make(right); " + "return f(2) * 100 + g(2);" }, + { false, + false, + 1, + 0, + 111, + "int k = left; auto held = [kept = k] \\ -> kept; k += 10; " + "return held() * 100 + k;" }, + { false, + false, + 3, + 0, + 12, + "auto pick = \\(int v) -> { int q = if (v > 0) { return 1; } " + "else { 2 }; return q; }; " + "return pick(left) * 10 + pick(0 - left);" }, + { false, + false, + 1, + 0, + 307, + "auto scan = \\(int v) -> { int q = while (true) { " + "if (v > 3) { return 7; } break 2; }; return q + v; }; " + "return scan(left) * 100 + scan(left + 3);" }, + { false, + false, + 3, + 0, + 6, + "int n = 0; int t = 0; while (if (n >= left) { break; } else { " + "true }) { auto step = [by = n] \\ -> by + 1; n = step(); " + "t += n; } return t;" }, + { false, + false, + 3, + 0, + 6, + "int t = 0; for (int i = 0; i < 10; i += if (i == left) { " + "break; } else { 1 }) { auto add = [by = i] \\(int w) -> w + " + "by; t = add(t); } return t;" }, + // A closure passed to a method, made by one and returned by one. + { false, false, 4, 0, 12, "return Apply(\\(int w) -> w * 3, left);" }, + { false, + false, + 5, + 2, + 711, + "auto add = Adder(left); auto ten = Adder(10); " + "return add(right) * 100 + ten(1);" }, + { false, + false, + 3, + 0, + 8, + "auto f = \\(int w) -> w + 1; auto g = Pass(f); " + "return g(left) + f(left);" }, + { false, + false, + 5, + 1, + 6, + "auto a = \\(int w) -> w + 1; auto b = \\(int w) -> w * 2; " + "auto c = Choose(left > right, a, b); return c(left);" }, + { false, + false, + 5, + 9, + 10, + "auto a = \\(int w) -> w + 1; auto b = \\(int w) -> w * 2; " + "auto c = Choose(left > right, a, b); return c(left);" }, + // A method named where a value is expected. + { false, + false, + 3, + 4, + 14, + "auto f = Plain; return Apply(f, left) + Apply(Plain, right);" }, + // A closure variable assigned again, in a loop and from itself. + { false, + false, + 3, + 0, + 103, + "auto f = Adder(0); for (int i = 1; i <= left; i += 1) { " + "f = Adder(i); } return f(100);" }, + { false, + false, + 0, + 0, + 100, + "auto f = Adder(0); for (int i = 1; i <= left; i += 1) { " + "f = Adder(i); } return f(100);" }, + { false, + false, + 5, + 0, + 7, + "auto f = Adder(1); f = Compose2(f); return f(left);" }, + // Closures that exist on one branch only. + { false, + false, + 4, + 1, + 5, + "int r = 0; if (left > right) { auto f = Adder(left); " + "r = f(1); } else { auto g = Adder(right); " + "auto h = Compose2(g); r = h(1); } return r;" }, + { false, + false, + 1, + 3, + 7, + "int r = 0; if (left > right) { auto f = Adder(left); " + "r = f(1); } else { auto g = Adder(right); " + "auto h = Compose2(g); r = h(1); } return r;" }, + // A closure result that nothing receives. + { false, + false, + 5, + 0, + 7, + "_ = Adder(left); auto f = Adder(2); _ = Adder(3); " + "return f(left);" }, + // A closure kept alive only by the closure that captured it. + { false, + false, + 5, + 1, + 12, + "auto inner = Adder(left); auto outer = \\(int w) -> " + "inner(w) * 2; return outer(right);" }, + // A closure made in every pass of a loop that returns from + // its middle. + { false, + false, + 4, + 0, + 103, + "for (int i = 0; i < 5; i += 1) { auto f = Adder(i); " + "if (f(left) > 6) { return f(100); } } return 0;" }, + { false, + false, + 0, + 0, + 0, + "for (int i = 0; i < 5; i += 1) { auto f = Adder(i); " + "if (f(left) > 6) { return f(100); } } return 0;" }, + }); + + // Methods that take, make and return closures. A parameter is borrowed + // and a result is owned, so `Pass` and `Choose` must hand out a + // reference of their own, and the closure of `Compose2` must keep the + // closure it was given. + constexpr std::string_view kClosureHelpers + = " public static int Apply(_ (int) -> int f, _ int v) { return " + "f(v); }\n" + " public static (int) -> int Adder(_ int n) { return \\(int w) " + "-> w + n; }\n" + " public static (int) -> int Pass(_ (int) -> int f) { return " + "f; }\n" + " public static (int) -> int Choose(_ bool c, _ (int) -> int a, " + "_ (int) -> int b) { if (c) { return a; } return b; }\n" + " public static (int) -> int Compose2(_ (int) -> int f) { " + "return \\(int w) -> f(f(w)); }\n" + " public static int Plain(_ int value) { return value + value; " + "}\n"; + + // Classic enums. A value of an enum is its underlying integer in the + // generated code, so these runs show that the numbering of members, the + // comparison of values and the selection of a match arm by member + // survive every native stage. + constexpr std::string_view kEnumDeclarations + = "enum Status { NONE, UNKNOWN = 0, READY }\n" + "enum Level = byte { LOW = 1, MID, HIGH = 10, TOP }\n" + // Values computed from earlier members: 1, 2, 4, 7, 8, 2, 250, 5. + "enum Flag = ubyte { READ = 1, WRITE = READ << 1, EXECUTE = WRITE * " + "2, ALL = READ | WRITE | EXECUTE, NEXT, HALF = EXECUTE / 2, REST = " + "!(EXECUTE + 1), ROUNDED = 9 // 2 }\n"; + constexpr std::string_view kEnumHelpers + = " public static Status Pick(_ int v) { if (v > 0) { return " + "Status.READY; } return Status.NONE; }\n" + " public static int Rank(_ Level l) { return match (l) { .LOW " + "-> 1, .MID -> 2, .HIGH -> 10, .TOP -> 11 }; }\n" + " public static Level Raise(_ Level l) { return match (l) { " + ".LOW -> Level.MID, .MID -> Level.HIGH, _ -> Level.TOP }; }\n" + " public static Flag Bit(_ int v) { return match (v) { 1 -> " + ".READ, 2 -> .WRITE, 4 -> .EXECUTE, 7 -> .ALL, 8 -> .NEXT, 250 -> " + ".REST, 5 -> .ROUNDED, _ -> .HALF }; }\n"; + constexpr auto kEnumCases = std::to_array< + Visual::XSharp::Fuzzing::ExecutionCase>({ + // A member has the value its expression computes from the + // members before it; a match over them is complete when every + // value is named. + { false, + false, + 1, + 0, + 1, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, + false, + 2, + 0, + 2, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, + false, + 4, + 0, + 3, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, + false, + 7, + 0, + 4, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, + false, + 8, + 0, + 5, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, + false, + 250, + 0, + 6, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, + false, + 5, + 0, + 7, + "return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, " + ".ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };" }, + { false, false, 0, 0, 1, "return Flag.HALF == Flag.WRITE ? 1 : 0;" }, + { false, false, 0, 0, 0, "return Flag.NEXT == Flag.ALL ? 1 : 0;" }, + { false, + false, + 3, + 0, + 1, + "Status s = Pick(left); return s == Status.READY ? 1 : 2;" }, + { false, + false, + 0, + 0, + 2, + "Status s = Pick(left); return s == Status.READY ? 1 : 2;" }, + { false, + false, + 0, + 0, + 2, + "Status s = Pick(left); return s \\= Status.NONE ? 1 : 2;" }, + { false, + false, + 0, + 0, + 1, + "return Status.NONE == Status.UNKNOWN ? 1 : 2;" }, + { false, + false, + 0, + 0, + 12021, + "return Rank(Level.LOW) + Rank(Level.MID) * 10 + " + "Rank(Level.HIGH) * 100 + Rank(Level.TOP) * 1000;" }, + { false, + false, + 3, + 0, + 20, + "return match (Pick(left)) { .NONE -> 10, .READY -> 20 };" }, + { false, + false, + 0, + 0, + 10, + "return match (Pick(left)) { .NONE -> 10, .READY -> 20 };" }, + { false, + false, + 0, + 0, + 10, + "return match (Pick(left)) { .UNKNOWN -> 10, .READY -> 20 };" }, + { false, + false, + 1, + 1, + 1, + "return match (Pick(left)) { .READY if (right > 0) -> 1, " + ".READY -> 2, .NONE -> 3 };" }, + { false, + false, + 1, + 0, + 2, + "return match (Pick(left)) { .READY if (right > 0) -> 1, " + ".READY -> 2, .NONE -> 3 };" }, + { false, + false, + 0, + 5, + 3, + "return match (Pick(left)) { .READY if (right > 0) -> 1, " + ".READY -> 2, .NONE -> 3 };" }, + { false, + false, + 0, + 0, + 1011, + "return Rank(Raise(Raise(Level.LOW))) * 100 + " + "Rank(Raise(Level.HIGH));" }, + { false, + false, + 1, + 0, + 1, + "Status s = match (left) { 1 -> Status.READY, _ -> Status.NONE " + "}; return s == Status.READY ? 1 : 0;" }, + { false, + false, + 0, + 1, + 1, + "return match (Pick(left)), (right > 0) { (.NONE), (true) -> 1, " + "(.NONE), (false) -> 2, (.READY), (true) -> 3, (.READY), (false) " + "-> 4 };" }, + { false, + false, + 1, + 0, + 4, + "return match (Pick(left)), (right > 0) { (.NONE), (true) -> 1, " + "(.NONE), (false) -> 2, (.READY), (true) -> 3, (.READY), (false) " + "-> 4 };" }, + { false, + false, + 0, + 0, + 3, + "Level l = Level.LOW; int n = 0; while (l \\= Level.TOP) { " + "l = Raise(l); n += 1; } return n;" }, + { false, + false, + 2, + 0, + 56, + "auto f = \\(Status s) -> s == Status.READY ? 5 : 6; " + "return f(Pick(left)) * 10 + f(Status.NONE);" }, + // The target-typed spelling, and enums as the value of a + // conditional and of a loop expression. + { false, + false, + 3, + 0, + 1, + "Status s = Pick(left); return s == .READY ? 1 : 2;" }, + { false, false, 0, 0, 11, "return Rank(.HIGH) + Rank(.LOW);" }, + { false, + false, + 1, + 0, + 1, + "Status s = left > 0 ? .READY : .NONE; " + "return s == .READY ? 1 : 0;" }, + { false, + false, + 1, + 0, + 11, + "Level l = while (true) { if (left > 0) { break Level.TOP; } " + "break Level.MID; }; return Rank(l);" }, + { false, + false, + 0, + 0, + 2, + "Level l = while (true) { if (left > 0) { break Level.TOP; } " + "break Level.MID; }; return Rank(l);" }, + }); + + // Evaluation by need. A value that is never needed is never computed, so + // a division by zero and a call that never returns do nothing when + // nothing reads their value; run through LLVM, the first would end the + // process and the second would never end. A store happens where it is + // written, and a value means what its variables held where it was + // bound. + // Arguments passed by need. A method computes an argument when it first + // needs it, at most once, and never when it does not need it; a value + // the caller needs as well is computed once for both. Such a value is + // an object of the runtime, so each case is a program of its own and + // must leave no allocation behind. + constexpr std::string_view kByNeedHelpers + = " public static int Never(_ int v) { return Never(v + 1); " + "}\n" + " public static int Half(_ int v) { return v / 2; }\n" + " public static int Step(_ int n) { return n > 0 ? 1 + " + "Step(n - 1) : 0; }\n" + " public static int Pick(_ int flag, _ int value) { if " + "(flag > 0) { return value; } return 7; }\n" + " public static int Pass(_ int flag, _ int value) { return" + " Pick(flag, value); }\n" + " public static int Both(_ int flag, _ int first, _ int " + "second) { if (flag > 0) { return first; } return second; }\n" + " public static int Kept(_ int flag, _ int value) { auto " + "read = \\(int more) -> value + more; if (flag > 0) { return " + "read(1); } return 3; }\n" + " public static int Often(_ int flag, _ int value) { int " + "total = 0; for (int i = 0; i < 50; i += 1) { if (flag > 0) {" + " total += value; } } return total; }\n" + " public static int Down(_ int n, _ int spare) { if (n <= " + "0) { return 0; } return 1 + Down(n - 1, spare / 0); }\n"; + constexpr auto kByNeedCases = std::to_array< + Visual::XSharp::Fuzzing::ExecutionCase>({ + // An argument that the method never needs is never computed. + { false, false, 6, 0, 7, "return Pick(right, left / right);" }, + { false, false, 6, 3, 2, "return Pick(right, left / right);" }, + { false, false, 1, 0, 7, "return Pick(right, Never(left));" }, + { false, + false, + 8, + 0, + 11, + "return Pick(0, Never(left)) + Pick(1, Half(left));" }, + { false, false, 5, 0, 6, "return Both(right, Never(left), left + 1);" }, + { false, + false, + 6, + 3, + 2, + "return Both(right, left / right, Never(left));" }, + // A value handed through one method to another, and through a method + // that calls itself. + { false, false, 6, 0, 7, "return Pass(right, left / right);" }, + { false, false, 6, 3, 2, "return Pass(right, left / right);" }, + { false, + false, + 6, + 2, + 4, + "return Pick(right, Pick(right, left / right) + 1);" }, + { false, false, 5, 0, 5, "return Down(left, right);" }, + // A local by need that is handed on, and that the caller needs as + // well. + { false, + false, + 1, + 0, + 7, + "int x = Never(left); int y = Pick(right, x); return y;" }, + { false, + false, + 6, + 0, + 14, + "int x = left / right; return Pick(right, x) + Pick(right, x " + "+ 1);" }, + { false, + false, + 6, + 3, + 5, + "int x = left / right; return Pick(right, x) + Pick(right, x " + "+ 1);" }, + { false, + false, + 30, + 1, + 60, + "int x = Step(left); int y = Pick(right, x); return right > 0" + " ? x + y : y;" }, + { false, + false, + 30, + 0, + 7, + "int x = Step(left); int y = Pick(right, x); return right > 0" + " ? x + y : y;" }, + { false, + false, + 6, + 3, + 6, + "int x = left / right; int y = x + 1; return Pass(right, y * " + "2);" }, + { false, + false, + 6, + 0, + 7, + "int x = left / right; int y = x + 1; return Pass(right, y * " + "2);" }, + // An argument means what its variables held at the call, and a store + // in it happens there. + { false, + false, + 8, + 1, + 12100, + "int a = left; int r = Pick(right, Half(a) + a); a = 100; " + "return r * 1000 + a;" }, + { false, + false, + 1, + 0, + 17, + "int n = 0; int r = Pick(0, (n += 1) + left); return n * 10 +" + " r;" }, + // An argument the method reads often, one a closure of the method + // captures, and one for each pass of a loop. + { false, false, 8, 1, 200, "return Often(right, Half(left));" }, + { false, false, 8, 0, 0, "return Often(right, Half(left));" }, + { false, false, 6, 3, 3, "return Kept(1, left / right);" }, + { false, + false, + 12, + 0, + 29, + "int t = 0; for (int i = 0; i <= 3; i += 1) { t += Pick(i, " + "left / i); } return t;" }, + }); + + constexpr std::string_view kLazyHelpers + = " public static int Never(_ int v) { return Never(v + 1); }\n" + " public static int Half(_ int v) { return v / 2; }\n"; + constexpr auto kLazyCases = std::to_array< + Visual::XSharp::Fuzzing::ExecutionCase>({ + { false, false, 1, 0, 5, "int x = left / right; return 5;" }, + // A value that needs another value twice computes it once, also + // through a chain of such values. + { false, + false, + 6, + 0, + 7, + "int z0 = left / right; int z1 = z0 / 1 - z0 / 2; int z2 = z1 / 1 - " + "z1 / 2; int z3 = z2 / 1 - z2 / 2; int z4 = z3 / 1 - z3 / 2; int z5 " + "= z4 / 1 - z4 / 2; int z6 = z5 / 1 - z5 / 2; int z7 = z6 / 1 - z6 / " + "2; int z8 = z7 / 1 - z7 / 2; int z9 = z8 / 1 - z8 / 2; int z10 = z9 " + "/ 1 - z9 / 2; int z11 = z10 / 1 - z10 / 2; int z12 = z11 / 1 - z11 " + "/ 2; int z13 = z12 / 1 - z12 / 2; int z14 = z13 / 1 - z13 / 2; int " + "z15 = z14 / 1 - z14 / 2; int z16 = z15 / 1 - z15 / 2; int z17 = z16 " + "/ 1 - z16 / 2; int z18 = z17 / 1 - z17 / 2; int z19 = z18 / 1 - z18 " + "/ 2; int z20 = z19 / 1 - z19 / 2; int z21 = z20 / 1 - z20 / 2; int " + "z22 = z21 / 1 - z21 / 2; int z23 = z22 / 1 - z22 / 2; int z24 = z23 " + "/ 1 - z23 / 2; if (right > 0) { return z24; } return 7;" }, + { false, + false, + 6, + 3, + 1, + "int z0 = left / right; int z1 = z0 / 1 - z0 / 2; int z2 = z1 / 1 - " + "z1 / 2; int z3 = z2 / 1 - z2 / 2; int z4 = z3 / 1 - z3 / 2; int z5 " + "= z4 / 1 - z4 / 2; int z6 = z5 / 1 - z5 / 2; int z7 = z6 / 1 - z6 / " + "2; int z8 = z7 / 1 - z7 / 2; int z9 = z8 / 1 - z8 / 2; int z10 = z9 " + "/ 1 - z9 / 2; int z11 = z10 / 1 - z10 / 2; int z12 = z11 / 1 - z11 " + "/ 2; int z13 = z12 / 1 - z12 / 2; int z14 = z13 / 1 - z13 / 2; int " + "z15 = z14 / 1 - z14 / 2; int z16 = z15 / 1 - z15 / 2; int z17 = z16 " + "/ 1 - z16 / 2; int z18 = z17 / 1 - z17 / 2; int z19 = z18 / 1 - z18 " + "/ 2; int z20 = z19 / 1 - z19 / 2; int z21 = z20 / 1 - z20 / 2; int " + "z22 = z21 / 1 - z21 / 2; int z23 = z22 / 1 - z22 / 2; int z24 = z23 " + "/ 1 - z23 / 2; if (right > 0) { return z24; } return 7;" }, + { false, + false, + 6, + 0, + 9, + "int x = left / right; int y = right > 0 ? x / 1 + x / 2 : 0; return " + "right > 1 ? y : 9;" }, + { false, + false, + 6, + 1, + 9, + "int x = left / right; int y = right > 0 ? x / 1 + x / 2 : 0; return " + "right > 1 ? y : 9;" }, + { false, + false, + 6, + 3, + 3, + "int x = left / right; int y = right > 0 ? x / 1 + x / 2 : 0; return " + "right > 1 ? y : 9;" }, + { false, false, 1, 0, 5, "int x = Never(left); return 5;" }, + { false, + false, + 6, + 0, + 7, + "int x = left / right; if (right > 0) { return x; } return 7;" }, + { false, + false, + 6, + 3, + 2, + "int x = left / right; if (right > 0) { return x; } return 7;" }, + { false, + false, + 6, + 0, + 1, + "int x = left / right; return match (right) { 0 -> 1, _ -> x };" }, + { false, + false, + 6, + 2, + 3, + "int x = left / right; return match (right) { 0 -> 1, _ -> x };" }, + { false, + false, + 6, + 0, + 2, + "int x = left / right; return right > 0 && x > 1 ? 1 : 2;" }, + { false, + false, + 6, + 3, + 1, + "int x = left / right; return right > 0 && x > 1 ? 1 : 2;" }, + { false, + false, + 6, + 0, + 9, + "int x = left / right; int y = x + 1; return right > 0 ? y : 9;" }, + { false, + false, + 6, + 3, + 3, + "int x = left / right; int y = x + 1; return right > 0 ? y : 9;" }, + { false, + false, + 0, + 1, + 1, + "int x = Never(left); int y = x + 1; int z = y * 2; " + "return right > 0 ? 1 : 2;" }, + { false, + false, + 12, + 0, + 10, + "int t = 0; for (int i = 0; i <= 3; i += 1) { int x = left / i; " + "if (i > 1) { t += x; } } return t;" }, + { false, + false, + 6, + 0, + 0, + "int limit = left / right; int n = 0; " + "while (right > 0 && n < limit) { n += 1; } return n;" }, + { false, + false, + 6, + 2, + 3, + "int limit = left / right; int n = 0; " + "while (right > 0 && n < limit) { n += 1; } return n;" }, + { false, + false, + 8, + 0, + 12100, + "int a = left; int x = Half(a) + a; a = 100; " + "return x * 1000 + a;" }, + { false, + false, + 8, + 1, + 4, + "int a = left; int x = Half(a); a = a + 100; " + "if (right > 0) { return x; } return a;" }, + { false, + false, + 8, + 0, + 108, + "int a = left; int x = Half(a); a = a + 100; " + "if (right > 0) { return x; } return a;" }, + { false, + false, + 1, + 0, + 10, + "int n = 0; int x = (n += 1) + left; return n * 10;" }, + // The body of a callable evaluates by need as well. + { false, + false, + 6, + 0, + 7, + "auto f = \\(int v, int d) -> { int q = v / d; if (d > 0) { " + "return q; } return 7; }; return f(left, right);" }, + { false, + false, + 6, + 3, + 2, + "auto f = \\(int v, int d) -> { int q = v / d; if (d > 0) { " + "return q; } return 7; }; return f(left, right);" }, + { false, + false, + 1, + 0, + 5, + "auto f = \\(int v) -> { int q = Never(v); return 5; }; " + "return f(left);" }, + { false, + false, + 6, + 0, + 9, + "auto f = \\(int v, int d) -> { auto g = \\(int w) -> { " + "int q = w / d; return d > 0 ? q : 9; }; return g(v); }; " + "return f(left, right);" }, + }); + + // Written here by hand, apart from the generated tables and from the + // frontend tests: the expected values are worked out from the methods + // above, so these runs do not share an expectation with any other + // table. + constexpr auto kInferredCases = std::to_array< + Visual::XSharp::Fuzzing::ExecutionCase>({ + { false, false, 3, 4, 30, "return Twice(left) + Factorial(right);" }, + { false, false, 0, 1, 1, "return Twice(left) + Factorial(right);" }, + { false, false, 4, 0, 19, "return First(left);" }, + { false, + false, + 4, + 0, + 12, + "return (Even(left) ? 10 : 20) + (Odd(left) ? 1 : 2);" }, + { false, + false, + 3, + 0, + 21, + "return (Even(left) ? 10 : 20) + (Odd(left) ? 1 : 2);" }, + { false, false, 2, 0, 66, "return Pick(left) * 10 + Pick(0 - left);" }, + { false, + false, + 1, + 0, + 307, + "return Scan(left) * 100 + Scan(left + 3);" }, + { false, + false, + 3, + 0, + 6, + "int x = Twice(left); bool b = Even(x); return b ? x : 0 - x;" }, + }); + + int + Smoke() + { + // Hand-written results for calls of methods whose return type is + // inferred. + Visual::XSharp::Fuzzing::ExerciseExecutionCases( + "Inferred return execution", + kInferredCases, + kInferredHelpers); + // Hand-written results for evaluation by need. + Visual::XSharp::Fuzzing::ExerciseExecutionCases("Lazy execution", + kLazyCases, + kLazyHelpers); + // Hand-written results for classic enums. + Visual::XSharp::Fuzzing::ExerciseExecutionCases("Enum execution", + kEnumCases, + kEnumHelpers, + kEnumDeclarations); + // The JIT finds the runtime of the closure cases in this process. + // The call also keeps the runtime in the program where the linker + // would otherwise leave an unreferenced library out. + if (vxs_aarc_abi_version() != VXS_AARC_ABI_VERSION) + { + llvm::errs() << "the linked AARC runtime has another ABI version\n"; + return 1; + } + // Hand-written results for closures: created, called, nested and + // returned, through LLVM and the runtime that owns them. + // Each case is a program of its own, and the runtime must hold no + // more allocations after it than before: a closure that is not + // released, or a capture its destructor does not release, is a + // failure here on every platform, with the program that leaked. + for (const auto &closureCase : kClosureCases) + { + const auto before + = Visual::XSharp::Runtime::Aarc::LiveAllocations(); + Visual::XSharp::Fuzzing::ExerciseExecutionCases( + "Closure execution", + std::span(&closureCase, 1U), + kClosureHelpers); + const auto after = Visual::XSharp::Runtime::Aarc::LiveAllocations(); + if (after != before) + { + llvm::errs() + << "closure case left " << (after - before) + << " AARC allocation(s) behind: " << closureCase.body + << '\n'; + return 1; + } + } + for (const auto &byNeedCase : kByNeedCases) + { + const auto before + = Visual::XSharp::Runtime::Aarc::LiveAllocations(); + Visual::XSharp::Fuzzing::ExerciseExecutionCases( + "By-need execution", + std::span(&byNeedCase, 1U), + kByNeedHelpers); + const auto after = Visual::XSharp::Runtime::Aarc::LiveAllocations(); + if (after != before) + { + llvm::errs() + << "by-need case left " << (after - before) + << " AARC allocation(s) behind: " << byNeedCase.body + << '\n'; + return 1; + } + } + for (const auto text : kOwnershipCases) + { + llvm::errs() << "Ownership verification: " << text << '\n'; + // The bytes of the text are the input, as a fuzz target gets it. + // NOLINTNEXTLINE(cppcoreguidelines-pro-type-reinterpret-cast) + const auto *const bytes + = reinterpret_cast(text.data()); + Visual::XSharp::Fuzzing::ExerciseAcceptedSource( + { bytes, text.size() }); + } + return 0; + } +} // namespace + +int +main() +{ + // The compiler runs on its own stack, as in `vxs`. + return Visual::XSharp::Support::RunOnCompilerStack([] { + return Smoke(); + }); +} diff --git a/Compiler/Fuzzing/SourceFuzz.cpp b/Compiler/Fuzzing/SourceFuzz.cpp index 51de9555..5fda5111 100644 --- a/Compiler/Fuzzing/SourceFuzz.cpp +++ b/Compiler/Fuzzing/SourceFuzz.cpp @@ -15,6 +15,7 @@ #include "SourceFuzz.hpp" #include "Visual/XSharp/Backend/LLVM.hpp" #include "Visual/XSharp/Pipeline.hpp" +#include "Visual/XSharp/Runtime/AARC.hpp" namespace Visual::XSharp::Fuzzing { @@ -637,6 +638,14 @@ namespace Visual::XSharp::Fuzzing Invoke(const Llvm::Artifact &artifact, std::string_view identifier) -> std::int64_t { + // A compiled program may call the AARC runtime: a closure does, + // and so does an argument that is passed by need. The JIT finds + // the runtime in this process. The call also keeps the runtime + // in the program where the linker would otherwise leave an + // unreferenced library out. + if (vxs_aarc_abi_version() != VXS_AARC_ABI_VERSION) + llvm::report_fatal_error(llvm::Twine( + "the linked AARC runtime has another ABI version")); // Each oracle variant owns an isolated ORC session so equal source // symbols in optimized and reference modules cannot collide. Llvm::JitSession session; diff --git a/Compiler/Fuzzing/SourceFuzzSmoke.cpp b/Compiler/Fuzzing/SourceFuzzSmoke.cpp index 3ce9b875..73da53e3 100644 --- a/Compiler/Fuzzing/SourceFuzzSmoke.cpp +++ b/Compiler/Fuzzing/SourceFuzzSmoke.cpp @@ -6,207 +6,241 @@ #include #include -#include "BranchingExecutionCases.hpp" -#include "ExpressionExecutionCases.hpp" #include "SourceFuzz.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +namespace +{ + int + Smoke(); +} // namespace int main() { - constexpr std::string_view sourceText - = "namespace Demo; public class Program { public static long " - "Evaluate() { return 5; } }"; - const auto source = std::span( - reinterpret_cast(sourceText.data()), - sourceText.size()); - constexpr std::array expressionSeed{ 1U, 2U, 3U, 4U, - 5U, 6U, 7U, 8U }; - Visual::XSharp::Fuzzing::ExerciseLexer(source); - Visual::XSharp::Fuzzing::ExerciseParser(source); - const std::span emptySource; - Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); - Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); - // Control-flow shapes whose frontend and native CorePrep lowerings must - // agree. They combine forms the generated programs below keep separate: - // loops inside loops, short-circuit operators as loop conditions, and - // several functions sharing one module-wide symbol numbering. - constexpr std::array accepted{ - "namespace Parity; class Program { public static int Evaluate() { " - "int total = 0; for (int outer = 0; outer < 4; outer++) { " - "if (outer == 2) { continue; } int inner = 0; " - "while (inner < 3) { if (inner == 1) { inner++; continue; } " - "total = total + outer + inner; inner++; } } return total; } }", - "namespace Parity; class Program { public static int Evaluate() { " - "int index = 0; int total = 0; " - "while (index < 9 && (index < 4 || total < 20)) { " - "total = total + index; index++; } " - "do { total++; } while (total < 40 && index \\= 0); " - "return total; } }", - "namespace Parity; class Program { " - "public static bool Down(_ int n) { return n == 0 || Down(n - 1); } " - "public static int Pick(_ int n) { if (Down(n) && n < 6) { " - "return n; } return 0 - 1; } " - "public static int Evaluate() { return Pick(3) + Pick(8); } }", - "namespace Parity; class First { public static long Evaluate() { " - "long result = 8; if (result > 4) { result = result - 2; } " - "return result; } } class Second { public static long Other() { " - "long value = 3; for (int step = 0; step < 2; step++) { " - "value = value * 2; } return value; } }", - // Conditional expressions: Boolean and numeric tests, nesting in - // either result and in the test, and a recursion the first result - // terminates. - "namespace Parity; class Program { " - "public static int Sum(_ int n) { return n == 0 ? 0 : n + Sum(n - 1); " - "} public static int Pick(_ bool flag, _ int left, _ int right) { " - "return flag ? left > right ? left : right : left ? 0 - left : " - "right; } public static int Evaluate() { return (Sum(4) > 5 ? " - "Pick(true, 3, 9) > 4 : false) ? Pick(false, 0, 7) : Sum(2); } }", - // Truthy coalescing binds its left operand once. A call on the left - // is the case where an adapter that copies the bound value through - // an extra temporary disagrees with one that binds the call itself. - "namespace Parity; class Program { " - "public static int Next(_ int n) { return n > 2 ? Next(n - 3) : n; } " - "public static int Evaluate() { int value = Next(7) ?: Next(5); " - "return value + (Next(9) ?: 4) + (value ?: Next(8) ?: 6); } }", - // Compound assignments in statement and loop-update position, with - // a conditional and a coalescing operand. - "namespace Parity; class Program { public static int Evaluate() { " - "int total = 3; total += 4; total -= 1; total *= 5; total /= 2; " - "total //= 2; total %= 7; total <<= 3; total >>= 1; total &= 127; " - "total ^= 9; total |= 64; for (int index = 0; index < 6; index += 2) " - "{ total += index ? index : 1; total -= index ?: 2; } " - "return total; } }", - // Discarded values: a call keeps its result-dropping instruction; an - // operator, a conditional and a division are computed into - // temporaries. - "namespace Parity; class Program { " - "public static int Down(_ int n) { return n > 0 ? Down(n - 1) : 0; } " - "public static int Evaluate() { int value = 5; _ = Down(value); " - "_ = !Down(2); _ = 12 / value; _ = value > 3 ? Down(1) : value; " - "value > 4 ? Down(3) : 0; _ = Down(value) ?: 7; return value; } }", - // Conditional forms as loop conditions and as the operands of - // short-circuit operators, in a module with several functions. - "namespace Parity; class Program { " - "public static bool Small(_ int n) { return n < 3 ? true : n == 9; } " - "public static int Evaluate() { int index = 0; int total = 0; " - "while (index < 4 ? Small(index) || total < 9 : false) { " - "total += index ?: 5; index += 1; } " - "do { total -= 1; } while (total > 3 && (total ?: 1) \\= 2); " - "return Small(total) && total > 0 ? total : 0 - total; } }", - // Assignments and increments used as values: an earlier operand - // held across a later store, chained and compound forms, and - // arguments evaluated in order. - "namespace Parity; class Program { " - "public static int Pick(_ int a, _ int b, _ int c) { return a * 100 " - "+ b * 10 + c; } public static int Next(_ int n) { return n > 2 ? " - "Next(n - 3) : n; } public static int Evaluate() { int a = Next(7); " - "int b = 0; int c = 0; a = b = c = a + 1; int r = a + (a = b + 2) * " - "(a += 1) - a++ + ++b; r += Pick(a, a = r, a++) + Pick(c++, c++, c); " - "return r - (b -= a) + Next(a = 8) + a; } }", - // Loop conditions that store: the stores run before every test, - // after `continue` too, and a do/while body runs before its first. - "namespace Parity; class Program { public static int Evaluate() { " - "int n = 0; int sum = 0; int v = 0; " - "while ((v = n++) < 9) { if (v == 3) { continue; } " - "if (v == 5) { n += 2; continue; } sum += v; } " - "do { sum += 100; if (sum > 400) { break; } } while ((n -= 3) > 4); " - "for (int i = 0; (v = i * 3) < 11; i++) { if (v == 6) { continue; } " - "sum += v; } int j = 0; while (j++ < 3) { int k = 0; " - "do { sum += j; } while (++k < j); } return sum * 10 + n + v; } }", - // Stores in lazily evaluated operands: conditional results, the - // right side of short-circuit operators and a coalescing fallback. - "namespace Parity; class Program { " - "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) : " - "0; } public static int Evaluate() { int a = Step(7); int b = " - "Step(0); int hits = 0; int r = a > 5 ? (a -= 5) : (b += 1); " - "bool both = a > 2 && (hits += 1) > 0; " - "bool either = b \\= 0 || (hits += 10) > 0; " - "int c = b ?: (hits += 100); " - "int d = (a = Step(3)) ? (b = a) ?: (hits += 1000) : (hits = 0); " - "bool chain = both && (a = 1) > 0 && (b = 2) > 0 || (c = 0) == 0; " - "return r + a * 3 + b * 5 + hits * 7 + c + d + (both ? 1000 : 0) + " - "(either ? 2000 : 0) + (chain ? 4000 : 0); } }", - // Stores as statements of their own and inside closures, which - // capture by value and store into their own copies. - "namespace Parity; class Program { public static int Evaluate() { " - "int a = 4; (a = 6); _ = (a += 1); _ = a++; _ = a++ > 0 ? 1 : 0; " - "auto bump = \\(int value) -> { int local = value; " - "return (local += 1) + local++; }; " - "auto held = [kept = a++] \\ -> kept; " - "return bump(a) + held() + a; } }", - // Loops used as expressions: `while` and `for` forms, a nested loop - // expression, a loop statement with its own bare break inside one, - // and loop expressions as operands and as a loop condition. - "namespace Parity; class Program { " - "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) : " - "0; } public static int Evaluate() { int n = Step(2); " - "int first = while (true) { n += 1; if (n * n > 30) { break n; } }; " - "int second = for (int i = 0; ; i++) { if (i == 2) { continue; } " - "int inner = while (true) { int k = 0; while (true) { k++; " - "if (k == 3) { break; } } break k + i; }; " - "if (inner > 6) { break inner * 2; } }; " - "int third = first + while (true) { first += 1; break first; } + " - "first; int sum = 0; while (for (int j = sum; ; j++) { " - "if (j >= sum) { break j; } } < 4) { sum += 1; } " - "bool big = second > 9 && while (true) { n -= 1; break n > 0; }; " - "return first + second + third + sum + n + (big ? 100 : 0); } }", - // Floating values in Boolean contexts. Both lowerings must compare - // them with a floating zero, not an integer one. - "namespace Parity; class Program { public static int Evaluate() { " - "double zero = 0.0; double half = 0.5; float small = 0.25; " - "bool both = half && small; bool either = zero || half; " - "double kept = zero ?: half; int picked = half ? 1 : 2; " - "return (both ? 1 : 0) + (either ? 2 : 0) + (kept ? 4 : 0) + " - "picked * 8 + (not zero ? 32 : 0); } }", - }; - for (const auto text : accepted) - Visual::XSharp::Fuzzing::ExerciseAcceptedSource( - std::span( - reinterpret_cast(text.data()), - text.size())); - // Hand-written results for assignments and increments used as values - // and for loops used as expressions. - Visual::XSharp::Fuzzing::ExerciseExpressionCases(); - // Hand-written results for match, if expressions and guard. - Visual::XSharp::Fuzzing::ExerciseBranchingCases(); - llvm::errs() << "Differential smoke: mixed seed\n"; - Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); - llvm::errs() << "Differential smoke: empty seed\n"; - Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(emptySource); - // Constant seeds force leaves, full-depth addition, subtraction and - // multiplication. Exercise the independent oracle before a mutation - // campaign so a missing generated-code route cannot appear as success. - constexpr std::array selectors{ 252U, 253U, 254U, 255U }; - for (const auto selector : selectors) - { - const std::array seed{ selector }; - llvm::errs() << "Differential smoke: selector " - << static_cast(selector) << '\n'; - Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); - } - // A leaf selector followed by an explicit mode and limit byte reaches - // every generated control-flow shape at every trip count, including the - // zero-trip, continue and break paths, instead of only the shapes the - // four cycling selectors above happen to select. The two leaves are the - // literals 0 and 4, so a form that tests its generated expression sees - // both a false and a true value. - constexpr std::uint8_t kModes = 14U; - constexpr std::uint8_t kLimits = 12U; - constexpr std::array leaves{ 0U, 4U }; - for (const auto leaf : leaves) + // The compiler runs here on the stack it runs on in `vxs`. + return Visual::XSharp::Support::RunOnCompilerStack([] { + return Smoke(); + }); +} + +namespace +{ + int + Smoke() { - for (std::uint8_t mode = 0U; mode < kModes; ++mode) + constexpr std::string_view sourceText + = "namespace Demo; public class Program { public static long " + "Evaluate() { return 5; } }"; + const auto source = std::span( + reinterpret_cast(sourceText.data()), + sourceText.size()); + constexpr std::array expressionSeed{ 1U, 2U, 3U, 4U, + 5U, 6U, 7U, 8U }; + Visual::XSharp::Fuzzing::ExerciseLexer(source); + Visual::XSharp::Fuzzing::ExerciseParser(source); + const std::span emptySource; + Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(emptySource); + Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(source); + // Control-flow shapes whose frontend and native CorePrep lowerings must + // agree. They combine forms the generated programs below keep separate: + // loops inside loops, short-circuit operators as loop conditions, and + // several functions sharing one module-wide symbol numbering. + constexpr std::array accepted{ + "namespace Parity; class Program { public static int Evaluate() { " + "int total = 0; for (int outer = 0; outer < 4; outer++) { " + "if (outer == 2) { continue; } int inner = 0; " + "while (inner < 3) { if (inner == 1) { inner++; continue; } " + "total = total + outer + inner; inner++; } } return total; } }", + "namespace Parity; class Program { public static int Evaluate() { " + "int index = 0; int total = 0; " + "while (index < 9 && (index < 4 || total < 20)) { " + "total = total + index; index++; } " + "do { total++; } while (total < 40 && index \\= 0); " + "return total; } }", + "namespace Parity; class Program { " + "public static bool Down(_ int n) { return n == 0 || Down(n - 1); " + "} " + "public static int Pick(_ int n) { if (Down(n) && n < 6) { " + "return n; } return 0 - 1; } " + "public static int Evaluate() { return Pick(3) + Pick(8); } }", + "namespace Parity; class First { public static long Evaluate() { " + "long result = 8; if (result > 4) { result = result - 2; } " + "return result; } } class Second { public static long Other() { " + "long value = 3; for (int step = 0; step < 2; step++) { " + "value = value * 2; } return value; } }", + // Conditional expressions: Boolean and numeric tests, nesting in + // either result and in the test, and a recursion the first result + // terminates. + "namespace Parity; class Program { " + "public static int Sum(_ int n) { return n == 0 ? 0 : n + Sum(n - " + "1); " + "} public static int Pick(_ bool flag, _ int left, _ int right) { " + "return flag ? left > right ? left : right : left ? 0 - left : " + "right; } public static int Evaluate() { return (Sum(4) > 5 ? " + "Pick(true, 3, 9) > 4 : false) ? Pick(false, 0, 7) : Sum(2); } }", + // Truthy coalescing binds its left operand once. A call on the left + // is the case where an adapter that copies the bound value through + // an extra temporary disagrees with one that binds the call itself. + "namespace Parity; class Program { " + "public static int Next(_ int n) { return n > 2 ? Next(n - 3) : n; " + "} " + "public static int Evaluate() { int value = Next(7) ?: Next(5); " + "return value + (Next(9) ?: 4) + (value ?: Next(8) ?: 6); } }", + // Compound assignments in statement and loop-update position, with + // a conditional and a coalescing operand. + "namespace Parity; class Program { public static int Evaluate() { " + "int total = 3; total += 4; total -= 1; total *= 5; total /= 2; " + "total //= 2; total %= 7; total <<= 3; total >>= 1; total &= 127; " + "total ^= 9; total |= 64; for (int index = 0; index < 6; index += " + "2) " + "{ total += index ? index : 1; total -= index ?: 2; } " + "return total; } }", + // Discarded values: a call keeps its result-dropping instruction; + // an operator, a conditional and a division are computed into + // temporaries. + "namespace Parity; class Program { " + "public static int Down(_ int n) { return n > 0 ? Down(n - 1) : 0; " + "} " + "public static int Evaluate() { int value = 5; _ = Down(value); " + "_ = !Down(2); _ = 12 / value; _ = value > 3 ? Down(1) : value; " + "value > 4 ? Down(3) : 0; _ = Down(value) ?: 7; return value; } }", + // Conditional forms as loop conditions and as the operands of + // short-circuit operators, in a module with several functions. + "namespace Parity; class Program { " + "public static bool Small(_ int n) { return n < 3 ? true : n == 9; " + "} " + "public static int Evaluate() { int index = 0; int total = 0; " + "while (index < 4 ? Small(index) || total < 9 : false) { " + "total += index ?: 5; index += 1; } " + "do { total -= 1; } while (total > 3 && (total ?: 1) \\= 2); " + "return Small(total) && total > 0 ? total : 0 - total; } }", + // Assignments and increments used as values: an earlier operand + // held across a later store, chained and compound forms, and + // arguments evaluated in order. + "namespace Parity; class Program { " + "public static int Pick(_ int a, _ int b, _ int c) { return a * " + "100 " + "+ b * 10 + c; } public static int Next(_ int n) { return n > 2 ? " + "Next(n - 3) : n; } public static int Evaluate() { int a = " + "Next(7); " + "int b = 0; int c = 0; a = b = c = a + 1; int r = a + (a = b + 2) " + "* " + "(a += 1) - a++ + ++b; r += Pick(a, a = r, a++) + Pick(c++, c++, " + "c); " + "return r - (b -= a) + Next(a = 8) + a; } }", + // Loop conditions that store: the stores run before every test, + // after `continue` too, and a do/while body runs before its first. + "namespace Parity; class Program { public static int Evaluate() { " + "int n = 0; int sum = 0; int v = 0; " + "while ((v = n++) < 9) { if (v == 3) { continue; } " + "if (v == 5) { n += 2; continue; } sum += v; } " + "do { sum += 100; if (sum > 400) { break; } } while ((n -= 3) > " + "4); " + "for (int i = 0; (v = i * 3) < 11; i++) { if (v == 6) { continue; " + "} " + "sum += v; } int j = 0; while (j++ < 3) { int k = 0; " + "do { sum += j; } while (++k < j); } return sum * 10 + n + v; } }", + // Stores in lazily evaluated operands: conditional results, the + // right side of short-circuit operators and a coalescing fallback. + "namespace Parity; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": " + "0; } public static int Evaluate() { int a = Step(7); int b = " + "Step(0); int hits = 0; int r = a > 5 ? (a -= 5) : (b += 1); " + "bool both = a > 2 && (hits += 1) > 0; " + "bool either = b \\= 0 || (hits += 10) > 0; " + "int c = b ?: (hits += 100); " + "int d = (a = Step(3)) ? (b = a) ?: (hits += 1000) : (hits = 0); " + "bool chain = both && (a = 1) > 0 && (b = 2) > 0 || (c = 0) == 0; " + "return r + a * 3 + b * 5 + hits * 7 + c + d + (both ? 1000 : 0) + " + "(either ? 2000 : 0) + (chain ? 4000 : 0); } }", + // Stores as statements of their own and inside closures, which + // capture by value and store into their own copies. + "namespace Parity; class Program { public static int Evaluate() { " + "int a = 4; (a = 6); _ = (a += 1); _ = a++; _ = a++ > 0 ? 1 : 0; " + "auto bump = \\(int value) -> { int local = value; " + "return (local += 1) + local++; }; " + "auto held = [kept = a++] \\ -> kept; " + "return bump(a) + held() + a; } }", + // Loops used as expressions: `while` and `for` forms, a nested loop + // expression, a loop statement with its own bare break inside one, + // and loop expressions as operands and as a loop condition. + "namespace Parity; class Program { " + "public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) " + ": " + "0; } public static int Evaluate() { int n = Step(2); " + "int first = while (true) { n += 1; if (n * n > 30) { break n; } " + "}; " + "int second = for (int i = 0; ; i++) { if (i == 2) { continue; } " + "int inner = while (true) { int k = 0; while (true) { k++; " + "if (k == 3) { break; } } break k + i; }; " + "if (inner > 6) { break inner * 2; } }; " + "int third = first + while (true) { first += 1; break first; } + " + "first; int sum = 0; while (for (int j = sum; ; j++) { " + "if (j >= sum) { break j; } } < 4) { sum += 1; } " + "bool big = second > 9 && while (true) { n -= 1; break n > 0; }; " + "return first + second + third + sum + n + (big ? 100 : 0); } }", + // Floating values in Boolean contexts. Both lowerings must compare + // them with a floating zero, not an integer one. + "namespace Parity; class Program { public static int Evaluate() { " + "double zero = 0.0; double half = 0.5; float small = 0.25; " + "bool both = half && small; bool either = zero || half; " + "double kept = zero ?: half; int picked = half ? 1 : 2; " + "return (both ? 1 : 0) + (either ? 2 : 0) + (kept ? 4 : 0) + " + "picked * 8 + (not zero ? 32 : 0); } }", + }; + for (const auto text : accepted) + Visual::XSharp::Fuzzing::ExerciseAcceptedSource( + std::span( + reinterpret_cast(text.data()), + text.size())); + // The programs with hand-written results run in + // `source_execution_smoke`, a program of its own. + llvm::errs() << "Differential smoke: mixed seed\n"; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(expressionSeed); + llvm::errs() << "Differential smoke: empty seed\n"; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(emptySource); + // Constant seeds force leaves, full-depth addition, subtraction and + // multiplication. Exercise the independent oracle before a mutation + // campaign so a missing generated-code route cannot appear as success. + constexpr std::array selectors{ 252U, + 253U, + 254U, + 255U }; + for (const auto selector : selectors) { - for (std::uint8_t limit = 0U; limit < kLimits; ++limit) + const std::array seed{ selector }; + llvm::errs() << "Differential smoke: selector " + << static_cast(selector) << '\n'; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + } + // A leaf selector followed by an explicit mode and limit byte reaches + // every generated control-flow shape at every trip count, including the + // zero-trip, continue and break paths, instead of only the shapes the + // four cycling selectors above happen to select. The two leaves are the + // literals 0 and 4, so a form that tests its generated expression sees + // both a false and a true value. + constexpr std::uint8_t kModes = 14U; + constexpr std::uint8_t kLimits = 12U; + constexpr std::array leaves{ 0U, 4U }; + for (const auto leaf : leaves) + { + for (std::uint8_t mode = 0U; mode < kModes; ++mode) { - const std::array seed{ leaf, mode, limit }; - llvm::errs() << "Differential smoke: leaf " - << static_cast(leaf) << " mode " - << static_cast(mode) << " limit " - << static_cast(limit) << '\n'; - Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + for (std::uint8_t limit = 0U; limit < kLimits; ++limit) + { + const std::array seed{ leaf, + mode, + limit }; + llvm::errs() << "Differential smoke: leaf " + << static_cast(leaf) << " mode " + << static_cast(mode) << " limit " + << static_cast(limit) << '\n'; + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(seed); + } } } + return 0; } - return 0; -} +} // namespace diff --git a/Compiler/Fuzzing/SourceLibFuzzer.cpp b/Compiler/Fuzzing/SourceLibFuzzer.cpp index 3f933312..b97d44bc 100644 --- a/Compiler/Fuzzing/SourceLibFuzzer.cpp +++ b/Compiler/Fuzzing/SourceLibFuzzer.cpp @@ -6,6 +6,7 @@ #include #include "SourceFuzz.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" #ifndef VXS_SOURCE_FUZZ_STAGE # error VXS_SOURCE_FUZZ_STAGE must select one dedicated fuzz target @@ -15,19 +16,25 @@ extern "C" int LLVMFuzzerTestOneInput(const std::uint8_t *data, std::size_t size) { const std::span input(data, size); + // The compiler runs on its own stack, as it does in `vxs`: an input + // nested up to the frontend's limits must compile here as well, and the + // stack libFuzzer calls this function on is the platform's default. + Visual::XSharp::Support::RunOnCompilerStack([input] { #if VXS_SOURCE_FUZZ_STAGE == 0 - Visual::XSharp::Fuzzing::ExerciseLexer(input); + Visual::XSharp::Fuzzing::ExerciseLexer(input); #elif VXS_SOURCE_FUZZ_STAGE == 1 - Visual::XSharp::Fuzzing::ExerciseParser(input); + Visual::XSharp::Fuzzing::ExerciseParser(input); #elif VXS_SOURCE_FUZZ_STAGE == 2 - // Arbitrary source and generated arithmetic have independent corpora and - // time budgets. An invalid source mutation should not pay for two JITs. - Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(input); + // Arbitrary source and generated arithmetic have independent corpora + // and time budgets. An invalid source mutation should not pay for + // two JITs. + Visual::XSharp::Fuzzing::ExerciseSourceToLlvm(input); #elif VXS_SOURCE_FUZZ_STAGE == 3 - Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(input); + Visual::XSharp::Fuzzing::ExerciseDifferentialOracle(input); #else # error Unsupported VXS_SOURCE_FUZZ_STAGE #endif + }); // Oracle invariant failures terminate directly. No exception can unwind // through the C ABI, and libFuzzer retains the reproducing input. return 0; diff --git a/Compiler/Fuzzing/SourceStackProbe.cpp b/Compiler/Fuzzing/SourceStackProbe.cpp new file mode 100644 index 00000000..fb038a78 --- /dev/null +++ b/Compiler/Fuzzing/SourceStackProbe.cpp @@ -0,0 +1,67 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include + +#include "SourceFuzz.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// Measures the stack a whole compilation uses. +// +// source_stack_probe +// +// compiles a source file through the frontend and the native pipeline on a +// thread whose stack has the given size, as `vxs` does on the compiler +// stack, and prints the stack that thread committed. That is an upper bound +// on what it used, to the page and with the guard page, not the exact number +// of bytes in use; frames a sanitizer keeps on a heap-allocated fake stack +// are not in it. The size on the command line is only reserved. A source of at +// most 64 KiB that the frontend rejects is a completed run as well; a larger +// one must be accepted. The process is terminated by the operating system when +// the stack is too small. The program is a measuring instrument, not a test. + +int +main(int argc, char **argv) +{ + if (argc != 3) + { + llvm::errs() << "usage: source_stack_probe \n"; + return 2; + } + // The arguments of a measuring tool, read once at the start. + // NOLINTBEGIN(cppcoreguidelines-pro-bounds-pointer-arithmetic) + const char *const path = argv[1]; + const auto stack + = static_cast(std::strtoull(argv[2], nullptr, 10)) * 1024U; + // NOLINTEND(cppcoreguidelines-pro-bounds-pointer-arithmetic) + auto buffer = llvm::MemoryBuffer::getFile(path); + if (!buffer || stack == 0U) + { + llvm::errs() << "source_stack_probe: cannot read " << path << '\n'; + return 3; + } + const auto text = (*buffer)->getBuffer(); + std::size_t committed = 0U; + Visual::XSharp::Support::RunOnStack(stack, [&] { + // The bytes of the file are the input, as a fuzz target receives it. + // NOLINTNEXTLINE(cppcoreguidelines-pro-type-reinterpret-cast) + const auto *const bytes + = reinterpret_cast(text.data()); + // The entry that tolerates a rejection ignores input above the + // size limit of the fuzz targets; a larger file must be accepted. + constexpr std::size_t kFuzzInputLimit = std::size_t{ 64U } * 1024U; + if (text.size() <= kFuzzInputLimit) + Visual::XSharp::Fuzzing::ExerciseSourceToLlvm( + { bytes, text.size() }); + else + Visual::XSharp::Fuzzing::ExerciseAcceptedSource( + { bytes, text.size() }); + committed = Visual::XSharp::Support::CommittedStackBytes(); + }); + llvm::outs() << "ok committed-kib " << committed / 1024U << '\n'; + return 0; +} diff --git a/Compiler/Fuzzing/WireFuzz.cpp b/Compiler/Fuzzing/WireFuzz.cpp index 17b1a7a1..4beba0b6 100644 --- a/Compiler/Fuzzing/WireFuzz.cpp +++ b/Compiler/Fuzzing/WireFuzz.cpp @@ -33,6 +33,7 @@ namespace Visual::XSharp::Fuzzing limits.maximumOperands = 32U; limits.maximumTypeDepth = 16U; limits.maximumExpressionDepth = 32U; + limits.maximumStatementDepth = 32U; limits.maximumNumericBytes = 128U; return limits; } diff --git a/Compiler/Haskell/Core/Benches/Main.hs b/Compiler/Haskell/Core/Benches/Main.hs index 09517eb8..3fc837b6 100644 --- a/Compiler/Haskell/Core/Benches/Main.hs +++ b/Compiler/Haskell/Core/Benches/Main.hs @@ -56,6 +56,10 @@ main = do bgroup "ContradictoryIntegerPaths" [benchAt size integerOptimizeDigest (coreModuleAt size modules) | size <- flowSizes] , env (pure (CoreModules (loopFixtures flowSizes))) $ \modules -> bgroup "LoopIntegerFacts" [benchAt size integerOptimizeDigest (coreModuleAt size modules) | size <- flowSizes] + , env (verifiedFixtures makeNestedLoopModule loopDepths) $ \modules -> + bgroup "NestedLoopIntegerFacts" [benchAt size integerOptimizeDigest (coreModuleAt size modules) | size <- loopDepths] + , env (verifiedFixtures makeChainModule chainSizes) $ \modules -> + bgroup "EncodeChain" [benchAt size encodeDigest (coreModuleAt size modules) | size <- chainSizes] ] , bgroup "CorePrep" @@ -69,6 +73,14 @@ main = do bgroup "Encode" [benchAt size encodePrepDigest (corePrepModuleAt size modules) | size <- sizes] , env (traverse preparedDocument sizes) $ \documents -> bgroup "Decode" [benchAt size decodePrepDigest (documentAt size documents) | size <- sizes] + , env (verifiedFixtures makeChainModule chainSizes) $ \modules -> + bgroup "PrepareChain" [benchAt size prepareDigest (coreModuleAt size modules) | size <- chainSizes] + , env (verifiedFixtures makeSequenceModule chainSizes) $ \modules -> + bgroup "PrepareSequence" [benchAt size prepareDigest (coreModuleAt size modules) | size <- chainSizes] + , env (CorePrepModules <$> traverse (preparedFrom makeChainModule) chainSizes) $ \modules -> + bgroup "VerifyChain" [benchAt size verifyPrepDigest (corePrepModuleAt size modules) | size <- chainSizes] + , env (CorePrepModules <$> traverse (preparedFrom makeSequenceModule) chainSizes) $ \modules -> + bgroup "VerifySequence" [benchAt size verifyPrepDigest (corePrepModuleAt size modules) | size <- chainSizes] ] ] where @@ -77,6 +89,13 @@ main = do floatingSizes = [8, 32, 128, 512] integerSizes = [8, 32, 128, 512, 1024, 2048] flowSizes = [8, 32, 128, 512, 1024] + -- Levels of loops nested in each other. The loop analysis once + -- repeated the analysis of an inner loop for every pass over the + -- loop around it, so each level multiplied the time. + loopDepths = [2, 4, 8, 12, 16] + -- Links of an `else if` chain and statements of a sequence of `if` + -- statements. Each doubling should double the time. + chainSizes = [256, 512, 1024, 2048] benchAt :: Int -> (a -> Int) -> a -> Benchmark benchAt size measure input = bench (show size) (whnf measure input) @@ -145,6 +164,92 @@ makeLoopModule count = ] in CoreFunction functionName [] intType body +{- | Fixtures that must verify: a benchmark of a rejected module would time +the rejection. +-} +verifiedFixtures :: (Int -> CoreModule) -> [Int] -> IO CoreModules +verifiedFixtures make requested = CoreModules <$> traverse verified requested + where + verified size = case verifyCore (make size) of + Left diagnostics -> fail (show diagnostics) + Right value -> pure (size, value) + +preparedFrom :: (Int -> CoreModule) -> Int -> IO (Int, CorePrepModule) +preparedFrom make size = case prepareCore (make size) of + Left diagnostics -> fail (show diagnostics) + Right value -> case verifyCorePrep value of + Left diagnostics -> fail (show diagnostics) + Right verified -> pure (size, verified) + +{- | Loops nested to the given depth, each counting to a parameter, around +one statement: the workload of the loop fixed point of the integer analysis. +-} +makeNestedLoopModule :: Int -> CoreModule +makeNestedLoopModule depth = + CoreModule + (QualifiedName [Identifier "NestedLoopBenchmark"]) + [ CoreFunction + (name 1 "Count") + [(limitName, intType)] + intType + (CoreBind (CoreBinding totalName intType True (integer 0)) : loopAt 0 ++ [CoreReturn total]) + ] + where + limitName = name 2 "limit" + totalName = name 3 "total" + total = CoreVariable totalName intType + counterName level = name (4 + level) "counter" + counter level = CoreVariable (counterName level) intType + increment value = CorePrimitive CoreAdd [value, integer 1] intType + loopAt level + | level == depth = [CoreAssign totalName (increment total)] + | otherwise = + [ CoreBind (CoreBinding (counterName level) intType True (integer 0)) + , CoreWhile + (CorePrimitive CoreLessThan [counter level, CoreVariable limitName intType] boolType) + (CoreAssign (counterName level) (increment (counter level)) : loopAt (level + 1)) + ] + +{- | An `else if` chain: every false branch holds the next link, so the +statements nest as deep as the chain is long. +-} +makeChainModule :: Int -> CoreModule +makeChainModule links = + CoreModule + (QualifiedName [Identifier "ChainBenchmark"]) + [CoreFunction (name 1 "Pick") [(valueName, intType)] intType (chainFrom 0)] + where + valueName = name 2 "value" + chainFrom index + | index == links = [CoreReturn (integer 0)] + | otherwise = + [ CoreIf + (CorePrimitive CoreEqual [CoreVariable valueName intType, integer (toInteger index)] boolType) + [CoreReturn (integer (toInteger index * 3 + 1))] + (chainFrom (index + 1)) + ] + +-- | The same tests as a sequence of `if` statements, which do not nest. +makeSequenceModule :: Int -> CoreModule +makeSequenceModule count = + CoreModule + (QualifiedName [Identifier "SequenceBenchmark"]) + [ CoreFunction + (name 1 "Sum") + [(valueName, intType)] + intType + (CoreBind (CoreBinding totalName intType True (integer 0)) : map test [0 .. count - 1] ++ [CoreReturn total]) + ] + where + valueName = name 2 "value" + totalName = name 3 "total" + total = CoreVariable totalName intType + test index = + CoreIf + (CorePrimitive CoreEqual [CoreVariable valueName intType, integer (toInteger index)] boolType) + [CoreAssign totalName (CorePrimitive CoreAdd [total, integer (toInteger index)] intType)] + [] + inlineFixtures :: [Int] -> [(Int, CoreModule)] inlineFixtures = map (\size -> (size, makeInlineModule size)) diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core.hs index 6eaac169..eae288b1 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core.hs @@ -88,6 +88,25 @@ data CorePrimitive CoreBitwiseNot | -- | Runtime type-membership predicate. CoreTypeIs + | {- | A callable that remembers its result. + + The operand is a callable without parameters. The result is a callable + of the same type that calls the operand the first time it is called, + keeps what the operand returned, and returns that again on every later + call without calling the operand. Every copy of the result shares the + one remembered value. This is the suspended computation of evaluation + by need: a value that is computed when it is first needed, at most + once, wherever the need arises. + -} + CoreMemoize + | {- | A call of a function of the runtime. + + The first operand is an integer literal, the identity of the function + in the catalog of "Visual.XSharp.RuntimeCall"; the operands after it + are the arguments. The function is fixed when the program is compiled: + the first operand is never a value that is computed. + -} + CoreRuntimeCall deriving (Eq, Ord, Read, Show) -- | Typed expression graph consumed by Core verification and optimization. diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep.hs index 959b65e3..3467f24c 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep.hs @@ -101,6 +101,9 @@ data PrepState = PrepState data OpenBlock = OpenBlock { openBlockId :: Int , openBlockInstructions :: [CorePrepInstruction] + {- ^ Newest first, so that appending one is constant time; 'closeBlock' + puts them in execution order. + -} } {- | Adapt one verified Core module without changing its source-level semantics. @@ -123,23 +126,32 @@ prepareCore moduleValue = (functions, _) = prepareFunctionQueue initial work pure (CorePrepModule (coreModuleName verified) functions (coreModuleSourceFiles verified)) --- Closure conversion appends lifted functions together with their source --- owner. Processing the queue to exhaustion also supports nested closures --- without a separate whole-module mutation pass or a filename guess. +-- Closure conversion yields lifted functions together with their source +-- owner. They are prepared after every function that was already waiting, in +-- the order they were found, and the functions lifted out of them after +-- those: that supports nested closures without a separate whole-module pass +-- or a filename guess. +-- +-- The lifted functions wait in batches of their own, newest first. Appending +-- them to the functions still waiting wrapped that list once more for every +-- function prepared, also when nothing was lifted, and taking the next +-- function then cost time with the number of functions before it. prepareFunctionQueue :: PrepState -> [(CoreFunction, FilePath)] -> ([CorePrepFunction], PrepState) -prepareFunctionQueue state [] = case pendingFunctions state of - [] -> ([], state) - pending -> prepareFunctionQueue (state {pendingFunctions = []}) pending -prepareFunctionQueue state ((function, sourceFile) : remaining) = - let (prepared, afterFunction) = prepareFunction (state {nextBlock = 1, currentSourceFile = sourceFile}) function - pending = pendingFunctions afterFunction - nextState = afterFunction {pendingFunctions = []} - (later, final) = prepareFunctionQueue nextState (remaining ++ pending) - in (prepared : later, final) +prepareFunctionQueue initial work = go initial work [] + where + go state [] [] = ([], state) + go state [] lifted = go state (concat (reverse lifted)) [] + go state ((function, sourceFile) : remaining) lifted = + let (prepared, afterFunction) = prepareFunction (state {nextBlock = 1, currentSourceFile = sourceFile}) function + pending = pendingFunctions afterFunction + nextState = afterFunction {pendingFunctions = []} + (later, final) = go nextState remaining (if null pending then lifted else pending : lifted) + in (prepared : later, final) prepareFunction :: PrepState -> CoreFunction -> (CorePrepFunction, PrepState) prepareFunction state function = - let (blocks, after) = prepareStatements (state {loopTargets = []}) (OpenBlock 0 []) (coreFunctionBody function) + let (blocks, after) = + prepareStatementsTo fallingOff (state {loopTargets = []}) (OpenBlock 0 []) (coreFunctionBody function) [] in ( CorePrepFunction (coreFunctionName function) (currentSourceFile state) @@ -149,63 +161,97 @@ prepareFunction state function = blocks , after ) - + where + -- A function without a result may end without a return: reaching + -- the end of its body returns. A function with a result returns on + -- every path, which Core verification has established, so the end + -- of its body is not reached and is marked as such. + fallingOff + | coreFunctionReturnType function == unitType = CorePrepReturn (CorePrepLiteral CoreUnit unitType) + | otherwise = CorePrepUnreachable + +{- | Every symbol identity a function mentions. + +The identities are prepended to an accumulator. Appending the lists of the +operands instead would copy the identities of a first operand once for every +operator above it, which is quadratic in the length of an operator chain. +-} symbolIds :: CoreFunction -> [Int] symbolIds function = symbolIdValue (resolvedSymbol (coreFunctionName function)) : map (symbolIdValue . resolvedSymbol . fst) (coreFunctionParameters function) - ++ concatMap statementSymbolIds (coreFunctionBody function) + ++ statementsSymbolIds (coreFunctionBody function) [] + +statementsSymbolIds :: [CoreStatement] -> [Int] -> [Int] +statementsSymbolIds statements rest = foldr statementSymbolIds rest statements -statementSymbolIds :: CoreStatement -> [Int] -statementSymbolIds statement = case statement of - CoreBind binding -> symbol (coreBindingName binding) : expressionSymbolIds (coreBindingValue binding) - CoreAssign name expression -> symbol name : expressionSymbolIds expression - CoreReturn expression -> expressionSymbolIds expression +statementSymbolIds :: CoreStatement -> [Int] -> [Int] +statementSymbolIds statement rest = case statement of + CoreBind binding -> symbol (coreBindingName binding) : expressionSymbolIds (coreBindingValue binding) rest + CoreAssign name expression -> symbol name : expressionSymbolIds expression rest + CoreReturn expression -> expressionSymbolIds expression rest CoreIf condition trueBranch falseBranch -> - expressionSymbolIds condition ++ concatMap statementSymbolIds trueBranch ++ concatMap statementSymbolIds falseBranch - CoreWhile condition body -> expressionSymbolIds condition ++ concatMap statementSymbolIds body - CoreDoWhile body condition -> concatMap statementSymbolIds body ++ expressionSymbolIds condition + expressionSymbolIds condition (statementsSymbolIds trueBranch (statementsSymbolIds falseBranch rest)) + CoreWhile condition body -> expressionSymbolIds condition (statementsSymbolIds body rest) + CoreDoWhile body condition -> statementsSymbolIds body (expressionSymbolIds condition rest) CoreFor condition body update -> - expressionSymbolIds condition ++ concatMap statementSymbolIds body ++ concatMap statementSymbolIds update - CoreBreak -> [] - CoreContinue -> [] - CoreEvaluate expression -> expressionSymbolIds expression + expressionSymbolIds condition (statementsSymbolIds body (statementsSymbolIds update rest)) + CoreBreak -> rest + CoreContinue -> rest + CoreEvaluate expression -> expressionSymbolIds expression rest where symbol = symbolIdValue . resolvedSymbol -expressionSymbolIds :: CoreExpression -> [Int] -expressionSymbolIds expression = case expression of - CoreVariable name _ -> [symbol name] - CoreLiteral _ _ -> [] - CoreApply callee arguments _ -> expressionSymbolIds callee ++ concatMap expressionSymbolIds arguments - CorePrimitive _ arguments _ -> concatMap expressionSymbolIds arguments - CoreLet name _ value body _ -> symbol name : expressionSymbolIds value ++ expressionSymbolIds body - CoreConditional condition whenTrue whenFalse _ -> - concatMap expressionSymbolIds [condition, whenTrue, whenFalse] +expressionSymbolIds :: CoreExpression -> [Int] -> [Int] +expressionSymbolIds expression rest = case expression of + CoreVariable name _ -> symbol name : rest + CoreLiteral _ _ -> rest + CoreApply callee arguments _ -> expressionSymbolIds callee (expressions arguments rest) + CorePrimitive _ arguments _ -> expressions arguments rest + CoreLet name _ value body _ -> symbol name : expressionSymbolIds value (expressionSymbolIds body rest) + CoreConditional condition whenTrue whenFalse _ -> expressions [condition, whenTrue, whenFalse] rest CoreClosure captures parameters _ body _ -> map (symbol . coreCaptureName) captures - ++ concatMap (expressionSymbolIds . coreCaptureValue) captures - ++ map (symbol . fst) parameters - ++ concatMap statementSymbolIds body + ++ expressions + (map coreCaptureValue captures) + (map (symbol . fst) parameters ++ statementsSymbolIds body rest) where symbol = symbolIdValue . resolvedSymbol - -prepareStatements :: PrepState -> OpenBlock -> [CoreStatement] -> ([CorePrepBlock], PrepState) -prepareStatements state open [] = ([closeBlock open CorePrepUnreachable], state) -prepareStatements state open (statement : remaining) = case statement of + expressions values after = foldr expressionSymbolIds after values + +{- | Lower statements into blocks, ending the last open block with the given +terminator and placing the given blocks after the ones produced. + +Both arguments exist so that the work is linear in the size of the body. A +branch or a loop body falls through to a known block; passing that jump down +ends the region with it directly, where patching the blocks afterwards would +visit every block of a nested region once per enclosing region. The blocks +that follow are passed in so that they are consed onto, where appending them +would copy the blocks of a nested region once per enclosing region; an +@else if@ chain is such a nest, one level per link. + +Only the last block of a region can be left open: a @return@, @break@ or +@continue@ closes its block and drops the statements after it, and every +nested region is closed by its own terminator. +-} +prepareStatementsTo :: + CorePrepTerminator -> PrepState -> OpenBlock -> [CoreStatement] -> [CorePrepBlock] -> ([CorePrepBlock], PrepState) +prepareStatementsTo end state open [] rest = (closeBlock open end : rest, state) +prepareStatementsTo end state open (statement : remaining) rest = case statement of CoreBind binding -> let (closed, continued, operation, after) = atomizeOperation state open (coreBindingValue binding) instruction = CorePrepBind (coreBindingName binding) (coreBindingType binding) (coreBindingMutable binding) operation - (later, final) = prepareStatements after (appendInstruction continued instruction) remaining + (later, final) = prepareStatementsTo end after (appendInstruction continued instruction) remaining rest in (closed ++ later, final) CoreAssign name value -> let (closed, continued, atom, after) = atomize state open value - (later, final) = prepareStatements after (appendInstruction continued (CorePrepAssign name atom)) remaining + (later, final) = prepareStatementsTo end after (appendInstruction continued (CorePrepAssign name atom)) remaining rest in (closed ++ later, final) CoreEvaluate value | discardsItsOperation value -> let (closed, continued, operation, after) = atomizeOperation state open value - (later, final) = prepareStatements after (appendInstruction continued (CorePrepEvaluate operation)) remaining + (later, final) = + prepareStatementsTo end after (appendInstruction continued (CorePrepEvaluate operation)) remaining rest in (closed ++ later, final) | otherwise -> -- Only a call may be an instruction whose result is dropped. @@ -213,11 +259,11 @@ prepareStatements state open (statement : remaining) = case statement of -- its operands still run and may trap, and the unused atom is -- ignored. let (closed, continued, _, after) = atomize state open value - (later, final) = prepareStatements after continued remaining + (later, final) = prepareStatementsTo end after continued remaining rest in (closed ++ later, final) CoreReturn value -> let (closed, continued, atom, after) = atomize state open value - in (closed ++ [closeBlock continued (CorePrepReturn atom)], after) + in (closed ++ closeBlock continued (CorePrepReturn atom) : rest, after) CoreIf condition trueBranch falseBranch -> let (conditionBlocks, conditionOpen, conditionAtom, afterCondition) = atomize state open condition (booleanOpen, booleanAtom, afterBoolean) = booleanizeAtom afterCondition conditionOpen conditionAtom @@ -225,25 +271,31 @@ prepareStatements state open (statement : remaining) = case statement of falseId = trueId + 1 joinId = falseId + 1 branchState = afterBoolean {nextBlock = joinId + 1} - (trueBlocks, afterTrue) = prepareBranch branchState trueId joinId trueBranch - (falseBlocks, afterFalse) = prepareBranch afterTrue falseId joinId falseBranch + -- The states are threaded in source order. The block lists are + -- built back to front: each region is consed onto the blocks + -- that follow it, which laziness allows although those depend + -- on a later state. + (trueBlocks, afterTrue) = prepareBranchTo branchState trueId joinId trueBranch falseBlocks + (falseBlocks, afterFalse) = prepareBranchTo afterTrue falseId joinId falseBranch tailBlocks header = closeBlock booleanOpen (CorePrepBranch booleanAtom trueId falseId) - (tailBlocks, final) = prepareStatements afterFalse (OpenBlock joinId []) remaining - in (conditionBlocks ++ [header] ++ trueBlocks ++ falseBlocks ++ tailBlocks, final) + (tailBlocks, final) = prepareStatementsTo end afterFalse (OpenBlock joinId []) remaining rest + in (conditionBlocks ++ header : trueBlocks, final) CoreWhile condition body -> let (loopBlocks, exitOpen, afterLoop) = prepareWhile state open condition body - (tailBlocks, final) = prepareStatements afterLoop exitOpen remaining + (tailBlocks, final) = prepareStatementsTo end afterLoop exitOpen remaining rest in (loopBlocks ++ tailBlocks, final) CoreDoWhile body condition -> let (loopBlocks, exitOpen, afterLoop) = prepareDoWhile state open body condition - (tailBlocks, final) = prepareStatements afterLoop exitOpen remaining + (tailBlocks, final) = prepareStatementsTo end afterLoop exitOpen remaining rest in (loopBlocks ++ tailBlocks, final) CoreFor condition body update -> let (loopBlocks, exitOpen, afterLoop) = prepareFor state open condition body update - (tailBlocks, final) = prepareStatements afterLoop exitOpen remaining + (tailBlocks, final) = prepareStatementsTo end afterLoop exitOpen remaining rest in (loopBlocks ++ tailBlocks, final) - CoreBreak -> closeLoopControl state open True - CoreContinue -> closeLoopControl state open False + CoreBreak -> onto (closeLoopControl state open True) + CoreContinue -> onto (closeLoopControl state open False) + where + onto (blocks, after) = (blocks ++ rest, after) {- | Close the current block at the innermost loop transfer destination. The Boolean selects @break@ (exit) versus @continue@ (continuation point); @@ -274,8 +326,7 @@ prepareWhile state incoming condition body = (booleanOpen, predicate, afterBoolean) = booleanizeAtom afterCondition conditionOpen atom branch = closeBlock booleanOpen (CorePrepBranch predicate bodyId exitId) bodyState = afterBoolean {loopTargets = (exitId, conditionId) : loopTargets afterBoolean} - (bodyBlocks, afterBody) = prepareStatements bodyState (OpenBlock bodyId []) body - bodyEnd = jumpOpenBlocks conditionId bodyBlocks + (bodyEnd, afterBody) = prepareStatementsTo (CorePrepJump conditionId) bodyState (OpenBlock bodyId []) body [] finalState = afterBody {loopTargets = loopTargets state} in ([entry] ++ conditionBlocks ++ [branch] ++ bodyEnd, OpenBlock exitId [], finalState) @@ -293,8 +344,7 @@ prepareDoWhile state incoming body condition = reserved = state {nextBlock = exitId + 1} entry = closeBlock incoming (CorePrepJump bodyId) bodyState = reserved {loopTargets = (exitId, conditionId) : loopTargets state} - (bodyBlocks, afterBody) = prepareStatements bodyState (OpenBlock bodyId []) body - bodyEnd = jumpOpenBlocks conditionId bodyBlocks + (bodyEnd, afterBody) = prepareStatementsTo (CorePrepJump conditionId) bodyState (OpenBlock bodyId []) body [] (conditionBlocks, conditionOpen, atom, afterCondition) = atomize afterBody (OpenBlock conditionId []) condition (booleanOpen, predicate, afterBoolean) = booleanizeAtom afterCondition conditionOpen atom @@ -322,11 +372,10 @@ prepareFor state incoming condition body update = (booleanOpen, predicate, afterBoolean) = booleanizeAtom afterCondition conditionOpen atom branch = closeBlock booleanOpen (CorePrepBranch predicate bodyId exitId) bodyState = afterBoolean {loopTargets = (exitId, updateId) : loopTargets afterBoolean} - (bodyBlocks, afterBody) = prepareStatements bodyState (OpenBlock bodyId []) body - bodyEnd = jumpOpenBlocks updateId bodyBlocks + (bodyEnd, afterBody) = prepareStatementsTo (CorePrepJump updateId) bodyState (OpenBlock bodyId []) body [] updateState = afterBody {loopTargets = (exitId, updateId) : loopTargets state} - (updateBlocks, afterUpdate) = prepareStatements updateState (OpenBlock updateId []) update - updateEnd = jumpOpenBlocks conditionId updateBlocks + (updateEnd, afterUpdate) = + prepareStatementsTo (CorePrepJump conditionId) updateState (OpenBlock updateId []) update [] finalState = afterUpdate {loopTargets = loopTargets state} in ([entry] ++ conditionBlocks ++ [branch] ++ bodyEnd ++ updateEnd, OpenBlock exitId [], finalState) @@ -340,21 +389,13 @@ discardsItsOperation expression = case expression of CoreApply {} -> True _ -> False -jumpOpenBlocks :: Int -> [CorePrepBlock] -> [CorePrepBlock] -jumpOpenBlocks target = map connect - where - connect block - | corePrepBlockTerminator block == CorePrepUnreachable = - block {corePrepBlockTerminator = CorePrepJump target} - | otherwise = block - appendInstruction :: OpenBlock -> CorePrepInstruction -> OpenBlock appendInstruction open instruction = - open {openBlockInstructions = openBlockInstructions open ++ [instruction]} + open {openBlockInstructions = instruction : openBlockInstructions open} closeBlock :: OpenBlock -> CorePrepTerminator -> CorePrepBlock closeBlock open terminator = - CorePrepBlock (openBlockId open) (openBlockInstructions open) terminator + CorePrepBlock (openBlockId open) (reverse (openBlockInstructions open)) terminator -- Numeric conditions are a source-language convenience. Core retains their -- numeric type for optimization, while CorePrep makes the zero comparison @@ -383,13 +424,10 @@ corePrepAtomType atom = case atom of CorePrepVariable _ valueType -> valueType CorePrepLiteral _ valueType -> valueType -prepareBranch :: PrepState -> Int -> Int -> [CoreStatement] -> ([CorePrepBlock], PrepState) -prepareBranch state blockId joinId statements = - let (blocks, after) = prepareStatements state (OpenBlock blockId []) statements - in (map addJump blocks, after) - where - addJump block | corePrepBlockTerminator block == CorePrepUnreachable = block {corePrepBlockTerminator = CorePrepJump joinId} - addJump block = block +-- | Lower one branch of a conditional: it falls through to the join block. +prepareBranchTo :: PrepState -> Int -> Int -> [CoreStatement] -> [CorePrepBlock] -> ([CorePrepBlock], PrepState) +prepareBranchTo state blockId joinId = + prepareStatementsTo (CorePrepJump joinId) state (OpenBlock blockId []) atomize :: PrepState -> OpenBlock -> CoreExpression -> ([CorePrepBlock], OpenBlock, CorePrepAtom, PrepState) atomize state open expression = case expression of diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Verifier.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Verifier.hs index e3649e50..dd5962e5 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Verifier.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Verifier.hs @@ -6,12 +6,14 @@ Verification is required after decoding artifacts and before native lowering. -} module Visual.XSharp.Core.CorePrep.Verifier (verifyCorePrep) where -import Data.List (nub) +import Data.IntSet qualified as IntSet +import Data.Set qualified as Set import Visual.XSharp.AST import Visual.XSharp.Core qualified as Core import Visual.XSharp.Core.CorePrep import Visual.XSharp.Core.Template import Visual.XSharp.Diagnostic +import Visual.XSharp.RuntimeCall -- | Return the unchanged module when valid, or every discovered invariant error. verifyCorePrep :: CorePrepModule -> Either [Diagnostic] CorePrepModule @@ -36,7 +38,7 @@ verifyFunction function = ++ duplicateIds "VXC0003" "duplicate CorePrep block id" blockIds ++ duplicateIds "VXC0004" "duplicate CorePrep parameter symbol" parameterIds ++ duplicateIds "VXC0005" "CorePrep symbol is defined more than once" (parameterIds ++ definitionIds) - ++ concatMap (verifyBlock blockIds) blocks + ++ concatMap (verifyBlock (IntSet.fromList blockIds)) blocks where verifyParameterType = verifyType "function parameter" missingEntry ids = @@ -47,7 +49,7 @@ verifyFunction function = blockDefinitions :: CorePrepBlock -> [SymbolId] blockDefinitions block = [resolvedSymbol name | CorePrepBind name _ _ _ <- corePrepBlockInstructions block] -verifyBlock :: [Int] -> CorePrepBlock -> [Diagnostic] +verifyBlock :: IntSet.IntSet -> CorePrepBlock -> [Diagnostic] verifyBlock blockIds block = concatMap verifyInstruction (corePrepBlockInstructions block) ++ verifyTerminator blockIds (corePrepBlockTerminator block) @@ -97,12 +99,22 @@ verifyCapture (CorePrepCapture mode name valueType atom) = _ -> False verifyPrimitive :: Core.CorePrimitive -> [CorePrepAtom] -> Type -> [Diagnostic] +verifyPrimitive Core.CoreRuntimeCall atoms resultType = + -- The first atom names the function; the rest are its arguments. + case runtimeCallDefect named (map atomType (drop 1 atoms)) resultType of + Just defect -> [problem "VXC0026" ("CorePrep " ++ runtimeDefectText defect)] + Nothing -> [] + where + named = case atoms of + CorePrepLiteral (Core.CoreInteger identity) valueType : _ | valueType == intType -> runtimeFunctionOf identity + _ -> Nothing verifyPrimitive primitive atoms resultType | length atoms /= expectedArity = [problem "VXC0007" "CorePrep primitive has the wrong arity"] | any ((== ErrorType) . atomType) atoms = [problem "VXC0008" "CorePrep primitive contains an unresolved type"] | otherwise = operandProblems ++ resultProblems where - expectedArity = if primitive `elem` [Core.CoreNegate, Core.CoreLogicalNot, Core.CoreBitwiseNot] then 1 else 2 + expectedArity = + if primitive `elem` [Core.CoreNegate, Core.CoreLogicalNot, Core.CoreBitwiseNot, Core.CoreMemoize] then 1 else 2 comparisonOrLogical = if primitive `elem` [ Core.CoreLessThan @@ -141,6 +153,13 @@ verifyPrimitive primitive atoms resultType | referenceLike subjectType && typeSpelling identityType == "uint" -> [] | otherwise -> [problem "VXC0024" "CorePrep type test requires a reference subject and uint identity"] _ -> [] + | primitive == Core.CoreMemoize = case operandTypes of + [FunctionType [] result] | result == boolType || isNumericType result -> [] + _ -> + [ problem + "VXC0025" + "CorePrep memoization requires a callable without parameters whose result is bool or numeric" + ] | not operandsAgree = [problem "VXC0019" "CorePrep primitive operand types do not agree"] | logical && not booleanContext = [problem "VXC0020" "CorePrep logical primitive requires bool or numeric operands"] | integerOnly && not integer = [problem "VXC0022" "CorePrep bitwise primitive requires integer operands"] @@ -164,14 +183,17 @@ verifyPrimitive primitive atoms resultType NamedType _ _ -> not (isNumericType valueType) && valueType /= unitType _ -> False -verifyTerminator :: [Int] -> CorePrepTerminator -> [Diagnostic] +verifyTerminator :: IntSet.IntSet -> CorePrepTerminator -> [Diagnostic] verifyTerminator blockIds terminator = case terminator of CorePrepReturn atom -> verifyAtom atom CorePrepBranch atom trueTarget falseTarget -> verifyAtom atom ++ requireBool atom ++ targets [trueTarget, falseTarget] CorePrepJump target -> targets [target] CorePrepUnreachable -> [] where - targets values = [problem "VXC0010" "CorePrep terminator targets a missing block" | any (`notElem` blockIds) values] + targets values = + [ problem "VXC0010" "CorePrep terminator targets a missing block" + | any (`IntSet.notMember` blockIds) values + ] requireBool atom = [problem "VXC0011" "CorePrep branch condition must be bool" | atomType atom /= boolType] verifyAtom :: CorePrepAtom -> [Diagnostic] @@ -279,8 +301,11 @@ verifyType context valueType = map templateProblem (validateTemplateType 128 val renderPath [] = "the type root" renderPath indexes = "argument " ++ concatMap (\index -> "[" ++ show index ++ "]") indexes -duplicateIds :: (Eq a) => String -> String -> [a] -> [Diagnostic] -duplicateIds code message values = [problem code message | length values /= length (nub values)] +-- A function may have tens of thousands of blocks and symbols, so the +-- distinct values are counted through a set rather than by comparing every +-- pair. +duplicateIds :: (Ord a) => String -> String -> [a] -> [Diagnostic] +duplicateIds code message values = [problem code message | length values /= Set.size (Set.fromList values)] problem :: String -> String -> Diagnostic problem code message = Diagnostic CorePrepStage Error code Nothing message diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Decode.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Decode.hs index b5958e6f..9305d2c4 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Decode.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Decode.hs @@ -362,6 +362,8 @@ tagPrimitive tag = case tag of 24 -> Just CoreBitwiseOr 25 -> Just CoreBitwiseNot 26 -> Just CoreTypeIs + 27 -> Just CoreMemoize + 28 -> Just CoreRuntimeCall _ -> Nothing invalidTag :: String -> Word8 -> Decoder a diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Encode.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Encode.hs index 32527fb4..6c055ea0 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Encode.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Encode.hs @@ -299,6 +299,8 @@ primitiveTag primitive = case primitive of CoreBitwiseOr -> 24 CoreBitwiseNot -> 25 CoreTypeIs -> 26 + CoreMemoize -> 27 + CoreRuntimeCall -> 28 encodeVector :: WireLimits -> String -> Int -> (a -> Encoder) -> [a] -> Encoder encodeVector limits context maximumCount encode values = do diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Format.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Format.hs index a0c8e2f9..ef62f2d2 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Format.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/CorePrep/Wire/Format.hs @@ -26,7 +26,7 @@ newtype WireVersion -- | Schema version emitted by current CorePrep encoders. currentWireVersion :: WireVersion -currentWireVersion = WireVersion 6 +currentWireVersion = WireVersion 8 -- | Four-byte ASCII identifier at the start of each CorePrep document. wireMagic :: [Word8] diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Analysis.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Analysis.hs index a00cad78..645b27a0 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Analysis.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Analysis.hs @@ -29,6 +29,7 @@ import Visual.XSharp.AST (ResolvedName, SymbolId, resolvedSymbol) import Visual.XSharp.Core import Visual.XSharp.Core.Optimizer.IntegerFacts import Visual.XSharp.Core.Scalar (isCoreFloatingType, isCoreIntegerType) +import Visual.XSharp.RuntimeCall (runtimeFunctionOf, runtimeObservable) -- Effect ordering is deliberately conservative. The optimizer currently needs -- one decisive property: only PureEffect may disappear when its value is dead. @@ -169,6 +170,16 @@ caller, so proving a divisor nonzero cannot erase a call in an operand. -} primitiveEffectWithProvenNonzeroDivisor :: Bool -> CorePrimitive -> [CoreExpression] -> Effect primitiveEffectWithProvenNonzeroDivisor provenNonzero primitive arguments + -- A runtime call that writes to the console is observed by whoever + -- reads the output: it is kept, kept once, and kept where it stands, + -- like a call of a function that nothing is known about. The others + -- compute a string and do nothing else. + | primitive == CoreRuntimeCall = case arguments of + CoreLiteral (CoreInteger identity) _ : _ + | Just function <- runtimeFunctionOf identity + , not (runtimeObservable function) -> + AllocationEffect + _ -> CallEffect | primitive `elem` [CoreDivide, CoreFloorDivide, CoreRemainder] , firstTypeIsInteger arguments , not (knownNonzeroDivisor arguments || provenNonzero) = diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Constant.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Constant.hs index e67ab136..06b6dcc1 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Constant.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/Constant.hs @@ -167,10 +167,15 @@ simplifyExpressionUsing environment facts expression = case expression of (map (simplifyExpressionUsing environment facts) arguments) valueType CorePrimitive primitive arguments valueType -> + -- The facts are consulted for Boolean results only. Asking them + -- about every node would evaluate the whole operand tree again at + -- each level of an operator chain, which is quadratic in its length. let simplified = foldPrimitive primitive (map (simplifyExpressionUsing environment facts) arguments) valueType - in case conditionTruthFromFacts facts simplified of - Just truth | valueType == boolType -> CoreLiteral (CoreBoolean truth) boolType - _ -> simplified + in if valueType /= boolType || not (withinFactQueryBudget simplified) + then simplified + else case conditionTruthFromFacts facts simplified of + Just truth -> CoreLiteral (CoreBoolean truth) boolType + Nothing -> simplified CoreLet name bindingType value body valueType -> let simplifiedValue = simplifyExpressionUsing environment facts value bodyEnvironment = @@ -213,6 +218,37 @@ simplifyExpressionUsing environment facts expression = case expression of simplifyCapture capture = capture {coreCaptureValue = simplifyExpressionWithFacts environment facts (coreCaptureValue capture)} +{- | The largest condition, in expression nodes, whose truth is looked up in +the integer facts while it is simplified. + +The lookup walks the condition, and it is made at every Boolean node from +the leaves up, so on a chain of @&&@ or @||@ its cost grows with a power of +the length of the chain: 200 comparisons took 47 seconds. A condition of +ordinary size is far below the budget. A longer one is still simplified by +folding; only the facts are not asked about its larger parts. +-} +factQueryNodeBudget :: Int +factQueryNodeBudget = 256 + +{- | Whether an expression has at most 'factQueryNodeBudget' nodes; the +count stops as soon as the budget is exceeded. +-} +withinFactQueryBudget :: CoreExpression -> Bool +withinFactQueryBudget root = go factQueryNodeBudget [root] + where + go _ [] = True + go remaining (expression : pending) + | remaining <= 0 = False + | otherwise = go (remaining - 1) (children expression ++ pending) + children expression = case expression of + CoreVariable {} -> [] + CoreLiteral {} -> [] + CoreApply callee arguments _ -> callee : arguments + CorePrimitive _ arguments _ -> arguments + CoreLet _ _ value body _ -> [value, body] + CoreConditional condition whenTrue whenFalse _ -> [condition, whenTrue, whenFalse] + CoreClosure captures _ _ _ _ -> map coreCaptureValue captures + foldPrimitive :: CorePrimitive -> [CoreExpression] -> Type -> CoreExpression foldPrimitive primitive arguments valueType = case evaluatePrimitive primitive arguments valueType of diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/IntegerFacts.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/IntegerFacts.hs index 350046fa..26491787 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/IntegerFacts.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Optimizer/IntegerFacts.hs @@ -339,12 +339,42 @@ the last, potentially under-approximating iteration. loopFactSummary :: IntegerFacts -> CoreStatement -> Maybe LoopFactSummary loopFactSummary input statement = case statement of CoreWhile condition body -> - Just $ solveLoop input (whileStep condition body []) + Just $ solve body (whileStep condition body []) CoreDoWhile body condition -> - Just $ solveLoop input (doWhileStep body condition) + Just $ solve body (doWhileStep body condition) CoreFor condition body update -> - Just $ solveLoop input (whileStep condition body update) + Just $ solve (body ++ update) (whileStep condition body update) _ -> Nothing + where + -- Every round of the fixed point analyses the loop body, and with it + -- every loop nested in the body, each of which iterates in turn: the + -- work is the product of the rounds of all enclosing loops, which is + -- exponential in the depth of the nest. Only the innermost levels + -- iterate; a loop with a deeper nest inside it is summarized in one + -- pass from the state that assumes nothing, which is always sound + -- and is what the iteration falls back to when it does not settle. + solve nested step + | loopNestingHeight nested >= iteratedLoopNesting = summarize 1 True emptyIntegerFacts (step emptyIntegerFacts) + | otherwise = solveLoop input step + +{- | How many levels of nested loops are iterated to a fixed point. A loop +whose body holds this many levels of loops or more is summarized in a single +pass, so the cost of a nest grows with its size and not exponentially with +its depth. +-} +iteratedLoopNesting :: Int +iteratedLoopNesting = 2 + +-- | The deepest nest of loops in the statements: 0 when they hold no loop. +loopNestingHeight :: [CoreStatement] -> Int +loopNestingHeight = foldr (max . statementHeight) 0 + where + statementHeight statement = case statement of + CoreIf _ whenTrue whenFalse -> max (loopNestingHeight whenTrue) (loopNestingHeight whenFalse) + CoreWhile _ body -> 1 + loopNestingHeight body + CoreDoWhile body _ -> 1 + loopNestingHeight body + CoreFor _ body update -> 1 + max (loopNestingHeight body) (loopNestingHeight update) + _ -> 0 loopIterationLimit :: Int loopIterationLimit = 8 diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Verifier.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Verifier.hs index 906abd34..80c4434a 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Verifier.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Verifier.hs @@ -9,11 +9,13 @@ module Visual.XSharp.Core.Verifier (verifyCore) where import Data.List (group, sort) import Data.Map.Strict qualified as Map +import Data.Set qualified as Set import Visual.XSharp.AST import Visual.XSharp.Core import Visual.XSharp.Core.Scalar import Visual.XSharp.Core.Template import Visual.XSharp.Diagnostic +import Visual.XSharp.RuntimeCall type Environment = Map.Map SymbolId (Type, Bool) @@ -41,14 +43,19 @@ moduleProblems moduleValue = ++ duplicates "VXC1032" "duplicate Core function source owner" (map fst sourceOwners) ++ [ problem "VXC1033" "Core function source owner is absent from the module source catalog" | (_, path) <- sourceOwners - , path `notElem` sourceFiles + , path `Set.notMember` sourceFileSet ] ++ [ problem "VXC1034" "Core function has no source owner" | not (null sourceFiles) , function <- functions , let identifier = symbolIdValue (resolvedSymbol (coreFunctionName function)) - , identifier `notElem` map fst sourceOwners + , identifier `Set.notMember` ownedFunctions ] + -- Sets, because both checks run once for every function: against + -- lists, a module of thousands of functions was verified in time + -- with the square of their number. + sourceFileSet = Set.fromList sourceFiles + ownedFunctions = Set.fromList (map fst sourceOwners) invalidPath path = null path || '\0' `elem` path functionEnvironment = Map.fromList @@ -282,12 +289,22 @@ callProblems callee arguments resultType = case expressionType callee of _ -> [problem "VXC1025" "Core call target is not a function"] primitiveProblems :: CorePrimitive -> [CoreExpression] -> Type -> [Diagnostic] +primitiveProblems CoreRuntimeCall arguments resultType = + -- The first operand names the function and the rest are checked against + -- the row of the catalog that function has. + case runtimeCallDefect named (map expressionType (drop 1 arguments)) resultType of + Just defect -> [problem "VXC1075" ("Core " ++ runtimeDefectText defect)] + Nothing -> [] + where + named = case arguments of + CoreLiteral (CoreInteger identity) valueType : _ | valueType == intType -> runtimeFunctionOf identity + _ -> Nothing primitiveProblems primitive arguments resultType = [problem "VXC1026" "Core primitive has the wrong operand count" | length arguments /= arity] ++ operandProblems ++ typeMismatch "VXC1028" "Core primitive result has the wrong type" expectedResult resultType where - unary = primitive `elem` [CoreNegate, CoreLogicalNot, CoreBitwiseNot] + unary = primitive `elem` [CoreNegate, CoreLogicalNot, CoreBitwiseNot, CoreMemoize] logical = primitive `elem` [CoreLogicalAnd, CoreLogicalOr, CoreLogicalNot] integerOnly = primitive `elem` [CoreShiftLeft, CoreShiftRight, CoreBitwiseAnd, CoreBitwiseXor, CoreBitwiseOr, CoreBitwiseNot] comparison = primitive `elem` [CoreLessThan, CoreLessEqual, CoreGreaterThan, CoreGreaterEqual, CoreEqual, CoreNotEqual] @@ -304,6 +321,14 @@ primitiveProblems primitive arguments resultType = | isReferenceLike subjectType && identityType == namedType "uint" -> [] | otherwise -> [problem "VXC1044" "Core type test requires a reference subject and uint identity"] _ -> [] + -- A remembered result lives in the callable itself, in a slot + -- that owns nothing: only a result without ownership is kept. + | primitive == CoreMemoize = case argumentTypes of + [FunctionType [] result] + | result == boolType || isCoreNumericType result -> [] + | otherwise -> [memoizeProblem] + [_] -> [memoizeProblem] + _ -> [] | logical && not operandsBoolean = [problem "VXC1027" "Core logical primitive requires bool or numeric operands"] | integerOnly && not operandsInteger = [problem "VXC1027" "Core bitwise primitive requires integer operands"] | primitive `elem` [CoreEqual, CoreNotEqual] @@ -317,6 +342,8 @@ primitiveProblems primitive arguments resultType = | logical || comparison || primitive == CoreTypeIs = boolType | primitive == CoreFloorDivide && isCoreFloatingType firstType = intType | otherwise = firstType + memoizeProblem = + problem "VXC1073" "Core memoization requires a callable without parameters whose result is bool or numeric" isReferenceLike valueType = case valueType of FunctionType _ _ -> True NamedType _ _ -> not (isCoreNumericType valueType) && valueType /= unitType @@ -338,6 +365,13 @@ literalProblems literal valueType = NamedType _ _ -> not (isCoreNumericType value) && value /= unitType _ -> False +{- | Whether control can never fall off the end of the statements. + +That is so when they return on every path, and also when they hold a loop +that cannot be left: its condition is the literal true and no @break@ +leaves it. Such a loop ends only through a @return@ inside it, or not at +all, so nothing after it is reached. +-} statementsAlwaysReturn :: [CoreStatement] -> Bool statementsAlwaysReturn [] = False statementsAlwaysReturn (statement : remaining) = case statement of @@ -345,7 +379,24 @@ statementsAlwaysReturn (statement : remaining) = case statement of CoreIf _ trueBranch falseBranch -> (not (null falseBranch) && statementsAlwaysReturn trueBranch && statementsAlwaysReturn falseBranch) || statementsAlwaysReturn remaining + CoreWhile condition body + | cannotBeLeft condition body -> True + CoreDoWhile body condition + | cannotBeLeft condition body -> True + CoreFor condition body update + | cannotBeLeft condition (body ++ update) -> True _ -> statementsAlwaysReturn remaining + where + cannotBeLeft condition body = isLiteralTrue condition && not (leftByBreak body) + isLiteralTrue expression = case expression of + CoreLiteral (CoreBoolean True) _ -> True + _ -> False + -- A break in a nested loop leaves that loop. + leftByBreak = any breaks + breaks nested = case nested of + CoreBreak -> True + CoreIf _ trueBranch falseBranch -> leftByBreak trueBranch || leftByBreak falseBranch + _ -> False emptyName :: QualifiedName -> [Diagnostic] emptyName (QualifiedName parts) = diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Wire.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Wire.hs index 14c251f0..c05d6b4a 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Core/Wire.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Core/Wire.hs @@ -33,7 +33,7 @@ newtype CoreWireVersion = CoreWireVersion {coreWireVersionNumber :: Word16} -- | Schema version emitted by the current Core writer. currentCoreWireVersion :: CoreWireVersion -currentCoreWireVersion = CoreWireVersion 8 +currentCoreWireVersion = CoreWireVersion 10 -- | Finite bounds for total bytes, recursion, and individual collections. data CoreWireLimits = CoreWireLimits @@ -89,15 +89,43 @@ data CoreWireError = CoreWireError } deriving (Eq, Ord, Read, Show) -type Encoder = Either CoreWireError [Word8] +{- | Bytes under construction. + +An encoded module is assembled from the encodings of its parts, and a part +may hold nearly all of the module: the body of a deeply nested statement, or +the rest of an @else if@ chain. Appending lists would copy such a part once +for every level around it, which makes encoding quadratic in the nesting +depth. A function that prepends its bytes to whatever follows is composed in +constant time, and the list is produced once, at the end. +-} +newtype Bytes = Bytes ([Word8] -> [Word8]) + +instance Semigroup Bytes where + Bytes first <> Bytes second = Bytes (first . second) + +instance Monoid Bytes where + mempty = Bytes id + +tagByte :: Word8 -> Bytes +tagByte value = Bytes (value :) + +rawBytes :: [Word8] -> Bytes +rawBytes values = Bytes (values ++) + +runBytes :: Bytes -> [Word8] +runBytes (Bytes prepend) = prepend [] + +type Encoder = Either CoreWireError Bytes -- | Encode one Core module under explicit resource limits. encodeCore :: CoreWireLimits -> CoreModule -> Either CoreWireError [Word8] encodeCore limits moduleValue = do payload <- encodeModule limits moduleValue - let bytes = magic ++ word16 (coreWireVersionNumber currentCoreWireVersion) ++ word16 0 ++ payload - requireEncode limits "wire byte length" (maximumCoreWireBytes limits) (length bytes) - pure bytes + let output = + runBytes + (rawBytes magic <> rawBytes (word16 (coreWireVersionNumber currentCoreWireVersion)) <> rawBytes (word16 0) <> payload) + requireEncode limits "wire byte length" (maximumCoreWireBytes limits) (length output) + pure output encodeModule :: CoreWireLimits -> CoreModule -> Encoder encodeModule limits moduleValue = do @@ -117,7 +145,7 @@ encodeModule limits moduleValue = do (maximumCoreFunctions limits) (encodeFunction limits owners) (coreModuleFunctions moduleValue) - pure (name ++ sources ++ functions) + pure (name <> sources <> functions) encodeFunction :: CoreWireLimits -> Map.Map Int FilePath -> CoreFunction -> Encoder encodeFunction limits owners function = do @@ -142,11 +170,11 @@ encodeFunction limits owners function = do (maximumCoreStatements limits) (encodeStatement limits) (coreFunctionBody function) - pure (name ++ source ++ parameters ++ result ++ body) + pure (name <> source <> parameters <> result <> body) encodeParameter :: CoreWireLimits -> (ResolvedName, Type) -> Encoder encodeParameter limits (name, valueType) = - (++) + (<>) <$> encodeResolvedName limits "parameter symbol" name <*> encodeType limits 0 valueType @@ -156,12 +184,12 @@ encodeStatement limits statement = case statement of name <- encodeResolvedName limits "binding symbol" (coreBindingName binding) valueType <- encodeType limits 0 (coreBindingType binding) value <- encodeExpression limits 0 (coreBindingValue binding) - pure ([0] ++ name ++ valueType ++ encodeBool (coreBindingMutable binding) ++ value) + pure (tagByte 0 <> name <> valueType <> rawBytes (encodeBool (coreBindingMutable binding)) <> value) CoreAssign name expression -> taggedExpression 1 <$> encodeResolvedName limits "assignment symbol" name <*> encodeExpression limits 0 expression - CoreReturn expression -> (2 :) <$> encodeExpression limits 0 expression + CoreReturn expression -> (tagByte 2 <>) <$> encodeExpression limits 0 expression CoreIf condition trueBranch falseBranch -> do encodedCondition <- encodeExpression limits 0 condition encodedTrue <- @@ -178,8 +206,8 @@ encodeStatement limits statement = case statement of (maximumCoreStatements limits) (encodeStatement limits) falseBranch - pure ([3] ++ encodedCondition ++ encodedTrue ++ encodedFalse) - CoreEvaluate expression -> (4 :) <$> encodeExpression limits 0 expression + pure (tagByte 3 <> encodedCondition <> encodedTrue <> encodedFalse) + CoreEvaluate expression -> (tagByte 4 <>) <$> encodeExpression limits 0 expression CoreWhile condition body -> do encodedCondition <- encodeExpression limits 0 condition encodedBody <- @@ -189,7 +217,7 @@ encodeStatement limits statement = case statement of (maximumCoreStatements limits) (encodeStatement limits) body - pure ([5] ++ encodedCondition ++ encodedBody) + pure (tagByte 5 <> encodedCondition <> encodedBody) CoreDoWhile body condition -> do encodedBody <- encodeVector @@ -199,7 +227,7 @@ encodeStatement limits statement = case statement of (encodeStatement limits) body encodedCondition <- encodeExpression limits 0 condition - pure ([6] ++ encodedBody ++ encodedCondition) + pure (tagByte 6 <> encodedBody <> encodedCondition) CoreFor condition body update -> do encodedCondition <- encodeExpression limits 0 condition encodedBody <- @@ -216,11 +244,11 @@ encodeStatement limits statement = case statement of (maximumCoreStatements limits) (encodeStatement limits) update - pure ([7] ++ encodedCondition ++ encodedBody ++ encodedUpdate) - CoreBreak -> pure [8] - CoreContinue -> pure [9] + pure (tagByte 7 <> encodedCondition <> encodedBody <> encodedUpdate) + CoreBreak -> pure (tagByte 8) + CoreContinue -> pure (tagByte 9) where - taggedExpression tag left right = [tag] ++ left ++ right + taggedExpression tag left right = tagByte tag <> left <> right encodeExpression :: CoreWireLimits -> Int -> CoreExpression -> Encoder encodeExpression limits depth expression @@ -229,11 +257,11 @@ encodeExpression limits depth expression CoreVariable name valueType -> do encodedType <- encodeType limits 0 valueType encodedName <- encodeResolvedName limits "variable symbol" name - pure ([0] ++ encodedType ++ encodedName) + pure (tagByte 0 <> encodedType <> encodedName) CoreLiteral literal valueType -> do encodedType <- encodeType limits 0 valueType encodedLiteral <- encodeLiteral limits literal - pure ([1] ++ encodedType ++ encodedLiteral) + pure (tagByte 1 <> encodedType <> encodedLiteral) CoreApply callee arguments valueType -> do encodedType <- encodeType limits 0 valueType encodedCallee <- encodeExpression limits (depth + 1) callee @@ -244,7 +272,10 @@ encodeExpression limits depth expression (maximumCoreOperands limits) (encodeExpression limits (depth + 1)) arguments - pure ([2] ++ encodedType ++ encodedCallee ++ encodedArguments) + pure (tagByte 2 <> encodedType <> encodedCallee <> encodedArguments) + -- The first operand of a primitive is at the depth of the + -- primitive: a chain of operators nests in it, as deep as the chain + -- is long, and every stage walks that chain in a loop. CorePrimitive primitive arguments valueType -> do encodedType <- encodeType limits 0 valueType encodedArguments <- @@ -252,9 +283,9 @@ encodeExpression limits depth expression limits "primitive operand count" (maximumCoreOperands limits) - (encodeExpression limits (depth + 1)) - arguments - pure ([3, primitiveTag primitive] ++ encodedType ++ encodedArguments) + (\(position, argument) -> encodeExpression limits (depth + position) argument) + (zip (0 : repeat 1) arguments) + pure (rawBytes [3, primitiveTag primitive] <> encodedType <> encodedArguments) CoreClosure captures parameters returnType body valueType -> do encodedType <- encodeType limits 0 valueType encodedCaptures <- @@ -279,27 +310,27 @@ encodeExpression limits depth expression (maximumCoreStatements limits) (encodeStatement limits) body - pure ([4] ++ encodedType ++ encodedCaptures ++ encodedParameters ++ encodedReturn ++ encodedBody) + pure (tagByte 4 <> encodedType <> encodedCaptures <> encodedParameters <> encodedReturn <> encodedBody) CoreLet name bindingType value body valueType -> do encodedType <- encodeType limits 0 valueType encodedName <- encodeResolvedName limits "let symbol" name encodedBindingType <- encodeType limits 0 bindingType encodedValue <- encodeExpression limits (depth + 1) value encodedBody <- encodeExpression limits (depth + 1) body - pure ([5] ++ encodedType ++ encodedName ++ encodedBindingType ++ encodedValue ++ encodedBody) + pure (tagByte 5 <> encodedType <> encodedName <> encodedBindingType <> encodedValue <> encodedBody) CoreConditional condition whenTrue whenFalse valueType -> do encodedType <- encodeType limits 0 valueType encodedCondition <- encodeExpression limits (depth + 1) condition encodedTrue <- encodeExpression limits (depth + 1) whenTrue encodedFalse <- encodeExpression limits (depth + 1) whenFalse - pure ([6] ++ encodedType ++ encodedCondition ++ encodedTrue ++ encodedFalse) + pure (tagByte 6 <> encodedType <> encodedCondition <> encodedTrue <> encodedFalse) encodeCoreCapture :: CoreWireLimits -> Int -> CoreCapture -> Encoder encodeCoreCapture limits depth capture = do encodedName <- encodeResolvedName limits "closure capture symbol" (coreCaptureName capture) encodedType <- encodeType limits 0 (coreCaptureType capture) encodedValue <- encodeExpression limits (depth + 1) (coreCaptureValue capture) - pure (captureModeTag (coreCaptureMode capture) : encodedName ++ encodedType ++ encodedValue) + pure (tagByte (captureModeTag (coreCaptureMode capture)) <> encodedName <> encodedType <> encodedValue) captureModeTag :: CaptureMode -> Word8 captureModeTag mode = case mode of @@ -309,21 +340,21 @@ captureModeTag mode = case mode of encodeLiteral :: CoreWireLimits -> CoreLiteral -> Encoder encodeLiteral limits literal = case literal of - CoreUnit -> pure [0] - CoreBoolean value -> pure (1 : encodeBool value) - CoreInteger value -> (4 :) <$> encodeInteger limits value - CoreFloating spelling -> (5 :) <$> encodeAscii limits "floating literal" spelling - CoreString value -> (3 :) <$> encodeText limits "string literal" value - CoreNull -> pure [6] + CoreUnit -> pure (tagByte 0) + CoreBoolean value -> pure (tagByte 1 <> rawBytes (encodeBool value)) + CoreInteger value -> (tagByte 4 <>) <$> encodeInteger limits value + CoreFloating spelling -> (tagByte 5 <>) <$> encodeAscii limits "floating literal" spelling + CoreString value -> (tagByte 3 <>) <$> encodeText limits "string literal" value + CoreNull -> pure (tagByte 6) encodeType :: CoreWireLimits -> Int -> Type -> Encoder encodeType limits depth valueType | depth > maximumCoreTypeDepth limits = failure CoreLimitExceeded "type" "type nesting exceeds limit" - | valueType == unitType = pure [0] - | valueType == boolType = pure [1] - | valueType == intType = pure [2] - | valueType == stringType = pure [3] - | Just scalarTag <- scalarTypeTag valueType = pure [scalarTag] + | valueType == unitType = pure (tagByte 0) + | valueType == boolType = pure (tagByte 1) + | valueType == intType = pure (tagByte 2) + | valueType == stringType = pure (tagByte 3) + | Just scalarTag <- scalarTypeTag valueType = pure (tagByte scalarTag) | NamedType name arguments <- valueType = do encodedName <- encodeQualifiedName limits name encodedArguments <- @@ -333,7 +364,7 @@ encodeType limits depth valueType (maximumCoreOperands limits) (encodeTemplateArgument limits (depth + 1)) arguments - pure ([4] ++ encodedName ++ encodedArguments) + pure (tagByte 4 <> encodedName <> encodedArguments) | FunctionType parameters result <- valueType = do encodedParameters <- encodeVector @@ -343,28 +374,28 @@ encodeType limits depth valueType (encodeType limits (depth + 1)) parameters encodedResult <- encodeType limits (depth + 1) result - pure ([5] ++ encodedParameters ++ encodedResult) - | TypeVariable name <- valueType = (6 :) <$> encodeResolvedName limits "type variable symbol" name + pure (tagByte 5 <> encodedParameters <> encodedResult) + | TypeVariable name <- valueType = (tagByte 6 <>) <$> encodeResolvedName limits "type variable symbol" name | ErrorType <- valueType = failure CoreUnsupportedType "type" "ErrorType cannot cross the Core boundary" encodeTemplateArgument :: CoreWireLimits -> Int -> TemplateArgument -> Encoder encodeTemplateArgument limits depth argument = case argument of - TypeTemplateArgument valueType -> (0 :) <$> encodeType limits depth valueType + TypeTemplateArgument valueType -> (tagByte 0 <>) <$> encodeType limits depth valueType ValueTemplateArgument value -> encodeTemplateValue limits value encodeTemplateValue :: CoreWireLimits -> TemplateValue -> Encoder encodeTemplateValue limits value = case value of - IntegerTemplateValue integer -> (1 :) <$> encodeInteger limits integer - BooleanTemplateValue boolean -> pure [2, if boolean then 1 else 0] - CharacterTemplateValue scalar -> (3 :) <$> encodeInteger limits scalar - TemplateValueParameter name -> (4 :) <$> encodeResolvedName limits "template value parameter" name + IntegerTemplateValue integer -> (tagByte 1 <>) <$> encodeInteger limits integer + BooleanTemplateValue boolean -> pure (rawBytes [2, if boolean then 1 else 0]) + CharacterTemplateValue scalar -> (tagByte 3 <>) <$> encodeInteger limits scalar + TemplateValueParameter name -> (tagByte 4 <>) <$> encodeResolvedName limits "template value parameter" name encodeResolvedName :: CoreWireLimits -> String -> ResolvedName -> Encoder encodeResolvedName limits context name | symbolIdValue (resolvedSymbol name) <= 0 = failure CoreInvalidSymbol context "symbol id must be positive" | otherwise = do spelling <- encodeText limits (context ++ " spelling") (identifierText (resolvedSpelling name)) - pure (word64 (fromIntegral (symbolIdValue (resolvedSymbol name))) ++ spelling) + pure (rawBytes (word64 (fromIntegral (symbolIdValue (resolvedSymbol name)))) <> spelling) encodeQualifiedName :: CoreWireLimits -> QualifiedName -> Encoder encodeQualifiedName limits (QualifiedName parts) = @@ -379,18 +410,18 @@ encodeVector :: CoreWireLimits -> String -> Int -> (a -> Encoder) -> [a] -> Enco encodeVector limits context maximumValue encode values = do requireEncode limits context maximumValue (length values) encoded <- traverse encode values - pure (word32 (fromIntegral (length values)) ++ concat encoded) + pure (rawBytes (word32 (fromIntegral (length values))) <> mconcat encoded) encodeText :: CoreWireLimits -> String -> String -> Encoder encodeText limits context value = do requireEncode limits context (maximumCoreTextScalars limits) (length value) codePoints <- traverse encodeScalar value - pure (word32 (fromIntegral (length value)) ++ concat codePoints) + pure (rawBytes (word32 (fromIntegral (length value))) <> mconcat codePoints) where encodeScalar character | code >= 0xd800 && code <= 0xdfff = failure CoreInvalidScalar context "surrogate is not a Unicode scalar" | code > 0x10ffff = failure CoreInvalidScalar context "code point exceeds Unicode range" - | otherwise = pure (word32 (fromIntegral code)) + | otherwise = pure (rawBytes (word32 (fromIntegral code))) where code = ord character @@ -419,7 +450,7 @@ encodeInteger :: CoreWireLimits -> Integer -> Encoder encodeInteger limits value = do let magnitude = integerMagnitude (abs value) requireEncode limits "integer magnitude" (maximumCoreNumericBytes limits) (length magnitude) - pure (encodeBool (value < 0) ++ word32 (fromIntegral (length magnitude)) ++ magnitude) + pure (rawBytes (encodeBool (value < 0)) <> rawBytes (word32 (fromIntegral (length magnitude))) <> rawBytes magnitude) integerMagnitude :: Integer -> [Word8] integerMagnitude 0 = [] @@ -431,8 +462,8 @@ integerMagnitude value = reverse (unfoldr step value) encodeAscii :: CoreWireLimits -> String -> String -> Encoder encodeAscii limits context value = do requireEncode limits context (maximumCoreNumericBytes limits) (length value) - bytes <- traverse ascii value - pure (word32 (fromIntegral (length bytes)) ++ bytes) + characters <- traverse ascii value + pure (rawBytes (word32 (fromIntegral (length characters))) <> rawBytes characters) where ascii character | ord character <= 0x7f = Right (fromIntegral (ord character)) @@ -561,7 +592,18 @@ decodeExpression depth = do 3 -> do primitive <- readWord8 "primitive tag" >>= decodePrimitive valueType <- decodeType 0 - arguments <- decodeVector "primitive operand count" maximumCoreOperands (decodeExpression (depth + 1)) + count <- readWord32 "primitive operand count" + requireDecode + (toInteger count <= toInteger (maximumCoreOperands limits)) + CoreLimitExceeded + "primitive operand count" + "count exceeds configured limit" + -- The first operand is at the depth of the primitive. + arguments <- + sequence + [ decodeExpression (depth + position) + | position <- take (fromIntegral count) (0 : repeat 1) + ] pure (CorePrimitive primitive arguments valueType) 4 -> do valueType <- decodeType 0 @@ -777,6 +819,8 @@ decodePrimitive tag = case drop (fromIntegral tag) primitives of , CoreBitwiseOr , CoreBitwiseNot , CoreTypeIs + , CoreMemoize + , CoreRuntimeCall ] primitiveTag :: CorePrimitive -> Word8 @@ -807,6 +851,8 @@ primitiveTag primitive = fromIntegral (index primitive primitives) , CoreBitwiseOr , CoreBitwiseNot , CoreTypeIs + , CoreMemoize + , CoreRuntimeCall ] index :: CorePrimitive -> [CorePrimitive] -> Int index value (candidate : remaining) = if value == candidate then 0 else 1 + index value remaining diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer.hs index af1856db..5ecb9f8a 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer.hs @@ -2,18 +2,30 @@ -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 -- | Lower the typed source AST into target-independent, source-attributed Core. -module Visual.XSharp.Desugarer (Desugarer (..), defaultDesugarer, runDesugarer) where +module Visual.XSharp.Desugarer (Desugarer (..), defaultDesugarer, desugarerWithin, runDesugarer) where +import Control.Monad (forM, zipWithM) import Control.Monad.State.Strict import Data.Bits (xor) import Data.Char (ord) import Data.List (nub) +import Data.Map.Strict (Map) +import Data.Map.Strict qualified as Map import Data.Word (Word64) import Visual.XSharp.AST +import Visual.XSharp.Completion import Visual.XSharp.Core +import Visual.XSharp.Core.Scalar (isCoreNumericType) import Visual.XSharp.Desugarer.Branching +import Visual.XSharp.Desugarer.Captures +import Visual.XSharp.Desugarer.Effects +import Visual.XSharp.Desugarer.Handing +import Visual.XSharp.Desugarer.Laziness import Visual.XSharp.Desugarer.Sequencing +import Visual.XSharp.Desugarer.Symbols +import Visual.XSharp.Desugarer.Workers import Visual.XSharp.Diagnostic (Diagnostic (..), DiagnosticSeverity (Error), DiagnosticStage (DesugarerStage)) +import Visual.XSharp.RuntimeCall (runtimeFunctionIdentity, runtimeFunctionOfName) -- | Pluggable desugaring pass from typed source semantics into Core IR. newtype Desugarer = Desugarer @@ -27,24 +39,59 @@ runDesugarer = desugarTypedAST -- | Default Core lowerer used by the compiler pipeline. defaultDesugarer :: Desugarer -defaultDesugarer = Desugarer lowerTree +defaultDesugarer = Desugarer (\typed -> lowerTree [typed] typed) -lowerTree :: TypedAST -> Either [Diagnostic] CoreModule -lowerTree (TypedAST tree@(SyntaxTree namespace declarations)) = - evalStateT lowerModule (1 + maximum (0 : syntaxSymbolIds tree)) +{- | The lowerer for one tree of a program that is lowered as several: the +methods of ordinary types and the specializations of templates. Which calls +have an effect is decided from all the trees together, because a method of +one may call a method of another. +-} +desugarerWithin :: [TypedAST] -> Desugarer +desugarerWithin program = Desugarer (lowerTree program) + +lowerTree :: [TypedAST] -> TypedAST -> Either [Diagnostic] CoreModule +lowerTree program (TypedAST tree@(SyntaxTree namespace declarations)) = + evalStateT lowerModule initial where + effects = programEffects [programDeclarations | TypedAST (SyntaxTree _ programDeclarations) <- program] defaultName = QualifiedName [Identifier "Main"] sourceFiles = nub (map portableSourcePath (concatMap declarationSourceFiles declarations)) + methods = methodDeclarations declarations + initial = + LowerState + { lowerNextSymbol = 1 + maximum (0 : syntaxSymbolIds tree) + , lowerBreakTarget = Nothing + , lowerContinue = [CoreContinue] + , lowerLeave = [CoreBreak] + , lowerSettled = Nothing + , lowerFollowing = [] + , lowerDeferred = [] + , lowerMethods = methods + , lowerNeeds = methodNeeds (scalarType . lowerBoundaryType) (expressionActs effects) methods + , lowerEffects = effects + , lowerHanded = [] + , lowerSuspended = [] + , lowerWorkers = Map.empty + , lowerPendingWorkers = [] + } lowerModule = do functions <- concat <$> mapM lowerTop declarations + -- The functions that receive suspended computations are lowered + -- after the methods, once it is known which of them a call asks + -- for; lowering one may ask for more. + workers <- lowerWorkersUntilNone pure ( CoreModuleWithSources (maybe defaultName id namespace) - functions + (functions ++ map fst workers) sourceFiles - (functionSources declarations) + (functionSources declarations ++ map snd workers) ) +-- | Whether a lowered type is one whose values are computed by need. +scalarType :: Type -> Bool +scalarType lowered = lowered == boolType || isCoreNumericType lowered + -- Keep the physical file catalog even for a declaration that does not yet -- lower to executable code. The project driver uses it to produce stable, -- one-source-per-artifact output names after optimization has removed dead @@ -56,6 +103,7 @@ declarationSourceFiles declaration = case declaration of portableSpanSource declaration : concatMap declarationSourceFiles members TemplateTypeDeclaration {typeMembers = members} -> portableSpanSource declaration : concatMap declarationSourceFiles members + EnumDeclaration {} -> [portableSpanSource declaration] -- Artifact paths always use portable separators, even when discovery ran on -- Windows; these names become stable wire and output identities. @@ -72,6 +120,7 @@ functionSources = concatMap declarationFunctionSources FunctionDeclaration {} -> owner declaration TypeDeclaration {typeMembers = members} -> concatMap declarationFunctionSources members TemplateTypeDeclaration {typeMembers = members} -> concatMap declarationFunctionSources members + EnumDeclaration {} -> [] owner declaration = [ ( symbolIdValue (resolvedSymbol (declarationName declaration)) @@ -79,7 +128,135 @@ functionSources = concatMap declarationFunctionSources ) ] -type Lower = StateT Int (Either [Diagnostic]) +{- | What the lowering carries besides the tree. + +The break target and the lowering of @continue@ belong to the statement +being lowered. They are state rather than arguments because a statement is +also reached through an expression: a block used as a value holds statements, +and a @break@ or @continue@ in it must reach the loop around the expression. +-} +data LowerState = LowerState + { lowerNextSymbol :: Int + -- ^ The identity of the next compiler-generated local. + , lowerBreakTarget :: BreakTarget + -- ^ Where a @break value;@ of the statement being lowered stores. + , lowerContinue :: [CoreStatement] + {- ^ What a @continue@ of the statement being lowered becomes: the Core + @continue@ in a loop body, and a @break@ of the enclosing one-pass loop + in an update clause, where it ends the update. + -} + , lowerLeave :: [CoreStatement] + {- ^ What a @break@ of the statement being lowered becomes, after the + store of its value if it carries one: the Core @break@, preceded in an + update clause that is wrapped in a one-pass loop by the store that makes + the loop around it leave as well. + -} + , lowerSettled :: Maybe [SymbolId] + {- ^ Whether bindings are evaluated by need, and if so which locals are + not: those that are assigned after their binding and those a closure + captures. Without a list every binding is evaluated where it stands. + -} + , lowerFollowing :: [Statement ResolvedName Type] + -- ^ The statements that follow the one being lowered in its block. + , lowerDeferred :: [(SymbolId, (ResolvedName, Expression ResolvedName Type))] + {- ^ The bindings whose value is computed by need: for each, the flag + that says whether it has been computed and the expression that computes + it. + -} + , lowerMethods :: Map SymbolId (Declaration ResolvedName Type) + -- ^ The methods a call can name directly. + , lowerNeeds :: MethodNeeds + , lowerEffects :: Effects + -- ^ Which calls of the program do something that can be observed. + -- ^ Which parameters of which methods are passed by need. + , lowerHanded :: [SymbolId] + {- ^ The locals of the function being lowered whose values may be handed + on by need. Such a local is suspended in a callable of its own. + -} + , lowerSuspended :: [(SymbolId, (ResolvedName, Type))] + {- ^ The names whose value is a suspended computation: for each, the + callable that computes it and remembers it, with its type. A read of + the name is a call of the callable. + -} + , lowerWorkers :: Map SymbolId (ResolvedName, Type) + {- ^ For each method that a call has handed a suspended computation, + the function that receives suspended computations in its place, with + its type. + -} + , lowerPendingWorkers :: [SymbolId] + -- ^ The methods whose such function has been named and is not lowered yet. + } + +type Lower = StateT LowerState (Either [Diagnostic]) + +{- | Lower with every binding evaluated where it stands. The body of a +function is lowered this way once, to learn which of its locals are assigned +and which a closure captures, and so is the body of a callable. +-} +inPlace :: Lower a -> Lower a +inPlace = evaluatingWith Nothing + +{- | Lower with bindings evaluated by need, except those of the given locals, +which are assigned later or captured by a closure. +-} +byNeed :: [SymbolId] -> Lower a -> Lower a +byNeed settled = evaluatingWith (Just settled) + +evaluatingWith :: Maybe [SymbolId] -> Lower a -> Lower a +evaluatingWith settled action = do + previousSettled <- gets lowerSettled + previousDeferred <- gets lowerDeferred + previousSuspended <- gets lowerSuspended + modify (\current -> current {lowerSettled = settled, lowerDeferred = [], lowerSuspended = []}) + result <- action + modify + ( \current -> + current {lowerSettled = previousSettled, lowerDeferred = previousDeferred, lowerSuspended = previousSuspended} + ) + pure result + +-- | Lower with the given locals as the ones that may be handed on by need. +handing :: [SymbolId] -> Lower a -> Lower a +handing handed action = do + previous <- gets lowerHanded + modify (\current -> current {lowerHanded = handed}) + result <- action + modify (\current -> current {lowerHanded = previous}) + pure result + +-- | Lower with the given break target, and restore the previous one after. +targeting :: BreakTarget -> Lower a -> Lower a +targeting target action = do + previous <- gets lowerBreakTarget + modify (\current -> current {lowerBreakTarget = target}) + result <- action + modify (\current -> current {lowerBreakTarget = previous}) + pure result + +{- | Lower with the given lowerings of @continue@ and of @break@, and restore +the previous ones after. +-} +transferringWith :: [CoreStatement] -> [CoreStatement] -> Lower a -> Lower a +transferringWith continues leaves action = do + previousContinue <- gets lowerContinue + previousLeave <- gets lowerLeave + modify (\current -> current {lowerContinue = continues, lowerLeave = leaves}) + result <- action + modify (\current -> current {lowerContinue = previousContinue, lowerLeave = previousLeave}) + pure result + +{- | Lower a part of a Core loop in which the transfers of the source loop +are the transfers of that Core loop: its body, and its condition. +-} +inCoreLoop :: Lower a -> Lower a +inCoreLoop = transferringWith [CoreContinue] [CoreBreak] + +{- | The branching forms as they are lowered at the current place: a +@break value;@ in a block used as a value stores where one in the statement +around it would. +-} +branching :: (BranchLowering Lower -> Lower a) -> Lower a +branching use = gets lowerBreakTarget >>= use . branchLowering freshPatternSubject :: Lower ResolvedName freshPatternSubject = freshGenerated "$pattern" @@ -89,8 +266,8 @@ freshCoalesceSubject = freshGenerated "$coalesce" freshGenerated :: String -> Lower ResolvedName freshGenerated prefix = do - identifier <- get - put (identifier + 1) + identifier <- gets lowerNextSymbol + modify (\current -> current {lowerNextSymbol = identifier + 1}) pure (ResolvedName (SymbolId identifier) (Identifier (prefix ++ show identifier))) lowerTop :: Declaration ResolvedName Type -> Lower [CoreFunction] @@ -99,11 +276,26 @@ lowerTop TypeDeclaration {typeMembers = members} = mapM lowerDeclaration members -- concrete arguments. Lowering them here would leak unresolved type variables -- into Core and create one fake unspecialized native function. lowerTop TemplateTypeDeclaration {} = pure [] +-- An enum has no code: its members are constants of its underlying type, +-- which the type checker has put where they are used. +lowerTop EnumDeclaration {} = pure [] lowerTop function@FunctionDeclaration {} = (: []) <$> lowerDeclaration function lowerDeclaration :: Declaration ResolvedName Type -> Lower CoreFunction lowerDeclaration declaration@FunctionDeclaration {} = do - body <- lowerFunctionBlock returnType (declarationBody declaration) + -- The body is lowered twice. The first lowering evaluates every binding + -- where it stands and is kept only for what it shows: the locals that + -- are assigned after their binding and the locals a closure captures, + -- which must have their values in place. The second evaluates every + -- other binding by need. + inPlaceBody <- inPlace (lowerFunctionBlock returnType (declarationBody declaration)) + needs <- gets lowerNeeds + acts <- gets (expressionActs . lowerEffects) + body <- + handing (handedLocals needs acts (declarationBody declaration)) $ + byNeed + (assignedSymbols inPlaceBody ++ capturedSymbols inPlaceBody) + (lowerFunctionBlock returnType (declarationBody declaration)) pure ( CoreFunction (declarationName declaration) @@ -117,6 +309,7 @@ lowerDeclaration declaration@FunctionDeclaration {} = do returnType = lowerBoundaryType $ case declarationAnnotation declaration of FunctionType _ result -> result; value -> value lowerDeclaration TypeDeclaration {} = error "type declarations are lowered through lowerTop" lowerDeclaration TemplateTypeDeclaration {} = error "template declarations require specialization before Core lowering" +lowerDeclaration EnumDeclaration {} = error "enum declarations are lowered through lowerTop" {- | Where a @break value;@ stores its value: the result slot of the loop expression it leaves. A loop statement has no slot, and its body is lowered @@ -126,8 +319,8 @@ type BreakTarget = Maybe (ResolvedName, Type) {- | The lowering of this module as the branching forms of "Visual.XSharp.Desugarer.Branching" receive it. The target is where a -@break value;@ in a statement arm stores its value; value blocks have none, -because nothing may leave them early. +@break value;@ among their statements stores its value: the result slot of +the loop expression around them, if there is one. -} branchLowering :: BreakTarget -> BranchLowering Lower branchLowering target = @@ -144,26 +337,133 @@ branchLowering target = lowerBlock :: Block ResolvedName Type -> Lower [CoreStatement] lowerBlock = lowerBlockInto Nothing +{- | Lower the statements of a block in order. + +A statement an expression of which never completes ends the block: it is +lowered as the statements that run until control leaves, without the store, +the binding or the test that would have received the value, and the +statements after it, which are never reached, are not lowered. Nothing is +invented for a value that does not exist. +-} lowerBlockInto :: BreakTarget -> Block ResolvedName Type -> Lower [CoreStatement] -lowerBlockInto target (Block statements) = concat <$> mapM (lowerStatementInto target) statements +lowerBlockInto target = lowerBlockBefore target [] + +{- | Lower the statements of a block that the given statements follow. The +followers are not lowered here; they are what a binding at the end of the +block looks at to learn whether its value is needed at once. +-} +lowerBlockBefore :: BreakTarget -> [Statement ResolvedName Type] -> Block ResolvedName Type -> Lower [CoreStatement] +lowerBlockBefore target followers (Block statements) = go statements + where + go [] = pure [] + go (statement : remaining) = case neverCompletingExpression statement of + Just (before, expression) -> do + leading <- concat <$> mapM (lowerStatementInto target) before + (leading ++) <$> lowerNeverCompleting (branchLowering target) expression + Nothing -> do + -- Only a statement lowered from here knows what follows it. + -- Any other lowering of a statement sees no follower, and a + -- binding there is computed by need. + modify (\current -> current {lowerFollowing = remaining ++ followers}) + lowered <- lowerStatementInto target statement + modify (\current -> current {lowerFollowing = []}) + (lowered ++) <$> go remaining + +{- | The expression of a statement that is always evaluated and never +completes, with the statements that run before it as part of the same +statement. +-} +neverCompletingExpression :: + Statement ResolvedName Type -> Maybe ([Statement ResolvedName Type], Expression ResolvedName Type) +neverCompletingExpression statement = case statement of + BindingStatement _ _ _ _ _ value -> only value + AssignmentStatement _ _ _ value -> only value + CompoundAssignmentStatement _ _ _ _ value -> only value + DiscardStatement _ value -> only value + ReturnStatement _ (Just value) -> only value + IfStatement _ condition _ _ -> only condition + GuardStatement _ condition _ -> only condition + -- A condition that leaves by a break of its own loop is lowered with + -- the loop, which that break needs around it. + WhileStatement _ condition _ + | not (any isBreak (expressionTransfers condition)) -> only condition + ForStatement _ initializer (Just condition) _ _ + | doesNotComplete condition + , not (any isBreak (expressionTransfers condition)) -> + Just (maybe [] (: []) initializer, condition) + ExpressionStatement _ value _ -> only value + _ -> Nothing + where + only value = if doesNotComplete value then Just ([], value) else Nothing lowerFunctionBlock :: Type -> Block ResolvedName Type -> Lower [CoreStatement] -lowerFunctionBlock returnType (Block statements) = case reverse statements of +lowerFunctionBlock returnType (Block statements) = case statementsFromEnd of ExpressionStatement _ expression False : remaining - | returnType /= unitType -> do - prefix <- concat <$> mapM lowerStatement (reverse remaining) + | returnType /= unitType + , not (blockCannotComplete (Block (reverse remaining))) + , not (doesNotComplete expression) -> do + -- The final expression is the value the function returns, so + -- it follows the statements before it as a return would. + prefix <- lowerBlockBefore Nothing (take 1 statementsFromEnd) (Block (reverse remaining)) (valuePrefix, value) <- lowerExpression expression pure (prefix ++ valuePrefix ++ [CoreReturn value]) - _ -> concat <$> mapM lowerStatement statements + _ -> lowerBlock (Block statements) + where + statementsFromEnd = reverse statements lowerStatement :: Statement ResolvedName Type -> Lower [CoreStatement] lowerStatement = lowerStatementInto Nothing lowerStatementInto :: BreakTarget -> Statement ResolvedName Type -> Lower [CoreStatement] -lowerStatementInto target statement = case statement of +lowerStatementInto target statement = targeting target (lowerStatementWith target statement) + +-- The break target is already the current one. +lowerStatementWith :: BreakTarget -> Statement ResolvedName Type -> Lower [CoreStatement] +lowerStatementWith target statement = case statement of BindingStatement _ kind _ name valueType value -> do - (prefix, lowered) <- lowerExpression value - pure (prefix ++ [CoreBind (CoreBinding name (lowerBoundaryType valueType) (kind == MutableBinding) lowered)]) + settled <- gets lowerSettled + deferred <- gets lowerDeferred + following <- gets lowerFollowing + suspended <- gets lowerSuspended + handed <- gets lowerHanded + needs <- gets (argumentNeeds . lowerNeeds) + acts <- gets (expressionActs . lowerEffects) + let loweredType = lowerBoundaryType valueType + -- A binding must compute its value where it stands when it is + -- assigned later or captured, when its value is not a scalar, + -- and when its initializer has an effect. + mustStand inPlaceLocals bound boundType initializer = + resolvedSymbol bound `elem` inPlaceLocals + || not (scalarType (lowerBoundaryType boundType)) + || not (deferrableExpression initializer) + || acts initializer + -- A binding is certain to be evaluated when it must stand in + -- place or the statement after it is certain to need it. Only + -- such a binding makes the values its initializer reads needed. + certain inPlaceLocals bound boundType initializer later = + mustStand inPlaceLocals bound boundType initializer + || neededNext needs (certain inPlaceLocals) bound later + -- Nothing can tell where a value is computed that cannot fail + -- and that reads no value computed by need. + indifferent initializer = + not (worthDeferring initializer) + && not + ( any + ((`elem` map fst deferred ++ map fst suspended) . resolvedSymbol . fst) + (expressionNames initializer) + ) + case settled of + Just inPlaceLocals + | not (certain inPlaceLocals name valueType value following) + , not (indifferent value) -> + -- A value that may be handed to another function is + -- suspended where that function can reach it. + if resolvedSymbol name `elem` handed + then suspendBinding name loweredType value + else deferBinding inPlaceLocals name loweredType value + _ -> do + (prefix, lowered) <- lowerExpression value + pure (prefix ++ [CoreBind (CoreBinding name loweredType (kind == MutableBinding) lowered)]) AssignmentStatement _ name _ value -> do (prefix, lowered) <- lowerExpression value pure (prefix ++ [CoreAssign name lowered]) @@ -177,9 +477,11 @@ lowerStatementInto target statement = case statement of whenFalse <- maybe (pure []) (lowerBlockInto target) falseBlock pure (prefix ++ [CoreIf loweredCondition whenTrue whenFalse]) WhileStatement {} -> lowerLoop Nothing statement + -- A loop statement has no slot for break values, and a continue in its + -- body or condition is its own. DoWhileStatement _ body condition -> do - loweredBody <- lowerBlock body - loweredCondition <- lowerExpression condition + loweredBody <- inCoreLoop (lowerBlock body) + loweredCondition <- inCoreLoop (targeting Nothing (lowerExpression condition)) if null (fst loweredCondition) then pure [CoreDoWhile loweredBody (snd loweredCondition)] else do @@ -202,11 +504,12 @@ lowerStatementInto target statement = case statement of in pure [stepStatement name loweredType (CoreVariable name loweredType)] CompoundAssignmentStatement _ operator name valueType value -> lowerCompound operator name valueType value DiscardStatement _ value -> lowerDiscarded value - BreakStatement _ Nothing -> pure [CoreBreak] + BreakStatement _ Nothing -> gets lowerLeave BreakStatement spanValue (Just value) -> case target of Just (slot, _) -> do (prefix, lowered) <- lowerExpression value - pure (prefix ++ [CoreAssign slot lowered, CoreBreak]) + leave <- gets lowerLeave + pure (prefix ++ [CoreAssign slot lowered] ++ leave) Nothing -> lift ( Left @@ -218,7 +521,7 @@ lowerStatementInto target statement = case statement of "a value-carrying break reached Core lowering outside a loop expression" ] ) - ContinueStatement _ -> pure [CoreContinue] + ContinueStatement _ -> gets lowerContinue -- The block runs when the condition is false and never completes -- normally, so the statements after the guard follow the empty branch. GuardStatement _ condition block -> do @@ -234,23 +537,125 @@ lowerStatementInto target statement = case statement of | annotation == voidType -> fst <$> lowerMatch (branchLowering target) annotation subjects arms ExpressionStatement _ value _ -> lowerDiscarded value -{- | Lower a @while@ or @for@ loop whose body stores break values into the -given target. Loops nested in the body are statements of their own and are -lowered without a target. +{- | Bind a local whose value is computed by need. + +The binding computes nothing. It declares the local, a flag that says the +value has not been computed, and a copy of each variable the initializer +reads that is assigned somewhere in the function: the initializer means the +values those variables have here, whenever it is evaluated. The first read +of the local that is reached computes the value and sets the flag, and no +later read computes it again; see the lowering of a name. +-} +deferBinding :: + [SymbolId] -> ResolvedName -> Type -> Expression ResolvedName Type -> Lower [CoreStatement] +deferBinding inPlaceLocals name loweredType value = do + let changing = + nub + [ (seen, lowerBoundaryType seenType) + | (seen, seenType) <- expressionNames value + , resolvedSymbol seen `elem` inPlaceLocals + ] + copies <- mapM (const (freshGenerated "$seen")) changing + known <- freshGenerated "$known" + let renames = zip (map (resolvedSymbol . fst) changing) copies + rename seen = maybe seen id (lookup (resolvedSymbol seen) renames) + false = CoreLiteral (CoreBoolean False) boolType + modify + ( \current -> + current {lowerDeferred = (resolvedSymbol name, (known, renameNames rename value)) : lowerDeferred current} + ) + pure + ( [ CoreBind (CoreBinding copy seenType False (CoreVariable seen seenType)) + | ((seen, seenType), copy) <- zip changing copies + ] + ++ [ CoreBind (CoreBinding known boolType True false) + , CoreBind (CoreBinding name loweredType True (neutralValue loweredType)) + ] + ) + +{- | Lower a @while@ or @for@ loop whose body and condition store break +values into the given target. Loops nested in the body are statements of +their own and are lowered without a target. + +A @break@ in the condition leaves the loop. The statements of a condition +that has any stand at the top of the Core loop body, so the Core @break@ +there leaves the right loop without further work. + +A @continue@ in the condition abandons the rest of the condition and +evaluates the condition again; in a @for@ loop the update clause does not +run. In a @while@ loop the condition stands at the top of the Core loop +body, where the Core @continue@ does exactly that. In a @for@ loop the Core +@continue@ would run the update, so a condition that holds a @continue@ is +evaluated in a loop of its own, which the @continue@ repeats and which is +left once the condition has a result. A @break@ in such a condition leaves +that inner loop with the result still false, which leaves the @for@. + +A @continue@ in the update clause ends the update, and the condition is +tested next. Core has no such transfer inside an update region, so an update +clause that holds one is wrapped in a loop that runs once: its statements, +then a @break@. The @continue@ becomes a @break@ of that loop. + +A @break@ in the update clause leaves the loop, which the Core @break@ in an +update region does. In an update clause that is wrapped, it first sets a +flag that is bound before the loop, and the loop is left after the wrapper +when the flag is set. -} lowerLoop :: BreakTarget -> Statement ResolvedName Type -> Lower [CoreStatement] lowerLoop target loop = case loop of WhileStatement _ condition body -> do - loweredCondition <- lowerExpression condition - loweredBody <- lowerBlockInto target body + loweredCondition <- inLoop (lowerExpression condition) + loweredBody <- inCoreLoop (lowerBlockInto target body) pure [whileLoop loweredCondition loweredBody] ForStatement _ initializer condition updates body -> do + modify (\current -> current {lowerFollowing = []}) loweredInitializer <- maybe (pure []) lowerStatement initializer - loweredCondition <- maybe (pure ([], CoreLiteral (CoreBoolean True) boolType)) lowerExpression condition - loweredBody <- lowerBlockInto target body - loweredUpdates <- concat <$> mapM lowerStatement updates - pure (loweredInitializer ++ [forLoop loweredCondition loweredBody loweredUpdates]) + evaluatedCondition <- maybe (pure ([], alwaysTrue)) (inLoop . lowerExpression) condition + loweredCondition <- + if maybe False (any isContinue . expressionTransfers) condition + then repeatable evaluatedCondition + else pure evaluatedCondition + loweredBody <- inCoreLoop (lowerBlockInto target body) + let updateTransfers = concatMap statementTransfers updates + inUpdate = lowerBlockInto target (Block updates) + (bindings, updateRegion) <- + if any isContinue updateTransfers + then + if any isBreak updateTransfers + then do + leave <- freshGenerated "$leave" + let leaving = CoreVariable leave boolType + loweredUpdates <- + transferringWith [CoreBreak] [CoreAssign leave alwaysTrue, CoreBreak] inUpdate + pure + ( [CoreBind (CoreBinding leave boolType True alwaysFalse)] + , + [ CoreWhile alwaysTrue (loweredUpdates ++ [CoreBreak]) + , CoreIf leaving [CoreBreak] [] + ] + ) + else do + loweredUpdates <- transferringWith [CoreBreak] [CoreBreak] inUpdate + pure ([], [CoreWhile alwaysTrue (loweredUpdates ++ [CoreBreak])]) + else do + loweredUpdates <- inCoreLoop inUpdate + pure ([], loweredUpdates) + pure (loweredInitializer ++ bindings ++ [forLoop loweredCondition loweredBody updateRegion]) _ -> lowerStatement loop + where + inLoop = inCoreLoop . targeting target + alwaysTrue = CoreLiteral (CoreBoolean True) boolType + alwaysFalse = CoreLiteral (CoreBoolean False) boolType + -- The condition in a loop of its own, which a continue repeats. Its + -- result is a flag, because the condition may be numeric. + repeatable (prefix, value) = do + holds <- freshGenerated "$holds" + pure + ( + [ CoreBind (CoreBinding holds boolType True alwaysFalse) + , CoreWhile alwaysTrue (prefix ++ [CoreIf value [CoreAssign holds alwaysTrue] [], CoreBreak]) + ] + , CoreVariable holds boolType + ) {- | Lower an expression whose value is dropped. @@ -327,8 +732,56 @@ An expression without assignment or increment operands has no statements, and its Core expression is the direct translation of the source. -} lowerExpression :: Expression ResolvedName Type -> Lower Lowered -lowerExpression expression = case expression of - NameExpression _ name valueType -> pure ([], CoreVariable name (lowerBoundaryType valueType)) +lowerExpression expression = do + deferred <- gets lowerDeferred + needs <- gets (argumentNeeds . lowerNeeds) + case neededTwice needs (map fst deferred) expression of + [] -> lowerOperands expression + shared -> do + -- Each read of a value by need carries the computation of that + -- value. An expression that is certain to read one several + -- times computes it once, ahead of itself, and then reads it as + -- an ordinary local; otherwise a chain of such values would + -- grow with a power of its length. + forced <- concat <$> mapM (fmap fst . lowerOperands . snd) shared + modify + ( \current -> + current {lowerDeferred = filter ((`notElem` map (resolvedSymbol . fst) shared) . fst) deferred} + ) + (prefix, lowered) <- lowerOperands expression + modify (\current -> current {lowerDeferred = deferred}) + pure (forced ++ prefix, lowered) + +lowerOperands :: Expression ResolvedName Type -> Lower Lowered +lowerOperands expression = case expression of + -- A read of a local whose value is computed by need computes it if no + -- earlier read has, and reads it. + NameExpression _ name valueType -> do + deferred <- gets lowerDeferred + suspended <- gets lowerSuspended + let loweredType = lowerBoundaryType valueType + case lookup (resolvedSymbol name) deferred of + -- A read of a suspended value calls its callable, which + -- computes the value if nothing has, here or anywhere else. + Nothing + | Just (computation, computationType) <- lookup (resolvedSymbol name) suspended -> + pure ([], CoreApply (CoreVariable computation computationType) [] loweredType) + Nothing -> pure ([], CoreVariable name loweredType) + Just (known, initializer) -> do + (prefix, computed) <- lowerExpression initializer + pure + ( + [ CoreIf + (CoreVariable known boolType) + [] + ( prefix + ++ [ CoreAssign name computed + , CoreAssign known (CoreLiteral (CoreBoolean True) boolType) + ] + ) + ] + , CoreVariable name loweredType + ) LiteralExpression _ literal valueType -> let loweredType = lowerBoundaryType valueType in pure ([], CoreLiteral (lowerLiteral loweredType literal) loweredType) @@ -343,6 +796,36 @@ lowerExpression expression = case expression of "an unresolved member selector reached Core lowering" ] ) + MethodReferenceExpression spanValue _ _ _ -> + lift + ( Left + [ Diagnostic + DesugarerStage + Error + "VXD0003" + (Just spanValue) + "an unresolved member selector reached Core lowering" + ] + ) + -- A call of a runtime function: its identity, and then its arguments + -- in order. + CallExpression _ (NameExpression _ callee _) arguments valueType + | Just function <- runtimeFunctionOfName callee -> do + loweredArguments <- mapM lowerExpression arguments + (prefix, values) <- sequenceOperands loweredArguments + let identity = CoreLiteral (CoreInteger (runtimeFunctionIdentity function)) intType + pure (prefix, CorePrimitive CoreRuntimeCall (identity : values) (lowerBoundaryType valueType)) + -- A call that hands a value on by need goes to the function that + -- receives suspended computations. + CallExpression _ (NameExpression _ callee _) arguments valueType -> do + byNeedFlags <- handsOnByNeed callee arguments + case byNeedFlags of + Just flags -> lowerHandingCall callee flags arguments (lowerBoundaryType valueType) + Nothing -> do + loweredCallee <- lowerExpression (calleeOf expression) + loweredArguments <- mapM lowerExpression arguments + (prefix, calleeValue, argumentValues) <- sequenceAfter loweredCallee loweredArguments + pure (prefix, CoreApply calleeValue argumentValues (lowerBoundaryType valueType)) CallExpression _ callee arguments valueType -> do loweredCallee <- lowerExpression callee loweredArguments <- mapM lowerExpression arguments @@ -353,6 +836,16 @@ lowerExpression expression = case expression of (prefix, lowered) <- lowerExpression value pure (prefix, CorePrimitive (lowerUnary operator) [lowered] (lowerBoundaryType valueType)) BinaryExpression _ operator left right valueType + -- A right operand that never completes leaves when it is + -- evaluated, so the operator yields a value only on the path that + -- skips it, and that value is known: false for `&&`, true for `||`. + | operator `elem` [LogicalAnd, LogicalOr] + , doesNotComplete right -> do + (leftPrefix, loweredLeft) <- lowerExpression left + leaving <- branching (`lowerNeverCompleting` right) + let conjunction = operator == LogicalAnd + skip = if conjunction then CoreIf loweredLeft leaving [] else CoreIf loweredLeft [] leaving + pure (leftPrefix ++ [skip], CoreLiteral (CoreBoolean (not conjunction)) boolType) | operator `elem` [LogicalAnd, LogicalOr] -> do (leftPrefix, loweredLeft) <- lowerExpression left loweredRight <- lowerExpression right @@ -378,12 +871,33 @@ lowerExpression expression = case expression of subjectRead = CoreVariable subjectName subjectType predicate = lowerPattern subjectRead subjectType patternValue pure (prefix, CoreLet subjectName subjectType loweredSubject predicate boolType) + -- A choice that never yields a value is reached here only where the + -- place that reads it survives its statements, which is the condition + -- of a do/while loop; a statement whose expression never completes is + -- lowered by 'lowerBlockInto' without that place. The condition is + -- never tested, so the constant is not a value of the program. + ConditionalExpression _ _ _ _ valueType + | valueType == voidType -> do + statements <- branching (`lowerNeverCompleting` expression) + pure (statements, CoreLiteral (CoreBoolean False) boolType) + MatchExpression _ _ arms valueType + | valueType == voidType && not (null arms) -> do + statements <- branching (`lowerNeverCompleting` expression) + pure (statements, CoreLiteral (CoreBoolean False) boolType) + -- A branch that does not complete has no value to lower. + ConditionalExpression _ condition first second valueType + | any doesNotComplete [first, second] -> do + loweredCondition <- lowerExpression condition + branching (\lowering -> lowerSelection lowering valueType loweredCondition first second) ConditionalExpression _ condition first second valueType -> do (conditionPrefix, loweredCondition) <- lowerExpression condition loweredFirst <- lowerExpression first loweredSecond <- lowerExpression second let loweredType = lowerBoundaryType valueType - if null (fst loweredFirst) && null (fst loweredSecond) + -- A string is an object that is owned: it is selected into a + -- variable, which the rules of ownership for an assignment cover, + -- and not by an expression that would hold two objects at once. + if null (fst loweredFirst) && null (fst loweredSecond) && loweredType /= stringType then pure ( conditionPrefix @@ -395,6 +909,22 @@ lowerExpression expression = case expression of pure (conditionPrefix ++ statements, value) -- The left operand is both the test and the first result. Binding it once -- keeps its effects single even though it is read twice. + -- A fallback that never completes leaves when the left value is false + -- in Boolean context, so the operator yields the left value or nothing. + CoalesceExpression _ left fallback valueType + | doesNotComplete fallback -> do + (leftPrefix, loweredLeft) <- lowerExpression left + leaving <- branching (`lowerNeverCompleting` fallback) + subjectName <- freshCoalesceSubject + let loweredType = lowerBoundaryType valueType + subjectRead = CoreVariable subjectName loweredType + pure + ( leftPrefix + ++ [ CoreBind (CoreBinding subjectName loweredType False loweredLeft) + , CoreIf subjectRead [] leaving + ] + , subjectRead + ) CoalesceExpression _ left fallback valueType -> do (leftPrefix, loweredLeft) <- lowerExpression left loweredFallback <- lowerExpression fallback @@ -449,6 +979,13 @@ lowerExpression expression = case expression of -- A loop expression runs its loop as statements. Each break that leaves -- it stores its value into the result slot first; the type checker has -- established that the loop cannot end any other way. + -- A loop that no break leaves never yields a value; like a choice + -- that never does, it is reached here only as the condition of a + -- do/while loop. + LoopExpression _ loop valueType + | valueType == voidType -> do + statements <- lowerLoop Nothing loop + pure (statements, CoreLiteral (CoreBoolean False) boolType) LoopExpression _ loop valueType -> do result <- freshGenerated "$loop" let loweredType = lowerBoundaryType valueType @@ -457,12 +994,23 @@ lowerExpression expression = case expression of ( CoreBind (CoreBinding result loweredType True (neutralValue loweredType)) : statements , CoreVariable result loweredType ) - BlockExpression _ block _ -> lowerValueBlock (branchLowering Nothing) block - MatchExpression _ subjects arms valueType -> lowerMatch (branchLowering Nothing) valueType subjects arms + BlockExpression _ block _ -> branching (`lowerValueBlock` block) + MatchExpression _ subjects arms valueType -> branching (\lowering -> lowerMatch lowering valueType subjects arms) CallableExpression _ explicit captures parameters body valueType -> do let loweredParameters = [(parameterName parameter, lowerBoundaryType (parameterAnnotation parameter)) | parameter <- parameters] - loweredBody <- lowerCallableBody body + -- A callable is a function of its own: no loop of the function + -- that creates it is around its body. + -- Its bindings are computed by need like those of a method, from + -- its own assigned and captured locals. While the function around + -- it is lowered in place, to learn its locals, so is the callable. + settled <- gets lowerSettled + let lowerBody = inCoreLoop (targeting Nothing (lowerCallableBody body)) + loweredBody <- case settled of + Nothing -> inPlace lowerBody + Just _ -> do + inPlaceBody <- inPlace lowerBody + byNeed (assignedSymbols inPlaceBody ++ capturedSymbols inPlaceBody) lowerBody (prefix, loweredCaptures) <- lowerCaptures captures let returnType = case valueType of FunctionType _ result -> lowerBoundaryType result @@ -474,6 +1022,264 @@ lowerExpression expression = case expression of then (prefix, closure loweredCaptures) else ([], closure (discoverImplicitCaptures loweredParameters loweredBody)) +-- | The callee of a call expression. +calleeOf :: Expression ResolvedName Type -> Expression ResolvedName Type +calleeOf expression = case expression of + CallExpression _ callee _ _ -> callee + _ -> expression + +-- | The type of a suspended computation of a value of the given lowered type. +suspendedType :: Type -> Type +suspendedType = FunctionType [] + +{- | Whether an argument is handed on by need when its parameter allows it: +computing it later gives what computing it now would give, and it either +may fail or run without end, or reads a value that is itself computed by +need. Any other argument is computed at the call, where it costs least. +-} +handedOn :: Expression ResolvedName Type -> Lower Bool +handedOn argument = do + deferred <- gets (map fst . lowerDeferred) + suspended <- gets (map fst . lowerSuspended) + acts <- gets (expressionActs . lowerEffects) + pure + ( deferrableExpression argument + && not (acts argument) + && ( worthDeferring argument + || any ((`elem` deferred ++ suspended) . resolvedSymbol . fst) (expressionNames argument) + ) + ) + +{- | Whether a call hands a value on by need, and if so which parameters of +the method are passed by need. + +Only a call lowered for its own sake does: while a body is lowered in place, +to learn its locals, every argument is computed at the call. +-} +handsOnByNeed :: ResolvedName -> [Expression ResolvedName Type] -> Lower (Maybe [Bool]) +handsOnByNeed callee arguments = do + settled <- gets lowerSettled + needs <- gets lowerNeeds + case (settled, Map.lookup (resolvedSymbol callee) needs) of + (Just _, Just flags) | length flags == length arguments -> do + handed <- mapM handedOn arguments + pure (if or (zipWith (&&) flags handed) then Just flags else Nothing) + _ -> pure Nothing + +{- | Lower a call of the function that receives suspended computations in +place of a method. + +An argument of a parameter that is passed by need becomes a callable. One +that is handed on by need is suspended: nothing of it is computed here. One +that is already a suspended value is passed as it is, so a value handed +through several methods is still computed once. Any other argument is +computed here, as it always was, and its callable returns that value. +-} +lowerHandingCall :: ResolvedName -> [Bool] -> [Expression ResolvedName Type] -> Type -> Lower Lowered +lowerHandingCall callee flags arguments resultType = do + (worker, workerType) <- workerOf callee flags + operands <- zipWithM operand flags arguments + (prefix, values) <- sequenceOperands operands + pure (prefix, CoreApply (CoreVariable worker workerType) values resultType) + where + operand False argument = lowerExpression argument + operand True argument = do + suspended <- gets lowerSuspended + lazily <- handedOn argument + let loweredType = lowerBoundaryType (expressionAnnotation argument) + callableType = suspendedType loweredType + case argument of + NameExpression _ name _ + | Just (computation, computationType) <- lookup (resolvedSymbol name) suspended -> + pure ([], CoreVariable computation computationType) + _ + | lazily -> do + (prefix, computation) <- suspend loweredType argument + pure (prefix, CoreVariable computation callableType) + | otherwise -> do + (prefix, value) <- lowerExpression argument + kept <- freshGenerated "$value" + computation <- freshGenerated "$computed" + let keptValue = CoreVariable kept loweredType + closure = + CoreClosure + [CoreCapture StrongCapture kept loweredType keptValue] + [] + loweredType + [CoreReturn keptValue] + callableType + pure + ( prefix + ++ [ CoreBind (CoreBinding kept loweredType False value) + , CoreBind (CoreBinding computation callableType False closure) + ] + , CoreVariable computation callableType + ) + +{- | Suspend an expression: bind a callable that computes it when it is first +called and remembers the result, and return the name of the callable. + +The callable takes the values of the variables the expression reads when it +is created, so the expression means what its variables hold here, whenever +it is computed. A suspended value the expression reads is taken as its +callable, and computed only if this one is. The functions of the module are +not taken: a callable calls them as any function does. +-} +suspend :: Type -> Expression ResolvedName Type -> Lower ([CoreStatement], ResolvedName) +suspend loweredType value = do + deferred <- gets lowerDeferred + -- A value in a flag and a slot of this frame cannot be computed from + -- another function. The locals an argument reads are suspended instead, + -- see 'handedLocals'; one that is not is computed before it is read. + let framed = + foldr + (\seen others -> if resolvedSymbol (fst seen) `elem` map (resolvedSymbol . fst) others then others else seen : others) + [] + [seen | seen <- expressionNames value, resolvedSymbol (fst seen) `elem` map fst deferred] + forced <- + concat <$> mapM (\(name, nameType) -> fst <$> lowerOperands (NameExpression (expressionSourceSpan value) name nameType)) framed + modify (\current -> current {lowerDeferred = filter ((`notElem` map (resolvedSymbol . fst) framed) . fst) deferred}) + following <- gets lowerFollowing + modify (\current -> current {lowerFollowing = []}) + (bodyPrefix, result) <- inCoreLoop (targeting Nothing (lowerExpression value)) + modify (\current -> current {lowerDeferred = deferred, lowerFollowing = following}) + methods <- gets (Map.keys . lowerMethods) + workers <- gets (map (resolvedSymbol . fst) . Map.elems . lowerWorkers) + computation <- freshGenerated "$suspended" + let body = bodyPrefix ++ [CoreReturn result] + callableType = suspendedType loweredType + captures = + [ capture + | capture <- discoverImplicitCaptures [] body + , resolvedSymbol (coreCaptureName capture) `notElem` methods ++ workers + ] + closure = CoreClosure captures [] loweredType body callableType + pure + ( forced ++ [CoreBind (CoreBinding computation callableType False (CorePrimitive CoreMemoize [closure] callableType))] + , computation + ) + +{- | Bind a local whose value may be handed to another function by need: its +value is a suspended computation, and every read of the local calls it. +-} +suspendBinding :: ResolvedName -> Type -> Expression ResolvedName Type -> Lower [CoreStatement] +suspendBinding name loweredType value = do + (prefix, computation) <- suspend loweredType value + modify + ( \current -> + current {lowerSuspended = (resolvedSymbol name, (computation, suspendedType loweredType)) : lowerSuspended current} + ) + pure prefix + +{- | The function that receives suspended computations in place of the given +method, named on the first call that asks for it and lowered later. +-} +workerOf :: ResolvedName -> [Bool] -> Lower (ResolvedName, Type) +workerOf method flags = do + workers <- gets lowerWorkers + case Map.lookup (resolvedSymbol method) workers of + Just worker -> pure worker + Nothing -> do + declaration <- gets ((Map.! resolvedSymbol method) . lowerMethods) + identifier <- gets lowerNextSymbol + let spelling = identifierText (resolvedSpelling method) ++ "$need" ++ show identifier + worker = ResolvedName (SymbolId identifier) (Identifier spelling) + parameterTypes = + [ if flag then suspendedType parameterType else parameterType + | (flag, parameter) <- zip flags (methodParameters declaration) + , let parameterType = lowerBoundaryType (parameterAnnotation parameter) + ] + workerType = FunctionType parameterTypes (declarationResultType declaration) + modify + ( \current -> + current + { lowerNextSymbol = identifier + 1 + , lowerWorkers = Map.insert (resolvedSymbol method) (worker, workerType) (lowerWorkers current) + , lowerPendingWorkers = lowerPendingWorkers current ++ [resolvedSymbol method] + } + ) + pure (worker, workerType) + +-- | The lowered result type of a method. +declarationResultType :: Declaration ResolvedName Type -> Type +declarationResultType declaration = + lowerBoundaryType $ case declarationAnnotation declaration of FunctionType _ result -> result; value -> value + +-- | Lower the functions that were asked for until no call asks for another. +lowerWorkersUntilNone :: Lower [(CoreFunction, (Int, FilePath))] +lowerWorkersUntilNone = do + pending <- gets lowerPendingWorkers + case pending of + [] -> pure [] + method : later -> do + modify (\current -> current {lowerPendingWorkers = later}) + worker <- lowerWorker method + (worker :) <$> lowerWorkersUntilNone + +{- | Lower the function that receives suspended computations in place of a +method. + +It is the method's body once more, with the parameters that are passed by +need as suspended values: a read of one calls its callable. A parameter that +a closure of the body captures is computed when the function is entered, +because a closure takes the values of its captures when it is created. +Everything the function defines takes a fresh symbol, since the method has +been lowered from the same statements. +-} +lowerWorker :: SymbolId -> Lower (CoreFunction, (Int, FilePath)) +lowerWorker method = do + declaration <- gets ((Map.! method) . lowerMethods) + flags <- gets ((Map.! method) . lowerNeeds) + (worker, _) <- gets ((Map.! method) . lowerWorkers) + needs <- gets lowerNeeds + effects <- gets lowerEffects + let returnType = declarationResultType declaration + body = methodBody declaration + inPlaceBody <- inPlace (lowerFunctionBlock returnType body) + let captured = capturedSymbols inPlaceBody + received <- + forM (zip flags (methodParameters declaration)) $ \(flag, parameter) -> do + let name = parameterName parameter + parameterType = lowerBoundaryType (parameterAnnotation parameter) + if flag + then do + computation <- freshGenerated "$argument" + pure (name, parameterType, Just computation) + else pure (name, parameterType, Nothing) + let entered = + [ CoreBind (CoreBinding name valueType False (CoreApply (CoreVariable computation (suspendedType valueType)) [] valueType)) + | (name, valueType, Just computation) <- received + , resolvedSymbol name `elem` captured + ] + suspendedParameters = + [ (resolvedSymbol name, (computation, suspendedType valueType)) + | (name, valueType, Just computation) <- received + , resolvedSymbol name `notElem` captured + ] + lowered <- + handing (handedLocals needs (expressionActs effects) body) $ + byNeed (assignedSymbols inPlaceBody ++ captured) $ do + modify (\current -> current {lowerSuspended = suspendedParameters}) + lowerFunctionBlock returnType body + let statements = entered ++ lowered + own = nub ([name | (name, _, Nothing) <- received] ++ definedNames statements) + replacements <- Map.fromList <$> mapM (\name -> (,) (resolvedSymbol name) <$> freshLike name) own + let parameters = + [ case suspendedBy of + Just computation -> (computation, suspendedType valueType) + Nothing -> (Map.findWithDefault name (resolvedSymbol name) replacements, valueType) + | (name, valueType, suspendedBy) <- received + ] + pure + ( CoreFunction worker parameters returnType (renameSymbols replacements statements) + , (symbolIdValue (resolvedSymbol worker), portableSpanSource declaration) + ) + where + freshLike name = do + identifier <- gets lowerNextSymbol + modify (\current -> current {lowerNextSymbol = identifier + 1}) + pure (ResolvedName (SymbolId identifier) (resolvedSpelling name)) + -- Capture initializers are evaluated in order when the closure is created, -- like the operands of any other expression. lowerCaptures :: [Capture ResolvedName Type] -> Lower ([CoreStatement], [CoreCapture]) @@ -492,9 +1298,11 @@ lowerCaptures captures = do lowerCallableBody :: CallableBody ResolvedName Type -> Lower [CoreStatement] lowerCallableBody body = case body of - CallableExpressionBody expression -> do - (prefix, value) <- lowerExpression expression - pure (prefix ++ [CoreReturn value]) + CallableExpressionBody expression + | doesNotComplete expression -> branching (`lowerNeverCompleting` expression) + | otherwise -> do + (prefix, value) <- lowerExpression expression + pure (prefix ++ [CoreReturn value]) CallableBlockBody block -> let returnType = maybe unitType id (callableFinalType block) in lowerFunctionBlock returnType block @@ -509,6 +1317,7 @@ expressionAnnotation expression = case expression of NameExpression _ _ valueType -> valueType LiteralExpression _ _ valueType -> valueType MemberAccessExpression _ _ _ valueType -> valueType + MethodReferenceExpression _ _ _ valueType -> valueType CallExpression _ _ _ valueType -> valueType UnaryExpression _ _ _ valueType -> valueType BinaryExpression _ _ _ _ valueType -> valueType @@ -584,58 +1393,6 @@ joinWith separator (value : rest) = value ++ separator ++ joinWith separator res -- analysis is deliberately performed after desugaring so syntactic sugar -- cannot hide a read. Locals introduced by the callable and its parameters -- are removed before stable first-use ordering is assigned. -discoverImplicitCaptures :: [(ResolvedName, Type)] -> [CoreStatement] -> [CoreCapture] -discoverImplicitCaptures parameters statements = - let bound = map (resolvedSymbol . fst) parameters ++ localSymbols statements - free = filter (\(name, _) -> resolvedSymbol name `notElem` bound) (statementReads statements) - in [CoreCapture StrongCapture name valueType (CoreVariable name valueType) | (name, valueType) <- uniqueReads free] - -localSymbols :: [CoreStatement] -> [SymbolId] -localSymbols = concatMap collect - where - collect statement = case statement of - CoreBind binding -> [resolvedSymbol (coreBindingName binding)] - CoreIf _ yes no -> localSymbols yes ++ localSymbols no - CoreWhile _ body -> localSymbols body - CoreDoWhile body _ -> localSymbols body - CoreFor _ body update -> localSymbols body ++ localSymbols update - _ -> [] - -statementReads :: [CoreStatement] -> [(ResolvedName, Type)] -statementReads = concatMap collect - where - collect statement = case statement of - CoreBind binding -> expressionReads (coreBindingValue binding) - CoreAssign _ value -> expressionReads value - CoreReturn value -> expressionReads value - CoreIf condition yes no -> expressionReads condition ++ statementReads yes ++ statementReads no - CoreWhile condition body -> expressionReads condition ++ statementReads body - CoreDoWhile body condition -> statementReads body ++ expressionReads condition - CoreFor condition body update -> - expressionReads condition ++ statementReads body ++ statementReads update - CoreEvaluate value -> expressionReads value - CoreBreak -> [] - CoreContinue -> [] - -expressionReads :: CoreExpression -> [(ResolvedName, Type)] -expressionReads expression = case expression of - CoreVariable name valueType -> [(name, valueType)] - CoreLiteral _ _ -> [] - CoreApply callee arguments _ -> expressionReads callee ++ concatMap expressionReads arguments - CorePrimitive _ arguments _ -> concatMap expressionReads arguments - CoreLet name _ value body _ -> - expressionReads value ++ filter ((/= resolvedSymbol name) . resolvedSymbol . fst) (expressionReads body) - CoreConditional condition whenTrue whenFalse _ -> - expressionReads condition ++ expressionReads whenTrue ++ expressionReads whenFalse - CoreClosure captures _ _ body _ -> concatMap (expressionReads . coreCaptureValue) captures ++ statementReads body - -uniqueReads :: [(ResolvedName, Type)] -> [(ResolvedName, Type)] -uniqueReads = foldl append [] - where - append output value@(name, _) - | any ((== resolvedSymbol name) . resolvedSymbol . fst) output = output - | otherwise = output ++ [value] - lowerLiteral :: Type -> Literal -> CoreLiteral lowerLiteral valueType literal = case literal of IntegerLiteral value @@ -655,6 +1412,10 @@ lowerBoundaryType valueType | valueType == voidType = unitType | FunctionType parameters result <- valueType = FunctionType (map lowerBoundaryType parameters) (lowerBoundaryType result) + -- A value of an enum is a value of its underlying integer type. An enum + -- must not cross into Core as a named type: named types are references + -- there, and an enum is not one. + | Just underlying <- enumUnderlyingType valueType = underlying | NamedType name arguments <- valueType = NamedType name (map lowerTemplateArgument arguments) | otherwise = valueType @@ -662,108 +1423,3 @@ lowerTemplateArgument :: TemplateArgument -> TemplateArgument lowerTemplateArgument argument = case argument of TypeTemplateArgument valueType -> TypeTemplateArgument (lowerBoundaryType valueType) ValueTemplateArgument value -> ValueTemplateArgument value - --- Generated Core bindings must never collide with source symbols. Gathering --- the complete typed tree once is cheaper and more robust than reserving a --- magic numeric range or deriving identities from source positions. -syntaxSymbolIds :: SyntaxTree ResolvedName Type -> [Int] -syntaxSymbolIds (SyntaxTree _ declarations) = concatMap declarationSymbolIds declarations - -declarationSymbolIds :: Declaration ResolvedName Type -> [Int] -declarationSymbolIds declaration = - symbolValue (declarationName declaration) - : case declaration of - TypeDeclaration {typeMembers = members} -> concatMap declarationSymbolIds members - TemplateTypeDeclaration {declarationTemplateParameters = parameters, typeMembers = members} -> - map (symbolValue . templateParameterName) parameters ++ concatMap declarationSymbolIds members - FunctionDeclaration {declarationParameters = parameters, declarationBody = body} -> - map (symbolValue . parameterName) parameters ++ blockSymbolIds body - -blockSymbolIds :: Block ResolvedName Type -> [Int] -blockSymbolIds (Block statements) = concatMap statementIds statements - -statementIds :: Statement ResolvedName Type -> [Int] -statementIds statement = case statement of - BindingStatement _ _ _ name _ value -> symbolValue name : expressionIds value - AssignmentStatement _ name _ value -> symbolValue name : expressionIds value - ReturnStatement _ value -> maybe [] expressionIds value - IfStatement _ condition yes no -> expressionIds condition ++ blockSymbolIds yes ++ maybe [] blockSymbolIds no - WhileStatement _ condition body -> expressionIds condition ++ blockSymbolIds body - DoWhileStatement _ body condition -> blockSymbolIds body ++ expressionIds condition - ForStatement _ initializer condition updates body -> - maybe [] statementIds initializer - ++ maybe [] expressionIds condition - ++ concatMap statementIds updates - ++ blockSymbolIds body - ForEachStatement _ _ _ name _ source body -> - symbolValue name : expressionIds source ++ blockSymbolIds body - IncrementStatement _ name _ -> [symbolValue name] - CompoundAssignmentStatement _ _ name _ value -> symbolValue name : expressionIds value - DiscardStatement _ value -> expressionIds value - BreakStatement _ value -> maybe [] expressionIds value - ContinueStatement {} -> [] - GuardStatement _ condition block -> expressionIds condition ++ blockSymbolIds block - BlockStatement _ block -> blockSymbolIds block - ExpressionStatement _ value _ -> expressionIds value - -expressionIds :: Expression ResolvedName Type -> [Int] -expressionIds expression = case expression of - NameExpression _ name _ -> [symbolValue name] - LiteralExpression {} -> [] - MemberAccessExpression _ receiver _ _ -> expressionIds receiver - CallExpression _ callee arguments _ -> expressionIds callee ++ concatMap expressionIds arguments - UnaryExpression _ _ value _ -> expressionIds value - BinaryExpression _ _ left right _ -> expressionIds left ++ expressionIds right - IsPatternExpression _ subject _ _ -> expressionIds subject - ConditionalExpression _ condition first second _ -> concatMap expressionIds [condition, first, second] - CoalesceExpression _ left fallback _ -> expressionIds left ++ expressionIds fallback - AssignmentExpression _ _ name value _ -> symbolValue name : expressionIds value - IncrementExpression _ _ name _ -> [symbolValue name] - LoopExpression _ loop _ -> statementIds loop - BlockExpression _ block _ -> blockSymbolIds block - MatchExpression _ subjects arms _ -> - [ symbolValue name - | arm <- arms - , Just name <- map matchPatternBinding (matchArmPatterns arm) - ] - ++ concatMap expressionIds (subjects ++ concatMap matchArmExpressions arms) - CallableExpression _ _ captures parameters body _ -> - map (symbolValue . captureName) captures - ++ concatMap (maybe [] expressionIds . captureInitializer) captures - ++ map (symbolValue . parameterName) parameters - ++ callableBodyIds body - -callableBodyIds :: CallableBody ResolvedName Type -> [Int] -callableBodyIds body = case body of - CallableExpressionBody expression -> expressionIds expression - CallableBlockBody block -> blockSymbolIds block - -symbolValue :: ResolvedName -> Int -symbolValue = symbolIdValue . resolvedSymbol -lowerUnary :: UnaryOperator -> CorePrimitive -lowerUnary UnaryNegate = CoreNegate -lowerUnary LogicalNot = CoreLogicalNot -lowerUnary BitwiseNot = CoreBitwiseNot -lowerUnary UnaryPlus = CoreAdd -lowerBinary :: BinaryOperator -> CorePrimitive -lowerBinary operator = case operator of - Add -> CoreAdd - Subtract -> CoreSubtract - Multiply -> CoreMultiply - Divide -> CoreDivide - FloorDivide -> CoreFloorDivide - Remainder -> CoreRemainder - Power -> CorePower - ShiftLeft -> CoreShiftLeft - ShiftRight -> CoreShiftRight - BitwiseAnd -> CoreBitwiseAnd - BitwiseXor -> CoreBitwiseXor - BitwiseOr -> CoreBitwiseOr - LessThan -> CoreLessThan - LessEqual -> CoreLessEqual - GreaterThan -> CoreGreaterThan - GreaterEqual -> CoreGreaterEqual - Equal -> CoreEqual - NotEqual -> CoreNotEqual - LogicalAnd -> CoreLogicalAnd - LogicalOr -> CoreLogicalOr diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Branching.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Branching.hs index 609353a5..d400b172 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Branching.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Branching.hs @@ -16,10 +16,12 @@ module Visual.XSharp.Desugarer.Branching ( BranchLowering (..) , lowerValueBlock , lowerMatch - , maximumNestedArms + , lowerSelection + , lowerNeverCompleting ) where import Visual.XSharp.AST +import Visual.XSharp.Completion import Visual.XSharp.Core import Visual.XSharp.Desugarer.Sequencing @@ -43,11 +45,90 @@ data BranchLowering lower = BranchLowering -- ^ The Core literal of a source literal of the given Core type. } +{- | Lower an expression that never yields a value, as the statements that +run until control leaves. + +Such an expression has no Core value, so none is made up for it: there is no +result slot, no placeholder and nothing for a join to read. The operands +that are evaluated before the one that never completes are evaluated for +their effects, in order; the operands after it are never reached and are not +lowered. A choice every branch of which leaves is a conditional statement +over the statements of its branches. +-} +lowerNeverCompleting :: + (Monad lower) => BranchLowering lower -> Expression ResolvedName Type -> lower [CoreStatement] +lowerNeverCompleting lowering expression = case expression of + BlockExpression _ (Block statements) _ -> case reverse statements of + ExpressionStatement _ value False : before + | not (blockCannotComplete (Block (reverse before))) -> do + leading <- branchStatements lowering (reverse before) + final <- lowerNeverCompleting lowering value + pure (leading ++ final) + _ -> branchStatements lowering statements + ConditionalExpression _ condition first second _ + | doesNotComplete condition -> lowerNeverCompleting lowering condition + | otherwise -> do + (prefix, test) <- branchExpression lowering condition + whenTrue <- lowerNeverCompleting lowering first + whenFalse <- lowerNeverCompleting lowering second + pure (prefix ++ [CoreIf test whenTrue whenFalse]) + MatchExpression _ subjects arms _ + | not (any doesNotComplete subjects) -> fst <$> lowerMatch lowering voidType subjects arms + -- A loop that no break leaves: the loop statement itself. + LoopExpression _ loop _ -> branchStatements lowering [loop] + _ -> case neverCompletingOperand expression of + Just (before, operand) -> do + -- Reading a name or a literal has no effect to keep. + evaluated <- mapM (branchDiscarded lowering) (filter hasEvaluation before) + final <- lowerNeverCompleting lowering operand + pure (concat evaluated ++ final) + -- Not reached for an expression that never completes; an + -- expression that does is evaluated for its effects. + Nothing -> branchDiscarded lowering expression + where + hasEvaluation operand = case operand of + NameExpression {} -> False + LiteralExpression {} -> False + _ -> True + +{- | Lower a two-way choice exactly one of whose branches completes. + +The result slot is assigned only on the branch that completes. The other +branch leaves with @return@, @break@ or @continue@, which are ordinary Core +statements wherever they stand; it is lowered as its statements alone and +never reaches the read of the slot. +-} +lowerSelection :: + (Monad lower) => + BranchLowering lower -> + Type -> + Lowered -> + Expression ResolvedName Type -> + Expression ResolvedName Type -> + lower Lowered +lowerSelection lowering valueType (conditionPrefix, condition) first second = do + result <- branchFresh lowering "$selected" + let resultType = branchType lowering valueType + branch value + | doesNotComplete value = lowerNeverCompleting lowering value + | otherwise = do + (prefix, lowered) <- branchExpression lowering value + pure (prefix ++ [CoreAssign result lowered]) + whenTrue <- branch first + whenFalse <- branch second + pure + ( conditionPrefix + ++ [ CoreBind (CoreBinding result resultType True (neutralValue resultType)) + , CoreIf condition whenTrue whenFalse + ] + , CoreVariable result resultType + ) + {- | Lower a block used as a value. -The leading statements run in order; the value is the final expression. The -type checker has established that the block ends with one and that nothing -in it leaves the block early. +The leading statements run in order; the value is the final expression. A +block that does not complete has no value; the forms that hold one lower it +through 'lowerNeverCompleting' and never ask for its value. -} lowerValueBlock :: (Monad lower) => BranchLowering lower -> Block ResolvedName Type -> lower Lowered lowerValueBlock lowering (Block statements) = case reverse statements of @@ -62,18 +143,19 @@ lowerValueBlock lowering (Block statements) = case reverse statements of {- | Lower a @match@. The subjects are evaluated once, left to right, and each is bound to a local -so that every arm tests the same values. The arms then nest: an arm that does -not accept continues with the arms after it, in its else branch. +so that every arm tests the same values. The names the arms bind are bound +next, each to its subject: binding a scalar that was already evaluated has no +effect of its own, and no arm can see a name of another arm. The arms then +form one chain of conditionals, each arm in the false branch of the one +before it. That is the shape of an @else if@ chain, which every stage walks +in a loop, so the lowering of a match is as deep as one arm whatever the +number of arms. A match whose type is @void@ is the statement form: its arm bodies run for -their effects and its value is the unit literal. Otherwise every arm stores -its value into a result slot, which the type checker guarantees is assigned -on every path, because some arm always accepts. - -The nesting is bounded. A match with more than 'maximumNestedArms' arms is -lowered in groups of that size: the groups follow each other in one statement -sequence, a Boolean slot records that an arm was taken, and every group after -the first runs only while the slot is still false. +their effects and its value is the unit literal. Otherwise every arm that +completes stores its value into a result slot, and an arm whose block leaves +stores nothing. The type checker guarantees that the slot is assigned on +every path that reaches its read, because some arm always accepts. -} lowerMatch :: (Monad lower) => @@ -92,103 +174,62 @@ lowerMatch lowering annotation subjects arms = do | (name, valueType, value) <- zip3 subjectNames subjectTypes subjectValues ] subjectReads = zipWith CoreVariable subjectNames subjectTypes + patternBindings = + [ CoreBind (CoreBinding name (expressionType subject) True subject) + | arm <- arms + , (MatchTypePattern _ _ (Just name) _, subject) <- zip (matchArmPatterns arm) subjectReads + ] + prefix = subjectPrefix ++ subjectBindings ++ patternBindings if annotation == voidType then do - chain <- lowerArmGroups lowering Nothing subjectReads arms - pure (subjectPrefix ++ subjectBindings ++ chain, CoreLiteral CoreUnit unitType) + chain <- lowerArms lowering Nothing subjectReads arms + pure (prefix ++ chain, CoreLiteral CoreUnit unitType) else do result <- branchFresh lowering "$matched" let resultType = branchType lowering annotation - chain <- lowerArmGroups lowering (Just result) subjectReads arms + chain <- lowerArms lowering (Just result) subjectReads arms pure - ( subjectPrefix - ++ subjectBindings - ++ CoreBind (CoreBinding result resultType True (neutralValue resultType)) - : chain + ( prefix ++ CoreBind (CoreBinding result resultType True (neutralValue resultType)) : chain , CoreVariable result resultType ) -{- | The most arms that are lowered as one chain of nested statements. +{- | Lower the arms from the given one on, as the false branch of the arm +before them. -Every arm of a chain adds one level of nesting to the Core it lowers to, and -the stages after Core walk nested statements recursively. A source with a -few hundred arms must not turn into a few hundred levels, so longer matches -are split into groups of this size. --} -maximumNestedArms :: Int -maximumNestedArms = 16 - -{- | Lower all arms, in groups of at most 'maximumNestedArms'. - -A match that fits in one group is a single chain and needs no slot. A longer -one binds a mutable @$taken@ slot; each body sets it before it runs, and -each group after the first is the else branch of a test of the slot. --} -lowerArmGroups :: - (Monad lower) => - BranchLowering lower -> - Maybe ResolvedName -> - [CoreExpression] -> - [MatchArm ResolvedName Type] -> - lower [CoreStatement] -lowerArmGroups lowering result subjects arms - | length arms <= maximumNestedArms = lowerArms lowering result Nothing subjects arms - | otherwise = do - taken <- branchFresh lowering "$taken" - groups <- mapM (lowerArms lowering result (Just taken) subjects) (groupsOf maximumNestedArms arms) - let takenRead = CoreVariable taken boolType - sequenced = case groups of - [] -> [] - first : later -> first ++ [CoreIf takenRead [] group | group <- later] - pure (CoreBind (CoreBinding taken boolType True (CoreLiteral (CoreBoolean False) boolType)) : sequenced) - -groupsOf :: Int -> [value] -> [[value]] -groupsOf size values = case splitAt size values of - ([], _) -> [] - (group, remaining) -> group : groupsOf size remaining - -{- | Lower the arms from the given one on. - -The names an arm binds are bound before its test: binding a scalar that was -already evaluated has no effect of its own, and the guard may read them. An -arm that always accepts ends the chain; the type checker has rejected any arm -after it. +An arm that always accepts ends the chain with its body; the type checker +has rejected any arm after it. A guard is the last operand of the arm's +test, behind the short-circuit conjunction, so it runs only when the +patterns accept. Only a guard that stores into a local needs statements of +its own; they go before the conditional of its arm, inside the false branch +of the arm before it, and that arm alone adds a level of nesting. -} lowerArms :: (Monad lower) => BranchLowering lower -> Maybe ResolvedName -> - Maybe ResolvedName -> [CoreExpression] -> [MatchArm ResolvedName Type] -> lower [CoreStatement] -lowerArms _ _ _ _ [] = pure [] -lowerArms lowering result taken subjects (arm : remaining) = do - let paired = zip (matchArmPatterns arm) subjects - bindings = - [ CoreBind (CoreBinding name (expressionType subject) False subject) - | (MatchTypePattern _ _ (Just name) _, subject) <- paired - ] - tests = concatMap (uncurry (patternTests lowering)) paired - armBody <- lowerArmBody lowering result (matchArmBody arm) - -- The slot is set before the body runs, because the body may leave the - -- function or the enclosing loop. - let body = [CoreAssign slot (CoreLiteral (CoreBoolean True) boolType) | Just slot <- [taken]] ++ armBody - later <- lowerArms lowering result taken subjects remaining +lowerArms _ _ _ [] = pure [] +lowerArms lowering result subjects (arm : remaining) = do + let tests = concatMap (uncurry (patternTests lowering)) (zip (matchArmPatterns arm) subjects) + body <- lowerArmBody lowering result (matchArmBody arm) + later <- lowerArms lowering result subjects remaining case (tests, matchArmGuard arm) of - ([], Nothing) -> pure (bindings ++ body) - (_, Nothing) -> pure (bindings ++ [CoreIf (conjunction tests) body later]) - ([], Just guard) -> do - (guardPrefix, guardValue) <- branchExpression lowering guard - pure (bindings ++ guardPrefix ++ [CoreIf guardValue body later]) - -- The guard runs only when the patterns accept, and its statements - -- with it. The decision is a Boolean slot, so a numeric guard is - -- tested in Boolean context and never stored. + ([], Nothing) -> pure body + (_, Nothing) -> pure [CoreIf (conjunction tests) body later] (_, Just guard) -> do - loweredGuard <- branchExpression lowering guard - accepted <- branchFresh lowering "$accepted" - let (statements, decision) = decideLogical True accepted (conjunction tests) loweredGuard - pure (bindings ++ statements ++ [CoreIf decision body later]) + (guardPrefix, guardValue) <- branchExpression lowering guard + if null guardPrefix + then pure [CoreIf (conjunction (tests ++ [guardValue])) body later] + else + if null tests + then pure (guardPrefix ++ [CoreIf guardValue body later]) + else do + accepted <- branchFresh lowering "$accepted" + let (statements, decision) = + decideLogical True accepted (conjunction tests) (guardPrefix, guardValue) + pure (statements ++ [CoreIf decision body later]) {- | The comparisons one pattern needs against its subject. @@ -207,8 +248,12 @@ patternTests lowering patternValue subject = case patternValue of MatchNullPattern {} -> [CoreLiteral (CoreBoolean False) boolType] MatchCasePattern {} -> [CoreLiteral (CoreBoolean False) boolType] --- Every test is a Boolean without effects, so the short-circuit primitive --- only decides how many comparisons run. +-- The comparisons are Booleans without effects, so for them the +-- short-circuit primitive only decides how many run. A guard, when there is +-- one, is the last operand and is therefore evaluated only when every +-- comparison held; it may be numeric, which the primitive tests in Boolean +-- context. A lone operand is used as it is: a conditional statement tests a +-- numeric condition in Boolean context as well. conjunction :: [CoreExpression] -> CoreExpression conjunction tests = case tests of [] -> CoreLiteral (CoreBoolean True) boolType @@ -224,8 +269,12 @@ its effects. lowerArmBody :: (Monad lower) => BranchLowering lower -> Maybe ResolvedName -> Expression ResolvedName Type -> lower [CoreStatement] lowerArmBody lowering result body = case (result, body) of - (Just slot, _) -> do - (prefix, value) <- branchExpression lowering body - pure (prefix ++ [CoreAssign slot value]) + (Just slot, _) + | doesNotComplete body -> lowerNeverCompleting lowering body + | otherwise -> do + (prefix, value) <- branchExpression lowering body + pure (prefix ++ [CoreAssign slot value]) (Nothing, BlockExpression _ (Block statements) _) -> branchStatements lowering statements - (Nothing, _) -> branchDiscarded lowering body + (Nothing, _) + | doesNotComplete body -> lowerNeverCompleting lowering body + | otherwise -> branchDiscarded lowering body diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Captures.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Captures.hs new file mode 100644 index 00000000..61c7c4eb --- /dev/null +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Captures.hs @@ -0,0 +1,87 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | What a closure takes from the place that creates it. + +A callable written without a capture list captures what its body reads and +does not define: the lowering finds those names in the lowered body. A +closure takes the value of each when it is created. +-} +module Visual.XSharp.Desugarer.Captures + ( discoverImplicitCaptures + , localSymbols + ) where + +import Visual.XSharp.AST +import Visual.XSharp.Core + +{- | The captures of a closure with the given parameters and body: every name +the body reads that is neither a parameter nor a local of the body, once, +in the order of the first read. +-} +discoverImplicitCaptures :: [(ResolvedName, Type)] -> [CoreStatement] -> [CoreCapture] +discoverImplicitCaptures parameters statements = + let bound = map (resolvedSymbol . fst) parameters ++ localSymbols statements + free = filter (\(name, _) -> resolvedSymbol name `notElem` bound) (statementReads statements) + in [CoreCapture StrongCapture name valueType (CoreVariable name valueType) | (name, valueType) <- uniqueReads free] + +{- | The symbols the statements bind, in every branch and loop body they +hold. The body of a closure among them is a scope of its own and is not +searched. +-} +localSymbols :: [CoreStatement] -> [SymbolId] +localSymbols = concatMap collect + where + collect statement = case statement of + CoreBind binding -> [resolvedSymbol (coreBindingName binding)] + CoreIf _ yes no -> localSymbols yes ++ localSymbols no + CoreWhile _ body -> localSymbols body + CoreDoWhile body _ -> localSymbols body + CoreFor _ body update -> localSymbols body ++ localSymbols update + _ -> [] + +statementReads :: [CoreStatement] -> [(ResolvedName, Type)] +statementReads = concatMap collect + where + collect statement = case statement of + CoreBind binding -> expressionReads (coreBindingValue binding) + CoreAssign _ value -> expressionReads value + CoreReturn value -> expressionReads value + CoreIf condition yes no -> expressionReads condition ++ statementReads yes ++ statementReads no + CoreWhile condition body -> expressionReads condition ++ statementReads body + CoreDoWhile body condition -> statementReads body ++ expressionReads condition + CoreFor condition body update -> + expressionReads condition ++ statementReads body ++ statementReads update + CoreEvaluate value -> expressionReads value + CoreBreak -> [] + CoreContinue -> [] + +expressionReads :: CoreExpression -> [(ResolvedName, Type)] +expressionReads expression = case expression of + CoreVariable name valueType -> [(name, valueType)] + CoreLiteral _ _ -> [] + CoreApply callee arguments _ -> expressionReads callee ++ concatMap expressionReads arguments + CorePrimitive _ arguments _ -> concatMap expressionReads arguments + CoreLet name _ value body _ -> + expressionReads value ++ filter ((/= resolvedSymbol name) . resolvedSymbol . fst) (expressionReads body) + CoreConditional condition whenTrue whenFalse _ -> + expressionReads condition ++ expressionReads whenTrue ++ expressionReads whenFalse + -- A closure reads, from the place that creates it, its capture + -- initializers and whatever its body reads that is not its own: its + -- parameters, its captures and its locals belong to the closure. Without + -- that, a closure around this one would capture them as if they were + -- names of its surroundings. + CoreClosure captures parameters _ body _ -> + let own = + map (resolvedSymbol . fst) parameters + ++ map (resolvedSymbol . coreCaptureName) captures + ++ localSymbols body + in concatMap (expressionReads . coreCaptureValue) captures + ++ filter ((`notElem` own) . resolvedSymbol . fst) (statementReads body) + +uniqueReads :: [(ResolvedName, Type)] -> [(ResolvedName, Type)] +uniqueReads = foldl append [] + where + append output value@(name, _) + | any ((== resolvedSymbol name) . resolvedSymbol . fst) output = output + | otherwise = output ++ [value] diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Effects.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Effects.hs new file mode 100644 index 00000000..c82f5e1b --- /dev/null +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Effects.hs @@ -0,0 +1,121 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Which calls do something that can be observed. + +Visual X# evaluates by need, and effects are not lazy: an effect happens +where it is written, whether or not the value of the expression that +contains it is ever needed. The source has no notation for an effect, so the +compiler has to know which expressions have one. A store is an effect that +shows in the expression itself. Output does not: @Log(value)@ writes to the +console only because of what the body of @Log@ does. + +This module finds the methods a call of which may write to the console. An +expression that contains such a call is evaluated where it stands, like an +expression that stores: its binding is not deferred, and it is not handed to +another function as a suspended computation. + +The answer errs on the side of an effect. A method acts when a call anywhere +in its body acts, in a callable it creates as well as in its own statements: +creating a callable that writes is taken for writing. A call through a +callable value acts when the body of any callable of the program holds a +call that acts, or when a method that acts is used as a value, because +which callable a value holds is not followed. A program in which no +callable writes keeps its calls through callables by need, and a program +without console output is unaffected altogether. +-} +module Visual.XSharp.Desugarer.Effects + ( Effects + , noEffects + , programEffects + , callActs + , expressionActs + ) where + +import Data.Map.Strict (Map) +import Data.Map.Strict qualified as Map +import Visual.XSharp.AST +import Visual.XSharp.Desugarer.Handing (blockExpressions, methodBody, methodDeclarations, within) +import Visual.XSharp.RuntimeCall + +-- | What is known about the calls of a program. +data Effects = Effects + { effectsMethods :: Map SymbolId Bool + -- ^ For each method of the program, whether a call of it may act. + , effectsCallables :: Bool + -- ^ Whether a call through a callable value may act. + } + deriving (Eq, Show) + +-- | A program in which no call acts. +noEffects :: Effects +noEffects = Effects Map.empty False + +{- | The effects of a program given as the declarations of every tree that is +lowered for it. All of them are read together because a method of one tree +may call a method of another: the methods of ordinary types and the +specializations of templates are lowered apart. + +A method starts out as not acting and is found to act when its body holds a +call that acts; the answer is computed again from itself until it no longer +changes, which it must, because a method that acts never stops acting. +-} +programEffects :: [[Declaration ResolvedName Type]] -> Effects +programEffects trees = settle noEffects {effectsMethods = Map.map (const False) bodies} + where + bodies = Map.map (blockExpressions . methodBody) (Map.unions (map methodDeclarations trees)) + -- Every expression of the body of every callable the program + -- creates, in whatever method. + callableBodies = + [ case body of + CallableExpressionBody result -> within result + CallableBlockBody block -> blockExpressions block + | expressions <- Map.elems bodies + , CallableExpression _ _ _ _ body _ <- expressions + ] + -- The methods that are used as a value somewhere: named more often + -- than they are called by name. + heldMethods = + Map.keys + ( Map.filter + (> 0) + ( Map.unionWith + (+) + (occurrences 1 [name | NameExpression _ name _ <- everything]) + (occurrences (-1) [name | CallExpression _ (NameExpression _ name _) _ _ <- everything]) + ) + ) + everything = concat (Map.elems bodies) + occurrences :: Int -> [ResolvedName] -> Map SymbolId Int + occurrences weight names = + Map.fromListWith (+) [(resolvedSymbol name, weight) | name <- names, Map.member (resolvedSymbol name) bodies] + settle current = + let next = + Effects + (Map.map (any (callActsIn current)) bodies) + ( any (any (callActsIn current)) callableBodies + || any (\held -> Map.findWithDefault False held (effectsMethods current)) heldMethods + ) + in if next == current then current else settle next + +-- | Whether a call of the given name may act. +callActs :: Effects -> ResolvedName -> Bool +callActs effects name = case runtimeFunctionOfName name of + Just function -> runtimeObservable function + -- A name that is not a method is a callable value. + Nothing -> Map.findWithDefault (effectsCallables effects) (resolvedSymbol name) (effectsMethods effects) + +{- | Whether evaluating an expression may act: whether any call in it does. +The statements of a block, a loop or a callable inside the expression are +not searched; an expression that holds one is evaluated where it stands for +that reason alone. +-} +expressionActs :: Effects -> Expression ResolvedName annotation -> Bool +expressionActs effects = any (callActsIn effects) . within + +callActsIn :: Effects -> Expression ResolvedName annotation -> Bool +callActsIn effects expression = case expression of + CallExpression _ (NameExpression _ callee _) _ _ -> callActs effects callee + -- The callee is itself computed: a callable value. + CallExpression {} -> effectsCallables effects + _ -> False diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Handing.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Handing.hs new file mode 100644 index 00000000..77559079 --- /dev/null +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Handing.hs @@ -0,0 +1,303 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | What the lowering needs to know to hand a value on by need. + +An argument is a value like any other: it is computed when the method that +receives it first needs it, at most once, and not at all when the method +never needs it. A value that is computed where it stands needs nothing for +that. A value by need that one function hands to another needs a place both +can reach: the caller may need it too, and whoever needs it first computes +it for both. That place is a callable that remembers its result, a +/suspended computation/; see 'Visual.XSharp.Core.CoreMemoize'. + +This module holds the two analyses that decide where suspended computations +are used. Both read the typed tree and neither changes it. + +* 'methodNeeds' finds the parameters a method may leave unused: those are + passed by need. A parameter the method is certain to read first of all is + passed as a value, which costs nothing and can be told apart from passing + it by need only by which of two failures a caller meets. + +* 'handedLocals' finds the locals of a function whose values may be handed + on by need. Such a local is suspended in a callable of its own instead of + in a flag and a slot of the frame, because a frame cannot be reached from + another function. +-} +module Visual.XSharp.Desugarer.Handing + ( MethodNeeds + , methodNeeds + , argumentNeeds + , methodDeclarations + , methodParameters + , methodBody + , handedLocals + , blockExpressions + , within + ) where + +import Data.Map.Strict (Map) +import Data.Map.Strict qualified as Map +import Data.Set (Set) +import Data.Set qualified as Set +import Visual.XSharp.AST +import Visual.XSharp.Desugarer.Laziness + +{- | For each method that has any, which of its parameters are passed by +need, in declaration order. A method all of whose parameters are passed as +values has no entry. +-} +type MethodNeeds = Map SymbolId [Bool] + +{- | The methods a call can name directly: the methods of the ordinary types +of a source set, by symbol. The methods of a template are lowered with its +specializations, in a lowering of their own. +-} +methodDeclarations :: [Declaration ResolvedName Type] -> Map SymbolId (Declaration ResolvedName Type) +methodDeclarations declarations = + Map.fromList + [ (resolvedSymbol (declarationName member), member) + | TypeDeclaration {typeMembers = members} <- declarations + , member@FunctionDeclaration {} <- members + ] + +-- | The parameters of a method; a declaration that is not a method has none. +methodParameters :: Declaration name annotation -> [Parameter name annotation] +methodParameters declaration = case declaration of + FunctionDeclaration {declarationParameters = parameters} -> parameters + _ -> [] + +-- | The body of a method; a declaration that is not a method has an empty one. +methodBody :: Declaration name annotation -> Block name annotation +methodBody declaration = case declaration of + FunctionDeclaration {declarationBody = body} -> body + _ -> Block [] + +{- | Which parameters of which methods are passed by need. + +A parameter is passed by need when its type can be suspended, as the given +predicate decides, and the method is not certain to need it before anything +else that can be observed; see 'neededFirst'. Computing such an argument at +the call then differs from computing it where the method needs it in +nothing but the place. + +Whether a statement reads a parameter depends on the methods it calls: a +call does not read an argument it passes by need. The answer therefore +starts from every parameter of a suspendable type being passed by need, +under which a call reads the fewest arguments, and is computed again from +its own result until it no longer changes. Each round can only find more +parameters that are certain to be read, so the rounds end. +-} +methodNeeds :: + (Type -> Bool) -> + -- | Whether evaluating an expression may do something observable. + (Expression ResolvedName Type -> Bool) -> + Map SymbolId (Declaration ResolvedName Type) -> + MethodNeeds +methodNeeds suspendable acts methods = Map.filter or (settle (Map.map suspendableParameters methods)) + where + suspendableParameters declaration = + [suspendable (parameterAnnotation parameter) | parameter <- methodParameters declaration] + settle current = + let next = Map.mapWithKey (refine current) current + in if next == current then current else settle next + refine current symbol flags = case Map.lookup symbol methods of + Just declaration@FunctionDeclaration {declarationBody = Block statements} -> + [ flag && not (neededFirst (argumentNeeds current) acts (parameterName parameter) statements) + | (flag, parameter) <- zip flags (methodParameters declaration) + ] + _ -> flags + +{- | Whether the first thing a body does that can be observed needs the given +parameter. + +The statements are followed in order for as long as nothing they do can be +observed. A binding whose initializer has no effect computes nothing where +it stands: it is passed over, and when its value always needs the parameter +its name is remembered, because a later need of that name is a need of the +parameter. A statement that evaluates an expression which can neither fail +nor run without end is passed over as well. The answer is found at the +first statement that needs the parameter or one of the remembered names, +and it is no at the first statement that could do anything else: fail, +never return, choose a path or leave. + +A parameter this holds for may be computed at the call. A caller can tell +only which of two failures it meets when the argument and the statement +that needs it both fail, and the language leaves that open. +-} +neededFirst :: + ArgumentNeeds ResolvedName -> + (Expression ResolvedName Type -> Bool) -> + ResolvedName -> + [Statement ResolvedName Type] -> + Bool +neededFirst needs acts parameter = go [parameter] + where + go _ [] = False + go names (statement : later) = case statement of + BindingStatement _ _ _ bound _ value + | deferrable value -> go (if needing names value then bound : names else names) later + | otherwise -> evaluated names value later + -- A remembered name that is stored into is a local that is + -- computed where it is bound, which is where it needed the + -- parameter. + AssignmentStatement _ target _ value -> target `elem` names || evaluated names value later + CompoundAssignmentStatement _ _ target _ value -> target `elem` names || evaluated names value later + IncrementStatement _ target _ -> target `elem` names || go names later + DiscardStatement _ value -> evaluated names value later + ExpressionStatement _ value _ -> evaluated names value later + ReturnStatement _ (Just value) -> needing names value + IfStatement _ condition _ _ -> needing names condition + GuardStatement _ condition _ -> needing names condition + WhileStatement _ condition _ -> needing names condition + _ -> False + evaluated names value later = needing names value || (quiet value && go names later) + quiet value = deferrable value && not (worthDeferring value) + -- An expression that acts is evaluated where it stands: what it + -- does happens before anything after it. + deferrable value = deferrableExpression value && not (acts value) + needing names value = any (\name -> alwaysReads needs name value) names + +-- | The needs of a method by its name, as the laziness analysis asks for them. +argumentNeeds :: MethodNeeds -> ArgumentNeeds ResolvedName +argumentNeeds needs name = Map.lookup (resolvedSymbol name) needs + +{- | The locals of a function body whose values may be handed on by need. + +A local is handed on when an argument that may be passed by need reads it, +and when the initializer of a local that is handed on reads it: the +suspended computation of the one reads the other, from wherever it runs. +The result may name locals that turn out to be computed where they stand; +for those it has no consequence. +-} +handedLocals :: MethodNeeds -> (Expression ResolvedName Type -> Bool) -> Block ResolvedName Type -> [SymbolId] +handedLocals needs acts body = Set.toList (close seeds) + where + expressions = blockExpressions body + initializers = + Map.fromList + [ (resolvedSymbol name, map (resolvedSymbol . fst) (expressionNames value)) + | (name, value) <- blockBindings body + , deferrableExpression value + , not (acts value) + ] + seeds = + Set.fromList + [ resolvedSymbol name + | CallExpression _ (NameExpression _ callee _) arguments _ <- expressions + , Just flags <- [Map.lookup (resolvedSymbol callee) needs] + , length flags == length arguments + , (True, argument) <- zip flags arguments + , deferrableExpression argument + , not (acts argument) + , (name, _) <- expressionNames argument + ] + close :: Set SymbolId -> Set SymbolId + close current = + let reached = + Set.fromList + [ source + | symbol <- Set.toList current + , source <- Map.findWithDefault [] symbol initializers + ] + next = Set.union current reached + in if Set.size next == Set.size current then current else close next + +-- | Every binding of a block with its initializer, at any depth. +blockBindings :: Block name annotation -> [(name, Expression name annotation)] +blockBindings block = + [(name, value) | BindingStatement _ _ _ name _ value <- statementsWithin block] + +{- | Every expression of a block: those its statements hold, and every +expression inside those, at any depth. +-} +blockExpressions :: Block name annotation -> [Expression name annotation] +blockExpressions block = concatMap within (concatMap statementExpressions (statementsWithin block)) + +{- | Every statement of a block, at any depth: the statements of the blocks +its statements hold, and of the blocks, loops and callables its expressions +hold. +-} +statementsWithin :: Block name annotation -> [Statement name annotation] +statementsWithin (Block statements) = concatMap statementsOf statements + where + statementsOf statement = + statement + : concatMap statementsWithin (statementBlocks statement) + ++ concatMap statementsOf (statementStatements statement) + ++ concatMap inExpression (statementExpressions statement) + inExpression expression = concatMap held (within expression) + held expression = case expression of + LoopExpression _ loop _ -> statementsOf loop + BlockExpression _ block _ -> statementsWithin block + CallableExpression _ _ _ _ (CallableBlockBody block) _ -> statementsWithin block + _ -> [] + +-- | The blocks a statement holds directly. +statementBlocks :: Statement name annotation -> [Block name annotation] +statementBlocks statement = case statement of + IfStatement _ _ whenTrue whenFalse -> whenTrue : maybe [] (: []) whenFalse + WhileStatement _ _ body -> [body] + DoWhileStatement _ body _ -> [body] + ForStatement _ _ _ _ body -> [body] + ForEachStatement _ _ _ _ _ _ body -> [body] + GuardStatement _ _ block -> [block] + BlockStatement _ block -> [block] + _ -> [] + +-- | The statements a statement holds directly, outside its blocks. +statementStatements :: Statement name annotation -> [Statement name annotation] +statementStatements statement = case statement of + ForStatement _ initializer _ updates _ -> maybe [] (: []) initializer ++ updates + _ -> [] + +-- | The expressions a statement holds directly. +statementExpressions :: Statement name annotation -> [Expression name annotation] +statementExpressions statement = case statement of + BindingStatement _ _ _ _ _ value -> [value] + AssignmentStatement _ _ _ value -> [value] + ReturnStatement _ value -> maybe [] (: []) value + IfStatement _ condition _ _ -> [condition] + WhileStatement _ condition _ -> [condition] + DoWhileStatement _ _ condition -> [condition] + ForStatement _ _ condition _ _ -> maybe [] (: []) condition + ForEachStatement _ _ _ _ _ source _ -> [source] + IncrementStatement {} -> [] + CompoundAssignmentStatement _ _ _ _ value -> [value] + DiscardStatement _ value -> [value] + BreakStatement _ value -> maybe [] (: []) value + ContinueStatement _ -> [] + GuardStatement _ condition _ -> [condition] + BlockStatement {} -> [] + ExpressionStatement _ value _ -> [value] + +{- | An expression and every expression inside it that belongs to the same +evaluation: operands, arms, capture initializers and the expression body of +a callable. The statements an expression holds are reached through +'statementsWithin'. +-} +within :: Expression name annotation -> [Expression name annotation] +within expression = expression : concatMap within (children expression) + where + children value = case value of + NameExpression {} -> [] + LiteralExpression {} -> [] + MemberAccessExpression _ receiver _ _ -> [receiver] + MethodReferenceExpression _ receiver _ _ -> [receiver] + CallExpression _ callee arguments _ -> callee : arguments + UnaryExpression _ _ operand _ -> [operand] + BinaryExpression _ _ left right _ -> [left, right] + IsPatternExpression _ subject _ _ -> [subject] + ConditionalExpression _ condition first second _ -> [condition, first, second] + CoalesceExpression _ left fallback _ -> [left, fallback] + AssignmentExpression _ _ _ assigned _ -> [assigned] + IncrementExpression {} -> [] + LoopExpression {} -> [] + BlockExpression {} -> [] + MatchExpression _ subjects arms _ -> + subjects ++ concat [maybe [] (: []) (matchArmGuard arm) ++ [matchArmBody arm] | arm <- arms] + CallableExpression _ _ captures _ body _ -> + [initializer | Capture {captureInitializer = Just initializer} <- captures] + ++ case body of + CallableExpressionBody result -> [result] + CallableBlockBody _ -> [] diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Laziness.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Laziness.hs new file mode 100644 index 00000000..12e3cb41 --- /dev/null +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Laziness.hs @@ -0,0 +1,254 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | What the lowering needs to know to evaluate a value by need. + +Visual X# is a lazy language: a value is computed when it is first needed and +at most once, and a value that is never needed is never computed. Its effects +are not lazy: a store into a variable and a transfer of control happen where +they are written. The source has no notation for either fact; the compiler +tells the two kinds of expression apart. + +This module holds that distinction for the initializer of a local binding. +An initializer may be deferred when evaluating it later gives what evaluating +it now would give: it stores nothing, transfers nowhere, and reads the +variables it names as they are where the binding stands. Deferring it is +observable only when it might fail or never return, so only then is it worth +a flag and a test at every use. +-} +module Visual.XSharp.Desugarer.Laziness + ( deferrableExpression + , worthDeferring + , expressionNames + , renameNames + , capturedSymbols + , ArgumentNeeds + , alwaysReads + , neededTwice + , neededNext + ) where + +import Visual.XSharp.AST +import Visual.XSharp.Core + +{- | Whether evaluating an expression at a later point of the same function +gives the value that evaluating it in place would give, provided the +variables it reads still have the values they had. + +That holds for expressions built from names, literals, operators, tests and +calls: a call cannot store into a local of its caller. It does not hold for +an expression that stores, that leaves, or that contains statements. +-} +deferrableExpression :: Expression name annotation -> Bool +deferrableExpression expression = case expression of + NameExpression {} -> True + LiteralExpression {} -> True + MemberAccessExpression _ receiver _ _ -> deferrableExpression receiver + MethodReferenceExpression _ receiver _ _ -> deferrableExpression receiver + CallExpression _ callee arguments _ -> all deferrableExpression (callee : arguments) + UnaryExpression _ _ value _ -> deferrableExpression value + BinaryExpression _ _ left right _ -> deferrableExpression left && deferrableExpression right + IsPatternExpression _ subject _ _ -> deferrableExpression subject + ConditionalExpression _ condition first second _ -> all deferrableExpression [condition, first, second] + CoalesceExpression {} -> False + AssignmentExpression {} -> False + IncrementExpression {} -> False + LoopExpression {} -> False + BlockExpression {} -> False + MatchExpression {} -> False + CallableExpression {} -> False + +{- | Whether an expression can fail or never return, which is when it matters +whether it is evaluated at all: it calls something, or it divides. +-} +worthDeferring :: Expression name annotation -> Bool +worthDeferring expression = case expression of + CallExpression {} -> True + BinaryExpression _ operator left right _ -> + operator `elem` [Divide, FloorDivide, Remainder] || worthDeferring left || worthDeferring right + MemberAccessExpression _ receiver _ _ -> worthDeferring receiver + MethodReferenceExpression _ receiver _ _ -> worthDeferring receiver + UnaryExpression _ _ value _ -> worthDeferring value + IsPatternExpression _ subject _ _ -> worthDeferring subject + ConditionalExpression _ condition first second _ -> any worthDeferring [condition, first, second] + _ -> False + +-- | The names an expression that may be deferred reads, with their types. +expressionNames :: Expression name annotation -> [(name, annotation)] +expressionNames expression = case expression of + NameExpression _ name annotation -> [(name, annotation)] + MemberAccessExpression _ receiver _ _ -> expressionNames receiver + MethodReferenceExpression _ receiver _ _ -> expressionNames receiver + CallExpression _ callee arguments _ -> concatMap expressionNames (callee : arguments) + UnaryExpression _ _ value _ -> expressionNames value + BinaryExpression _ _ left right _ -> expressionNames left ++ expressionNames right + IsPatternExpression _ subject _ _ -> expressionNames subject + ConditionalExpression _ condition first second _ -> concatMap expressionNames [condition, first, second] + _ -> [] + +{- | Replace the names an expression that may be deferred reads. A name that +has no replacement is kept. +-} +renameNames :: (name -> name) -> Expression name annotation -> Expression name annotation +renameNames rename expression = case expression of + NameExpression spanValue name annotation -> NameExpression spanValue (rename name) annotation + MemberAccessExpression spanValue receiver member annotation -> + MemberAccessExpression spanValue (renameNames rename receiver) member annotation + MethodReferenceExpression spanValue receiver member annotation -> + MethodReferenceExpression spanValue (renameNames rename receiver) member annotation + CallExpression spanValue callee arguments annotation -> + CallExpression spanValue (renameNames rename callee) (map (renameNames rename) arguments) annotation + UnaryExpression spanValue operator value annotation -> + UnaryExpression spanValue operator (renameNames rename value) annotation + BinaryExpression spanValue operator left right annotation -> + BinaryExpression spanValue operator (renameNames rename left) (renameNames rename right) annotation + IsPatternExpression spanValue subject patternValue annotation -> + IsPatternExpression spanValue (renameNames rename subject) patternValue annotation + ConditionalExpression spanValue condition first second annotation -> + ConditionalExpression + spanValue + (renameNames rename condition) + (renameNames rename first) + (renameNames rename second) + annotation + _ -> expression + +{- | For the name of a method whose parameters may be passed by need, which +of them are, in declaration order; nothing for any other name. A call of +such a method does not read an argument it passes by need: the method may +never need it. +-} +type ArgumentNeeds name = name -> Maybe [Bool] + +{- | Whether evaluating an expression always reads the given name, whatever +values it meets: the name stands where neither a short-circuit operator, nor +a conditional, nor a parameter that is passed by need can skip it. Blocks, +matches, loops and callables used as values are not searched; the answer for +them is that it is not known. +-} +alwaysReads :: (Eq name) => ArgumentNeeds name -> name -> Expression name annotation -> Bool +alwaysReads needs name expression = case expression of + NameExpression _ found _ -> found == name + MemberAccessExpression _ receiver _ _ -> always receiver + MethodReferenceExpression _ receiver _ _ -> always receiver + CallExpression _ (NameExpression _ callee _) arguments _ + | Just flags <- needs callee + , length flags == length arguments -> + any always [argument | (False, argument) <- zip flags arguments] + CallExpression _ callee arguments _ -> any always (callee : arguments) + UnaryExpression _ _ value _ -> always value + BinaryExpression _ operator left right _ + | operator `elem` [LogicalAnd, LogicalOr] -> always left + | otherwise -> always left || always right + IsPatternExpression _ subject _ _ -> always subject + ConditionalExpression _ condition _ _ _ -> always condition + AssignmentExpression _ _ _ value _ -> always value + _ -> False + where + always = alwaysReads needs name + +{- | The reads of the given values by need that an expression is certain to +evaluate and holds more than once, one read for each such value. + +Computing those values ahead of the expression changes nothing that can be +observed except which of two computations fails first when both fail: the +expression has no effect, and it evaluates each of them whenever it is +evaluated. +-} +neededTwice :: + ArgumentNeeds ResolvedName -> + [SymbolId] -> + Expression ResolvedName annotation -> + [(ResolvedName, Expression ResolvedName annotation)] +neededTwice needs deferred expression + | null deferred || not (deferrableExpression expression) = [] + | otherwise = go [] (nameReads expression) + where + go _ [] = [] + go seen ((name, found) : remaining) + | symbol `elem` seen = go seen remaining + | symbol `elem` deferred + , symbol `elem` map (resolvedSymbol . fst) remaining + , alwaysReads needs name expression = + (name, found) : go (symbol : seen) remaining + | otherwise = go (symbol : seen) remaining + where + symbol = resolvedSymbol name + nameReads value = case value of + NameExpression _ name _ -> [(name, value)] + MemberAccessExpression _ receiver _ _ -> nameReads receiver + MethodReferenceExpression _ receiver _ _ -> nameReads receiver + CallExpression _ callee arguments _ -> concatMap nameReads (callee : arguments) + UnaryExpression _ _ operand _ -> nameReads operand + BinaryExpression _ _ left right _ -> nameReads left ++ nameReads right + IsPatternExpression _ subject _ _ -> nameReads subject + ConditionalExpression _ condition first second _ -> concatMap nameReads [condition, first, second] + _ -> [] + +{- | Whether the statement that follows a binding is certain to read it. + +A value that is certain to be needed by the very next statement may be +computed where its binding stands: nothing can be observed between the two +places except which of two computations fails first when both fail. That +saves the flag and the test for the common case of a value that is used at +once. + +The next statement reads the name for certain when the expression it +evaluates first always reads it. A binding evaluates its initializer first +only when it is itself evaluated in place, which the second argument +decides for it from the statements after it. +-} +neededNext :: + (Eq name) => + ArgumentNeeds name -> + -- | Whether a later binding is evaluated in place, given what follows it. + (name -> annotation -> Expression name annotation -> [Statement name annotation] -> Bool) -> + name -> + [Statement name annotation] -> + Bool +neededNext needs inPlaceBinding name following = case following of + next : later -> case next of + BindingStatement _ _ _ bound annotation value -> + always value && inPlaceBinding bound annotation value later + AssignmentStatement _ _ _ value -> always value + CompoundAssignmentStatement _ _ _ _ value -> always value + ReturnStatement _ (Just value) -> always value + IfStatement _ condition _ _ -> always condition + GuardStatement _ condition _ -> always condition + WhileStatement _ condition _ -> always condition + DiscardStatement _ value -> always value + ExpressionStatement _ value _ -> always value + _ -> False + [] -> False + where + always = alwaysReads needs name + +{- | Every local that a closure created by the statements captures, at any +depth. A closure takes the value of a capture when it is created, so a +binding that a closure captures has to have its value by then. +-} +capturedSymbols :: [CoreStatement] -> [SymbolId] +capturedSymbols = concatMap statement + where + statement value = case value of + CoreBind binding -> expression (coreBindingValue binding) + CoreAssign _ assigned -> expression assigned + CoreReturn returned -> expression returned + CoreEvaluate evaluated -> expression evaluated + CoreIf condition whenTrue whenFalse -> expression condition ++ capturedSymbols whenTrue ++ capturedSymbols whenFalse + CoreWhile condition body -> expression condition ++ capturedSymbols body + CoreDoWhile body condition -> capturedSymbols body ++ expression condition + CoreFor condition body update -> expression condition ++ capturedSymbols body ++ capturedSymbols update + CoreBreak -> [] + CoreContinue -> [] + expression value = case value of + CoreVariable {} -> [] + CoreLiteral {} -> [] + CoreApply callee arguments _ -> concatMap expression (callee : arguments) + CorePrimitive _ arguments _ -> concatMap expression arguments + CoreLet _ _ bound body _ -> expression bound ++ expression body + CoreConditional condition whenTrue whenFalse _ -> concatMap expression [condition, whenTrue, whenFalse] + CoreClosure captures _ _ body _ -> + map (resolvedSymbol . coreCaptureName) captures + ++ concatMap (expression . coreCaptureValue) captures + ++ capturedSymbols body diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Sequencing.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Sequencing.hs index 1ea00222..644f48c8 100644 --- a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Sequencing.hs +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Sequencing.hs @@ -27,7 +27,7 @@ module Visual.XSharp.Desugarer.Sequencing , decideLogical ) where -import Visual.XSharp.AST (ResolvedName (..), SymbolId, Type, boolType) +import Visual.XSharp.AST (ResolvedName (..), SymbolId, Type, boolType, stringType) import Visual.XSharp.Core import Visual.XSharp.Core.Scalar (isCoreFloatingType) @@ -84,6 +84,7 @@ neutralValue :: Type -> CoreExpression neutralValue valueType | valueType == boolType = CoreLiteral (CoreBoolean False) valueType | isCoreFloatingType valueType = CoreLiteral (CoreFloating "0") valueType + | valueType == stringType = CoreLiteral (CoreString "") valueType | otherwise = CoreLiteral (CoreInteger 0) valueType -- | Leave the innermost loop unless the condition holds. diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Symbols.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Symbols.hs new file mode 100644 index 00000000..63e39a5a --- /dev/null +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Symbols.hs @@ -0,0 +1,129 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Tables the lowering to Core reads and does not change: the symbols a +typed tree already uses, and the Core primitive each source operator stands +for. +-} +module Visual.XSharp.Desugarer.Symbols + ( syntaxSymbolIds + , lowerUnary + , lowerBinary + ) where + +import Visual.XSharp.AST +import Visual.XSharp.Core + +{- | Every symbol a typed tree declares or binds. + +Generated Core bindings must never collide with source symbols. Gathering +the complete typed tree once is cheaper and more robust than reserving a +magic numeric range or deriving identities from source positions. +-} +syntaxSymbolIds :: SyntaxTree ResolvedName Type -> [Int] +syntaxSymbolIds (SyntaxTree _ declarations) = concatMap declarationSymbolIds declarations + +declarationSymbolIds :: Declaration ResolvedName Type -> [Int] +declarationSymbolIds declaration = + symbolValue (declarationName declaration) + : case declaration of + TypeDeclaration {typeMembers = members} -> concatMap declarationSymbolIds members + TemplateTypeDeclaration {declarationTemplateParameters = parameters, typeMembers = members} -> + map (symbolValue . templateParameterName) parameters ++ concatMap declarationSymbolIds members + FunctionDeclaration {declarationParameters = parameters, declarationBody = body} -> + map (symbolValue . parameterName) parameters ++ blockSymbolIds body + EnumDeclaration {} -> [] + +blockSymbolIds :: Block ResolvedName Type -> [Int] +blockSymbolIds (Block statements) = concatMap statementIds statements + +statementIds :: Statement ResolvedName Type -> [Int] +statementIds statement = case statement of + BindingStatement _ _ _ name _ value -> symbolValue name : expressionIds value + AssignmentStatement _ name _ value -> symbolValue name : expressionIds value + ReturnStatement _ value -> maybe [] expressionIds value + IfStatement _ condition yes no -> expressionIds condition ++ blockSymbolIds yes ++ maybe [] blockSymbolIds no + WhileStatement _ condition body -> expressionIds condition ++ blockSymbolIds body + DoWhileStatement _ body condition -> blockSymbolIds body ++ expressionIds condition + ForStatement _ initializer condition updates body -> + maybe [] statementIds initializer + ++ maybe [] expressionIds condition + ++ concatMap statementIds updates + ++ blockSymbolIds body + ForEachStatement _ _ _ name _ source body -> + symbolValue name : expressionIds source ++ blockSymbolIds body + IncrementStatement _ name _ -> [symbolValue name] + CompoundAssignmentStatement _ _ name _ value -> symbolValue name : expressionIds value + DiscardStatement _ value -> expressionIds value + BreakStatement _ value -> maybe [] expressionIds value + ContinueStatement {} -> [] + GuardStatement _ condition block -> expressionIds condition ++ blockSymbolIds block + BlockStatement _ block -> blockSymbolIds block + ExpressionStatement _ value _ -> expressionIds value + +expressionIds :: Expression ResolvedName Type -> [Int] +expressionIds expression = case expression of + NameExpression _ name _ -> [symbolValue name] + LiteralExpression {} -> [] + MemberAccessExpression _ receiver _ _ -> expressionIds receiver + MethodReferenceExpression _ receiver _ _ -> expressionIds receiver + CallExpression _ callee arguments _ -> expressionIds callee ++ concatMap expressionIds arguments + UnaryExpression _ _ value _ -> expressionIds value + BinaryExpression _ _ left right _ -> expressionIds left ++ expressionIds right + IsPatternExpression _ subject _ _ -> expressionIds subject + ConditionalExpression _ condition first second _ -> concatMap expressionIds [condition, first, second] + CoalesceExpression _ left fallback _ -> expressionIds left ++ expressionIds fallback + AssignmentExpression _ _ name value _ -> symbolValue name : expressionIds value + IncrementExpression _ _ name _ -> [symbolValue name] + LoopExpression _ loop _ -> statementIds loop + BlockExpression _ block _ -> blockSymbolIds block + MatchExpression _ subjects arms _ -> + [ symbolValue name + | arm <- arms + , Just name <- map matchPatternBinding (matchArmPatterns arm) + ] + ++ concatMap expressionIds (subjects ++ concatMap matchArmExpressions arms) + CallableExpression _ _ captures parameters body _ -> + map (symbolValue . captureName) captures + ++ concatMap (maybe [] expressionIds . captureInitializer) captures + ++ map (symbolValue . parameterName) parameters + ++ callableBodyIds body + +callableBodyIds :: CallableBody ResolvedName Type -> [Int] +callableBodyIds body = case body of + CallableExpressionBody expression -> expressionIds expression + CallableBlockBody block -> blockSymbolIds block + +symbolValue :: ResolvedName -> Int +symbolValue = symbolIdValue . resolvedSymbol + +-- | The Core primitive of a unary operator. +lowerUnary :: UnaryOperator -> CorePrimitive +lowerUnary UnaryNegate = CoreNegate +lowerUnary LogicalNot = CoreLogicalNot +lowerUnary BitwiseNot = CoreBitwiseNot +lowerUnary UnaryPlus = CoreAdd + +-- | The Core primitive of a binary operator. +lowerBinary :: BinaryOperator -> CorePrimitive +lowerBinary operator = case operator of + Add -> CoreAdd + Subtract -> CoreSubtract + Multiply -> CoreMultiply + Divide -> CoreDivide + FloorDivide -> CoreFloorDivide + Remainder -> CoreRemainder + Power -> CorePower + ShiftLeft -> CoreShiftLeft + ShiftRight -> CoreShiftRight + BitwiseAnd -> CoreBitwiseAnd + BitwiseXor -> CoreBitwiseXor + BitwiseOr -> CoreBitwiseOr + LessThan -> CoreLessThan + LessEqual -> CoreLessEqual + GreaterThan -> CoreGreaterThan + GreaterEqual -> CoreGreaterEqual + Equal -> CoreEqual + NotEqual -> CoreNotEqual + LogicalAnd -> CoreLogicalAnd + LogicalOr -> CoreLogicalOr diff --git a/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Workers.hs b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Workers.hs new file mode 100644 index 00000000..36123439 --- /dev/null +++ b/Compiler/Haskell/Core/src/Visual/XSharp/Desugarer/Workers.hs @@ -0,0 +1,101 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Giving a second lowering of a method body symbols of its own. + +A method whose parameters may be passed by need is lowered twice: once as +the method itself, which receives values, and once as the function that +receives suspended computations, for the callers that hand them on. Both +come from the same source statements, whose locals carry the symbols the +renamer gave them. Symbols are unique in a module, so the second lowering +takes fresh ones for everything it defines. +-} +module Visual.XSharp.Desugarer.Workers + ( definedNames + , renameSymbols + ) where + +import Data.Map.Strict (Map) +import Data.Map.Strict qualified as Map +import Visual.XSharp.AST +import Visual.XSharp.Core + +{- | Every name the statements define, at any depth: bindings, expression +bindings, and the parameters of closures. A capture is not listed: it is +the name it captures. +-} +definedNames :: [CoreStatement] -> [ResolvedName] +definedNames = concatMap statement + where + statement value = case value of + CoreBind binding -> coreBindingName binding : expression (coreBindingValue binding) + CoreAssign _ assigned -> expression assigned + CoreReturn returned -> expression returned + CoreEvaluate evaluated -> expression evaluated + CoreIf condition whenTrue whenFalse -> expression condition ++ definedNames whenTrue ++ definedNames whenFalse + CoreWhile condition body -> expression condition ++ definedNames body + CoreDoWhile body condition -> definedNames body ++ expression condition + CoreFor condition body update -> expression condition ++ definedNames body ++ definedNames update + CoreBreak -> [] + CoreContinue -> [] + expression value = case value of + CoreVariable {} -> [] + CoreLiteral {} -> [] + CoreApply callee arguments _ -> concatMap expression (callee : arguments) + CorePrimitive _ arguments _ -> concatMap expression arguments + CoreLet name _ bound body _ -> name : expression bound ++ expression body + CoreConditional condition whenTrue whenFalse _ -> concatMap expression [condition, whenTrue, whenFalse] + CoreClosure captures parameters _ body _ -> + map fst parameters + ++ concatMap (expression . coreCaptureValue) captures + ++ definedNames body + +{- | Replace every occurrence of the given symbols, where they are defined and +where they are used. A symbol without a replacement is kept: the functions +of the module, and whatever the statements read from outside. +-} +renameSymbols :: Map SymbolId ResolvedName -> [CoreStatement] -> [CoreStatement] +renameSymbols replacements = map statement + where + name original = Map.findWithDefault original (resolvedSymbol original) replacements + statement value = case value of + CoreBind binding -> + CoreBind + binding + { coreBindingName = name (coreBindingName binding) + , coreBindingValue = expression (coreBindingValue binding) + } + CoreAssign target assigned -> CoreAssign (name target) (expression assigned) + CoreReturn returned -> CoreReturn (expression returned) + CoreEvaluate evaluated -> CoreEvaluate (expression evaluated) + CoreIf condition whenTrue whenFalse -> + CoreIf (expression condition) (map statement whenTrue) (map statement whenFalse) + CoreWhile condition body -> CoreWhile (expression condition) (map statement body) + CoreDoWhile body condition -> CoreDoWhile (map statement body) (expression condition) + CoreFor condition body update -> + CoreFor (expression condition) (map statement body) (map statement update) + CoreBreak -> CoreBreak + CoreContinue -> CoreContinue + expression value = case value of + CoreVariable variable valueType -> CoreVariable (name variable) valueType + CoreLiteral {} -> value + CoreApply callee arguments valueType -> + CoreApply (expression callee) (map expression arguments) valueType + CorePrimitive primitive arguments valueType -> + CorePrimitive primitive (map expression arguments) valueType + CoreLet bound bindingType boundValue body valueType -> + CoreLet (name bound) bindingType (expression boundValue) (expression body) valueType + CoreConditional condition whenTrue whenFalse valueType -> + CoreConditional (expression condition) (expression whenTrue) (expression whenFalse) valueType + CoreClosure captures parameters returnType body valueType -> + CoreClosure + [ capture + { coreCaptureName = name (coreCaptureName capture) + , coreCaptureValue = expression (coreCaptureValue capture) + } + | capture <- captures + ] + [(name parameter, parameterType) | (parameter, parameterType) <- parameters] + returnType + (map statement body) + valueType diff --git a/Compiler/Haskell/Core/visual-xsharp-core.cabal b/Compiler/Haskell/Core/visual-xsharp-core.cabal index bd81a12d..22d3fd0b 100644 --- a/Compiler/Haskell/Core/visual-xsharp-core.cabal +++ b/Compiler/Haskell/Core/visual-xsharp-core.cabal @@ -2,7 +2,7 @@ cabal-version: 3.12 -- SPDX-FileCopyrightText: 2026 Progmasoft -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 name: visual-xsharp-core -version: 0.4.1 +version: 0.5.0 synopsis: Visual X# Core and CorePrep models, verification, and wire codecs license: MPL-2.0 author: Progmasoft @@ -30,7 +30,13 @@ library Visual.XSharp.Core.Wire Visual.XSharp.Desugarer Visual.XSharp.Desugarer.Branching + Visual.XSharp.Desugarer.Captures + Visual.XSharp.Desugarer.Effects + Visual.XSharp.Desugarer.Handing + Visual.XSharp.Desugarer.Laziness Visual.XSharp.Desugarer.Sequencing + Visual.XSharp.Desugarer.Symbols + Visual.XSharp.Desugarer.Workers other-modules: Visual.XSharp.Core.Optimizer.Analysis Visual.XSharp.Core.Optimizer.Constant diff --git a/Compiler/Haskell/Driver/src/Visual/XSharp/Compiler.hs b/Compiler/Haskell/Driver/src/Visual/XSharp/Compiler.hs index a55503f6..d17d5dd9 100644 --- a/Compiler/Haskell/Driver/src/Visual/XSharp/Compiler.hs +++ b/Compiler/Haskell/Driver/src/Visual/XSharp/Compiler.hs @@ -144,8 +144,12 @@ compileSemanticToCorePrep semantic = do templateSpecializationDiagnostics (planTemplateSpecializations defaultTemplateSpecializationLimits typed (discoveredTemplateDemands templateDiscovery)) >>= mapLeft templatePlanDiagnostics . verifyTemplateSpecializationPlan - ordinaryCore <- runDesugarer defaultDesugarer typed >>= verifyCore - specializationCore <- runDesugarer defaultDesugarer (specializationTypedAST templatePlan) >>= verifyCore + -- The two trees are one program: which calls have an effect is + -- decided from both. + let specializations = specializationTypedAST templatePlan + desugarer = desugarerWithin [typed, specializations] + ordinaryCore <- runDesugarer desugarer typed >>= verifyCore + specializationCore <- runDesugarer desugarer specializations >>= verifyCore let core = CoreModuleWithSources (coreModuleName ordinaryCore) diff --git a/Compiler/Haskell/Driver/test/BranchingDiagnosticTests.hs b/Compiler/Haskell/Driver/test/BranchingDiagnosticTests.hs index 94008798..72757088 100644 --- a/Compiler/Haskell/Driver/test/BranchingDiagnosticTests.hs +++ b/Compiler/Haskell/Driver/test/BranchingDiagnosticTests.hs @@ -113,12 +113,37 @@ positionTests = , reportedAt "VXT0046" (3, 19) ["int r = if (flag) { } else { 2 };", "return r;"] ) , - ( "a return in a value block is reported at the block" - , reportedAt "VXT0047" (3, 19) ["int r = if (flag) { return 1; 2 } else { 3 };", "return r;"] + ( "a return of the wrong type in a value block is reported at the return" + , reportedAt "VXT0005" (3, 21) ["int r = if (flag) { return true; } else { 3 };", "return r;"] ) , - ( "a break out of a value block is reported at the break" - , reportedAt "VXT0059" (4, 21) ["while (true) {", "int r = if (flag) { break; 1 } else { 2 };", "}", "return 0;"] + ( "a break without a loop in a value block is reported at the break" + , reportedAt "VXT0025" (3, 21) ["int r = if (flag) { break; } else { 2 };", "return r;"] + ) + , + ( "a return of the wrong type in a value block of a loop expression is reported at the return" + , reportedAt + "VXT0005" + (4, 21) + ["int r = while (true) {", "int q = if (flag) { return true; } else { 2 };", "break q;", "};", "return r;"] + ) + , + ( "a break without a value out of a loop expression is reported at the break" + , reportedAt + "VXT0040" + (4, 21) + ["int r = while (true) {", "int q = if (flag) { break; } else { 2 };", "break q;", "};", "return r;"] + ) + , + ( "a continue inside a callable in a loop is reported at the continue" + , reportedAt + "VXT0027" + (4, 37) + ["while (flag) {", "auto f = \\(int v) -> { if (v > 0) { continue; } return v; };", "}", "return 0;"] + ) + , + ( "a value-carrying break in the update of a loop statement is reported at the break" + , reportedAt "VXT0026" (3, 41) ["for (int i = 0; i < 3; i += if (flag) { break 1; } else { 1 }) { }", "return 0;"] ) , ( "a guard block that does not leave is reported at the guard" @@ -145,7 +170,7 @@ specimen = , "guard (left >= 0) else { return 0; }" , "{ int scoped = left; }" , "int kind = match (left), (flag) { (0), (_) -> 10, (1), (true) -> 20," - , "(int low), (_) if low < 5 -> { int doubled = low * 2; doubled }, (_), (_) -> 30 };" + , "(int low), (_) if low < 5 -> { int doubled = low * 2; doubled } (_), (_) -> 30 };" , "int total = 0;" , "for (int index = 0; index < right; index++) {" , "match (index % 3) { 0 -> { continue; }, 1 -> total += 10, _ -> { total += index; } }" diff --git a/Compiler/Haskell/Driver/test/BranchingEvaluationCases.hs b/Compiler/Haskell/Driver/test/BranchingEvaluationCases.hs new file mode 100644 index 00000000..9a7e5457 --- /dev/null +++ b/Compiler/Haskell/Driver/test/BranchingEvaluationCases.hs @@ -0,0 +1,466 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | The programs that "BranchingTests" runs, with the value each must return. + +Generated by @go -C helpers run ./cmd/execution-cases generate@. Do not edit: +the cases are written in the files under @Compiler/Fuzzing/Cases@, which also +generate the tables the native smoke program @source_execution_smoke@ runs +through LLVM. + +A case is the body of @int Run(bool flag, bool other, int left, int right)@ +and the runs of that body. The expected values are written by hand from the +language rules. +-} +module BranchingEvaluationCases + ( EvaluationCase + , evaluationCases + , selectionCases + , leavingCases + ) where + +-- | A body, and for each run its arguments and the value it must return. +type EvaluationCase = (String, [((Bool, Bool, Integer, Integer), Integer)]) + +-- | Every case, in the order the tests report them. +evaluationCases :: [EvaluationCase] +evaluationCases = selectionCases ++ leavingCases + +-- | Which arm, block or body is selected, and how often each part runs. From @Compiler/Fuzzing/Cases/Selection.cases@. +selectionCases :: [EvaluationCase] +selectionCases = + [ + ( "return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };" + , [((False, False, 1, 0), 10), ((False, False, 2, 0), 20), ((False, False, 5, 0), 30)] + ) + , -- Negative constants. + ( "int v = 0 - left; return match (v) { -1 -> 10, -2 -> 20, 0 -> 5, _ -> 7 };" + , [((False, False, 1, 0), 10), ((False, False, 2, 0), 20), ((False, False, 0, 0), 5), ((False, False, 3, 0), 7)] + ) + , + ( "int v = 0 - left - 9223372036854775807; return match (v) { -9223372036854775808 -> 1, -9223372036854775807 -> 2, _ -> 0 };" + , [((False, False, 1, 0), 1), ((False, False, 0, 0), 2)] + ) + , + ( "int v = right - left; return match (v), (flag) { (-1), (true) -> 1, (-1), (_) -> 2, (_), (_) -> 3 };" + , [((True, False, 3, 2), 1), ((False, False, 3, 2), 2), ((True, False, 2, 2), 3)] + ) + , -- Two bool subjects covered without a catch-all arm. + ( "return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 };" + , [((True, True, 0, 0), 1), ((True, False, 0, 0), 1), ((False, True, 0, 0), 2), ((False, False, 0, 0), 3)] + ) + , + ( "return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 };" + , [((True, True, 0, 0), 3), ((True, False, 0, 0), 2), ((False, True, 0, 0), 1), ((False, False, 0, 0), 0)] + ) + , -- The subject is evaluated once, whichever arm accepts. + ( "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> 300 }; return r + n;" + , [((False, False, 0, 0), 101), ((False, False, 1, 0), 202), ((False, False, 5, 0), 306)] + ) + , -- Subjects are evaluated left to right, before any arm is tested. + ( "int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, (_), (_) -> 0 };" + , [((False, False, 1, 0), 1), ((False, False, 2, 0), 0)] + ) + , -- A guard runs only when the patterns of its arm accept. + ( "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> 10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + calls;" + , [((False, False, 1, 0), 1001), ((False, False, 2, 0), 2010), ((False, False, 3, 0), 3000)] + ) + , -- A guard that is false passes the value on to the arms after it. + ( "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" + , [((True, False, 1, 0), 1), ((False, False, 1, 0), 2), ((True, False, 9, 0), 3)] + ) + , -- A numeric guard is tested in Boolean context. + ( "return match (left) { 1 if right -> 1, _ -> 0 };" + , [((False, False, 1, 5), 1), ((False, False, 1, 0), 0), ((False, False, 2, 5), 0)] + ) + , -- Only the body of the accepting arm runs. + ( "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += 10), _ -> (n += 100) }; return r * 1000 + n;" + , [((False, False, 1, 0), 1001), ((False, False, 2, 0), 10010), ((False, False, 3, 0), 100100)] + ) + , + ( "return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };" + , [((False, False, 2, 0), 1), ((False, False, 3, 0), 2), ((False, False, 4, 0), 0)] + ) + , + ( "long wide = 5; return match (wide) { 5 -> 1, _ -> 0 };" + , [((False, False, 0, 0), 1)] + ) + , + ( "long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 ? 1 : 0;" + , [((False, False, 1, 0), 0), ((False, False, 2, 0), 1)] + ) + , + ( "int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } return r;" + , [((False, False, 1, 0), 5), ((False, False, 4, 0), 8)] + ) + , + ( "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } };" + , [((True, False, 1, 0), 5), ((False, False, 1, 0), 6), ((False, False, 2, 0), 7), ((False, False, 2, 3), 8)] + ) + , + ( "return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { int t = right * 3; t - 1 } };" + , [((False, False, 1, 4), 9), ((False, False, 2, 4), 11)] + ) + , -- A statement arm may continue or leave the enclosing loop. + ( "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 -> { continue; }, 5 -> { break; }, _ -> { total += i; } } } return total;" + , [((False, False, 10, 0), 8), ((False, False, 3, 0), 1), ((False, False, 0, 0), 0)] + ) + , -- A statement arm may supply the value of an enclosing loop expression. + ( "int n = 0; int r = while (true) { n += 1; match (n) { 4 -> { break n * 10; }, _ -> { } } }; return r;" + , [((False, False, 0, 0), 40)] + ) + , + ( "match (left) { 1 -> { return 100; }, _ -> { } } return 5;" + , [((False, False, 1, 0), 100), ((False, False, 2, 0), 5)] + ) + , + ( "match (left) { } return 5;" + , [((False, False, 1, 0), 5)] + ) + , + ( "int r = if (left > right) { left } else { right }; return r;" + , [((False, False, 3, 5), 5), ((False, False, 9, 2), 9)] + ) + , + ( "int r = if (flag) { int t = left * 2; t + 1 } else { int t = right * 3; t - 1 }; return r;" + , [((True, False, 4, 0), 9), ((False, False, 0, 5), 14)] + ) + , -- Only the selected block of an if expression runs. + ( "int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 }; return r * 1000 + n;" + , [((True, False, 0, 0), 10001), ((False, False, 0, 0), 20100)] + ) + , + ( "return (if (flag) { left } else { right }) + (if (other) { 100 } else { 200 });" + , [((True, True, 1, 2), 101), ((False, False, 1, 2), 202)] + ) + , + ( "guard (left > 0) else { return 0 - 1; } return left * 2;" + , [((False, False, 3, 0), 6), ((False, False, 0, 0), -1)] + ) + , + ( "int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == 0) else { continue; } total += i; } return total;" + , [((False, False, 6, 0), 6), ((False, False, 1, 0), 0)] + ) + , + ( "int n = 0; while (true) { guard (n < left) else { break; } n += 1; } return n;" + , [((False, False, 4, 0), 4), ((False, False, 0, 0), 0)] + ) + , + ( "int total = 0; { int part = left * 2; total += part; } { int part = right * 3; total += part; } return total;" + , [((False, False, 2, 3), 13), ((False, False, 0, 0), 0)] + ) + , + ( "int n = 0; while (true) { { n += 1; if (n > left) { break; } } } { { return n * 10; } }" + , [((False, False, 3, 0), 40), ((False, False, 0, 0), 10)] + ) + , + ( "int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { continue; } } { int step = i * 2; total += step; } } return total;" + , [((False, False, 4, 0), 10), ((False, False, 1, 0), 0)] + ) + , + ( "return match (left) { 1 -> 10 2 -> 20 _ -> 30 };" + , [((False, False, 1, 0), 10), ((False, False, 2, 0), 20), ((False, False, 9, 0), 30)] + ) + , -- A pattern binding is a local of its arm and may be assigned. + ( "return match (left) { int value if value > 2 -> { value = value * 2; value }, int value -> { value += 1; value } };" + , [((False, False, 5, 0), 10), ((False, False, 1, 0), 2)] + ) + , -- Assigning a binding does not change the subject the later arms test. + ( "int r = 0; match (left) { int value if (value = 7) > 9 -> { r = 1; }, 3 -> { r = 2; }, _ -> { r = 3; } } return r;" + , [((False, False, 3, 0), 2), ((False, False, 4, 0), 3)] + ) + , + ( "int r = if (flag) { if (other) { 1 } else { 2 } } else { match (left) { 1 -> 10, _ -> 20 } }; return r;" + , [((True, True, 0, 0), 1), ((True, False, 0, 0), 2), ((False, False, 1, 0), 10), ((False, False, 2, 0), 20)] + ) + , + ( "int n = 0; int r = if (flag) { if (other) { n += 1; } else { n += 2; } match (left) { 1 -> { n += 10; } } n } else { 0 }; return r;" + , [((True, True, 1, 0), 11), ((True, False, 2, 0), 2), ((False, False, 1, 0), 0)] + ) + , + ( "match (left) { 1 -> { match (right) { 2 -> { return 12; } } }, _ -> { } } return 5;" + , [((False, False, 1, 2), 12), ((False, False, 1, 3), 5), ((False, False, 2, 2), 5)] + ) + , -- A guard condition with a store runs once, before the block decision. + ( "int n = left; guard ((n += 1) > 3) else { return n * 10; } return n;" + , [((False, False, 5, 0), 6), ((False, False, 1, 0), 20)] + ) + ] + +-- | Expressions that leave instead of yielding a value. From @Compiler/Fuzzing/Cases/Leaving.cases@. +leavingCases :: [EvaluationCase] +leavingCases = + [ -- An expression every branch of which leaves never yields a value. + ( "int r = if (flag) { return 1; } else { return 2; }; return r + 50;" + , [((True, False, 0, 0), 1), ((False, False, 0, 0), 2)] + ) + , + ( "int r = match (left) { 0 -> { return 10; }, _ -> { return 20; } }; return r + 1;" + , [((False, False, 0, 0), 10), ((False, False, 5, 0), 20)] + ) + , + ( "return Twice(if (flag) { return 7; } else { return 9; });" + , [((True, False, 0, 0), 7), ((False, False, 0, 0), 9)] + ) + , + ( "int r = 1 + (if (flag) { return 1; } else { return 2; }); return r;" + , [((True, False, 0, 0), 1), ((False, False, 0, 0), 2)] + ) + , + ( "int t = 0; int i = 0; while (i < 5) { i += 1; int q = if (i > left) { break; } else { continue; }; t += q; } return t * 10 + i;" + , [((False, False, 2, 0), 3), ((False, False, 9, 0), 5)] + ) + , + ( "bool b = flag && (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" + , [((True, False, 5, 0), 1), ((True, False, 0, 0), 2), ((False, False, 5, 0), 4)] + ) + , + ( "bool b = flag || (if (left > 0) { return 1; } else { return 2; }); return b ? 3 : 4;" + , [((True, False, 5, 0), 3), ((False, False, 5, 0), 1), ((False, False, 0, 0), 2)] + ) + , + ( "int n = left ?: (if (flag) { return 100; } else { return 200; }); return n;" + , [((False, False, 5, 0), 5), ((True, False, 0, 0), 100), ((False, False, 0, 0), 200)] + ) + , + ( "int r = if (left > 0) { 5 } else { if (flag) { return 1; } else { return 2; } }; return r;" + , [((False, False, 3, 0), 5), ((True, False, 0, 0), 1), ((False, False, 0, 0), 2)] + ) + , + ( "int n = left; do { n += 1; } while (if (n > 3) { return n; } else { return 0 - n; }); return 99;" + , [((False, False, 5, 0), 6), ((False, False, 1, 0), -2)] + ) + , -- The condition of a do/while that never completes still runs after + -- the body and after a continue, with its effects, and is skipped by + -- a break. + ( "int t = 0; int n = 0; do { n += 1; if (n == left) { continue; } t += 10; } while (if ((t += 1) > 100) { return 1; } else { return t * 100 + n; }); return 99;" + , [((False, False, 1, 0), 101), ((False, False, 5, 0), 1101)] + ) + , + ( "int t = 0; do { if (left > 0) { break; } t += 5; } while (if ((t += 1) > 0) { return t; } else { return 0; }); return t + 1000;" + , [((False, False, 1, 0), 1000), ((False, False, 0, 0), 6)] + ) + , + ( "int t = 0; for (int i = 0; i < 3; i += 1) { do { t += 1; } while (if (t > left) { return t * 10 + i; } else { return t; }); } return 77;" + , [((False, False, 0, 0), 10), ((False, False, 5, 0), 1)] + ) + , + ( "int t = 0; int i = 0; while (i < 3) { i += 1; int j = 0; do { j += 1; if (j < 3) { continue; } t += j; } while (if (i > left) { return t * 10 + i; } else { j < 4 }); } return t;" + , [((False, False, 0, 0), 1), ((False, False, 9, 0), 21)] + ) + , -- A transfer in a value block targets the innermost loop only. + ( "int t = 0; int i = 0; while (i < 5) { i += 1; int j = 0; while (j < 3) { j += 1; t += if (j == 2) { break; } else { 1 }; } } return t * 10 + i;" + , [((False, False, 0, 0), 55)] + ) + , + ( "int t = 0; int i = 0; while (i < 5) { i += 1; int j = 0; while (j < 3) { j += 1; t += if (j == 2) { continue; } else { 1 }; } } return t * 10 + i;" + , [((False, False, 0, 0), 105)] + ) + , + ( "int t = 0; while (t < 10) { t += 1; int q = while (true) { break t; }; if (q > left) { break; } } return t;" + , [((False, False, 3, 0), 4), ((False, False, 20, 0), 10)] + ) + , -- A short-circuit operator evaluates its left operand, with its + -- effects, and reaches a right operand that never completes only + -- when the left one does not decide. + ( "int t = 0; bool b = (t += 1) > 0 && (if (left > 0) { return t * 10; } else { return t * 100; }); return 7;" + , [((False, False, 1, 0), 10), ((False, False, 0, 0), 100)] + ) + , + ( "int t = 0; bool b = (t += 1) > 5 && (if (left > 0) { return t * 10; } else { return t * 100; }); return t + (b ? 1000 : 0);" + , [((False, False, 1, 0), 1), ((False, False, 0, 0), 1)] + ) + , + ( "int t = 0; bool b = (t += left) > 0 || (if (flag) { return t * 10; } else { return 50 + t; }); return b ? t + 1 : 7;" + , [((False, False, 3, 0), 4), ((True, False, 0, 0), 0), ((False, False, 0, 0), 50)] + ) + , + ( "int t = 0; int q = (t += left) ? t * 2 : (if (flag) { return 0 - 1; } else { return 0 - 2; }); return q + t;" + , [((False, False, 4, 0), 12), ((True, False, 0, 0), -1), ((False, False, 0, 0), -2)] + ) + , -- Operands are evaluated left to right up to the one that never + -- completes; the ones after it are not evaluated. + ( "int t = 0; int r = (t += 1) + (if (flag) { return t * 10; } else { return t * 100; }) + (t += 50); return r;" + , [((True, False, 0, 0), 10), ((False, False, 0, 0), 100)] + ) + , + ( "int t = 0; return Twice(Twice(t += 3) + (if (flag) { return t; } else { return t * 2; }));" + , [((True, False, 0, 0), 3), ((False, False, 0, 0), 6)] + ) + , + ( "int t = left; t = Twice(t += 1) + (if (flag) { return t; } else { return t + 100; }); return 0 - 1;" + , [((True, False, 4, 0), 5), ((False, False, 4, 0), 105)] + ) + , -- A guard runs only when its patterns accept, and a guarded arm that + -- leaves does not make the match leave. + ( "int t = 0; int r = match (left) { 0 if (t += 1) > 5 -> { return 1; }, 0 -> { return 10 + t; }, _ -> { return 20 + t; } }; return r;" + , [((False, False, 0, 0), 11), ((False, False, 3, 0), 20)] + ) + , + ( "int r = match (left) { _ if flag -> { return 1; }, _ -> 5 }; return r + 1;" + , [((True, False, 0, 0), 1), ((False, False, 0, 0), 6)] + ) + , -- A break in the condition of a loop leaves that loop, once, with the + -- effects of the condition up to it. + ( "int n = 0; while (if (n >= left) { break; } else { true }) { n += 1; } return n * 10 + 1;" + , [((False, False, 3, 0), 31), ((False, False, 0, 0), 1)] + ) + , + ( "int c = 0; int n = 0; while (if ((c += 1) > left) { break; } else { true }) { n += 1; } return c * 100 + n;" + , [((False, False, 2, 0), 302), ((False, False, 0, 0), 100)] + ) + , + ( "int n = 0; do { n += 1; } while (if (n >= left) { break; } else { true }); return n;" + , [((False, False, 3, 0), 3), ((False, False, 0, 0), 1)] + ) + , + ( "int t = 0; for (int i = 0; if (i > left) { break; } else { i < 10 }; i += 1) { t += i; } return t;" + , [((False, False, 2, 0), 3), ((False, False, 20, 0), 45)] + ) + , + ( "int t = 0; for (int i = 0; i < 3; i += 1) { int j = 0; while (if (j == 2) { break; } else { true }) { j += 1; t += 1; } t += 10; } return t;" + , [((False, False, 0, 0), 36)] + ) + , + ( "int t = 0; int i = 0; while (i < 4) { i += 1; int j = 0; while (if (j >= i) { break; } else { true }) { j += 1; if (j == 2) { continue; } t += 1; } } return t * 10 + i;" + , [((False, False, 0, 0), 74)] + ) + , -- A continue in the update of a loop ends the update; the condition + -- is tested next, and the update is not run again for that pass. + ( "int t = 0; int skips = 0; for (int i = 0; i < 6; i += if (i == left && skips == 0) { skips += 1; continue; } else { 1 }) { t += 1; } return t * 10 + skips;" + , [((False, False, 2, 0), 71), ((False, False, 9, 0), 60)] + ) + , + ( "int t = 0; int u = 0; for (int i = 0; i < 4; i += 1, u += if (i == left) { continue; } else { 10 }) { t += 1; } return u * 10 + t;" + , [((False, False, 2, 0), 304), ((False, False, 9, 0), 404)] + ) + , + ( "int t = 0; for (int i = 0; i < 3; i += 1) { int c = 0; for (int j = 0; j < 4; j += if (c == 0 && j == left) { c += 1; continue; } else { 1 }) { t += 1; } t += c * 100; } return t;" + , [((False, False, 1, 0), 315), ((False, False, 7, 0), 12)] + ) + , -- A continue in the condition of a loop abandons the rest of the + -- condition and evaluates the condition of that loop again. The body + -- does not run in between, and neither does the update of a for. + ( "int c = 0; int n = 0; while (if ((c += 1) < left) { continue; } else { n < 2 }) { n += 1; } return c * 100 + n;" + , [((False, False, 3, 0), 502), ((False, False, 0, 0), 302)] + ) + , + ( "int u = 0; int c = 0; int t = 0; for (int i = 0; if ((c += 1) == left) { continue; } else { i < 3 }; i += 1, u += 1) { t += 10; } return c * 1000 + u * 100 + t;" + , [((False, False, 2, 0), 5330), ((False, False, 9, 0), 4330)] + ) + , + ( "int c = 0; int n = 0; do { n += 1; } while (if ((c += 1) < left) { continue; } else { n < 3 }); return c * 100 + n;" + , [((False, False, 3, 0), 503), ((False, False, 0, 0), 303)] + ) + , -- It targets the loop the condition belongs to, not one around it. + ( "int t = 0; int c = 0; for (int i = 0; i < 3; i += 1) { int j = 0; int once = 0; while (if (once == 0) { once = 1; c += 1; continue; } else { j < 2 }) { j += 1; t += 1; } t += 10; } return t * 100 + c;" + , [((False, False, 0, 0), 3603)] + ) + , -- A condition every path of which leaves: it repeats until it breaks. + ( "int c = 0; while (if ((c += 1) < left) { continue; } else { break; }) { c += 100; } return c;" + , [((False, False, 4, 0), 4), ((False, False, 0, 0), 1)] + ) + , + ( "int c = 0; int u = 0; for (int i = 0; if ((c += 1) < left) { continue; } else { break; }; i += 1, u += 1) { c += 100; } return c * 10 + u;" + , [((False, False, 4, 0), 40), ((False, False, 0, 0), 10)] + ) + , -- A break in the update of a loop leaves that loop, with the effects + -- of the update up to it and without the rest of the update. + ( "int t = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }) { t += 1; } return t;" + , [((False, False, 2, 0), 3), ((False, False, 20, 0), 10)] + ) + , + ( "int t = 0; int u = 0; for (int i = 0; i < 10; u += 1, i += if (i == left) { break; } else { 1 }, u += 100) { t += 1; } return u * 10 + t;" + , [((False, False, 1, 0), 1022), ((False, False, 20, 0), 10110)] + ) + , -- A break and a continue in one update clause. + ( "int t = 0; int s = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }, s += if (i == 2) { continue; } else { 1 }, s += 10) { t += 1; } return s * 100 + t;" + , [((False, False, 4, 0), 3305), ((False, False, 20, 0), 9910)] + ) + , -- It leaves the loop the update belongs to, not one around it. + ( "int t = 0; for (int a = 0; a < 3; a += 1) { for (int b = 0; b < 5; b += if (b == 1) { break; } else { 1 }) { t += 1; } t += 10; } return t;" + , [((False, False, 0, 0), 36)] + ) + , -- In the update of a loop used as an expression it carries the value. + ( "int r = for (int i = 0; ; i += if (i == left) { break i * 10; } else { 1 }) { if (i > 5) { break 99; } }; return r;" + , [((False, False, 3, 0), 30), ((False, False, 9, 0), 99)] + ) + , -- A value block may carry a value out of the loop expression around + -- it; the value goes to that loop, not to one around it. + ( "int r = while (true) { int q = if (flag) { break left + 1; } else { 2 }; break q; }; return r;" + , [((True, False, 4, 0), 5), ((False, False, 4, 0), 2)] + ) + , + ( "int r = while (true) { int q = match (left) { 0 -> { break 100; }, int n -> n * 2 }; break q; }; return r;" + , [((False, False, 0, 0), 100), ((False, False, 4, 0), 8)] + ) + , + ( "int t = 0; int r = for (int i = 0; ; i += 1) { int inner = while (true) { t += 1; int q = if (t > left) { break t * 2; } else { 0 }; t += q; }; if (inner > 0) { break inner + i; } }; return r * 10 + t;" + , [((False, False, 2, 0), 63), ((False, False, 0, 0), 21)] + ) + , + ( "int t = 0; int r = Twice(while (true) { t += 1; int q = (t += 10) + (if (t > left) { break t; } else { 1 }); t += q; }); return r * 100 + t;" + , [((False, False, 5, 0), 2211), ((False, False, 30, 0), 6834)] + ) + , -- A return leaves the method from a loop expression, directly and + -- through a value block. + ( "int r = while (true) { int q = if (flag) { return 77; } else { 2 }; break q + left; }; return r;" + , [((True, False, 0, 0), 77), ((False, False, 3, 0), 5)] + ) + , + ( "int r = while (true) { if (flag) { return 1; } break 2; }; return r + 10;" + , [((True, False, 0, 0), 1), ((False, False, 0, 0), 12)] + ) + , + ( "int r = while (true) { return left; }; return r + 1;" + , [((False, False, 6, 0), 6)] + ) + , + ( "int t = 0; int r = while (true) { t += 1; int q = for (int i = 0; ; i += 1) { if (i + t > left) { return i * 10 + t; } if (i == 2) { break i; } }; if (t == 3) { break q; } }; return 0 - r;" + , [((False, False, 1, 0), 11), ((False, False, 2, 0), 21), ((False, False, 99, 0), -2)] + ) + , -- A block used as a value may leave instead of yielding one. + ( "int r = if (left > 5) { return 100; } else { left * 2 }; return r + 1;" + , [((False, False, 9, 0), 100), ((False, False, 3, 0), 7)] + ) + , + ( "int r = match (left) { 0 -> { return 50; }, int n -> n + 1 }; return r * 2;" + , [((False, False, 0, 0), 50), ((False, False, 4, 0), 10)] + ) + , + ( "return Twice(if (flag) { return 7; } else { left });" + , [((True, False, 4, 0), 7), ((False, False, 4, 0), 8)] + ) + , + ( "int t = 0; int i = 0; while (i < 10) { i += 1; t += if (i > left) { break; } else { i }; } return t * 100 + i;" + , [((False, False, 3, 0), 604), ((False, False, 0, 0), 1)] + ) + , + ( "int t = 0; for (int i = 0; i < 6; i += 1) { t += match (i) { 2 -> { continue; }, int n -> { if (n == left) { continue; } else { n } } }; } return t;" + , [((False, False, 4, 0), 9), ((False, False, 9, 0), 13)] + ) + , -- A guard block that leaves through a match, and through continue. + ( "guard (left > 0) else { match (flag) { true -> { return 1; }, false -> { return 2; } } } return left + 10;" + , [((True, False, 0, 0), 1), ((False, False, 0, 0), 2), ((False, False, 5, 0), 15)] + ) + , + ( "int t = 0; int i = 0; while (i < left) { i += 1; guard (i \\= 2) else { continue; } t += i; } return t;" + , [((False, False, 4, 0), 8), ((False, False, 1, 0), 1)] + ) + , + ( "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } return r;" + , [((False, False, 1, 0), 10), ((False, False, 2, 0), 20), ((False, False, 7, 0), 0)] + ) + , + ( "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 -> n + 1, _ -> 0 };" + , [((False, False, 20, 0), 40), ((False, False, 7, 0), 8), ((False, False, 3, 0), 0)] + ) + , + ( "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 };" + , [((False, False, 1, 1), 11), ((False, False, 1, 5), 10), ((False, False, 4, 1), 1), ((False, False, 4, 4), 0)] + ) + , + ( "return match (flag) { true -> 1, false -> 2 };" + , [((True, False, 0, 0), 1), ((False, False, 0, 0), 2)] + ) + ] diff --git a/Compiler/Haskell/Driver/test/BranchingTests.hs b/Compiler/Haskell/Driver/test/BranchingTests.hs index ad20ae08..00e604d1 100644 --- a/Compiler/Haskell/Driver/test/BranchingTests.hs +++ b/Compiler/Haskell/Driver/test/BranchingTests.hs @@ -15,6 +15,7 @@ should not run is observed as a wrong result. -} module BranchingTests (branchingTests) where +import BranchingEvaluationCases (evaluationCases) import CoreInterpreter import Data.List (isInfixOf, isPrefixOf) import Visual.XSharp.AST @@ -97,12 +98,42 @@ parserTests = , ("the comma after a block body is optional", parses (body "match (left) { 1 -> { } 2 -> { } } return 0;")) , ("the comma after the last arm is optional", parses (body "return match (left) { 1 -> 10, _ -> 20, };")) , ("a match without arms parses", parses (body "match (left) { } return 0;")) + , ("the comma after an expression body is optional", parses (body "return match (left) { 1 -> 10 2 -> 20 _ -> 30 };")) , - ( "an expression body must be separated from the next arm" - , parseFailsWith "VXP0038" (body "return match (left) { 1 -> 10 _ -> 20 };") + ( "a parenthesized pattern after an expression body without a comma is read as a call" + , parseFailsWith "VXP0038" (body "return match (left) { 1 -> Twice (2) -> 20, _ -> 30 };") + ) + , ("a comma keeps a parenthesized pattern apart from the body before it", parses (body "return match (left) { 1 -> Twice(left), (2) -> 20, _ -> 30 };")) + , ("an if in last position of a value block is the value of the block", nestedIfIsValue) + , ("a match in last position of a value block is the value of the block", nestedMatchIsValue) + , ("an if that is not last in a value block is a statement", ifInsideValueBlockIsStatement) + , ("a match that is not last in a value block is a statement", matchInsideValueBlockIsStatement) + , + ( "a statement block inside a value block may end with a match" + , parses (body "int r = if (flag) { if (other) { match (left) { 1 -> { Twice(left); } } } 1 } else { 2 }; return r;") + ) + , + ( "a value in a block that is not used as a value needs its semicolon" + , parseFailsWith "VXP0006" (body "int r = if (flag) { if (other) { 1 } 2 } else { 3 }; return r;") ) , ("a bare name is not a pattern", parseFailsWith "VXP0036" (body "match (left) { right -> { } } return 0;")) - , ("a negative literal is not a pattern", parseFailsWith "VXP0036" (body "match (left) { -1 -> { } } return 0;")) + , ("a minus sign and a numeric literal are a pattern", parses (body "match (left) { -1 -> { } } return 0;")) + , ("a negative pattern may stand in parentheses", parses (body "match (left), (right) { (-1), (- 2) -> { } } return 0;")) + , ("a negative floating literal is a pattern", parses (body "double d = 1.5; return match (d) { -1.5 -> 1, _ -> 0 };")) + , ("a minus sign before a name is not a pattern", parseFailsWith "VXP0036" (body "match (left) { -right -> { } } return 0;")) + , ("a minus sign before a Boolean literal is not a pattern", parseFailsWith "VXP0036" (body "match (left) { -true -> { } } return 0;")) + , ("a minus sign before a parenthesis is not a pattern", parseFailsWith "VXP0036" (body "match (left) { -(1) -> { } } return 0;")) + , ("two minus signs are not a pattern", parseFailsWith "VXP0036" (body "match (left) { - -1 -> { } } return 0;")) + , ("an arithmetic expression is not a pattern", not (parses (body "match (left) { 1 + 1 -> { } } return 0;"))) + , ("a plus sign before a literal is not a pattern", not (parses (body "match (left) { +1 -> { } } return 0;"))) + , + ( "an if whose first block leaves is the value of a value block it ends" + , parses (body "int r = match (left) { 0 -> { if (flag) { return 9; } else { 4 } }, _ -> 1 }; return r;") + ) + , -- The body of a method is not a block used as a value. + ( "an if in last position of a method body stays a statement" + , parseFailsWith "VXP0006" (body "if (flag) { 1 } else { 2 }") + ) , ("an unterminated match is reported", not (parses (body "match (left) { 1 -> { }"))) , ("an if in operand position is a conditional over two value blocks", ifExpressionParses) , ("an if at the start of a statement stays an if statement", ifStatementKeepsItsNode) @@ -238,6 +269,33 @@ nestedBlockParses = case firstStatements (body "int a = 1; { int b = 2; a = b; } Just [BindingStatement {}, BlockStatement _ (Block [BindingStatement {}, AssignmentStatement {}]), ReturnStatement {}] -> True _ -> False +-- The statements of the first block of `int r = if (flag) { ... } else { 0 };`. +innerBlock :: String -> Maybe [Statement Identifier ()] +innerBlock inner = case firstStatements (body ("int r = if (flag) { " ++ inner ++ " } else { 0 }; return r;")) of + Just (BindingStatement _ _ _ _ () (ConditionalExpression _ _ (BlockExpression _ (Block statements) ()) _ ()) : _) -> + Just statements + _ -> Nothing + +nestedIfIsValue :: Bool +nestedIfIsValue = case innerBlock "if (other) { 1 } else { 2 }" of + Just [ExpressionStatement _ (ConditionalExpression _ _ (BlockExpression {}) (BlockExpression {}) ()) False] -> True + _ -> False + +nestedMatchIsValue :: Bool +nestedMatchIsValue = case innerBlock "match (left) { 1 -> 10, _ -> 20 }" of + Just [ExpressionStatement _ (MatchExpression {}) False] -> True + _ -> False + +ifInsideValueBlockIsStatement :: Bool +ifInsideValueBlockIsStatement = case innerBlock "if (other) { Twice(left); } else { Twice(right); } 1" of + Just [IfStatement _ _ (Block [ExpressionStatement _ _ True]) (Just (Block [ExpressionStatement _ _ True])), ExpressionStatement _ _ False] -> True + _ -> False + +matchInsideValueBlockIsStatement :: Bool +matchInsideValueBlockIsStatement = case innerBlock "match (left) { 1 -> { Twice(left); } } 1" of + Just [ExpressionStatement _ (MatchExpression {}) True, ExpressionStatement _ _ False] -> True + _ -> False + guardParses :: Bool guardParses = case firstStatements (body "guard (left > 0) else { return 0; } return 1;") of Just [GuardStatement _ (BinaryExpression _ GreaterThan _ _ ()) (Block [ReturnStatement {}]), _] -> True @@ -347,6 +405,44 @@ typeTests = ( "a type pattern must name the type of its subject" , rejectedWith "VXT0057" (body "match (left) { long value -> { } } return 0;") ) + , -- No numeric conversion is applied to a type pattern. + ( "a wider scalar type pattern does not make a match complete by conversion" + , rejectedWith "VXT0057" (body "return match (left) { long value -> 1 };") + ) + , + ( "a narrower scalar type pattern is rejected as well" + , rejectedWith "VXT0057" (body "return match (left) { short value -> 1, _ -> 0 };") + ) + , ("a type pattern of the subject's own type accepts every value", accepted (body "return match (left) { int value -> value };")) + , -- Negative constants are checked against the type of the subject. + ("the most negative value of a type is a pattern", accepted (body "return match (left) { -9223372036854775808 -> 1, _ -> 0 };")) + , + ( "a negative literal below the range of the subject is rejected" + , rejectedWith "VXT0016" (body "return match (left) { -9223372036854775809 -> 1, _ -> 0 };") + ) + , + ( "a positive literal above the range of the subject is still rejected" + , rejectedWith "VXT0016" (body "return match (left) { 9223372036854775808 -> 1, _ -> 0 };") + ) + , ("the range of a narrow signed subject includes its minimum", accepted (body "byte small = 1; return match (small) { -128 -> 1, 127 -> 2, _ -> 0 };")) + , + ( "a negative literal below the range of a narrow subject is rejected" + , rejectedWith "VXT0016" (body "byte small = 1; return match (small) { -129 -> 1, _ -> 0 };") + ) + , + ( "a positive literal above the range of a narrow subject is rejected" + , rejectedWith "VXT0016" (body "byte small = 1; return match (small) { 128 -> 1, _ -> 0 };") + ) + , + ( "a negative literal is outside every unsigned type" + , rejectedWith "VXT0016" (body "ubyte small = 1; return match (small) { -1 -> 1, _ -> 0 };") + ) + , ("a second arm for the same negative value is unreachable", rejectedWith "VXT0053" (body "return match (left) { -1 -> 1, -1 -> 2, _ -> 0 };")) + , ("minus zero is the value zero", rejectedWith "VXT0053" (body "return match (left) { 0 -> 1, -0 -> 2, _ -> 0 };")) + , + ( "a negative literal does not make a match complete" + , rejectedWith "VXT0052" (body "return match (left) { -1 -> 1, 0 -> 2, 1 -> 3 };") + ) , ("a literal pattern of another type is rejected", rejectedWith "VXT0054" (body "match (left) { \"a\" -> { } } return 0;")) , ("a guard must be bool or numeric", rejectedWith "VXT0049" (body "match (left) { 1 if \"x\" -> { } } return 0;")) , ("a numeric guard is accepted", accepted (body "match (left) { 1 if right -> { } } return 0;")) @@ -358,7 +454,23 @@ typeTests = ( "an effectful expression body in a statement match is accepted" , accepted (body "int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } return r;") ) - , ("a pattern binding is immutable", rejectedWith "VXT0003" (body "match (left) { int value -> { value = 5; } } return 0;")) + , ("a pattern binding may be assigned", accepted (body "match (left) { int value -> { value = 5; return value; } } return 0;")) + , + ( "a nested if in last position supplies the value of its block" + , accepted (body "int r = if (flag) { if (other) { 1 } else { 2 } } else { 3 }; return r;") + ) + , + ( "a match in last position supplies the value of its block" + , accepted (body "int r = if (flag) { match (left) { 1 -> 10, _ -> 20 } } else { 3 }; return r;") + ) + , + ( "a match in last position of a value block must accept every value" + , rejectedWith "VXT0052" (body "int r = if (flag) { match (left) { 1 -> 10 } } else { 3 }; return r;") + ) + , + ( "a match in last position of a statement arm is a statement" + , accepted (body "match (left) { 1 -> { match (right) { 2 -> { return 2; } } } } return 0;") + ) , ( "a pattern binding is in scope in its guard" , accepted (body "return match (left) { int value if value > 3 -> value, _ -> 0 };") @@ -396,20 +508,282 @@ typeTests = ) , ("an empty value block has no value", rejectedWith "VXT0046" (body "int r = if (flag) { } else { 2 }; return r;")) , - ( "a return inside a value block is rejected" - , rejectedWith "VXT0047" (body "int r = if (flag) { return 1; 2 } else { 3 }; return r;") + ( "a value block may return from the method" + , accepted (body "int r = if (flag) { return 1; } else { 3 }; return r;") + ) + , + ( "a return nested inside a value block is accepted" + , accepted (body "int r = match (left) { 1 -> { if (flag) { return 1; } 2 }, _ -> 3 }; return r;") + ) + , + ( "a return in a value block must carry the return type of the method" + , rejectedWith "VXT0005" (body "int r = if (flag) { return true; } else { 3 }; return r;") + ) + , + ( "a value block that leaves gives the if expression the type of the other block" + , rejectedWith "VXT0002" (body "bool r = if (flag) { return 1; } else { left }; return 0;") + ) + , + ( "an if expression whose blocks both leave is valid" + , accepted (body "int r = if (flag) { return 1; } else { return 2; }; return r;") + ) + , + ( "a match expression whose arms all leave is valid" + , accepted (body "int r = match (left) { 1 -> { return 1; }, _ -> { return 2; } }; return r;") + ) + , + ( "an expression that never completes is not held to the type of its receiver" + , accepted (body "bool r = if (flag) { return 1; } else { return 2; }; return r ? 1 : 0;") + ) + , + ( "an inferred binding may be initialized by an expression that never completes" + , accepted (body "auto r = if (flag) { return 1; } else { return 2; }; return 0;") + ) + , + ( "the returns of an expression that never completes are still checked" + , rejectedWith "VXT0005" (body "int r = if (flag) { return true; } else { return 2; }; return r;") + ) + , + ( "the statements after an expression that never completes are still checked" + , rejectedWith "VXT0012" (body "int r = if (flag) { return 1; } else { return 2; }; int s = r + true; return s;") + ) + , + ( "a match whose arms all leave must still be exhaustive" + , rejectedWith "VXT0052" (body "int r = match (left) { 1 -> { return 1; } }; return r;") + ) + , + ( "an expression that never completes may be an operand" + , accepted (body "int r = 1 + (if (flag) { return 1; } else { return 2; }); return r;") + ) + , + ( "an expression that never completes may be a condition" + , accepted (body "if (if (flag) { return 1; } else { return 2; }) { return 3; } return 4;") + ) + , + ( "a block that ends with an expression that never completes leaves" + , accepted (body "int r = if (left > 0) { 5 } else { if (flag) { return 1; } else { return 2; } }; return r;") + ) + , + ( "a right operand of && that never completes does not end the statement" + , rejectedWith "VXT0061" (body "guard (left > 0) else { bool b = flag && (if (other) { return 1; } else { return 2; }); } return 5;") + ) + , + ( "a right operand of || that never completes does not end the statement" + , rejectedWith "VXT0061" (body "guard (left > 0) else { bool b = flag || (if (other) { return 1; } else { return 2; }); } return 5;") + ) + , + ( "a conditional result that never completes does not end the statement" + , rejectedWith "VXT0061" (body "guard (left > 0) else { int q = flag ? 1 : (if (other) { return 1; } else { return 2; }); } return 5;") + ) + , + ( "a coalescing fallback that never completes does not end the statement" + , rejectedWith "VXT0061" (body "guard (left > 0) else { int q = right ?: (if (other) { return 1; } else { return 2; }); } return 5;") + ) + , + ( "a left operand of && that never completes ends the statement" + , accepted (body "guard (left > 0) else { bool b = (if (other) { return 1; } else { return 2; }) && flag; } return 5;") + ) + , + ( "an argument that never completes ends the statement" + , accepted (body "guard (left > 0) else { int q = Twice(if (other) { return 1; } else { return 2; }); } return 5;") + ) + , + ( "a match whose only leaving arm is guarded completes" + , rejectedWith "VXT0061" (body "guard (left > 0) else { match (right) { _ if flag -> { return 1; } } } return 5;") + ) + , + ( "a match with a guarded arm and a catch-all that both leave does not complete" + , accepted (body "guard (left > 0) else { match (right) { 1 if flag -> { return 1; }, _ -> { return 2; } } } return 5;") + ) + , + ( "an if without an else whose block leaves completes" + , rejectedWith "VXT0061" (body "guard (left > 0) else { if (flag) { return 1; } } return 5;") + ) + , + ( "a loop with a condition completes even when its body leaves" + , rejectedWith "VXT0061" (body "guard (left > 0) else { while (flag) { return 1; } } return 5;") + ) + , + ( "a break of an inner loop expression does not leave the endless loop around it" + , accepted (body "guard (left > 0) else { while (true) { int q = while (true) { break 1; }; } } return 5;") + ) + , + ( "a break in a value block of an inner loop does not leave the endless loop around it" + , accepted (body "guard (left > 0) else { while (true) { while (flag) { int q = if (other) { break; } else { 1 }; } } } return 5;") + ) + , + ( "a break in a match arm leaves the endless loop it stands in" + , rejectedWith "VXT0061" (body "guard (left > 0) else { while (true) { match (right) { 1 -> { break; }, _ -> { } } } } return 5;") + ) + , -- The statements kept and dropped around an operand that never completes. + ( "the operands before one that never completes keep their effects and the ones after it are dropped" + , case loweredBody (body "int t = 0; int r = Twice(t += 3) + (if (flag) { return t; } else { return 2; }) + Twice(9); return r;") of + Just [CoreBind _, CoreAssign _ _, CoreEvaluate (CoreApply _ _ _), CoreIf _ [CoreReturn _] [CoreReturn _]] -> True + _ -> False + ) + , + ( "a right operand of && that never completes lowers to a conditional without a slot" + , case loweredBody (body "bool b = flag && (if (other) { return 1; } else { return 2; }); return b ? 3 : 4;") of + Just (CoreIf _ [CoreIf _ [CoreReturn _] [CoreReturn _]] [] : rest) -> not (any bindsResultSlot rest) + _ -> False + ) + , + ( "a guard block may leave through an initializer that never completes" + , accepted (body "guard (left > 0) else { int q = if (flag) { return 1; } else { return 2; }; } return 5;") + ) + , -- Nothing is stored for a value that does not exist: no slot, no + -- binding, and none of the statements that are never reached. + ( "an initializer that never completes lowers to its branches alone" + , case loweredBody (body "int r = if (flag) { return 1; } else { return 2; }; return r;") of + Just [CoreIf _ [CoreReturn _] [CoreReturn _]] -> True + _ -> False + ) + , + ( "a returned expression that never completes lowers to its branches alone" + , case loweredBody (body "return Twice(if (flag) { return 1; } else { return 2; });") of + Just [CoreIf _ [CoreReturn _] [CoreReturn _]] -> True + _ -> False + ) + , + ( "a match that never completes lowers without a result slot" + , case loweredBody (body "int r = match (left) { 1 -> { return 1; }, _ -> { return 2; } }; return r;") of + Just statements -> not (any bindsResultSlot statements) + Nothing -> False + ) + , + ( "a value block that can complete must end with its value" + , rejectedWith "VXT0046" (body "int r = if (flag) { if (other) { return 1; } } else { 3 }; return r;") + ) + , + ( "an arm that leaves does not make a match complete" + , rejectedWith "VXT0052" (body "int r = match (left) { 1 -> { return 1; }, 2 -> 5 }; return r;") + ) + , + ( "a value block may break out of the loop around its expression" + , accepted (body "while (true) { int r = if (flag) { break; } else { 2 }; } return 0;") + ) + , + ( "a value block may continue the loop around its expression" + , accepted (body "while (flag) { int r = match (left) { 1 -> { continue; }, _ -> 2 }; } return 0;") + ) + , ("a break in a value block needs a loop", rejectedWith "VXT0025" (body "int r = if (flag) { break; } else { 2 }; return r;")) + , ("a continue in a value block needs a loop", rejectedWith "VXT0027" (body "int r = if (flag) { continue; } else { 2 }; return r;")) + , + ( "a break in a value block does not leave a loop from inside a closure" + , not (accepted (body "while (flag) { auto f = \\() -> { int r = if (other) { break; } else { 2 }; return r; }; } return 0;")) + ) + , + ( "a break in the condition of a loop leaves that loop" + , accepted (body "int n = 0; while (if (n > left) { break; } else { true }) { n += 1; } return n;") + ) + , ("a break in the condition of a loop needs no loop around it", accepted (body "while (if (flag) { break; } else { true }) { } return 0;")) + , + ( "a continue in the update of a loop is accepted" + , accepted (body "int t = 0; for (int i = 0; i < 3; i += if (t > 9) { t = 0; continue; } else { 1 }) { t += 5; } return t;") + ) + , + ( "a value block may carry a value out of the loop expression around it" + , accepted (body "int r = while (true) { int q = if (flag) { break 1; } else { 2 }; break q; }; return r;") ) , - ( "a return nested inside a value block is rejected" - , rejectedWith "VXT0047" (body "int r = match (left) { 1 -> { if (flag) { return 1; } 2 }, _ -> 3 }; return r;") + ( "a value block may return from a loop expression" + , accepted (body "int r = while (true) { int q = if (flag) { return 1; } else { 2 }; break q; }; return r;") + ) + , -- Real misuse of the same forms. + ( "a break that leaves a statement loop from a value block carries no value" + , rejectedWith "VXT0026" (body "while (flag) { int q = if (other) { break 1; } else { 2 }; } return 0;") ) , - ( "a break cannot leave a value block" - , rejectedWith "VXT0059" (body "while (true) { int r = if (flag) { break; 1 } else { 2 }; } return 0;") + ( "a break that leaves a loop expression from a value block must carry a value" + , rejectedWith "VXT0040" (body "int r = while (true) { int q = if (flag) { break; } else { 2 }; break q; }; return r;") ) , - ( "a continue cannot leave a value block" - , rejectedWith "VXT0059" (body "while (flag) { int r = match (left) { 1 -> { continue; 1 }, _ -> 2 }; } return 0;") + ( "the values carried out of value blocks must have the type of the other break values" + , rejectedWith "VXT0043" (body "long wide = 3; int r = while (true) { int q = if (flag) { break wide; } else { 2 }; break q; }; return r;") + ) + , + ( "a break in the condition of a statement loop carries no value" + , rejectedWith "VXT0026" (body "while (if (flag) { break 1; } else { true }) { } return 0;") + ) + , + ( "a return in a value block of a loop expression carries the return type of the method" + , rejectedWith "VXT0005" (body "int r = while (true) { int q = if (flag) { return true; } else { 2 }; break q; }; return r;") + ) + , + ( "a break in a value block inside a closure does not reach a loop expression around the closure" + , not (accepted (body "int r = while (true) { auto f = \\() -> { int q = if (flag) { break 1; } else { 2 }; return q; }; break f(); }; return r;")) + ) + , -- The condition of a loop expression is the constant true, so a break + -- cannot stand in it. + ( "a loop expression whose condition is not the constant true is still rejected" + , rejectedWith "VXT0041" (body "int t = 0; int r = while (if (t > left) { break t; } else { true }) { t += 1; }; return r;") + ) + , -- Return types are inferred through expressions, and not across the + -- edge of a nested callable. + ( "a callable infers its result from a return in a value block" + , accepted (body "auto f = \\(int v) -> { int q = if (v > 0) { return 1; } else { 2 }; return q; }; return f(left);") + ) + , + ( "the returns of a callable with an inferred result must agree, also through a value block" + , rejectedWith "VXT0062" (body "auto f = \\(int v) -> { int q = if (v > 0) { return true; } else { 2 }; return q; }; return 0;") + ) + , + ( "the returns of a callable do not count for the method that creates it" + , accepted (body "auto g = \\(int w) -> { bool b = if (w > 0) { return true; } else { false }; return b; }; return g(left) ? 1 : 2;") + ) + , + ( "the returns of a nested callable do not count for the callable around it" + , "VXT0062" + `notElem` codesOf + (body "auto f = \\(int v) -> { auto g = \\(int w) -> { return w > 0; }; int q = if (v > 0) { return 1; } else { 2 }; return q; }; return f(left);") + ) + , + ( "a method with an inferred return type infers it from a return in a value block" + , accepted + ( unlines + [ "class Program {" + , " public static auto Pick(_ int v) { int q = if (v > 0) { return 1; } else { 2 }; return q; }" + , "}" + ] + ) + ) + , -- Callables are compiled and verified, through Core and CorePrep, but + -- not run: neither the reference evaluator nor the JIT of the native + -- smoke program executes closures. + ( "a callable that returns from a loop expression is accepted" + , accepted (body "auto f = \\(int v) -> { int q = while (true) { if (v > 3) { return 7; } break 2; }; return q + v; }; return f(left);") + ) + , + ( "a callable every path of which returns from a value block is accepted" + , accepted (body "auto f = \\(int v) -> { int q = if (v > 0) { return v * 2; } else { return 0 - v; }; return q; }; return f(left);") + ) + , + ( "a break in a value block does not leave a loop from inside a callable in that loop" + , rejectedWith "VXT0025" (body "while (flag) { auto f = \\(int v) -> { int q = if (v > 0) { break; } else { 2 }; return q; }; } return 0;") + ) + , + ( "a continue in a value block does not reach a loop from inside a callable in that loop" + , rejectedWith "VXT0027" (body "while (flag) { auto f = \\(int v) -> { int q = if (v > 0) { continue; } else { 2 }; return q; }; } return 0;") + ) + , + ( "a continue in the condition of a loop is accepted" + , accepted (body "int n = 0; while (if ((n += 1) < left) { continue; } else { n < right }) { } return n;") + ) + , + ( "a break in the update of a loop is accepted" + , accepted (body "for (int i = 0; i < 3; i += if (flag) { break; } else { 1 }) { } return 0;") + ) + , + ( "a break in the update of a loop statement carries no value" + , rejectedWith "VXT0026" (body "for (int i = 0; i < 3; i += if (flag) { break 1; } else { 1 }) { } return 0;") + ) + , + ( "a break in the update of a loop expression carries a value" + , rejectedWith "VXT0040" (body "int r = for (int i = 0; ; i += if (flag) { break; } else { 1 }) { break 2; }; return r;") + ) + , + ( "a continue in the condition of a loop needs that loop and no other" + , rejectedWith "VXT0027" (body "int q = if (flag) { continue; } else { 2 }; return q;") ) , ( "a loop inside a value block may still be left" @@ -436,6 +810,41 @@ typeTests = ) , ("a guard cannot break outside a loop", rejectedWith "VXT0025" (body "guard (flag) else { break; } return 0;")) , ("a guard block may end with a nested block that leaves", accepted (body "guard (flag) else { { return 1; } } return 0;")) + , -- The rule is about control flow, not about the last statement. + ("a guard block may leave before its last statement", accepted (body "int n = left; guard (flag) else { return 1; n += 1; } return n;")) + , ("a guard block that never ends is accepted", accepted (body "guard (flag) else { while (true) { } } return 0;")) + , ("a guard block with an endless for loop is accepted", accepted (body "guard (flag) else { for (int i = 0; ; i += 1) { } } return 0;")) + , + ( "a guard block whose loop can be left is rejected" + , rejectedWith "VXT0061" (body "guard (flag) else { while (true) { if (other) { break; } } } return 0;") + ) + , + ( "a guard block whose loop is left from a value block is rejected" + , rejectedWith "VXT0061" (body "guard (flag) else { while (true) { int q = if (other) { break; } else { 1 }; } } return 0;") + ) + , + ( "a break of an inner loop does not leave the outer endless loop" + , accepted (body "guard (flag) else { while (true) { while (other) { break; } } } return 0;") + ) + , + ( "a guard block whose loop has a condition is rejected" + , rejectedWith "VXT0061" (body "guard (flag) else { while (other) { } } return 0;") + ) + , + ( "a guard block may end with a match all of whose arms leave" + , accepted (body "guard (flag) else { match (other) { true -> { return 1; }, false -> { return 2; } } } return 0;") + ) + , + ( "a guard block that ends with a match that may select no arm is rejected" + , rejectedWith "VXT0061" (body "guard (flag) else { match (left) { 1 -> { return 1; } } } return 0;") + ) + , + ( "a guard block that ends with a match one arm of which completes is rejected" + , rejectedWith "VXT0061" (body "guard (flag) else { match (left) { 1 -> { return 1; }, _ -> { } } } return 0;") + ) + , ("a guard may continue a loop", accepted (body "int n = 0; while (n < left) { n += 1; guard (n > 1) else { continue; } } return n;")) + , ("a guard cannot continue outside a loop", rejectedWith "VXT0027" (body "guard (flag) else { continue; } return 0;")) + , ("a binding in a guard condition is recognized and not implemented", parseFailsWith "VXP0035" (body "guard (auto n = left) else { return 0; } return 1;")) , ("a nested block sees the names declared before it", accepted (body "int a = left; { a += 1; } return a;")) , ( "a name declared in a nested block ends with the block" @@ -468,17 +877,17 @@ loweringTests = , ("a statement match lowers without a result slot", matchStatementLowers) , ("a catch-all arm ends the chain without a test", catchAllEndsChain) , ("patterns for several subjects are tested together", severalSubjectsLower) - , ("a guard after a literal is decided in its own slot", guardedLiteralLowers) + , ("a guard after a literal is the last operand of the arm's test", guardedLiteralLowers) + , ("a guard that stores is decided in its own slot", storingGuardLowers) , ("a guard on a catch-all is the test itself", guardedCatchAllLowers) , ("a pattern binding is bound to the subject before its guard", bindingLowers) , ("an if expression over pure blocks is one lazy conditional", pureIfExpressionLowers) , ("an if expression whose block has statements selects into a slot", ifExpressionWithStatementsLowers) , ("a guard runs its block when the condition is false", guardLowers) , ("the statements of a nested block join the enclosing sequence", nestedBlockLowers) - , ("a match within the nesting bound has no taken slot", not (usesTakenSlot (wideMatch 16))) - , ("a match beyond the nesting bound records the taken arm in a slot", usesTakenSlot (wideMatch 17)) - , ("a wide match nests no deeper than one group", all ((<= 18) . nestingOf . wideMatch) [17, 40, 200]) - , ("a wide match is split into groups in one sequence", wideMatchIsGrouped) + , ("a match of any width is one chain of conditionals", all (isOneChain . wideMatch) [2, 17, 40, 200]) + , ("a chain of arms is as deep as one arm", all ((== 1) . nestingOf . wideMatch) [2, 17, 40, 200]) + , ("the names of all arms are bound before the chain", bindingsPrecedeChain) ] -- | A match expression with the given number of literal arms and a catch-all. @@ -490,41 +899,51 @@ wideMatch count = ++ "_ -> 0 };" ) -usesTakenSlot :: String -> Bool -usesTakenSlot text = case loweredBody text of - Just statements -> or [generated "$taken" name | CoreBind (CoreBinding name _ _ _) <- statements] - Nothing -> False +-- | Whether the body is a subject, a result slot, one chain of conditionals and a return. +isOneChain :: String -> Bool +isOneChain text = case loweredBody text of + Just [CoreBind _, CoreBind _, chain@CoreIf {}, CoreReturn _] -> linksOnly chain + _ -> False + where + -- Every false branch is the next link until the body of the catch-all. + linksOnly statement = case statement of + CoreIf _ [CoreAssign _ _] [next@CoreIf {}] -> linksOnly next + CoreIf _ [CoreAssign _ _] [CoreAssign _ _] -> True + _ -> False --- | The deepest nesting of conditional statements in the lowered body. +{- | The deepest nesting of conditional statements in the lowered body, where +the links of a chain share one level: a false branch that is exactly one +conditional continues the chain, as it does in every stage after Core. +-} nestingOf :: String -> Int nestingOf text = maybe 0 depth (loweredBody text) where depth :: [CoreStatement] -> Int depth statements = maximum (0 : map statementDepth statements) statementDepth statement = case statement of + CoreIf _ whenTrue [next@CoreIf {}] -> max (1 + depth whenTrue) (statementDepth next) CoreIf _ whenTrue whenFalse -> 1 + max (depth whenTrue) (depth whenFalse) _ -> 0 --- Forty arms are three groups: the first chain, then two guarded groups. -wideMatchIsGrouped :: Bool -wideMatchIsGrouped = case loweredBody (wideMatch 40) of +bindingsPrecedeChain :: Bool +bindingsPrecedeChain = case loweredBody (body "return match (left) { int low if low < 3 -> low, int high if high > 9 -> high, _ -> 0 };") of Just [ CoreBind (CoreBinding subject _ False _) + , CoreBind (CoreBinding first _ True (CoreVariable firstSource _)) + , CoreBind (CoreBinding second _ True (CoreVariable secondSource _)) , CoreBind (CoreBinding slot _ True _) - , CoreBind (CoreBinding taken _ True (CoreLiteral (CoreBoolean False) _)) - , CoreIf _ [CoreAssign firstTaken _, CoreAssign firstStore _] _ - , CoreIf (CoreVariable secondTest _) [] [CoreIf {}] - , CoreIf (CoreVariable thirdTest _) [] [CoreIf {}] - , CoreReturn (CoreVariable result _) + , CoreIf _ [CoreAssign _ _] [CoreIf _ [CoreAssign _ _] [CoreAssign _ _]] + , CoreReturn _ ] -> - generated "$subject" subject + map spelling [first, second] == ["low", "high"] + && all (== subject) [firstSource, secondSource] && generated "$matched" slot - && generated "$taken" taken - && firstTaken == taken - && firstStore == slot - && secondTest == taken - && thirdTest == taken - && result == slot + _ -> False + +-- | Whether a statement binds the result slot of a match or of a choice. +bindsResultSlot :: CoreStatement -> Bool +bindsResultSlot statement = case statement of + CoreBind binding -> any (`isPrefixOf` identifierText (resolvedSpelling (coreBindingName binding))) ["$matched", "$selected"] _ -> False loweredBody :: String -> Maybe [CoreStatement] @@ -576,8 +995,9 @@ matchStatementLowers = case loweredBody (body "match (left) { 1 -> { return 1; } _ -> False catchAllEndsChain :: Bool +-- The match never completes, so the statement after it is not lowered. catchAllEndsChain = case loweredBody (body "match (left) { _ -> { return 7; } } return 0;") of - Just [CoreBind (CoreBinding subject _ False _), CoreReturn _, CoreReturn _] -> generated "$subject" subject + Just [CoreBind (CoreBinding subject _ False _), CoreReturn _] -> generated "$subject" subject _ -> False severalSubjectsLower :: Bool @@ -591,18 +1011,33 @@ severalSubjectsLower = case loweredBody (body "match (left), (right) { (1), (2) comparesWith first 1 firstTest && comparesWith second 2 secondTest && comparesWith second 3 laterTest _ -> False +-- The guard is the last operand of the arm's test, behind the short-circuit +-- conjunction, so it is evaluated only when the comparison holds. guardedLiteralLowers :: Bool guardedLiteralLowers = case loweredBody (body "match (left) { 1 if flag -> { return 1; } } return 0;") of Just [ CoreBind (CoreBinding subject _ False _) + , CoreIf (CorePrimitive CoreLogicalAnd [test, CoreVariable guard _] _) [CoreReturn _] [] + , CoreReturn _ + ] -> + comparesWith subject 1 test && spelling guard == "flag" + _ -> False + +-- A guard that stores into a local runs as statements, and only when the +-- comparison holds: the decision is taken in a slot of its own. +storingGuardLowers :: Bool +storingGuardLowers = case loweredBody (body "int hits = 0; match (left) { 1 if (hits += 1) > 0 -> { return 1; } } return hits;") of + Just + [ _ + , CoreBind (CoreBinding subject _ False _) , CoreBind (CoreBinding decision _ True (CoreLiteral (CoreBoolean False) _)) - , CoreIf test [CoreIf (CoreVariable guard _) [CoreAssign decided _] []] [] + , CoreIf test [CoreAssign stored _, CoreIf _ [CoreAssign decided _] []] [] , CoreIf (CoreVariable taken _) [CoreReturn _] [] , CoreReturn _ ] -> generated "$accepted" decision && comparesWith subject 1 test - && spelling guard == "flag" + && spelling stored == "hits" && decided == decision && taken == decision _ -> False @@ -616,7 +1051,7 @@ bindingLowers :: Bool bindingLowers = case loweredBody (body "match (left) { int value if value > 3 -> { return value; } } return 0;") of Just [ CoreBind (CoreBinding subject _ False _) - , CoreBind (CoreBinding bound _ False (CoreVariable source _)) + , CoreBind (CoreBinding bound _ True (CoreVariable source _)) , CoreIf (CorePrimitive CoreGreaterThan [CoreVariable tested _, _] _) [CoreReturn (CoreVariable returned _)] [] , CoreReturn _ ] -> @@ -652,123 +1087,6 @@ nestedBlockLowers = case loweredBody (body "int a = left; { int b = a + 1; a = b -- ------------------------------------------------------------- evaluation --- | Source bodies, the arguments to run them on, and the value each run must return. -evaluationCases :: [(String, [((Bool, Bool, Integer, Integer), Integer)])] -evaluationCases = - [ ("return match (left) { 1 -> 10, 2 -> 20, _ -> 30 };", [(plain 1 0, 10), (plain 2 0, 20), (plain 5 0, 30)]) - , - ( "int r = 0; match (left) { 1 -> { r = 10; }, 2 -> { r = 20; } } return r;" - , [(plain 1 0, 10), (plain 2 0, 20), (plain 7 0, 0)] - ) - , - ( "return match (left) { int n if n > 10 -> n * 2, int n if n > 5 -> n + 1, _ -> 0 };" - , [(plain 20 0, 40), (plain 7 0, 8), (plain 3 0, 0)] - ) - , - ( "return match (left), (right) { (1), (1) -> 11, (1), (_) -> 10, (_), (1) -> 1, (_), (_) -> 0 };" - , [(plain 1 1, 11), (plain 1 5, 10), (plain 4 1, 1), (plain 4 4, 0)] - ) - , ("return match (flag) { true -> 1, false -> 2 };", [(flags True False 0 0, 1), (flags False False 0 0, 2)]) - , -- Two bool subjects covered without a catch-all arm. - ( "return match (flag), (other) { (true), (_) -> 1, (false), (true) -> 2, (false), (false) -> 3 };" - , [(flags True True 0 0, 1), (flags True False 0 0, 1), (flags False True 0 0, 2), (flags False False 0 0, 3)] - ) - , - ( "return match (flag), (other) { (true), (true) -> 3, (true), (_) -> 2, (_), (true) -> 1, (_), (_) -> 0 };" - , [(flags True True 0 0, 3), (flags True False 0 0, 2), (flags False True 0 0, 1), (flags False False 0 0, 0)] - ) - , -- The subject is evaluated once, whichever arm accepts. - ( "int n = left; int r = match (n += 1) { 1 -> 100, 2 -> 200, _ -> 300 }; return r + n;" - , [(plain 0 0, 101), (plain 1 0, 202), (plain 5 0, 306)] - ) - , -- Subjects are evaluated left to right, before any arm is tested. - ( "int n = left; return match (n += 1), (n * 10) { (2), (20) -> 1, (_), (_) -> 0 };" - , [(plain 1 0, 1), (plain 2 0, 0)] - ) - , -- A guard runs only when the patterns of its arm accept. - ( "int calls = 0; int r = match (left) { 1 if (calls += 1) > 0 -> 10, 2 if (calls += 10) > 0 -> 20, _ -> 30 }; return r * 100 + calls;" - , [(plain 1 0, 1001), (plain 2 0, 2010), (plain 3 0, 3000)] - ) - , -- A guard that is false passes the value on to the arms after it. - ( "return match (left) { 1 if flag -> 1, 1 -> 2, _ -> 3 };" - , [(flags True False 1 0, 1), (flags False False 1 0, 2), (flags True False 9 0, 3)] - ) - , -- A numeric guard is tested in Boolean context. - ( "return match (left) { 1 if right -> 1, _ -> 0 };" - , [(plain 1 5, 1), (plain 1 0, 0), (plain 2 5, 0)] - ) - , -- Only the body of the accepting arm runs. - ( "int n = 0; int r = match (left) { 1 -> (n += 1), 2 -> (n += 10), _ -> (n += 100) }; return r * 1000 + n;" - , [(plain 1 0, 1001), (plain 2 0, 10010), (plain 3 0, 100100)] - ) - , ("return match (Twice(left)) { 4 -> 1, 6 -> 2, _ -> 0 };", [(plain 2 0, 1), (plain 3 0, 2), (plain 4 0, 0)]) - , ("long wide = 5; return match (wide) { 5 -> 1, _ -> 0 };", [(plain 0 0, 1)]) - , ("long wide = match (left) { 1 -> 10, _ -> 20 }; return wide > 15 ? 1 : 0;", [(plain 1 0, 0), (plain 2 0, 1)]) - , ("int r = 0; match (left) { 1 -> r = 5, _ -> r = Twice(left) } return r;", [(plain 1 0, 5), (plain 4 0, 8)]) - , - ( "return match (left) { 1 -> if (flag) { 5 } else { 6 }, _ -> match (right) { 0 -> 7, _ -> 8 } };" - , [(flags True False 1 0, 5), (flags False False 1 0, 6), (plain 2 0, 7), (plain 2 3, 8)] - ) - , - ( "return match (left) { 1 -> { int t = right * 2; t + 1 }, _ -> { int t = right * 3; t - 1 } };" - , [(plain 1 4, 9), (plain 2 4, 11)] - ) - , -- A statement arm may continue or leave the enclosing loop. - ( "int total = 0; for (int i = 0; i < left; i++) { match (i) { 2 -> { continue; }, 5 -> { break; }, _ -> { total += i; } } } return total;" - , [(plain 10 0, 8), (plain 3 0, 1), (plain 0 0, 0)] - ) - , -- A statement arm may supply the value of an enclosing loop expression. - ( "int n = 0; int r = while (true) { n += 1; match (n) { 4 -> { break n * 10; }, _ -> { } } }; return r;" - , [(plain 0 0, 40)] - ) - , ("match (left) { 1 -> { return 100; }, _ -> { } } return 5;", [(plain 1 0, 100), (plain 2 0, 5)]) - , ("match (left) { } return 5;", [(plain 1 0, 5)]) - , ("int r = if (left > right) { left } else { right }; return r;", [(plain 3 5, 5), (plain 9 2, 9)]) - , - ( "int r = if (flag) { int t = left * 2; t + 1 } else { int t = right * 3; t - 1 }; return r;" - , [(flags True False 4 0, 9), (flags False False 0 5, 14)] - ) - , -- Only the selected block of an if expression runs. - ( "int n = 0; int r = if (flag) { n += 1; 10 } else { n += 100; 20 }; return r * 1000 + n;" - , [(flags True False 0 0, 10001), (flags False False 0 0, 20100)] - ) - , - ( "return (if (flag) { left } else { right }) + (if (other) { 100 } else { 200 });" - , [(flags True True 1 2, 101), (flags False False 1 2, 202)] - ) - , ("guard (left > 0) else { return 0 - 1; } return left * 2;", [(plain 3 0, 6), (plain 0 0, -1)]) - , - ( "int total = 0; for (int i = 0; i < left; i++) { guard (i % 2 == 0) else { continue; } total += i; } return total;" - , [(plain 6 0, 6), (plain 1 0, 0)] - ) - , - ( "int n = 0; while (true) { guard (n < left) else { break; } n += 1; } return n;" - , [(plain 4 0, 4), (plain 0 0, 0)] - ) - , - ( "int total = 0; { int part = left * 2; total += part; } { int part = right * 3; total += part; } return total;" - , [(plain 2 3, 13), (plain 0 0, 0)] - ) - , - ( "int n = 0; while (true) { { n += 1; if (n > left) { break; } } } { { return n * 10; } }" - , [(plain 3 0, 40), (plain 0 0, 10)] - ) - , - ( "int total = 0; for (int i = 0; i < left; i++) { { if (i == 1) { continue; } } { int step = i * 2; total += step; } } return total;" - , [(plain 4 0, 10), (plain 1 0, 0)] - ) - , -- A guard condition with a store runs once, before the block decision. - ( "int n = left; guard ((n += 1) > 3) else { return n * 10; } return n;" - , [(plain 5 0, 6), (plain 1 0, 20)] - ) - ] - -plain :: Integer -> Integer -> (Bool, Bool, Integer, Integer) -plain = flags False False - -flags :: Bool -> Bool -> Integer -> Integer -> (Bool, Bool, Integer, Integer) -flags flag other left right = (flag, other, left, right) - evaluationTests :: [(String, Bool)] evaluationTests = concat diff --git a/Compiler/Haskell/Driver/test/ClosureTests.hs b/Compiler/Haskell/Driver/test/ClosureTests.hs index 4286cd95..46967d1c 100644 --- a/Compiler/Haskell/Driver/test/ClosureTests.hs +++ b/Compiler/Haskell/Driver/test/ClosureTests.hs @@ -62,8 +62,8 @@ closureTests = , ("Core verifier accepts a well-formed closure", coreVerifierAcceptsClosure) , ("Core verifier rejects mismatched closure type", coreVerifierRejectsTypeMismatch) , ("Core verifier rejects capture initializer mismatch", coreVerifierRejectsCaptureMismatch) - , ("Core wire v8 round-trips closure values and ownership", coreWireClosureRoundTrip) - , ("CorePrep wire v6 round-trips closure creation and ownership", corePrepWireClosureRoundTrip) + , ("Core wire v10 round-trips closure values and ownership", coreWireClosureRoundTrip) + , ("CorePrep wire v7 round-trips closure creation and ownership", corePrepWireClosureRoundTrip) , ("CorePrep verifier accepts converted closure", corePrepVerifierAcceptsClosure) , ("CorePrep verifier rejects primitive weak capture", corePrepVerifierRejectsWeakPrimitive) ] @@ -331,6 +331,7 @@ declarationCallable declaration = case declaration of TypeDeclaration {typeMembers = members} -> firstJust (map declarationCallable members) TemplateTypeDeclaration {typeMembers = members} -> firstJust (map declarationCallable members) FunctionDeclaration {declarationBody = body} -> blockCallable body + EnumDeclaration {} -> Nothing blockCallable :: Block name annotation -> Maybe (Expression name annotation) blockCallable (Block statements) = firstJust (map statementCallable statements) @@ -404,6 +405,7 @@ symbols expression = case expression of NameExpression _ name _ -> [resolvedSymbol name] LiteralExpression {} -> [] MemberAccessExpression _ receiver _ _ -> symbols receiver + MethodReferenceExpression _ receiver _ _ -> symbols receiver CallExpression _ callee arguments _ -> symbols callee ++ concatMap symbols arguments UnaryExpression _ _ value _ -> symbols value BinaryExpression _ _ left right _ -> symbols left ++ symbols right @@ -609,7 +611,7 @@ invalidPreparedWeakCapture = -- Keep a textual assertion near the wire tests so failures caused by an -- accidental version rollback explain themselves in the test output. _wireVersionContext :: String -_wireVersionContext = "closures require Core wire version 8 and CorePrep wire version 6" +_wireVersionContext = "closures require Core wire version 10 and CorePrep wire version 8" _diagnosticContext :: Diagnostic -> Bool _diagnosticContext diagnostic = "closure" `isInfixOf` diagnosticMessage diagnostic diff --git a/Compiler/Haskell/Driver/test/ConditionalExpressionTests.hs b/Compiler/Haskell/Driver/test/ConditionalExpressionTests.hs index f22b2e8f..223cafb2 100644 --- a/Compiler/Haskell/Driver/test/ConditionalExpressionTests.hs +++ b/Compiler/Haskell/Driver/test/ConditionalExpressionTests.hs @@ -269,7 +269,8 @@ typeTests = , ("a conditional test may be numeric", accepted (returning "left ? left : right")) , ("a conditional test must be bool or numeric", rejectedWith "VXT0036" (returning "\"text\" ? left : right")) , ("conditional results must have one type", rejectedWith "VXT0037" (returning "flag ? left : other")) - , ("conditional results are limited to scalars for now", rejectedWith "VXT0039" (body "bool same = (flag ? \"a\" : \"b\") == \"a\"; return 0;")) + , ("a conditional selects one of two strings", accepted (body "bool same = (flag ? \"a\" : \"b\") == \"a\"; return 0;")) + , ("a string and a number are not results of one conditional", rejectedWith "VXT0037" (body "bool same = (flag ? \"a\" : 1) == \"a\"; return 0;")) , ("a void call is not a conditional result", rejected (body "flag ? Touch() : Touch(); return 0;")) , ("a literal first result takes the type of the second", accepted (longBody "long chosen = flag ? 1 : wide; return 0;")) , ("a literal second result takes the type of the first", accepted (longBody "long chosen = flag ? wide : 1; return 0;")) diff --git a/Compiler/Haskell/Driver/test/ConsoleTests.hs b/Compiler/Haskell/Driver/test/ConsoleTests.hs new file mode 100644 index 00000000..84b4d3a7 --- /dev/null +++ b/Compiler/Haskell/Driver/test/ConsoleTests.hs @@ -0,0 +1,484 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Console output, strings and the output format grammar. + +@Spec\/StandardLibrary\/IO\/ConsoleIO.vxs@ specifies the console. A program +writes with @Console.Print@, @Console.Println@, @Console.Printf@ and +@Console.Printfn@, to standard error with @Console.Error@ and its three +siblings, and builds a string with @Console.Format@. A format is a +compile-time string, checked against its arguments when the program is +compiled: a conversion that does not exist, a flag a conversion does not +take, a missing argument and an argument of the wrong type are errors of the +program and not of a run. + +Every program here is run in the reference Core evaluator, on the Core the +frontend lowered and on the Core the optimizer left, and what it wrote is +compared with a string written by hand. The line terminator of the +reference is a line feed. + +Output is an effect. The second half of this module holds what follows from +that under evaluation by need: an expression that writes is evaluated where +it stands and in the order it is written, whether or not its value is ever +needed, while a value that only computes is still computed by need. +-} +module ConsoleTests (consoleTests) where + +import CoreInterpreter +import Data.List (isInfixOf) +import Visual.XSharp.Compiler +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep.Verifier +import Visual.XSharp.Core.Verifier +import Visual.XSharp.Diagnostic + +consoleTests :: [(String, Bool)] +consoleTests = + concat + [ [ ("unoptimized Core writes " ++ show expected ++ " for " ++ body, written artifactCore body == Just (expected, "")) + , ("optimized Core writes " ++ show expected ++ " for " ++ body, written artifactOptimizedCore body == Just (expected, "")) + , ("every stage verifies " ++ body, verifies body) + ] + | (body, expected) <- outputs + ] + ++ [ ("standard error receives " ++ show expected ++ " for " ++ body, written artifactCore body == Just ("", expected)) + | (body, expected) <- errors + ] + ++ [ ("rejected with " ++ code ++ ": " ++ body, rejectedWith code body) + | (body, code) <- rejections + ] + ++ [ ("both streams keep their own text", written artifactCore mixed == Just ("ac", "b\n")) + , ("the optimizer keeps both streams", written artifactOptimizedCore mixed == Just ("ac", "b\n")) + , -- A program that writes nothing has no runtime call in it. + ("a program without output has no runtime call", not (mentions "CoreRuntimeCall" artifactCore "int x = 1;")) + , ("a write is a runtime call", mentions "CoreRuntimeCall" artifactCore "Console.Print(\"a\");") + , -- The optimizer may not drop a write whose result nothing uses: + -- the write is the point. + ("the optimizer keeps a write", mentions "CoreRuntimeCall" artifactOptimizedCore "Console.Print(\"a\");") + , ("the optimizer keeps a write in a method nothing reads the result of", written artifactOptimizedCore "Log(4);" == Just ("4\n", "")) + , -- A value that only computes is still by need beside output. + ("a value that is never needed is not computed beside output", written artifactCore unusedFailure == Just ("ok\n", "")) + , ("the same holds after optimization", written artifactOptimizedCore unusedFailure == Just ("ok\n", "")) + , -- A binding whose initializer writes has no flag: it is not deferred. + ("a binding that writes is not deferred", not (mentions "$known" artifactCore "int x = Log(1); Console.Println(2);")) + , ("a binding that only computes is deferred as before", mentions "$known" artifactCore "int x = Half(8); if (Zero() > 0) { Console.Println(x); }") + , -- An argument that writes is not suspended; one that only + -- computes still is. + ("an argument that writes is not suspended", not (mentions "CoreMemoize" artifactCore "Console.Println(Pick(0, Log(5)));")) + , ("an argument that only computes is suspended as before", mentions "CoreMemoize" artifactCore "Console.Println(Pick(0, 8 / Zero()));") + , -- A class of the program named Console is the program's own. + ("a class named Console shadows the console", shadowed) + ] + where + mixed = "Console.Print(\"a\"); Console.Errorln(\"b\"); Console.Print(\"c\");" + unusedFailure = "int z = 1 / Zero(); Console.Println(\"ok\");" + +{- | Programs and what they write to standard output. The body is the body of +@Main@; the helpers are the methods of 'program'. +-} +outputs :: [(String, String)] +outputs = + -- Plain output. Print writes the text; Println ends the line after it. + [ ("Console.Print(\"a\"); Console.Print(\"b\");", "ab") + , ("Console.Println(\"Hello\");", "Hello\n") + , ("Console.Println(\"\");", "\n") + , ("Console.Print(\"\");", "") + , ("Console.Println(\"one\"); Console.Println(\"two\");", "one\ntwo\n") + , ("System.Console.Println(\"qualified\");", "qualified\n") + , -- A percent sign has no meaning outside a format. + ("Console.Println(\"%d\");", "%d\n") + , ("Console.Println(\"100%\");", "100%\n") + , -- A value that is not a string is written as text. + ("Console.Println(42);", "42\n") + , ("Console.Println(0);", "0\n") + , ("Console.Println(0 - 7);", "-7\n") + , ("Console.Println(true);", "true\n") + , ("Console.Println(false);", "false\n") + , ("Console.Println('x');", "x\n") + , ("Console.Println(Half(10));", "5\n") + , ("int n = 9223372036854775807; Console.Println(n);", "9223372036854775807\n") + , ("uint u = 18446744073709551615; Console.Println(u);", "18446744073709551615\n") + , ("byte small = 100; Console.Println(small);", "100\n") + , ("ushort medium = 65535; Console.Println(medium);", "65535\n") + , -- A string is a value: bound, passed, returned. + ("String s = \"kept\"; Console.Println(s);", "kept\n") + , ("Console.Println(Name());", "Visual X#\n") + , ("Console.Println(Twice(\"ab\"));", "abab\n") + , -- + joins two strings, and writes a value that is not one as text. + ("Console.Println(\"Hello from \" + \"Visual X#\" + \"!\");", "Hello from Visual X#!\n") + , ("String language = \"Visual X#\"; Console.Println(\"Hello from \" + language + \"!\");", "Hello from Visual X#!\n") + , ("Console.Println(\"n=\" + 5);", "n=5\n") + , ("Console.Println(5 + \"n\");", "5n\n") + , ("Console.Println(\"b=\" + true);", "b=true\n") + , ("Console.Println(\"c=\" + 'z');", "c=z\n") + , ("Console.Println(\"\" + (1 + 2));", "3\n") + , -- + associates to the left: the string takes the 1, then the 2. + ("Console.Println(\"a\" + 1 + 2);", "a12\n") + , ("Console.Println(1 + 2 + \"a\");", "3a\n") + , ("uint u = 7; Console.Println(\"u=\" + u);", "u=7\n") + , ("Console.Println(\"\" + \"\");", "\n") + , ("String t = \"x\"; t += \"y\"; t += 3; Console.Println(t);", "xy3\n") + , ("String t = \"\"; for (int i = 0; i < 4; i += 1) { t += i; } Console.Println(t);", "0123\n") + , -- Strings are compared by what they hold. + ("Console.Println(\"a\" == \"a\");", "true\n") + , ("Console.Println(\"a\" == \"b\");", "false\n") + , ("Console.Println(\"a\" \\= \"b\");", "true\n") + , ("Console.Println(\"a\" \\= \"a\");", "false\n") + , ("String a = \"ab\"; String b = \"a\" + \"b\"; Console.Println(a == b);", "true\n") + , ("String a = \"ab\"; String b = \"a\" + \"c\"; Console.Println(a == b);", "false\n") + , ("Console.Println(\"\" == \"\");", "true\n") + , ("Console.Println(\"a\" == \"ab\");", "false\n") + , ("if (Name() == \"Visual X#\") { Console.Println(\"same\"); } else { Console.Println(\"other\"); }", "same\n") + , -- Printf writes its format; Printfn ends the line after it. + ("Console.Printf(\"plain\");", "plain") + , ("Console.Printfn(\"plain\");", "plain\n") + , ("Console.Printf(\"\");", "") + , ("Console.Printfn(\"\");", "\n") + , ("Console.Printf(\"%d\", 42);", "42") + , ("Console.Printfn(\"%d\", 42);", "42\n") + , ("Console.Printf(\"Count: %d\", 3);", "Count: 3") + , ("Console.Printf(\"%d and %d\", 1, 2);", "1 and 2") + , ("Console.Printfn(\"gcd(%d, %d) = %d\", 1071, 462, 21);", "gcd(1071, 462) = 21\n") + , -- %d: a signed integer in decimal. + ("Console.Printf(\"%d\", 0 - 255);", "-255") + , ("Console.Printf(\"%5d|\", 42);", " 42|") + , ("Console.Printf(\"%-5d|\", 42);", "42 |") + , ("Console.Printf(\"%05d\", 42);", "00042") + , ("Console.Printf(\"%05d\", 0 - 42);", "-0042") + , ("Console.Printf(\"%08d\", 123);", "00000123") + , ("Console.Printf(\"%+d\", 42);", "+42") + , ("Console.Printf(\"%+d\", 0 - 42);", "-42") + , ("Console.Printf(\"% d\", 42);", " 42") + , ("Console.Printf(\"%2d\", 12345);", "12345") + , -- The ' flag groups the integer digits in threes with apostrophes. + ("Console.Printf(\"%'d\", 1234567);", "1'234'567") + , ("Console.Printf(\"%'d\", 123);", "123") + , ("Console.Printf(\"%'d\", 1234);", "1'234") + , ("Console.Printf(\"%'d\", 0 - 1234567);", "-1'234'567") + , ("byte small = 0 - 100; Console.Printf(\"%d\", small);", "-100") + , -- %x: hexadecimal. A negative number is a minus sign and its + -- magnitude, never the bit pattern of its representation. + ("Console.Printf(\"%x\", 255);", "ff") + , ("Console.Printf(\"%x\", 0 - 255);", "-ff") + , ("Console.Printf(\"%#x\", 255);", "0xff") + , ("Console.Printf(\"%#x\", 0 - 255);", "-0xff") + , ("Console.Printf(\"%08x\", 48879);", "0000beef") + , ("Console.Printf(\"%#08x\", 255);", "0x0000ff") + , ("uint u = 4294967295; Console.Printf(\"%x\", u);", "ffffffff") + , ("Console.Printf(\"%x\", 0);", "0") + , -- %u: an unsigned integer. + ("uint u = 42; Console.Printf(\"%u\", u);", "42") + , ("Console.Printf(\"%u\", 42);", "42") + , ("ubyte tiny = 255; Console.Printf(\"%u\", tiny);", "255") + , ("uint u = 1234567; Console.Printf(\"%'u\", u);", "1'234'567") + , -- %s and %c. + ("Console.Printf(\"%s\", \"abc\");", "abc") + , ("Console.Printf(\"%10s|\", \"abc\");", " abc|") + , ("Console.Printf(\"%-10s|\", \"abc\");", "abc |") + , ("Console.Printf(\"%.2s\", \"abcdef\");", "ab") + , ("Console.Printf(\"%.5s\", \"abc\");", "abc") + , ("Console.Printf(\"%5.1s|\", \"abc\");", " a|") + , ("Console.Printf(\"%s: %d\", \"total\", 7);", "total: 7") + , ("Console.Printf(\"%s\", Name());", "Visual X#") + , ("Console.Printf(\"%c\", 'q');", "q") + , ("Console.Printf(\"%3c|\", 'q');", " q|") + , ("Console.Printf(\"%-3c|\", 'q');", "q |") + , -- %b: a Boolean. + ("Console.Printf(\"%b\", true);", "true") + , ("Console.Printf(\"%b\", 1 > 2);", "false") + , ("Console.Printf(\"%6b|\", false);", " false|") + , ("Console.Printf(\"%-6b|\", true);", "true |") + , -- %% is a percent sign and %n the line terminator; neither takes an + -- argument. + ("Console.Printf(\"100%%\");", "100%") + , ("Console.Printf(\"%d%%\", 50);", "50%") + , ("Console.Printf(\"A%nB\");", "A\nB") + , ("Console.Printf(\"%n\");", "\n") + , -- %f: a fixed number of digits after the point, six unless a + -- precision says otherwise; a tie rounds to the even digit. + ("Console.Printf(\"%f\", 12.5);", "12.500000") + , ("Console.Printf(\"%.2f\", 12.5);", "12.50") + , ("Console.Printf(\"%.0f\", 2.5);", "2") + , ("Console.Printf(\"%.0f\", 3.5);", "4") + , ("Console.Printf(\"%.1f\", 0.25);", "0.2") + , ("Console.Printf(\"%.1f\", 0.75);", "0.8") + , ("Console.Printf(\"%.3f\", 3.14159);", "3.142") + , ("Console.Printf(\"%10.3f|\", 3.14159);", " 3.142|") + , ("Console.Printf(\"%-10.3f|\", 3.14159);", "3.142 |") + , ("Console.Printf(\"%010.3f\", 3.14159);", "000003.142") + , ("Console.Printf(\"%+.1f\", 1.5);", "+1.5") + , ("Console.Printf(\"%'.2f\", 1234567.5);", "1'234'567.50") + , ("Console.Printf(\"%.2f\", 0.0);", "0.00") + , -- The digits are those of the binary number, not of the literal. + ("Console.Printf(\"%.20f\", 0.1);", "0.10000000000000000555") + , ("Console.Printf(\"%.1f\", 1000000.0);", "1000000.0") + , ("float price = 19.99; Console.Printfn(\"Price: %.2f\", price);", "Price: 19.99\n") + , -- A width or a precision written as * is the int argument before + -- the value. + ("Console.Printf(\"%*d|\", 5, 42);", " 42|") + , ("Console.Printf(\"%-*d|\", 5, 42);", "42 |") + , ("Console.Printf(\"%*s|\", 6, \"ab\");", " ab|") + , ("Console.Printf(\"%.*f\", 2, 3.14159);", "3.14") + , ("Console.Printf(\"%*.*f|\", 8, 2, 3.14159);", " 3.14|") + , ("int width = 4; Console.Printf(\"%*d|\", width, 7);", " 7|") + , -- Console.Format applies a format and writes nothing. + ("String t = Console.Format(\"%d-%d\", 1, 2); Console.Print(t);", "1-2") + , ("String t = Console.Format(\"Name: %s, Age: %d\", \"Ada\", 36); Console.Println(t);", "Name: Ada, Age: 36\n") + , ("String t = Console.Format(\"plain\"); Console.Print(t + \"!\");", "plain!") + , ("Console.Print(Console.Format(\"%05d\", 42) + Console.Format(\"%x\", 255));", "00042ff") + , ("String t = Console.Format(\"%d\", 1); Console.Println(\"x\");", "x\n") + , -- Output in loops, branches and methods. + ("for (int i = 0; i < 3; i += 1) { Console.Print(i); }", "012") + , ("for (int i = 1; i <= 3; i += 1) { Console.Printfn(\"%d squared is %d\", i, i * i); }", "1 squared is 1\n2 squared is 4\n3 squared is 9\n") + , ("if (Half(4) == 2) { Console.Println(\"yes\"); } else { Console.Println(\"no\"); }", "yes\n") + , ("Greet(\"Ada\"); Greet(\"Alan\");", "Hello, Ada!\nHello, Alan!\n") + , ("Console.Println(Fizz(3)); Console.Println(Fizz(5)); Console.Println(Fizz(15)); Console.Println(Fizz(7));", "Fizz\nBuzz\nFizzBuzz\n7\n") + , -- Output is an effect: it happens where it is written and in the + -- order it is written, whether or not a value is needed. + ("int x = Log(1); Console.Println(2);", "1\n2\n") + , ("int x = Log(1); int y = Log(2); Console.Println(y + x);", "1\n2\n3\n") + , ("int x = Log(1) + Log(2);", "1\n2\n") + , ("Console.Println(Pick(0, Log(5)));", "5\n7\n") + , ("Console.Println(Pick(1, Log(5)));", "5\n5\n") + , ("Console.Printf(\"%d %d\", Log(1), Log(2));", "1\n2\n1 2") + , -- The width of a conversion is written before its value and is + -- evaluated before it. + ("Console.Printf(\"%*d\", Log(3), Log(4));", "3\n4\n 4") + , ("Console.Println(\"a\" + Log(1) + Log(2));", "1\n2\na12\n") + , ("_ = Log(8);", "8\n") + , ("Log(8);", "8\n") + , ("int v = Zero() > 0 ? Log(1) : Log(2);", "2\n") + , ("bool both = Log(0) > 0 && Log(1) > 0;", "0\n") + , ("auto say = \\(int v) -> Log(v); int x = say(1); Console.Println(9);", "1\n9\n") + , ("int total = 0; for (int i = 0; i < 3; i += 1) { total += Log(i); } Console.Println(total);", "0\n1\n2\n3\n") + , ("int x = Twofold(3); Console.Println(\"after\");", "3\n3\nafter\n") + , -- A callable may write itself, and reads what it captured. + ("auto say = \\(int v) -> Console.Println(v); say(4); say(5);", "4\n5\n") + , ("auto say = \\(String s) -> Console.Printfn(\"<%s>\", s); say(\"a\");", "\n") + , ("int base = 3; auto say = \\(int v) -> { Console.Println(v + base); return v; }; int x = say(1); Console.Println(\"end\");", "4\nend\n") + , -- A value that only computes is still computed by need. + ("int z = 1 / Zero(); Console.Println(\"ok\");", "ok\n") + , ("int z = 8 / Zero(); Console.Println(Pick(0, z));", "7\n") + , ("String s = \"x\"; int z = 8 / Zero(); Console.Println(s);", "x\n") + , -- Strings that callables take, capture, make and return. + ("auto greet = \\(String who) -> \"Hi \" + who; Console.Println(greet(\"Ada\"));", "Hi Ada\n") + , ("String prefix = Name() + \": \"; auto label = \\(int n) -> prefix + n; Console.Println(label(1)); Console.Println(label(2));", "Visual X#: 1\nVisual X#: 2\n") + , ("auto twice = \\(int v) -> { Console.Printf(\"%d,\", v); return v * 2; }; Console.Println(twice(twice(1)));", "1,2,4\n") + , ("auto make = \\(int n) -> Console.Format(\"<%03d>\", n); String all = \"\"; for (int i = 0; i < 3; i += 1) { all += make(i); } Console.Println(all);", "<000><001><002>\n") + , ("String kept = Twice(\"ab\"); auto never = \\(int n) -> kept + n; Console.Println(\"x\");", "x\n") + , -- A method is a callable value: it writes where its call stands. + ("auto f = Log; int x = f(1); Console.Println(9);", "1\n9\n") + , -- Appending makes a new string; another name keeps the old one. + ("String s = \"a\"; String t = s; s += \"b\"; Console.Println(s + t);", "aba\n") + , ("String s = \"\"; for (int i = 0; i < 5; i += 1) { String old = s; s += i; if (old == s) { s += \"!\"; } } Console.Println(s);", "01234\n") + , -- String operations in arguments that are and are not needed. + ("Console.Println(Pick(1, Half(8)) + Name());", "4Visual X#\n") + , ("String made = Twice(Name() + \"!\"); Console.Println(Pick(0, 8 / Zero()));", "7\n") + , ("Console.Println(Console.Format(\"[%s]\", Console.Format(\"%5s\", Console.Format(\"%d\", 42))));", "[ 42]\n") + , ("Console.Println(Twice(Twice(\"ab\")) == \"abababab\");", "true\n") + , -- A conditional selects one of two strings; the other is never made. + ("String s = Zero() > 0 ? \"a\" : \"b\"; Console.Println(s);", "b\n") + , ("Console.Println(Zero() == 0 ? Name() : Twice(\"x\"));", "Visual X#\n") + , ("String a = \"x\"; String b = \"y\"; String c = Zero() > 0 ? a : b; Console.Println(c + a + b);", "yxy\n") + , ("Console.Println(Zero() > 0 ? \"a\" : Zero() == 0 ? \"b\" : \"c\");", "b\n") + , ("String s = Zero() > 0 ? \"\" + Log(1) : \"\" + Log(2); Console.Println(s);", "2\n2\n") + , ("String s = \"\"; for (int i = 0; i < 4; i += 1) { s += i % 2 == 0 ? \"e\" : \"o\"; } Console.Println(s);", "eoeo\n") + , ("Console.Println(\"<\" + (Zero() == 0 ? \"yes\" : \"no\") + \">\");", "\n") + , ("Greet(Zero() == 0 ? Name() : \"nobody\");", "Hello, Visual X#!\n") + , ("String s = \"keep\"; s = Zero() > 0 ? \"lost\" : s; Console.Println(s);", "keep\n") + ] + +-- | Programs and what they write to standard error. +errors :: [(String, String)] +errors = + [ ("Console.Error(\"e\");", "e") + , ("Console.Errorln(\"failed\");", "failed\n") + , ("Console.Error(\"a\"); Console.Errorln(\"b\");", "ab\n") + , ("Console.Errorf(\"Code: %d\", 7);", "Code: 7") + , ("Console.Errorfn(\"Code: %d\", 7);", "Code: 7\n") + , ("Console.Errorln(404);", "404\n") + , ("System.Console.Errorln(\"q\");", "q\n") + ] + +-- | Programs the type checker rejects, with the code it reports. +rejections :: [(String, String)] +rejections = + -- The console has the methods the specification gives it. + [ ("Console.Shout(\"x\");", "VXT0071") + , ("Console.println(\"x\");", "VXT0071") + , ("System.Console.Nothing();", "VXT0071") + , -- Print takes one value. + ("Console.Print();", "VXT0072") + , ("Console.Println();", "VXT0072") + , ("Console.Print(\"a\", \"b\");", "VXT0072") + , ("Console.Printf();", "VXT0072") + , ("Console.Format();", "VXT0072") + , -- What cannot be written as text. + ("auto f = \\(int v) -> v; Console.Println(f);", "VXT0073") + , ("auto f = \\(int v) -> v; String s = \"a\" + f;", "VXT0073") + , -- A format is a string literal. + ("String f = \"%d\"; Console.Printf(f, 1);", "VXT0074") + , ("Console.Printf(Name());", "VXT0074") + , ("Console.Printf(\"%d\" + \"%d\", 1, 2);", "VXT0074") + , ("Console.Format(42);", "VXT0074") + , -- A conversion that does not exist, or is not finished. + ("Console.Printf(\"%q\", 1);", "VXT0075") + , ("Console.Printf(\"%\");", "VXT0075") + , ("Console.Printf(\"abc%\");", "VXT0075") + , ("Console.Printf(\"%5\");", "VXT0075") + , ("Console.Printf(\"%.d\", 1);", "VXT0075") + , ("Console.Printf(\"%D\", 1);", "VXT0075") + , ("Console.Printf(\"%i\", 1);", "VXT0075") + , ("Console.Printf(\"%e\", 1.5);", "VXT0075") + , ("Console.Printf(\"%1234567890d\", 1);", "VXT0075") + , -- A flag a conversion does not take, and flags that exclude each + -- other; examples 16 to 18 of the specification file. + ("Console.Printf(\"%+s\", \"a\");", "VXT0076") + , ("Console.Printf(\"%0s\", \"a\");", "VXT0076") + , ("Console.Printf(\"%#d\", 1);", "VXT0076") + , ("Console.Printf(\"%'x\", 1);", "VXT0076") + , ("Console.Printf(\"%-08d\", 1);", "VXT0076") + , ("Console.Printf(\"%+ d\", 1);", "VXT0076") + , ("Console.Printf(\"%++d\", 1);", "VXT0076") + , ("Console.Printf(\"%+u\", 1);", "VXT0076") + , ("Console.Printf(\"% u\", 1);", "VXT0076") + , ("Console.Printf(\"%#s\", \"a\");", "VXT0076") + , ("Console.Printf(\"%'s\", \"a\");", "VXT0076") + , ("Console.Printf(\"%0c\", 'a');", "VXT0076") + , ("Console.Printf(\"%+b\", true);", "VXT0076") + , ("Console.Printf(\"%#f\", 1.5);", "VXT0076") + , ("Console.Printf(\"%.2d\", 1);", "VXT0076") + , ("Console.Printf(\"%.2x\", 1);", "VXT0076") + , ("Console.Printf(\"%.1c\", 'a');", "VXT0076") + , ("Console.Printf(\"%.1b\", true);", "VXT0076") + , ("Console.Printf(\"%5%\");", "VXT0076") + , ("Console.Printf(\"%-n\");", "VXT0076") + , -- The arguments are the ones the format names, no more and no fewer. + ("Console.Printf(\"%d\");", "VXT0077") + , ("Console.Printf(\"%d\", 1, 2);", "VXT0077") + , ("Console.Printf(\"plain\", 1);", "VXT0077") + , ("Console.Printf(\"%d %d\", 1);", "VXT0077") + , ("Console.Printf(\"%*d\", 1);", "VXT0077") + , ("Console.Printf(\"%*.*f\", 1, 2);", "VXT0077") + , ("Console.Printf(\"%n\", 1);", "VXT0077") + , ("Console.Printf(\"%%\", 1);", "VXT0077") + , ("Console.Format(\"%s\");", "VXT0077") + , -- An argument has the type its conversion takes. Nothing is + -- converted for formatting: examples 11 and 12. + ("uint value = 42; Console.Printf(\"%d\", value);", "VXT0078") + , ("int value = 42; Console.Printf(\"%u\", value);", "VXT0078") + , ("int value = 42; Console.Printf(\"%f\", value);", "VXT0078") + , ("float value = 1.5; Console.Printf(\"%d\", value);", "VXT0078") + , ("Console.Printf(\"%s\", 42);", "VXT0078") + , ("Console.Printf(\"%d\", \"text\");", "VXT0078") + , ("Console.Printf(\"%c\", 65);", "VXT0078") + , ("Console.Printf(\"%c\", \"a\");", "VXT0078") + , ("Console.Printf(\"%b\", 1);", "VXT0078") + , ("Console.Printf(\"%d\", true);", "VXT0078") + , ("Console.Printf(\"%x\", 1.5);", "VXT0078") + , ("Console.Printf(\"%x\", \"ff\");", "VXT0078") + , ("Console.Printf(\"%*d\", true, 1);", "VXT0078") + , ("Console.Printf(\"%*d\", \"5\", 1);", "VXT0078") + , ("uint width = 5; Console.Printf(\"%*d\", width, 1);", "VXT0078") + , ("Console.Printf(\"%.*f\", 1.5, 1.5);", "VXT0078") + , -- Specified, and not implemented yet. + ("Console.Printf(\"%A\", 1);", "VXT0079") + , ("Console.Printf(\"%O\", 1);", "VXT0079") + , ("Console.Println(1.5);", "VXT0079") + , ("String s = \"x\" + 1.5;", "VXT0079") + , ("longint wide = 1; Console.Println(wide);", "VXT0079") + , ("longint wide = 1; Console.Printf(\"%d\", wide);", "VXT0079") + , ("ulongint wide = 1; Console.Printf(\"%u\", wide);", "VXT0079") + , ("double wide = 1.5; Console.Printf(\"%f\", wide);", "VXT0079") + , ("Console.Stdout();", "VXT0079") + , ("Console.Stdin();", "VXT0079") + , ("Console.Stderr();", "VXT0079") + , -- Strings have + and the two equalities, and no other operator. + ("String s = \"a\" - \"b\";", "VXT0012") + , ("String s = \"a\" * 2;", "VXT0012") + , ("bool b = \"a\" < \"b\";", "VXT0012") + , -- An immutable string is not appended to. + ("final String s = \"a\"; s += \"b\";", "VXT0003") + ] + +{- | A class of the program named @Console@ is found before the console of +the language: its method is called, and nothing is written. +-} +shadowed :: Bool +shadowed = case compileSource text of + Right artifacts -> + not (any (("CoreRuntimeCall" `isInfixOf`) . show . coreFunctionBody) (coreModuleFunctions (artifactCore artifacts))) + && (runFunctionWriting 10000 (artifactCore artifacts) "Main" [] == Just (UnitValue, Written "" "")) + Left _ -> False + where + text = + unlines + [ "namespace Demo;" + , "public class Console {" + , " public static int Println(_ int value) { return value + 1; }" + , "}" + , "public class Program {" + , " public static void Main() { int kept = Console.Println(41); }" + , "}" + ] + +program :: String -> String +program body = + unlines + [ "namespace Demo;" + , "public class Program {" + , " public static int Zero() { return 0; }" + , " public static int Half(_ int v) { return v / 2; }" + , " public static int Pick(_ int flag, _ int value) { if (flag > 0) { return value; } return 7; }" + , " public static int Log(_ int v) { Console.Println(v); return v; }" + , " public static int Twofold(_ int v) { return Log(v) + Log(v); }" + , " public static String Name() { return \"Visual X#\"; }" + , " public static String Twice(_ String s) { return s + s; }" + , " public static void Greet(_ String who) { Console.Println(\"Hello, \" + who + \"!\"); }" + , " public static String Fizz(_ int n) {" + , " if (n % 15 == 0) { return \"FizzBuzz\"; }" + , " if (n % 3 == 0) { return \"Fizz\"; }" + , " if (n % 5 == 0) { return \"Buzz\"; }" + , " return \"\" + n;" + , " }" + , " public static void Main() {" + , " " ++ body + , " }" + , "}" + ] + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "console.vxs" text) + +-- | What @Main@ writes to standard output and to standard error. +written :: (FrontendArtifacts -> CoreModule) -> String -> Maybe (String, String) +written select body = case compileSource (program body) of + Right artifacts -> case runFunctionWriting 200000 (select artifacts) "Main" [] of + Just (_, Written output failure) -> Just (output, failure) + Nothing -> Nothing + Left _ -> Nothing + +rejectedWith :: String -> String -> Bool +rejectedWith code body = case compileSource (program body) of + Left diagnostics -> any ((== code) . diagnosticCode) diagnostics + Right _ -> False + +-- | Whether the Core of @Main@ mentions the given constructor or generated name. +mentions :: String -> (FrontendArtifacts -> CoreModule) -> String -> Bool +mentions needle select body = case compileSource (program body) of + Right artifacts -> + any + ((needle `isInfixOf`) . show . coreFunctionBody) + [ function + | function <- coreModuleFunctions (select artifacts) + , "Main" `isInfixOf` show (coreFunctionName function) + ] + Left _ -> False + +verifies :: String -> Bool +verifies body = case compileSource (program body) of + Right artifacts -> + verifyCore (artifactCore artifacts) == Right (artifactCore artifacts) + && verifyCore (artifactOptimizedCore artifacts) == Right (artifactOptimizedCore artifacts) + && verifyCorePrep (artifactCorePrep artifacts) == Right (artifactCorePrep artifacts) + Left _ -> False diff --git a/Compiler/Haskell/Driver/test/CoreInterpreter.hs b/Compiler/Haskell/Driver/test/CoreInterpreter.hs index 89004c7a..f95bb08b 100644 --- a/Compiler/Haskell/Driver/test/CoreInterpreter.hs +++ b/Compiler/Haskell/Driver/test/CoreInterpreter.hs @@ -11,37 +11,90 @@ shares no lowering or rewriting code with the compiler. The subset is the one the tests need: integer and Boolean values, the arithmetic, comparison, bitwise and logical primitives, expression-local -bindings, conditional expressions, direct calls of module functions and all -statement forms. Integers are unbounded; a test that depends on overflow +bindings, conditional expressions, direct calls of module functions, +closures, callables that remember their result, strings, the runtime calls +for text and the console, and all statement forms. What a program writes to +the console is kept and returned with its result. Integers are unbounded; a test that depends on overflow belongs to the native execution tests instead. Anything outside the subset, and any run that exceeds its step budget, yields 'Nothing' rather than a guess. -} module CoreInterpreter ( Value (..) + , Written (..) , runFunction , runFunctionWithBudget + , runFunctionWriting ) where import Data.Bits (complement, shiftL, shiftR, xor, (.&.), (.|.)) +import Data.Char (chr, isDigit) +import Data.Ratio ((%)) +import RuntimeText import Visual.XSharp.AST import Visual.XSharp.Core +import Visual.XSharp.RuntimeCall -- | A runtime value of the supported subset. data Value = IntegerValue Integer | BooleanValue Bool | UnitValue - deriving (Eq, Ord, Show) + | {- | A closure: the values its captures had when it was created, the + symbols of its parameters, and its body. A capture initializer is + evaluated once, where the closure is created. + -} + ClosureValue [(Int, Value)] [Int] [CoreStatement] + | {- | A callable that remembers its result: the cell that holds the + result once there is one, and the callable that computes it. Copies of + the value name the same cell, which is how they share the result. + -} + MemoValue Int Value + | -- | A string. + TextValue String + | {- | A floating-point number: whether it is negative zero, and its exact + value, which is that of the binary number its type holds. + -} + FloatingValue Bool Rational + deriving (Eq, Show) + +-- | What a run wrote to the console, in the order it was written. +data Written = Written + { writtenOutput :: String + -- ^ Standard output. + , writtenError :: String + -- ^ Standard error. + } + deriving (Eq, Show) -- Local values keyed by symbol identity. Core symbols are unique within a -- function, so one flat table is a faithful model of its locals. type Locals = [(Int, Value)] --- Remaining statement and call budget. It bounds every run, so a lowering --- error that turns a loop into an endless one fails the test instead of --- hanging the suite. -type Budget = Int +{- | What a run carries from step to step besides its locals. + +The steps left bound every run, so a lowering error that turns a loop into +an endless one fails the test instead of hanging the suite. The cells hold +the results that callables have remembered; they belong to the run and not +to a function, because a callable is passed between functions and must find +its result wherever it is called. +-} +data Budget = Budget + { stepsLeft :: Int + , cells :: [(Int, Value)] + , nextCell :: Int + , writes :: [(Bool, String)] + -- ^ Every console write, latest first: whether it went to standard + -- error, and the text. + } + +-- | Whether no step is left. +exhausted :: Budget -> Bool +exhausted budget = stepsLeft budget <= 0 + +-- | The budget after one step. +step :: Budget -> Budget +step budget = budget {stepsLeft = stepsLeft budget - 1} data Flow = Proceed @@ -56,10 +109,24 @@ runFunction = runFunctionWithBudget 200000 {- | Run a function by name; 'Nothing' when it is missing, leaves the supported subset, or exhausts the budget. -} -runFunctionWithBudget :: Budget -> CoreModule -> String -> [Value] -> Maybe Value -runFunctionWithBudget budget moduleValue name arguments = do +runFunctionWithBudget :: Int -> CoreModule -> String -> [Value] -> Maybe Value +runFunctionWithBudget steps moduleValue name arguments = fst <$> runFunctionWriting steps moduleValue name arguments + +{- | Run a function by name and return what it wrote to the console with its +result. Nothing is returned for a run that does not finish: what such a run +wrote before it stopped is not a result a test may rely on here. +-} +runFunctionWriting :: Int -> CoreModule -> String -> [Value] -> Maybe (Value, Written) +runFunctionWriting steps moduleValue name arguments = do function <- firstJust [function | function <- coreModuleFunctions moduleValue, functionSpelling function == name] - fst <$> call moduleValue budget function arguments + (value, budget) <- call moduleValue (Budget steps [] 0 []) function arguments + let inOrder = reverse (writes budget) + Just + ( value + , Written + (concat [written | (False, written) <- inOrder]) + (concat [written | (True, written) <- inOrder]) + ) functionSpelling :: CoreFunction -> String functionSpelling = identifierText . resolvedSpelling . coreFunctionName @@ -74,11 +141,11 @@ symbolOf = symbolIdValue . resolvedSymbol call :: CoreModule -> Budget -> CoreFunction -> [Value] -> Maybe (Value, Budget) call moduleValue budget function arguments - | budget <= 0 = Nothing + | exhausted budget = Nothing | length arguments /= length (coreFunctionParameters function) = Nothing | otherwise = do let locals = zip (map (symbolOf . fst) (coreFunctionParameters function)) arguments - (flow, _, remaining) <- executeAll moduleValue (budget - 1) locals (coreFunctionBody function) + (flow, _, remaining) <- executeAll moduleValue (step budget) locals (coreFunctionBody function) case flow of ReturnValue value -> Just (value, remaining) Proceed -> Just (UnitValue, remaining) @@ -94,7 +161,7 @@ executeAll moduleValue budget locals (statement : remaining) = do execute :: CoreModule -> Budget -> Locals -> CoreStatement -> Maybe (Flow, Locals, Budget) execute moduleValue budget locals statement - | budget <= 0 = Nothing + | exhausted budget = Nothing | otherwise = case statement of CoreBind binding -> do (value, afterLocals, afterBudget) <- evaluate moduleValue spent locals (coreBindingValue binding) @@ -121,7 +188,7 @@ execute moduleValue budget locals statement CoreBreak -> Just (BreakLoop, locals, spent) CoreContinue -> Just (ContinueLoop, locals, spent) where - spent = budget - 1 + spent = step budget {- | Run one structured loop. @@ -139,7 +206,7 @@ loop :: Bool -> Maybe (Flow, Locals, Budget) loop moduleValue budget locals condition body update testFirst - | budget <= 0 = Nothing + | exhausted budget = Nothing | testFirst = do (enter, afterLocals, afterBudget) <- test budget locals if enter then iteration afterBudget afterLocals else Just (Proceed, afterLocals, afterBudget) @@ -152,7 +219,7 @@ loop moduleValue budget locals condition body update testFirst selected <- truth value Just (selected, afterLocals, afterBudget) iteration currentBudget currentLocals = do - (flow, afterBody, bodyBudget) <- executeAll moduleValue (currentBudget - 1) currentLocals body + (flow, afterBody, bodyBudget) <- executeAll moduleValue (step currentBudget) currentLocals body case flow of BreakLoop -> Just (Proceed, afterBody, bodyBudget) ReturnValue _ -> Just (flow, afterBody, bodyBudget) @@ -160,9 +227,13 @@ loop moduleValue budget locals condition body update testFirst (updateFlow, afterUpdate, updateBudget) <- executeAll moduleValue bodyBudget afterBody update case updateFlow of Proceed -> loop moduleValue updateBudget afterUpdate condition body update True - -- Leaving the loop from its update clause has no - -- defined meaning in this evaluator. - _ -> Nothing + -- A break in the update clause leaves the loop, + -- and a return leaves the function. + BreakLoop -> Just (Proceed, afterUpdate, updateBudget) + ReturnValue _ -> Just (updateFlow, afterUpdate, updateBudget) + -- The Core verifier rejects a continue placed + -- directly in an update clause. + ContinueLoop -> Nothing store :: Int -> Value -> Locals -> Locals store symbol value locals = (symbol, value) : filter ((/= symbol) . fst) locals @@ -172,14 +243,31 @@ truth value = case value of BooleanValue flag -> Just flag IntegerValue number -> Just (number /= 0) UnitValue -> Nothing + ClosureValue {} -> Nothing + MemoValue {} -> Nothing + TextValue {} -> Nothing + FloatingValue {} -> Nothing evaluate :: CoreModule -> Budget -> Locals -> CoreExpression -> Maybe (Value, Locals, Budget) evaluate moduleValue budget locals expression - | budget <= 0 = Nothing + | exhausted budget = Nothing | otherwise = case expression of - CoreVariable name _ -> do - value <- lookup (symbolOf name) locals - Just (value, locals, budget) + CoreVariable name _ -> case lookup (symbolOf name) locals of + Just value -> Just (value, locals, budget) + -- A method named where a value is expected is a closure + -- without captures. + Nothing -> do + method <- + firstJust + [ candidate + | candidate <- coreModuleFunctions moduleValue + , symbolOf (coreFunctionName candidate) == symbolOf name + ] + Just + ( ClosureValue [] (map (symbolOf . fst) (coreFunctionParameters method)) (coreFunctionBody method) + , locals + , budget + ) CoreLiteral literal valueType -> do value <- literalValue literal valueType Just (value, locals, budget) @@ -192,23 +280,73 @@ evaluate moduleValue budget locals expression evaluate moduleValue afterBudget afterLocals (if selected then whenTrue else whenFalse) CorePrimitive CoreLogicalAnd [left, right] _ -> shortCircuit False left right CorePrimitive CoreLogicalOr [left, right] _ -> shortCircuit True left right + -- Remembering takes a cell that no other callable has. Nothing is + -- computed here: the callable is not called. + CorePrimitive CoreMemoize [operand] _ -> do + (target, afterLocals, afterBudget) <- evaluate moduleValue budget locals operand + case target of + ClosureValue _ [] _ -> + Just + ( MemoValue (nextCell afterBudget) target + , afterLocals + , afterBudget {nextCell = nextCell afterBudget + 1} + ) + _ -> Nothing + -- A runtime call: the function its first operand names, applied + -- to the values of the operands after it. + CorePrimitive CoreRuntimeCall (CoreLiteral (CoreInteger identity) _ : operands) _ -> do + function <- runtimeFunctionOf identity + (values, afterLocals, afterBudget) <- evaluateMany moduleValue budget locals operands + (value, afterCall) <- applyRuntime function values afterBudget + Just (value, afterLocals, afterCall) CorePrimitive primitive operands valueType -> do (values, afterLocals, afterBudget) <- evaluateMany moduleValue budget locals operands value <- applyPrimitive primitive valueType values Just (value, afterLocals, afterBudget) - CoreApply (CoreVariable callee _) arguments _ -> do - function <- - firstJust - [ candidate - | candidate <- coreModuleFunctions moduleValue - , symbolOf (coreFunctionName candidate) == symbolOf callee - ] - (values, afterLocals, afterBudget) <- evaluateMany moduleValue budget locals arguments - (value, callBudget) <- call moduleValue afterBudget function values + CoreApply (CoreVariable callee _) arguments _ + | function : _ <- + [ candidate + | candidate <- coreModuleFunctions moduleValue + , symbolOf (coreFunctionName candidate) == symbolOf callee + ] -> do + (values, afterLocals, afterBudget) <- evaluateMany moduleValue budget locals arguments + (value, callBudget) <- call moduleValue afterBudget function values + Just (value, afterLocals, callBudget) + -- The callee is evaluated before the arguments, like any operand. + CoreApply callee arguments _ -> do + (target, calleeLocals, calleeBudget) <- evaluate moduleValue budget locals callee + (values, afterLocals, afterBudget) <- evaluateMany moduleValue calleeBudget calleeLocals arguments + (value, callBudget) <- callClosure target values afterBudget Just (value, afterLocals, callBudget) - CoreApply {} -> Nothing - CoreClosure {} -> Nothing + CoreClosure captures parameters _ body _ -> do + (values, afterLocals, afterBudget) <- evaluateMany moduleValue budget locals (map coreCaptureValue captures) + Just + ( ClosureValue (zip (map (symbolOf . coreCaptureName) captures) values) (map (symbolOf . fst) parameters) body + , afterLocals + , afterBudget + ) where + -- A closure runs on its captures and its arguments alone: it sees + -- no local of the function that calls it. + callClosure target values currentBudget = case target of + ClosureValue captured parameters body + | not (exhausted currentBudget) && length parameters == length values -> do + (flow, _, remaining) <- + executeAll moduleValue (step currentBudget) (zip parameters values ++ captured) body + case flow of + ReturnValue value -> Just (value, remaining) + Proceed -> Just (UnitValue, remaining) + _ -> Nothing + -- The first call computes the result and keeps it in the cell; + -- every later call, of any copy, reads the cell. A call that + -- does not finish leaves the cell empty. + MemoValue cell computation + | not (exhausted currentBudget) && null values -> case lookup cell (cells currentBudget) of + Just remembered -> Just (remembered, step currentBudget) + Nothing -> do + (value, remaining) <- callClosure computation [] (step currentBudget) + Just (value, remaining {cells = (cell, value) : cells remaining}) + _ -> Nothing -- The right operand runs only when the left one does not decide. shortCircuit decidingValue left right = do (leftValue, afterLeft, leftBudget) <- evaluate moduleValue budget locals left @@ -234,10 +372,73 @@ literalValue literal valueType = case literal of | otherwise -> Just (IntegerValue number) CoreBoolean flag -> Just (BooleanValue flag) CoreUnit -> Just UnitValue - CoreFloating _ -> Nothing - CoreString _ -> Nothing + CoreFloating spelling -> floatingValue valueType spelling + CoreString value -> Just (TextValue value) CoreNull -> Nothing +{- | The value a floating-point literal has in its type: the decimal number +that was written, rounded to the nearest number the type holds. Only the +plain decimal spellings the tests use are read. +-} +floatingValue :: Type -> String -> Maybe Value +floatingValue valueType spelling = do + written <- decimal (filter (/= '_') spelling) + case valueType of + NamedType (QualifiedName [Identifier "float"]) [] -> + Just (FloatingValue False (toRational (fromRational written :: Double))) + NamedType (QualifiedName [Identifier "lfloat"]) [] -> + Just (FloatingValue False (toRational (fromRational written :: Float))) + _ -> Nothing + where + decimal value = + let (whole, afterWhole) = span isDigit value + (fraction, afterFraction) = case afterWhole of + '.' : rest -> span isDigit rest + _ -> ("", afterWhole) + mantissa = read ('0' : whole ++ fraction) :: Integer + scaled = mantissa % (10 ^ length fraction) + in case afterFraction of + [] | not (null whole) -> Just scaled + 'e' : exponentText -> scaleBy scaled exponentText + 'E' : exponentText -> scaleBy scaled exponentText + _ -> Nothing + scaleBy scaled exponentText = case exponentText of + '-' : digits | all isDigit digits, not (null digits) -> Just (scaled / (10 ^ (read digits :: Integer))) + '+' : digits | all isDigit digits, not (null digits) -> Just (scaled * (10 ^ (read digits :: Integer))) + digits | all isDigit digits, not (null digits) -> Just (scaled * (10 ^ (read digits :: Integer))) + _ -> Nothing + +{- | Apply a function of the runtime. The text functions are those of +"RuntimeText"; a console write adds to what the run has written. +-} +applyRuntime :: RuntimeFunction -> [Value] -> Budget -> Maybe (Value, Budget) +applyRuntime function values budget = case (function, values) of + (TextConcat, [TextValue left, TextValue right]) -> text (left ++ right) + (TextEquals, [TextValue left, TextValue right]) -> Just (BooleanValue (left == right), budget) + (TextFromSigned, [IntegerValue value]) -> text (textOfSigned value) + (TextFromUnsigned, [IntegerValue value]) -> text (textOfSigned value) + (TextFromBool, [BooleanValue value]) -> text (textOfBool value) + (TextFromChar, [IntegerValue value]) -> text [chr (fromInteger value)] + (TextFormatSigned, [IntegerValue flags, IntegerValue width, IntegerValue precision, IntegerValue value]) -> + text (formatInteger (Conversion flags width precision) value) + (TextFormatUnsigned, [IntegerValue flags, IntegerValue width, IntegerValue precision, IntegerValue value]) -> + text (formatInteger (Conversion flags width precision) value) + (TextFormatFloating, [IntegerValue flags, IntegerValue width, IntegerValue precision, FloatingValue negativeZero value]) -> + text (formatFloating (Conversion flags width precision) negativeZero value) + (TextFormatString, [IntegerValue flags, IntegerValue width, IntegerValue precision, TextValue value]) -> + text (formatText (Conversion flags width precision) value) + (TextFormatChar, [IntegerValue flags, IntegerValue width, IntegerValue _, IntegerValue value]) -> + text (formatText (Conversion flags width absent) [chr (fromInteger value)]) + (TextNewline, []) -> text lineTerminator + (ConsoleWrite, [TextValue value, IntegerValue target]) + | target `elem` [consoleOutput, consoleOutputLine, consoleError, consoleErrorLine] -> + let toError = target `elem` [consoleError, consoleErrorLine] + ended = if target `elem` [consoleOutputLine, consoleErrorLine] then value ++ lineTerminator else value + in Just (UnitValue, budget {writes = (toError, ended) : writes budget}) + _ -> Nothing + where + text value = Just (TextValue value, budget) + applyPrimitive :: CorePrimitive -> Type -> [Value] -> Maybe Value applyPrimitive primitive valueType values = case (primitive, values) of (CoreAdd, [IntegerValue left, IntegerValue right]) -> integer (left + right) @@ -259,6 +460,8 @@ applyPrimitive primitive valueType values = case (primitive, values) of (CoreEqual, [left, right]) -> Just (BooleanValue (left == right)) (CoreNotEqual, [left, right]) -> Just (BooleanValue (left /= right)) (CoreLogicalNot, [operand]) -> BooleanValue . not <$> truth operand + -- Remembering needs the cells of the run; 'evaluate' handles it. + (CoreMemoize, _) -> Nothing _ -> Nothing where -- A primitive whose result type is bool yields a Boolean even when diff --git a/Compiler/Haskell/Driver/test/CoreOptimizerSourceTests.hs b/Compiler/Haskell/Driver/test/CoreOptimizerSourceTests.hs index affed31e..47b5c315 100644 --- a/Compiler/Haskell/Driver/test/CoreOptimizerSourceTests.hs +++ b/Compiler/Haskell/Driver/test/CoreOptimizerSourceTests.hs @@ -3,6 +3,7 @@ module CoreOptimizerSourceTests (coreOptimizerSourceTests) where +import CoreInterpreter import Visual.XSharp.AST import Visual.XSharp.Compiler import Visual.XSharp.Core @@ -40,7 +41,7 @@ coreOptimizerSourceTests = , ("source immutable helper locals inline", sourceInlineLocal) , ("source dependent helper locals inline", sourceInlineDependentLocals) , ("source primitive arguments remain single evaluations", sourceInlinePrimitiveArgument) - , ("source failing unused arguments remain explicit", sourceInlineFailingArgument) + , ("source failing unused arguments are never computed", sourceUnusedFailingArgument) , ("source argument order survives generated lets", sourceInlineArgumentOrder) , ("source mutable helper bodies retain calls", sourceRejectsMutableHelper) , ("source branching helper bodies retain calls", sourceRejectsBranchHelper) @@ -297,15 +298,16 @@ sourceInlinePrimitiveArgument = case compiledProgram members of , "int Value() { return Identity(20 + 22); }" ] -sourceInlineFailingArgument :: Bool -sourceInlineFailingArgument = case compiledProgram members of - Just artifacts -> case lastReturn artifacts of - Just (CoreLet _ bindingType value result resultType) -> - bindingType == intType - && value == CorePrimitive CoreDivide [integer 1, integer 0] intType - && result == integer 42 - && resultType == intType - _ -> False +{- | An argument the method never needs is never computed. The call yields +the method's result, in the Core the Desugarer produced and in the Core the +optimizer left: no division is carried out on the way. +-} +sourceUnusedFailingArgument :: Bool +sourceUnusedFailingArgument = case compiledProgram members of + Just artifacts -> + all + (\core -> runFunction core "Value" [] == Just (IntegerValue 42)) + [artifactCore artifacts, artifactOptimizedCore artifacts] Nothing -> False where members = @@ -500,7 +502,7 @@ prepFunctionCalls value = any blockCalls (CorePrep.corePrepFunctionBlocks value) sourceGuardedDeadDivision :: Bool sourceGuardedDeadDivision = - case compiledProgram ["void Guarded(_ int divisor) { if (divisor \\= 0) { int result = 10 / divisor; } }"] of + case compiledProgram ["void Guarded(_ int divisor) { if (divisor \\= 0) { int result = 10 / divisor; if (result > 0) { } } }"] of Just artifacts -> case coreModuleFunctions (artifactOptimizedCore artifacts) of [function] -> not (containsIntegerDivision (coreFunctionBody function)) @@ -509,7 +511,7 @@ sourceGuardedDeadDivision = sourceUnguardedDeadDivision :: Bool sourceUnguardedDeadDivision = - case compiledProgram ["void Guarded(_ int divisor) { int result = 10 / divisor; }"] of + case compiledProgram ["void Guarded(_ int divisor) { int result = 10 / divisor; if (result > 0) { } }"] of Just artifacts -> case coreModuleFunctions (artifactOptimizedCore artifacts) of [function] -> containsIntegerDivision (coreFunctionBody function) @@ -519,7 +521,7 @@ sourceUnguardedDeadDivision = sourceAssignmentReplacesGuard :: Bool sourceAssignmentReplacesGuard = case compiledProgram - ["void Guarded(_ int input) { int divisor = input; if (divisor \\= 0) { divisor = 0; int result = 10 / divisor; } }"] of + ["void Guarded(_ int input) { int divisor = input; if (divisor \\= 0) { divisor = 0; int result = 10 / divisor; if (result > 0) { } } }"] of Just artifacts -> case coreModuleFunctions (artifactOptimizedCore artifacts) of [function] -> containsIntegerDivision (coreFunctionBody function) @@ -544,13 +546,13 @@ sourceVariableEqualityGuard :: Bool sourceVariableEqualityGuard = sourceDivisionExpectation False - ["void Guarded(_ int known, _ int divisor) { if (known \\= 0 && divisor == known) { int unused = 24 / divisor; } }"] + ["void Guarded(_ int known, _ int divisor) { if (known \\= 0 && divisor == known) { int probed = 24 / divisor; if (probed > 0) { } } }"] sourceVariableOrderingGuard :: Bool sourceVariableOrderingGuard = sourceDivisionExpectation False - [ "void Guarded(_ int lowerBound, _ int divisor) { if (lowerBound >= 0 && divisor > lowerBound) { int unused = 24 / divisor; } }" + [ "void Guarded(_ int lowerBound, _ int divisor) { if (lowerBound >= 0 && divisor > lowerBound) { int probed = 24 / divisor; if (probed > 0) { } } }" ] sourceFalseDisjunctionGuard :: Bool @@ -559,7 +561,7 @@ sourceFalseDisjunctionGuard = False [ "void Guarded(_ int divisor, _ int other) {" , " if (divisor == 0 || other == 0) { return; }" - , " int unused = 24 / divisor;" + , " int probed = 24 / divisor; if (probed > 0) { }" , "}" ] @@ -570,7 +572,7 @@ sourceProductRangeGuard = [ "void Guarded(_ int value) {" , " if (value > 0 && value < 5) {" , " int divisor = value * 2;" - , " int unused = 24 / divisor;" + , " int probed = 24 / divisor; if (probed > 0) { }" , " }" , "}" ] @@ -582,7 +584,7 @@ sourceDifferenceRangeGuard = [ "void Guarded(_ int value) {" , " if (value >= 8) {" , " int divisor = value - 5;" - , " int unused = 24 / divisor;" + , " int probed = 24 / divisor; if (probed > 0) { }" , " }" , "}" ] @@ -595,7 +597,7 @@ sourceNonzeroBranchJoin = , " if (input \\= 0) {" , " int divisor = input;" , " if (choose) { divisor = 7; } else { divisor = -1; }" - , " int unused = 24 / divisor;" + , " int probed = 24 / divisor; if (probed > 0) { }" , " }" , "}" ] @@ -608,7 +610,7 @@ sourceConflictingBranchJoin = , " if (input \\= 0) {" , " int divisor = input;" , " if (choose) { divisor = 7; } else { divisor = 0; }" - , " int unused = 24 / divisor;" + , " int probed = 24 / divisor; if (probed > 0) { }" , " }" , "}" ] @@ -669,7 +671,7 @@ sourceComparisonCase index (operator, boundary, trueEdge, reversed) = if reversed then (show boundary, "value") else ("value", show boundary) - selectedBody = " int unused = 24 / value; " + selectedBody = " int probed = 24 / value; if (probed > 0) { } " trueBody = if trueEdge then selectedBody else "" falseBody = if trueEdge then "" else selectedBody diff --git a/Compiler/Haskell/Driver/test/CoreVerifierTests.hs b/Compiler/Haskell/Driver/test/CoreVerifierTests.hs index 49bd4a63..ab69acec 100644 --- a/Compiler/Haskell/Driver/test/CoreVerifierTests.hs +++ b/Compiler/Haskell/Driver/test/CoreVerifierTests.hs @@ -20,6 +20,26 @@ coreVerifierTests = , ("Core verifier rejects an unresolved function result", rejectedWith "VXC1003" unresolvedFunctionResult) , ("Core verifier rejects duplicate parameter symbols", rejectedWith "VXC1004" duplicateParameters) , ("Core verifier rejects a missing value return", rejectedWith "VXC1005" missingValueReturn) + , -- A loop that cannot be left never falls through to the end. + ("Core verifier accepts a body that ends in a loop that cannot be left", accepted (endsInLoop (CoreWhile trueValue []))) + , ("Core verifier accepts an endless do/while at the end of a body", accepted (endsInLoop (CoreDoWhile [] trueValue))) + , ("Core verifier accepts an endless for at the end of a body", accepted (endsInLoop (CoreFor trueValue [] []))) + , + ( "Core verifier rejects a body that ends in a loop a break leaves" + , rejectedWith "VXC1005" (endsInLoop (CoreWhile trueValue [CoreIf trueValue [CoreBreak] []])) + ) + , + ( "Core verifier rejects a body that ends in a for loop whose update breaks" + , rejectedWith "VXC1005" (endsInLoop (CoreFor trueValue [] [CoreBreak])) + ) + , + ( "Core verifier rejects a body that ends in a loop with a condition" + , rejectedWith "VXC1005" (endsInLoop (CoreWhile (CoreLiteral (CoreBoolean False) boolType) [])) + ) + , + ( "a break of a nested loop does not leave the loop around it" + , accepted (endsInLoop (CoreWhile trueValue [CoreWhile trueValue [CoreBreak]])) + ) , ("Core verifier rejects a zero function symbol", rejectedWith "VXC1006" zeroFunctionSymbol) , ("Core verifier rejects a negative parameter symbol", rejectedWith "VXC1006" negativeParameterSymbol) , ("Core verifier rejects an unresolved parameter type", rejectedWith "VXC1007" unresolvedParameterType) @@ -157,6 +177,13 @@ duplicateParameters = missingValueReturn :: CoreModule missingValueReturn = coreModule [function mainName [] intType []] +-- | A function that returns a value and whose body is the given loop alone. +endsInLoop :: CoreStatement -> CoreModule +endsInLoop loop = coreModule [function mainName [] intType [loop]] + +trueValue :: CoreExpression +trueValue = CoreLiteral (CoreBoolean True) boolType + zeroFunctionSymbol :: CoreModule zeroFunctionSymbol = coreModule [unitFunction (resolved 0 "Main") []] diff --git a/Compiler/Haskell/Driver/test/EffectTests.hs b/Compiler/Haskell/Driver/test/EffectTests.hs new file mode 100644 index 00000000..59635802 --- /dev/null +++ b/Compiler/Haskell/Driver/test/EffectTests.hs @@ -0,0 +1,214 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Which expressions the compiler takes to have an effect. + +Evaluation is by need and effects are not: an expression that writes to the +console is evaluated where it stands. The source does not say which +expressions write, so the frontend infers it from the bodies of the methods +a program has; see "Visual.XSharp.Desugarer.Effects". These cases hold that +inference from both sides: + +* it must find every write, however far down the calls it is, or a program + would write later than it says, or not at all; +* it must not find writes that are not there, or a program without output + would lose the evaluation by need it had. + +Each program is checked twice: for what it writes, in the reference +evaluator before and after optimization, and for the shape of its Core, +where a flag named @$known@ marks a binding that was deferred and +@CoreMemoize@ an argument that was suspended. + +Where the inference errs on purpose, the case says so and pins it, so that +making it more exact shows as a change here. +-} +module EffectTests (effectTests) where + +import CoreInterpreter +import Data.List (isInfixOf) +import Visual.XSharp.Compiler +import Visual.XSharp.Core +import Visual.XSharp.Diagnostic + +effectTests :: [(String, Bool)] +effectTests = + concat + [ [ ("unoptimized: " ++ label, written artifactCore members body == Just expected) + , ("optimized: " ++ label, written artifactOptimizedCore members body == Just expected) + ] + | (label, members, body, expected) <- orders + ] + ++ [ (label, deferred members body == expected) | (label, members, body, expected) <- deferrals + ] + ++ [ (label, suspended members body == expected) | (label, members, body, expected) <- suspensions + ] + +-- | A method that writes, and methods that reach it through one and two calls. +chain :: String +chain = + unlines + [ " public static int Log(_ int v) { Console.Println(v); return v; }" + , " public static int Once(_ int v) { return Log(v) + 1; }" + , " public static int Twice(_ int v) { return Once(v) + 1; }" + , " public static int Quiet(_ int v) { return v + 1; }" + , " public static int Deep(_ int v) { return Quiet(Quiet(v)); }" + , " public static int Zero() { return 0; }" + ] + +-- | Two methods that call each other, one of which writes. +mutual :: String +mutual = + unlines + [ " public static int Ping(_ int n) { if (n <= 0) { return 0; } return Pong(n - 1) + 1; }" + , " public static int Pong(_ int n) { Console.Print(n); if (n <= 0) { return 0; } return Ping(n - 1) + 1; }" + , " public static int Even(_ int n) { if (n <= 0) { return 1; } return Odd(n - 1); }" + , " public static int Odd(_ int n) { if (n <= 0) { return 0; } return Even(n - 1); }" + , " public static int Zero() { return 0; }" + ] + +-- | Methods with parameters that are read first, after a write, and never. +parameters :: String +parameters = + unlines + [ " public static int Log(_ int v) { Console.Println(v); return v; }" + , " public static int Zero() { return 0; }" + , " public static int Pick(_ int flag, _ int value) { if (flag > 0) { return value; } return 7; }" + , " public static int Announce(_ int flag, _ int value) { Console.Print(\"a\"); if (flag > 0) { return value; } return 7; }" + , " public static int Through(_ int flag, _ int value) { return Announce(flag, value); }" + ] + +-- | Callables that write and callables that do not. +callables :: String +callables = + unlines + [ " public static int Log(_ int v) { Console.Println(v); return v; }" + , " public static int Zero() { return 0; }" + , " public static int Apply(_ int v) { auto step = \\(int w) -> w + 1; return step(v); }" + , " public static int Run(_ (int) -> int f) { return f(3); }" + , " public static int Quiet(_ int v) { return v + 1; }" + ] + +{- | What a program writes: the label, the members beside @Main@, the body of +@Main@ and the text. An expression that writes is carried out where it +stands, in order, whether or not anything needs its value. +-} +orders :: [(String, String, String, String)] +orders = + [ ("a write one call down happens at the binding", chain, "int x = Once(1); Console.Println(9);", "1\n9\n") + , ("a write two calls down happens at the binding", chain, "int x = Twice(1); Console.Println(9);", "1\n9\n") + , ("bindings that write keep their order", chain, "int a = Twice(1); int b = Once(2); int c = Log(3); Console.Println(a + b + c);", "1\n2\n3\n9\n") + , ("a binding that only computes does not move a write", chain, "int a = Deep(1); int b = Log(2); Console.Println(a);", "2\n3\n") + , ("a value that is never needed is not computed beside writes", chain, "int a = Deep(1) / Zero(); int b = Log(2);", "2\n") + , ("a write in a method that calls itself through another", mutual, "int x = Ping(4); Console.Println(\"!\");", "31!\n") + , ("methods that call each other and never write stay quiet", mutual, "int x = Even(5); Console.Println(x);", "0\n") + , ("a recursion that is never needed does not run", mutual, "int x = Even(4) / Zero(); Console.Println(\"ok\");", "ok\n") + , ("an argument that writes is carried out at the call", parameters, "Console.Println(Pick(0, Log(5)));", "5\n7\n") + , ("an argument that writes is carried out before the method writes", parameters, "Console.Println(Announce(0, Log(5)));", "5\na7\n") + , ("a method that writes may still leave an argument unused", parameters, "Console.Println(Announce(0, 8 / Zero()));", "a7\n") + , ("an argument handed through a method that writes may stay unused", parameters, "Console.Println(Through(0, 8 / Zero()));", "a7\n") + , ("an argument handed through is computed when it is needed", parameters, "Console.Println(Through(1, 8 / 2));", "a4\n") + , ("arguments that write keep the order they are written in", parameters, "Console.Println(Pick(Log(1), Log(2)));", "1\n2\n2\n") + , ("a callable that writes is called where its call stands", callables, "auto say = \\(int v) -> Log(v); int x = say(1); Console.Println(9);", "1\n9\n") + , ("a callable that writes directly is called where its call stands", callables, "auto say = \\(int v) -> { Console.Println(v); return v; }; int x = say(1); Console.Println(9);", "1\n9\n") + , ("creating a callable that writes writes nothing", callables, "auto say = \\(int v) -> Log(v); Console.Println(9);", "9\n") + , ("a callable is called as often as its call is reached", callables, "auto say = \\(int v) -> Log(v); int t = 0; for (int i = 0; i < 3; i += 1) { t += say(i); } Console.Println(t);", "0\n1\n2\n3\n") + , -- A method is a callable value too, and no callable expression is + -- written anywhere in these programs. + ("a method that writes held as a value is called where its call stands", callables, "auto f = Log; int x = f(1); Console.Println(9);", "1\n9\n") + , ("a method that writes handed to a method is called where that call stands", callables, "int x = Run(Log); Console.Println(9);", "3\n9\n") + , ("a quiet method handed to a method is called when its value is needed", callables, "int x = Run(Quiet); Console.Println(x);", "4\n") + , ("a quiet method handed to a method is not called for a value nothing needs", callables, "int x = Run(Quiet) / Zero(); Console.Println(9);", "9\n") + , ("a method that writes called through its class writes at the binding", callables, "int x = Program.Log(1); Console.Println(9);", "1\n9\n") + , ("a method with a quiet callable computes", callables, "Console.Println(Apply(4));", "5\n") + , ("a write in a condition happens when the condition is evaluated", chain, "if (Log(1) > 0) { Console.Println(\"yes\"); }", "1\nyes\n") + , ("a write behind a short circuit happens only when it is reached", chain, "bool b = Zero() > 0 && Log(1) > 0; Console.Println(\"end\");", "end\n") + , ("a write in the taken arm of a conditional happens", chain, "int v = Zero() > 0 ? Log(1) : Log(2); Console.Println(v);", "2\n2\n") + , ("a write in a loop condition happens on every pass", chain, "int i = 0; while (Log(i) < 2) { i += 1; }", "0\n1\n2\n") + , ("a discarded value that writes is carried out", chain, "_ = Twice(4);", "4\n") + , ("a statement that writes is carried out", chain, "Twice(4);", "4\n") + ] + +{- | Whether the Core of @Main@ holds a deferred binding. A binding whose +initializer may write is never deferred; one that only computes still is. +-} +deferrals :: [(String, String, String, Bool)] +deferrals = + [ ("a binding of a method that writes is not deferred", chain, "int x = Log(1); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding one call above a write is not deferred", chain, "int x = Once(1); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding two calls above a write is not deferred", chain, "int x = Twice(1); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding of a method that only computes is deferred", chain, "int x = Quiet(1); if (Zero() > 0) { Console.Println(x); }", True) + , ("a binding two calls above nothing is deferred", chain, "int x = Deep(1); if (Zero() > 0) { Console.Println(x); }", True) + , ("a binding with a write in one operand is not deferred", chain, "int x = Quiet(1) + Log(2); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding with a write in a conditional arm is not deferred", chain, "int x = Zero() > 0 ? Log(1) : 2; if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding of a recursion that writes is not deferred", mutual, "int x = Ping(2); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding of a recursion that only computes is deferred", mutual, "int x = Even(2); if (Zero() > 0) { Console.Println(x); }", True) + , -- No callable of this program writes, so a call through one computes. + ("a binding of a callable is deferred when no callable writes", callables, "auto step = \\(int w) -> w + 1; int x = step(1); if (Zero() > 0) { Console.Println(x); }", True) + , ("a binding of a callable is not deferred when a callable writes", callables, "auto say = \\(int v) -> Log(v); int x = say(1); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding of a method value that writes is not deferred", callables, "auto f = Log; int x = f(1); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding of a call handing on a method that writes is not deferred", callables, "int x = Run(Log); if (Zero() > 0) { Console.Println(x); }", False) + , ("a binding of a call handing on a quiet method is deferred", callables, "int x = Run(Quiet); if (Zero() > 0) { Console.Println(x); }", True) + , -- Calling a method by name does not make it a value. + ("a method that writes and is only called by name leaves callables quiet", callables, "Log(0); auto step = \\(int w) -> w + 1; int x = step(1); if (Zero() > 0) { Console.Println(x); }", True) + , -- The inference does not follow which callable a value holds: once + -- any callable of the program writes, a call through any callable + -- is taken to write. Pinned, so that a more exact answer shows. + ("a quiet callable is taken to write beside one that does", callables, "auto say = \\(int v) -> Log(v); auto step = \\(int w) -> w + 1; int x = step(1); if (Zero() > 0) { Console.Println(x); }", False) + ] + +{- | Whether the Core of @Main@ suspends an argument. An argument that may +write is computed at the call; one that only computes, and may fail, is +suspended when the method may leave it unused. +-} +suspensions :: [(String, String, String, Bool)] +suspensions = + [ ("an argument that only computes is suspended", parameters, "Console.Println(Pick(0, 8 / Zero()));", True) + , ("an argument that writes is not suspended", parameters, "Console.Println(Pick(0, Log(5)));", False) + , ("an argument with a write in one operand is not suspended", parameters, "Console.Println(Pick(0, 8 / Zero() + Log(5)));", False) + , -- A method that writes before it reads a parameter is not certain to + -- need the parameter first, so the argument may stay unused. + ("an argument of a method that writes first is suspended", parameters, "Console.Println(Announce(0, 8 / Zero()));", True) + , ("an argument handed through a method is suspended", parameters, "Console.Println(Through(0, 8 / Zero()));", True) + , ("an argument that cannot fail is passed as a value", parameters, "Console.Println(Pick(0, 8 + 1));", False) + ] + +program :: String -> String -> String +program members body = + unlines + [ "namespace Demo;" + , "public class Program {" + , members + , " public static void Main() {" + , " " ++ body + , " }" + , "}" + ] + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "effects.vxs" text) + +-- | What @Main@ writes to standard output. +written :: (FrontendArtifacts -> CoreModule) -> String -> String -> Maybe String +written select members body = case compileSource (program members body) of + Right artifacts -> case runFunctionWriting 200000 (select artifacts) "Main" [] of + Just (_, Written output "") -> Just output + _ -> Nothing + Left _ -> Nothing + +-- | Whether the unoptimized Core of @Main@ mentions the given text. +mainMentions :: String -> String -> String -> Bool +mainMentions needle members body = case compileSource (program members body) of + Right artifacts -> + any + ((needle `isInfixOf`) . show . coreFunctionBody) + [ function + | function <- coreModuleFunctions (artifactCore artifacts) + , "Main" `isInfixOf` show (coreFunctionName function) + ] + Left _ -> False + +deferred :: String -> String -> Bool +deferred = mainMentions "$known" + +suspended :: String -> String -> Bool +suspended = mainMentions "CoreMemoize" diff --git a/Compiler/Haskell/Driver/test/EnumTests.hs b/Compiler/Haskell/Driver/test/EnumTests.hs new file mode 100644 index 00000000..dd3a1cda --- /dev/null +++ b/Compiler/Haskell/Driver/test/EnumTests.hs @@ -0,0 +1,283 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Tests for classic enums. + +A classic enum is a value type whose members are named integers. The tests +pin its declaration, the numbering of its members, the one comparison its +values have, the absence of any conversion to or from an integer, and +@match@ over it: a case pattern names a member, two members with one value +are one case, and a match that names every value needs no catch-all arm. + +The expected values are written by hand. Every accepted program runs in the +reference evaluator on the unoptimized and on the optimized Core and is +verified as Core and as CorePrep; an enum must reach Core as its underlying +integer type and never as a named type. +-} +module EnumTests (enumTests) where + +import CoreInterpreter +import Data.List (isInfixOf) +import Visual.XSharp.AST +import Visual.XSharp.Compiler +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep.Verifier +import Visual.XSharp.Core.Verifier +import Visual.XSharp.Diagnostic +import Visual.XSharp.Lexer +import Visual.XSharp.Parser + +enumTests :: [(String, Bool)] +enumTests = + parserTests + ++ concat + [ [ ("unoptimized Core computes " ++ label, runs artifactCore statements arguments expected) + , ("optimized Core computes " ++ label, runs artifactOptimizedCore statements arguments expected) + ] + | (statements, runsOfCase) <- evaluationCases + , (arguments, expected) <- runsOfCase + , let label = show expected ++ " for " ++ show arguments ++ ": " ++ statements + ] + ++ [ ("every enum program verifies as Core and as CorePrep", all (verifies . fst) evaluationCases) + , ("no enum reaches Core as a named type", all (erased . fst) evaluationCases) + ] + ++ [(label, rejectedWith code (program statements)) | (label, code, statements) <- rejectedBodies] + ++ [(label, rejectedWith code (declarations ++ emptyProgram)) | (label, code, declarations) <- rejectedDeclarations] + ++ [ ("an enum without members is a declaration", accepted ("enum Empty { }\n" ++ emptyProgram)) + , ("an enum may be declared after the class that follows it in use", accepted (emptyProgram ++ "enum Later { ONE }\n")) + , ("a negative member value is accepted in a signed type", accepted ("enum Signed { LOW = -5, NEXT }\n" ++ emptyProgram)) + ] + +-- ---------------------------------------------------------------- sources + +enums :: String +enums = + unlines + [ "enum Status { NONE, UNKNOWN = 0, READY }" + , "enum Level = byte { LOW = 1, MID, HIGH = 10, TOP }" + , -- Values computed from earlier members: 1, 2, 4, 7, 8, 2, 250, 5. + "enum Flag = ubyte { READ = 1, WRITE = READ << 1, EXECUTE = WRITE * 2, ALL = READ | WRITE | EXECUTE, NEXT," + ++ " HALF = EXECUTE / 2, REST = !(EXECUTE + 1), ROUNDED = 9 // 2 }" + , "enum Offset = int { BASE = -3, ABOVE = BASE + 5, POWER = 2 ** 10, MODULO = -7 % 4, MIX = (ABOVE ^ 7) & 6 }" + ] + +-- | A program around the given body of @Evaluate@. +program :: String -> String +program statements = + "namespace Test;\n" + ++ enums + ++ unlines + [ "class Program {" + , " public static Status Pick(_ int v) { if (v > 0) { return Status.READY; } return Status.NONE; }" + , " public static int Rank(_ Level l) { return match (l) { .LOW -> 1, .MID -> 2, .HIGH -> 10, .TOP -> 11 }; }" + , " public static Level Raise(_ Level l) { return match (l) { .LOW -> Level.MID, .MID -> Level.HIGH, _ -> Level.TOP }; }" + , " public static Flag Bit(_ int v) { return match (v) { 1 -> .READ, 2 -> .WRITE, 4 -> .EXECUTE, 7 -> .ALL, 8 -> .NEXT, 250 -> .REST, 5 -> .ROUNDED, _ -> .HALF }; }" + , " public static Offset At(_ int v) { return match (v) { -3 -> .BASE, 2 -> .ABOVE, 1024 -> .POWER, _ -> .MODULO }; }" + , " public static int Evaluate(_ int left, _ int right) {" + , " " ++ statements + , " }" + , "}" + ] + +emptyProgram :: String +emptyProgram = "class Program { public static int Evaluate(_ int left, _ int right) { return 0; } }\n" + +-- ----------------------------------------------------------------- parser + +parserTests :: [(String, Bool)] +parserTests = + [ ("an enum declaration keeps its members in order", memberNames "enum Direction { NORTH, EAST, SOUTH, WEST, }" == Just ["NORTH", "EAST", "SOUTH", "WEST"]) + , ("a member keeps the value written for it", memberValues "enum Value = int { FIRST = 5, SECOND, THIRD = 10, FOURTH }" == Just [Just "5", Nothing, Just "10", Nothing]) + , ("a member value may be negative", memberValues "enum Signed { LOW = -5 }" == Just [Just "-5"]) + , -- A value is an expression; a comma ends it and the next member begins. + ("a member value is an expression", memberValues "enum Bits { A = 1, B = A << 1, C = (A | B) + 2 * 3, D }" == Just [Just "1", Just "A<<1", Just "A|B+2*3", Nothing]) + , ("a member value is not an assignment", not (parses "enum Bad { A = 1, B = A = 2 }")) + , ("the underlying type is kept", underlying "enum Status = byte { PENDING = 0 }" == Just (Just (ExplicitType (Identifier "byte")))) + , ("the underlying type may be absent", underlying "enum Status { PENDING }" == Just Nothing) + , ("enum is a reserved word", not (parses "class enum { }")) + ] + where + -- The underlying type and the members of the first declaration. + declaration text = case parseSource text of + Right (ParsedAST (SyntaxTree _ (EnumDeclaration _ _ _ written members : _))) -> Just (written, members) + _ -> Nothing + memberNames text = map (identifierText . enumCaseName) . snd <$> declaration text + memberValues text = map (fmap spelled . enumCaseValue) . snd <$> declaration text + -- The operators and operands of a value in the order they are written. + spelled :: Expression Identifier () -> String + spelled value = case value of + LiteralExpression _ (IntegerLiteral number) _ -> show number + NameExpression _ name _ -> identifierText name + UnaryExpression _ UnaryNegate operand _ -> "-" ++ spelled operand + BinaryExpression _ operator left right _ -> spelled left ++ sign operator ++ spelled right + _ -> "?" + sign operator = case operator of + Add -> "+" + Multiply -> "*" + ShiftLeft -> "<<" + BitwiseOr -> "|" + _ -> "?" + underlying text = fst <$> declaration text + +-- -------------------------------------------------------------- evaluation + +-- | Bodies of @Evaluate@, the arguments to run them on, and the value each run must return. +evaluationCases :: [(String, [((Integer, Integer), Integer)])] +evaluationCases = + [ -- Values of an enum are compared with == and \=. + ("Status s = Pick(left); return s == Status.READY ? 1 : 2;", [((3, 0), 1), ((0, 0), 2)]) + , ("Status s = Pick(left); return s \\= Status.NONE ? 1 : 2;", [((3, 0), 1), ((0, 0), 2)]) + , -- Two members with one value are equal. + ("return Status.NONE == Status.UNKNOWN ? 1 : 2;", [((0, 0), 1)]) + , -- Members are numbered from zero; one without a value follows the one before it. + ("return Rank(Level.LOW) + Rank(Level.MID) * 10 + Rank(Level.HIGH) * 100 + Rank(Level.TOP) * 1000;", [((0, 0), 12021)]) + , -- A match that names every value needs no catch-all arm. + ("return match (Pick(left)) { .NONE -> 10, .READY -> 20 };", [((3, 0), 20), ((0, 0), 10)]) + , -- A member is named through either of its names. + ("return match (Pick(left)) { .UNKNOWN -> 10, .READY -> 20 };", [((0, 0), 10)]) + , ("Level l = Level.MID; return match (l) { .LOW -> 1, _ -> 9 };", [((0, 0), 9)]) + , -- A guard does not count towards completeness, and a catch-all is then needed. + ("return match (Pick(left)) { .READY if (right > 0) -> 1, .READY -> 2, .NONE -> 3 };", [((1, 1), 1), ((1, 0), 2), ((0, 5), 3)]) + , -- An enum is assigned, passed, returned and yielded by a match. + ("Status s = Status.NONE; s = Status.READY; return s == Status.READY ? 7 : 8;", [((0, 0), 7)]) + , ("return Rank(Raise(Raise(Level.LOW))) * 100 + Rank(Raise(Level.HIGH));", [((0, 0), 1011)]) + , ("Status s = match (left) { 1 -> Status.READY, _ -> Status.NONE }; return s == Status.READY ? 1 : 0;", [((1, 0), 1), ((2, 0), 0)]) + , -- The target-typed spelling names a member of the enum the place + -- expects: a declared type, a parameter, a return type, the other + -- operand of a comparison, the variable assigned to. + ("Status s = .READY; return s == Status.READY ? 1 : 2;", [((0, 0), 1)]) + , ("Status s = Pick(left); return s == .READY ? 1 : 2;", [((3, 0), 1), ((0, 0), 2)]) + , ("return Rank(.HIGH) + Rank(.LOW);", [((0, 0), 11)]) + , ("Level l = .LOW; l = .TOP; return Rank(l);", [((0, 0), 11)]) + , ("Status s = match (left) { 1 -> .READY, _ -> .NONE }; return s == .READY ? 1 : 0;", [((1, 0), 1), ((2, 0), 0)]) + , -- A conditional, an if expression and a loop expression yield an enum. + ("Status s = left > 0 ? .READY : .NONE; return s == .READY ? 1 : 0;", [((1, 0), 1), ((0, 0), 0)]) + , ("Level l = if (left > 0) { Level.HIGH } else { Level.LOW }; return Rank(l);", [((1, 0), 10), ((0, 0), 1)]) + , ("Level l = while (true) { if (left > 0) { break Level.TOP; } break Level.MID; }; return Rank(l);", [((1, 0), 11), ((0, 0), 2)]) + , -- Two subjects with closed sets of values are complete together. + ( "return match (Pick(left)), (right > 0) { (.NONE), (true) -> 1, (.NONE), (false) -> 2, (.READY), (true) -> 3, (.READY), (false) -> 4 };" + , [((0, 1), 1), ((0, 0), 2), ((1, 1), 3), ((1, 0), 4)] + ) + , -- A member has the value its expression computes from the members + -- before it. Two members are equal exactly when their values are, so + -- comparing a member with the one a number selects reads its value. + ("return Bit(left) == Flag.WRITE ? 1 : 0;", [((2, 0), 1), ((1, 0), 0), ((4, 0), 0)]) + , ("return Bit(left) == Flag.EXECUTE ? 1 : 0;", [((4, 0), 1), ((2, 0), 0)]) + , ("return Bit(left) == Flag.ALL ? 1 : 0;", [((7, 0), 1), ((4, 0), 0)]) + , -- A member without a value follows a computed one: ALL is 7, NEXT is 8. + ("return Bit(left) == Flag.NEXT ? 1 : 0;", [((8, 0), 1), ((7, 0), 0)]) + , -- HALF is 4 / 2, the value of WRITE: the two names are one value. + ("return Flag.HALF == Flag.WRITE ? 1 : 0;", [((0, 0), 1)]) + , -- The complement is that of the underlying type: !5 in ubyte is 250. + ("return Bit(left) == Flag.REST ? 1 : 0;", [((250, 0), 1), ((5, 0), 0)]) + , -- 9 // 2 rounds to the nearest integer, away from zero at a half. + ("return Bit(left) == Flag.ROUNDED ? 1 : 0;", [((5, 0), 1), ((4, 0), 0)]) + , ("return At(left) == Offset.ABOVE ? 1 : 0;", [((2, 0), 1), ((-3, 0), 0)]) + , ("return At(left) == Offset.POWER ? 1 : 0;", [((1024, 0), 1), ((2, 0), 0)]) + , -- -7 % 4 is -3, the value of BASE; (2 ^ 7) & 6 is 4. + ("return Offset.MODULO == Offset.BASE ? 1 : 0;", [((0, 0), 1)]) + , ("return Offset.MIX == Offset.ABOVE ? 1 : 0;", [((0, 0), 0)]) + , -- A match over an enum with computed values is complete when every + -- value is named, through whichever of its names. + ("return match (Bit(left)) { .READ -> 1, .HALF -> 2, .EXECUTE -> 3, .ALL -> 4, .NEXT -> 5, .REST -> 6, .ROUNDED -> 7 };", [((1, 0), 1), ((2, 0), 2), ((4, 0), 3), ((7, 0), 4), ((8, 0), 5), ((250, 0), 6), ((5, 0), 7)]) + , -- An enum in a callable and in a loop. + ("auto f = \\(Status s) -> s == Status.READY ? 5 : 6; return f(Pick(left)) * 10 + f(Status.NONE);", [((2, 0), 56)]) + , ("Level l = Level.LOW; int n = 0; while (l \\= Level.TOP) { l = Raise(l); n += 1; } return n;", [((0, 0), 3)]) + ] + +-- ----------------------------------------------------------------- rejects + +rejectedBodies :: [(String, String, String)] +rejectedBodies = + [ ("a match over an enum must name every value", "VXT0052", "return match (Pick(left)) { .NONE -> 10 };") + , ("a guarded arm does not complete a match over an enum", "VXT0052", "return match (Pick(left)) { .NONE -> 1, .READY if (right > 0) -> 2 };") + , ("a second arm for the same value can never be selected", "VXT0053", "return match (Pick(left)) { .NONE -> 10, .UNKNOWN -> 11, .READY -> 20 };") + , ("an enum has only the members it declares", "VXT0064", "Status s = Status.MISSING; return 0;") + , ("a case pattern names a member of the subject's enum", "VXT0064", "return match (Pick(left)) { .LOW -> 1, _ -> 2 };") + , ("a case pattern needs a subject of an enum type", "VXT0056", "return match (left) { .NONE -> 1, _ -> 2 };") + , ("values of an enum are not ordered", "VXT0065", "return Status.NONE < Status.READY ? 1 : 2;") + , ("values of an enum are not added", "VXT0065", "Status s = Status.NONE + Status.READY; return 0;") + , ("values of different enums are not compared", "VXT0065", "return Status.NONE == Level.LOW ? 1 : 2;") + , ("a value of an enum is not compared with an integer", "VXT0065", "return Status.NONE == 0 ? 1 : 2;") + , ("a value of an enum is not an integer", "VXT0002", "int x = Status.READY; return x;") + , ("an integer is not a value of an enum", "VXT0002", "Status s = 1; return 0;") + , ("a value of an enum is not a condition", "VXT0006", "if (Status.READY) { return 1; } return 2;") + , ("the target-typed spelling needs a known target type", "VXT0069", "auto s = .READY; return 0;") + , ("the target-typed spelling needs an enum as its target", "VXT0069", "int x = .READY; return x;") + , ("the left operand of a comparison has no target type", "VXT0069", "Status s = Pick(left); return .NONE == s ? 1 : 2;") + , ("the target-typed spelling names a member of the expected enum", "VXT0064", "Status s = .LOW; return 0;") + , ("a call through the target-typed spelling has no type to call on", "VXT0032", ".Pick(left); return 0;") + , ("a value of one enum is not a value of another", "VXT0002", "Level l = Status.NONE; return 0;") + ] + +rejectedDeclarations :: [(String, String, String)] +rejectedDeclarations = + [ ("the underlying type of an enum is an integer type", "VXT0066", "enum A = float { X }\n") + , ("a member is named once", "VXT0067", "enum A { X, X }\n") + , ("a member value fits the underlying type", "VXT0068", "enum A = byte { X = 300 }\n") + , ("a numbered member fits the underlying type", "VXT0068", "enum A = byte { X = 127, Y }\n") + , ("an unsigned underlying type has no negative member", "VXT0068", "enum A = ubyte { X = -1 }\n") + , ("a computed member value fits the underlying type", "VXT0068", "enum A = byte { X = 100, Y = X + X }\n") + , ("a member value is an integer", "VXT0070", "enum A { X = 1.5 }\n") + , ("a member value names earlier members only", "VXT0070", "enum A { X = Y, Y = 1 }\n") + , ("a member value does not name the member itself", "VXT0070", "enum A { X = X + 1 }\n") + , ("a member value does not name a member of another enum", "VXT0070", "enum A { X = 1 }\nenum B { Y = A.X }\n") + , ("a member value is not a call", "VXT0070", "enum A { X = Program.Evaluate(1, 2) }\n") + , ("a member value is not a comparison", "VXT0070", "enum A { X = 1, Y = X < 2 }\n") + , ("a member value does not divide by zero", "VXT0070", "enum A { X = 0, Y = 4 / X }\n") + , ("a member value has no negative exponent", "VXT0070", "enum A { X = 2 ** -1 }\n") + , -- The value of a member is an expression like any other for the + -- nesting limit: 1100 negations in one another are deeper than 1024 + -- levels. + ( "a member value is within the expression nesting limit" + , "VXP0040" + , "enum A { X = " ++ concat (replicate 1100 "-(") ++ "1" ++ replicate 1100 ')' ++ " }\n" + ) + , ("the match of a member with a computed value is unreachable after its other name", "VXT0053", "enum A { X = 1, Y = X }\nclass P { public static int F(_ A a) { return match (a) { .X -> 1, .Y -> 2 }; } }\n") + , ("an enum is declared once", "VXR0001", "enum A { X }\nenum A { Y }\n") + , ("an enum and a class do not share a name", "VXR0001", "enum A { X }\nclass A { }\n") + ] + +-- ----------------------------------------------------------------- helpers + +parseSource :: String -> Either [Diagnostic] ParsedAST +parseSource text = do + tokens <- runLexer defaultLexer (LexerInput "enum.vxs" text) + runParser defaultParser (ParserInput "enum.vxs" tokens) + +parses :: String -> Bool +parses = either (const False) (const True) . parseSource + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "enum.vxs" text) + +accepted :: String -> Bool +accepted = either (const False) (const True) . compileSource + +rejectedWith :: String -> String -> Bool +rejectedWith code text = case compileSource text of + Left problems -> any ((== code) . diagnosticCode) problems + Right _ -> False + +runs :: (FrontendArtifacts -> CoreModule) -> String -> (Integer, Integer) -> Integer -> Bool +runs select statements (left, right) expected = case compileSource (program statements) of + Right artifacts -> + runFunction (select artifacts) "Evaluate" [IntegerValue left, IntegerValue right] == Just (IntegerValue expected) + Left _ -> False + +verifies :: String -> Bool +verifies statements = case compileSource (program statements) of + Right artifacts -> + verifyCore (artifactCore artifacts) == Right (artifactCore artifacts) + && verifyCore (artifactOptimizedCore artifacts) == Right (artifactOptimizedCore artifacts) + && verifyCorePrep (artifactCorePrep artifacts) == Right (artifactCorePrep artifacts) + Left _ -> False + +{- | Whether the Core of a program mentions no enum. The reserved root of an +enum type cannot be spelled in source, so its absence from the printed Core +shows that every enum was lowered to its underlying type. +-} +erased :: String -> Bool +erased statements = case compileSource (program statements) of + Right artifacts -> not ("Identifier \"enum\"" `isInfixOf` show (artifactCore artifacts)) + Left _ -> False diff --git a/Compiler/Haskell/Driver/test/FallThroughTests.hs b/Compiler/Haskell/Driver/test/FallThroughTests.hs new file mode 100644 index 00000000..d4f1a3c4 --- /dev/null +++ b/Compiler/Haskell/Driver/test/FallThroughTests.hs @@ -0,0 +1,151 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | How a function ends when its body ends. + +A function without a result may end without a @return@: reaching the end of +its body returns. A function with a result returns on every path, so the end +of its body is never reached, and the block that stands there is marked as +unreachable rather than given a value nobody wrote. + +The two must not be confused. A function without a result whose last block +was marked unreachable had no meaning once it was called, and a native +program whose @Main@ ended that way stopped instead of ending. +-} +module FallThroughTests (fallThroughTests) where + +import Visual.XSharp.AST +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep +import Visual.XSharp.Core.CorePrep.Verifier (verifyCorePrep) + +fallThroughTests :: [(String, Bool)] +fallThroughTests = + [ ("an empty body without a result returns", terminators (prepared unitType []) == [returnsNothing]) + , ("a body without a result that ends in a binding returns", terminators (prepared unitType [bind 2 1]) == [returnsNothing]) + , ("a body without a result that ends in a call returns", terminators (prepared unitType [CoreEvaluate callee]) == [returnsNothing]) + , ("a body without a result never ends in an unreachable block", all (noneUnreachable . prepared unitType) resultless) + , ("every body without a result passes verification", all (verified . prepared unitType) resultless) + , -- One return was written; the other is the end of the body. + ("the block after a conditional return returns", returns (prepared unitType [returnWhen]) == 2) + , ("the block after a loop returns", returns (prepared unitType [CoreWhile condition [bind 3 1]]) == 1) + , ("the block after a loop that is left early returns", returns (prepared unitType [CoreWhile condition [CoreBreak]]) == 1) + , ("the block after nested conditionals returns", returns (prepared unitType [nested]) == 2) + , ("an explicit return keeps its own value", terminators (prepared unitType [CoreReturn nothing]) == [returnsNothing]) + , ("a loop body still continues with its loop", loopBodyJumps) + , ("a body with a result is not given a value at its end", notElem returnsNothing (terminators (prepared intType [bothReturn]))) + , ("the end of a body with a result stays unreachable", elem CorePrepUnreachable (terminators (prepared intType [bothReturn]))) + , ("a body with a result that returns directly has one block", terminators (prepared intType [CoreReturn (integer 1)]) == [CorePrepReturn (CorePrepLiteral (CoreInteger 1) intType)]) + , ("a closure without a result returns at the end of its body", closureReturns) + , ("a closure with a result is not given a value at its end", closureWithResultIsNotCompleted) + ] + where + resultless = + [ [] + , [bind 2 1] + , [CoreEvaluate callee] + , [returnWhen] + , [returnWhen, bind 2 1] + , [CoreWhile condition [bind 3 1]] + , [CoreWhile condition [CoreBreak]] + , [CoreWhile condition [CoreContinue]] + , [CoreDoWhile [bind 3 1] condition] + , [CoreFor condition [bind 3 1] [bind 4 1]] + , [nested] + , [nested, returnWhen, CoreWhile condition []] + , [CoreIf condition [CoreReturn nothing] [CoreReturn nothing]] + ] + +-- | What reaching the end of a body without a result does. +returnsNothing :: CorePrepTerminator +returnsNothing = CorePrepReturn (CorePrepLiteral CoreUnit unitType) + +nothing :: CoreExpression +nothing = CoreLiteral CoreUnit unitType + +integer :: Integer -> CoreExpression +integer value = CoreLiteral (CoreInteger value) intType + +condition :: CoreExpression +condition = CoreVariable (name 20 "flag") boolType + +callee :: CoreExpression +callee = CoreApply (CoreVariable (name 21 "Act") (FunctionType [] unitType)) [] unitType + +bind :: Int -> Integer -> CoreStatement +bind symbol value = CoreBind (CoreBinding (name symbol "local") intType False (integer value)) + +-- | @if (flag) { return; }@ +returnWhen :: CoreStatement +returnWhen = CoreIf condition [CoreReturn nothing] [] + +-- | @if (flag) { if (flag) { return; } } else { local = 1; }@ +nested :: CoreStatement +nested = CoreIf condition [returnWhen] [bind 5 1] + +-- | @if (flag) { return 1; } else { return 2; }@ +bothReturn :: CoreStatement +bothReturn = CoreIf condition [CoreReturn (integer 1)] [CoreReturn (integer 2)] + +prepared :: Type -> [CoreStatement] -> CorePrepModule +prepared returnType statements = case prepareCore moduleValue of + Right moduleResult -> moduleResult + Left _ -> error "prepareCore cannot fail on these bodies" + where + moduleValue = CoreModuleWithSources (QualifiedName [Identifier "FallThrough"]) [function] [] [] + function = + CoreFunction + (name 1 "Evaluate") + [(name 20 "flag", boolType), (name 21 "Act", FunctionType [] unitType)] + returnType + statements + +terminators :: CorePrepModule -> [CorePrepTerminator] +terminators moduleValue = + [ corePrepBlockTerminator block + | function <- corePrepModuleFunctions moduleValue + , block <- corePrepFunctionBlocks function + ] + +-- | How many blocks return without a value. +returns :: CorePrepModule -> Int +returns = length . filter (== returnsNothing) . terminators + +noneUnreachable :: CorePrepModule -> Bool +noneUnreachable = notElem CorePrepUnreachable . terminators + +verified :: CorePrepModule -> Bool +verified = either (const False) (const True) . verifyCorePrep + +{- | The end of a loop body goes back to the loop, whatever the function +returns: only the end of the function's own body returns. +-} +loopBodyJumps :: Bool +loopBodyJumps = + length [() | CorePrepJump _ <- found] >= 2 + && length (filter (== returnsNothing) found) == 1 + where + found = terminators (prepared unitType [CoreWhile condition [bind 3 1]]) + +closureReturns :: Bool +closureReturns = case corePrepModuleFunctions moduleValue of + [_, lifted] -> map corePrepBlockTerminator (corePrepFunctionBlocks lifted) == [returnsNothing] + _ -> False + where + moduleValue = prepared unitType [CoreBind (CoreBinding (name 6 "act") callable False closure)] + callable = FunctionType [] unitType + closure = CoreClosure [] [] unitType [bind 7 1] callable + +closureWithResultIsNotCompleted :: Bool +closureWithResultIsNotCompleted = case corePrepModuleFunctions moduleValue of + [_, lifted] -> + let found = map corePrepBlockTerminator (corePrepFunctionBlocks lifted) + in notElem returnsNothing found && elem CorePrepUnreachable found + _ -> False + where + moduleValue = prepared unitType [CoreBind (CoreBinding (name 6 "pick") callable False closure)] + callable = FunctionType [] intType + closure = CoreClosure [] [] intType [bothReturn] callable + +name :: Int -> String -> ResolvedName +name symbol spelling = ResolvedName (SymbolId symbol) (Identifier spelling) diff --git a/Compiler/Haskell/Driver/test/FormatSweepTests.hs b/Compiler/Haskell/Driver/test/FormatSweepTests.hs new file mode 100644 index 00000000..ebd3d376 --- /dev/null +++ b/Compiler/Haskell/Driver/test/FormatSweepTests.hs @@ -0,0 +1,246 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Every conversion that can be written, against the rules stated a second time. + +"RuntimeCallTests" holds formats chosen by hand. This module writes all of +them: every set of flags, with and without a width and a precision, each +written as a number and as a star, for every conversion letter. What the +reader of formats makes of each is compared with a prediction that is +written here from the rules alone and shares nothing with the reader: + +* a conversion takes the flags of its row and no other; +* @-@ excludes @0@, and @+@ excludes the space; +* a precision is taken by @%f@ and @%s@; +* @%n@ and @%%@ take nothing at all. + +The same sweep is then run through the reference text functions, where the +properties of a field hold whatever the conversion: it is never shorter than +its width, padding is all it adds, and a value at the left is followed by +spaces only. +-} +module FormatSweepTests (formatSweepTests) where + +import Data.Char (isDigit) +import Data.List (isPrefixOf, isSuffixOf, subsequences) +import RuntimeText qualified as Reference +import Visual.XSharp.RuntimeCall +import Visual.XSharp.TypeChecker.Format + +formatSweepTests :: [(String, Bool)] +formatSweepTests = + [ ("every way to write %" ++ [letter] ++ " is read as the rules say", all agrees (written letter)) + | letter <- letters + ] + ++ [ ("the sweep writes every combination", length (concatMap written letters) == 9 * 64 * 3 * 3) + , ("the sweep holds formats that are accepted", any ((== Accepted) . predicted) (concatMap written letters)) + , ("the sweep holds formats that are rejected", any ((== Rejected) . predicted) (concatMap written letters)) + , ("a letter outside the grammar is never a conversion", all (unknown . (: [])) (filter (`notElem` letters ++ "AO") ['a' .. 'z'] ++ ['B' .. 'N'])) + , ("an integer field is never shorter than its width", and [length (integerField value) >= maybe 0 id width | value@(_, width, _) <- integerCases]) + , ("an integer field without a width is the number itself", and [integerField (flags, Nothing, value) == unpadded flags value | (flags, _, value) <- integerCases]) + , ("padding is all a width adds to an integer", all paddingOnly integerCases) + , ("a value at the left is followed by spaces only", all leftAligned integerCases) + , ("zeros stand after the sign", all zerosAfterSign integerCases) + , ("grouping adds apostrophes and nothing else", all groupingOnly [value | (_, _, value) <- integerCases]) + , ("groups have three digits, the first at most three", all groupsOfThree [value | (_, _, value) <- integerCases]) + , ("hexadecimal digits are those of the magnitude", all hexadecimalDigits [value | (_, _, value) <- integerCases]) + , ("a text field is never shorter than its width", and [length (textField width Nothing value) >= width | width <- [0 .. 9], value <- texts]) + , ("a precision is the most characters a text field keeps", and [textField 0 (Just precision) value == take precision value | precision <- [0 .. 9], value <- texts]) + , ("a fixed-point field has its precision's digits after the point", all fractionDigits fixedCases) + , ("a fixed-point number with no digits after the point has no point", and ['.' `notElem` fixed 0 value | value <- rationals]) + , ("more digits never change the digits before them by more than rounding", all prefixStable rationals) + ] + where + letters = "duxfscbn%" + +-- ------------------------------------------------------------ the reader + +-- | What the rules say of a format. +data Verdict = Accepted | Rejected + deriving (Eq, Show) + +-- | One conversion as it is written: flags, width, precision and letter. +type Written = (String, String, String, Char) + +-- | Every way to write a conversion with the given letter. +written :: Char -> [Written] +written letter = + [ (flags, width, precision, letter) + | flags <- subsequences "-0+ #'" + , width <- ["", "7", "*"] + , precision <- ["", ".3", ".*"] + ] + +spelled :: Written -> String +spelled (flags, width, precision, letter) = "%" ++ flags ++ width ++ precision ++ [letter] + +-- | The rules, from the specification, without the reader. +predicted :: Written -> Verdict +predicted (flags, width, precision, letter) + | letter `elem` "n%" = if null flags && null width && null precision then Accepted else Rejected + | any (`notElem` taken) flags = Rejected + | '-' `elem` flags && '0' `elem` flags = Rejected + | '+' `elem` flags && ' ' `elem` flags = Rejected + | not (null precision) && letter `notElem` "fs" = Rejected + | otherwise = Accepted + where + taken = case letter of + 'd' -> "-0+ '" + 'u' -> "-0'" + 'x' -> "-0#" + 'f' -> "-0+ '" + _ -> "-" + +{- | Whether the reader agrees with the rules, and, for an accepted +conversion, reads what was written: the flags as their bits, and one +argument more for each star. +-} +agrees :: Written -> Bool +agrees value@(flags, width, precision, letter) = case (predicted value, parseFormat (spelled value)) of + (Rejected, Left problem) -> formatProblemCode problem == "VXT0076" + (Accepted, Right [ConversionPiece conversion]) -> + letter `notElem` "n%" + && conversionLetter (conversionKind conversion) == letter + && conversionFlagBits conversion == sum (map bit flags) + (if letter == 'x' then flagHexadecimal else 0) + && conversionArgumentCount conversion == 1 + length (filter (== "*") [width, drop 1 precision]) + && conversionWidth conversion == size width + && conversionPrecision conversion == size (drop 1 precision) + (Accepted, Right [NewlinePiece]) -> letter == 'n' + (Accepted, Right [LiteralPiece "%"]) -> letter == '%' + _ -> False + where + bit flag = case flag of + '-' -> flagLeft + '0' -> flagZero + '+' -> flagPlus + ' ' -> flagSpace + '#' -> flagAlternate + _ -> flagGroup + size text + | null text = NoSize + | text == "*" = ArgumentSize + | otherwise = FixedSize (read text) + +unknown :: String -> Bool +unknown letter = case parseFormat ('%' : letter) of + Left problem -> formatProblemCode problem == "VXT0075" + Right _ -> False + +-- --------------------------------------------------- the reference fields + +-- | Flags, a width when there is one, and a value. +type IntegerCase = (Integer, Maybe Int, Integer) + +integerCases :: [IntegerCase] +integerCases = + [ (flags, width, value) + | flags <- [0, flagLeft, flagZero, flagPlus, flagSpace, flagGroup, flagPlus + flagZero, flagGroup + flagLeft, flagHexadecimal, flagHexadecimal + flagAlternate, flagHexadecimal + flagZero] + , width <- [Nothing, Just 0, Just 1, Just 5, Just 12, Just 30] + , value <- values + ] + where + values = + [0, 1, -1, 9, 10, -10, 99, 100, 999, 1000, -1000, 12345, 123456, 1234567, -1234567] + ++ [2 ^ (63 :: Int) - 1, negate (2 ^ (63 :: Int)), 2 ^ (64 :: Int) - 1] + +fieldOf :: Integer -> Maybe Int -> Reference.Conversion +fieldOf flags width = Reference.Conversion flags (maybe absent toInteger width) absent + +integerField :: IntegerCase -> String +integerField (flags, width, value) = Reference.formatInteger (fieldOf flags width) value + +unpadded :: Integer -> Integer -> String +unpadded flags = Reference.formatInteger (fieldOf flags Nothing) + +has :: Integer -> Integer -> Bool +has flags flag = (flags `div` flag) `mod` 2 == 1 + +-- | A field is the number without a width, with padding and nothing else. +paddingOnly :: IntegerCase -> Bool +paddingOnly value@(flags, width, number) = + length field == max (maybe 0 id width) (length plain) + && filter (`notElem` " 0") plain `isSubsequence` field + && length field >= maybe 0 id width + where + field = integerField value + plain = unpadded flags number + isSubsequence [] _ = True + isSubsequence _ [] = False + isSubsequence (x : xs) (y : ys) + | x == y = isSubsequence xs ys + | otherwise = isSubsequence (x : xs) ys + +leftAligned :: IntegerCase -> Bool +leftAligned value@(flags, _, number) + | has flags flagLeft = plain `isPrefixOf` field && all (== ' ') (drop (length plain) field) + | otherwise = True + where + field = integerField value + plain = unpadded flags number + +-- | With the zero flag the field ends in the digits and begins with the sign. +zerosAfterSign :: IntegerCase -> Bool +zerosAfterSign value@(flags, _, number) + | has flags flagZero && not (has flags flagLeft) = + ' ' `notElem` dropWhile (== ' ') (drop (length sign) field) + && sign `isPrefixOf` field + && digits `isSuffixOf` field + | otherwise = True + where + field = integerField value + plain = unpadded flags number + (sign, digits) = span (`elem` "+- ") plain + +groupingOnly :: Integer -> Bool +groupingOnly value = filter (/= '\'') (unpadded flagGroup value) == unpadded 0 value + +groupsOfThree :: Integer -> Bool +groupsOfThree value = case groups (dropWhile (== '-') (unpadded flagGroup value)) of + first : rest -> not (null first) && length first <= 3 && all ((== 3) . length) rest && all (all isDigit) (first : rest) + [] -> False + where + groups text = case break (== '\'') text of + (group, _ : more) -> group : groups more + (group, []) -> [group] + +hexadecimalDigits :: Integer -> Bool +hexadecimalDigits value = + all (`elem` "0123456789abcdef") digits + && foldl (\total digit -> total * 16 + toInteger (position digit)) 0 digits == abs value + && (value < 0) == ("-" `isPrefixOf` field) + where + field = unpadded flagHexadecimal value + digits = dropWhile (== '-') field + position digit = length (takeWhile (/= digit) "0123456789abcdef") + +texts :: [String] +texts = ["", "a", "abc", "abcdefghij", "\233\8364"] + +textField :: Int -> Maybe Int -> String -> String +textField width precision = Reference.formatText (Reference.Conversion 0 (toInteger width) (maybe absent toInteger precision)) + +rationals :: [Rational] +rationals = [0, 1, 1 / 2, 1 / 4, 1 / 8, 3 / 8, 5 / 2, 7 / 2, 12345 / 8, 1 / 1024, 999999 / 1000, 1234567 / 2] + +fixed :: Integer -> Rational -> String +fixed precision = Reference.formatFloating (Reference.Conversion 0 absent precision) False + +fixedCases :: [(Integer, Rational)] +fixedCases = [(precision, value) | precision <- [1 .. 12], value <- rationals] + +fractionDigits :: (Integer, Rational) -> Bool +fractionDigits (precision, value) = case break (== '.') (fixed precision value) of + (whole, _ : fraction) -> not (null whole) && all isDigit whole && length fraction == fromInteger precision && all isDigit fraction + _ -> False + +{- | With a precision large enough to hold every digit of the value, a +larger precision only appends zeros. These values are sums of powers of two +with at most ten binary places, or thousandths, so twelve digits hold the +binary ones exactly. +-} +prefixStable :: Rational -> Bool +prefixStable value + | exact = fixed 20 value == fixed 12 value ++ replicate 8 '0' + | otherwise = True + where + exact = fromRational (value * 1024) == (fromInteger (round (value * 1024)) :: Rational) diff --git a/Compiler/Haskell/Driver/test/InferredReturnTests.hs b/Compiler/Haskell/Driver/test/InferredReturnTests.hs new file mode 100644 index 00000000..5f20c317 --- /dev/null +++ b/Compiler/Haskell/Driver/test/InferredReturnTests.hs @@ -0,0 +1,221 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Tests for return types that are inferred, and for callables that are +called. + +A method declared with @auto@ takes its return type from its returns, and a +caller must see that type wherever the method is declared: later in the +class, in another class, or behind a chain of other such methods. A method +all of whose results depend on itself has none. A callable takes its return +type the same way, and may be created inside another callable, whose +captures it reads. + +The expected values are written by hand. Every program runs in the reference +evaluator of "CoreInterpreter" on the unoptimized and on the optimized Core, +and is verified as Core and as CorePrep. +-} +module InferredReturnTests (inferredReturnTests) where + +import CoreInterpreter +import Visual.XSharp.Compiler +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep.Verifier +import Visual.XSharp.Core.Verifier +import Visual.XSharp.Diagnostic + +inferredReturnTests :: [(String, Bool)] +inferredReturnTests = + concat + [ [ ("unoptimized Core computes " ++ label, runs artifactCore source left right expected) + , ("optimized Core computes " ++ label, runs artifactOptimizedCore source left right expected) + ] + | (source, runsOfCase) <- map inMethods methodCases ++ map inMethods closureCases ++ classCases + , ((left, right), expected) <- runsOfCase + , let label = show expected ++ " for " ++ show (left, right) ++ ": " ++ lastLine source + ] + ++ [ ("every program verifies as Core and as CorePrep", all verifies (map (fst . inMethods) (methodCases ++ closureCases) ++ map fst classCases)) + , ("a method whose only result is a call of itself has no return type", rejectedWith "VXT0063" (methods "return 0;" ++ loops)) + , ("two methods that only return each other have no return type", rejectedWith "VXT0063" (methods "return 0;" ++ cycle2)) + , ("returns of different types in an inferred method are reported", rejectedWith "VXT0062" (methods "return 0;" ++ mixed)) + , ("an inferred return type is checked at the call", rejected (methods "int x = Even(2); return x;")) + , ("a declared return type is still checked against its returns", rejectedWith "VXT0001" (methods "return 0;" ++ declared)) + , ("a method without a result is inferred as void", accepted (methods "Nothing(); return 0;")) + ] + where + inMethods (statements, runsOfCase) = (methods statements, runsOfCase) + lastLine = last . lines + +-- | A class around the given body of @Evaluate@, closed by the caller. +methods :: String -> String +methods statements = + unlines + [ "namespace Test;" + , "class Helper {" + , " public static auto Five() { return Program.Base() + 4; }" + , "}" + , "class Program {" + , " public static auto Twice(_ int value) { return value + value; }" + , " public static auto Factorial(_ int value) { if (value <= 1) { return 1; } return value * Factorial(value - 1); }" + , " public static auto First(_ int v) { return Second(v) + 1; }" + , " public static auto Second(_ int v) { return Third(v) + 10; }" + , " public static auto Third(_ int v) { return v * 2; }" + , " public static auto Even(_ int v) { if (v == 0) { return true; } return Odd(v - 1); }" + , " public static auto Odd(_ int v) { if (v == 0) { return false; } return Even(v - 1); }" + , " public static auto Pick(_ int v) { int q = if (v > 0) { return v * 3; } else { 5 }; return q + 1; }" + , " public static auto Scan(_ int v) { int q = while (true) { if (v > 3) { return 7; } break 2; }; return q + v; }" + , " public static auto Base() { return 1; }" + , " public static auto Nothing() { }" + , " public static int Apply(_ (int) -> int f, _ int v) { return f(v); }" + , " public static (int) -> int Adder(_ int n) { return \\(int w) -> w + n; }" + , " public static (int) -> int Pass(_ (int) -> int f) { return f; }" + , " public static (int) -> int Choose(_ bool c, _ (int) -> int a, _ (int) -> int b) { if (c) { return a; } return b; }" + , " public static (int) -> int Compose2(_ (int) -> int f) { return \\(int w) -> f(f(w)); }" + , " public static int Plain(_ int value) { return value + value; }" + , " public static int Evaluate(_ int left, _ int right) {" + , " " ++ statements + , " }" + , "}" + ] + +-- Declarations that are appended as further classes. +loops, cycle2, mixed, declared :: String +loops = "class Loops { public static auto Loop() { return Loop(); } }\n" +cycle2 = "class Cycle { public static auto A() { return B(); } public static auto B() { return A(); } }\n" +mixed = "class Mixed { public static auto Pick(_ bool f) { if (f) { return 1; } return true; } }\n" +declared = "class Declared { public static int Wrong() { return true; } }\n" + +-- | Bodies of @Evaluate@ that call methods with inferred return types. +methodCases :: [(String, [((Integer, Integer), Integer)])] +methodCases = + [ ("return Twice(left) + Factorial(right);", [((3, 4), 30), ((0, 1), 1)]) + , -- A chain is inferred from its end, against the order of declaration. + ("return First(left);", [((4, 0), 19)]) + , -- Mutual recursion is inferred from the base cases. + ("return (Even(left) ? 10 : 20) + (Odd(left) ? 1 : 2);", [((4, 0), 12), ((3, 0), 21)]) + , -- Returns that stand in a value block and in a loop expression. + ("return Pick(left) * 10 + Pick(0 - left);", [((2, 0), 66)]) + , ("return Scan(left) * 100 + Scan(left + 3);", [((1, 0), 307)]) + , -- The inferred type is the type of the call. + ("int x = Twice(left); bool b = Even(x); return b ? x : 0 - x;", [((3, 0), 6)]) + , -- A method of another class, which calls back into this one. + ("return Helper.Five() * left;", [((3, 0), 15)]) + ] + +-- | Bodies of @Evaluate@ that create callables and call them. +closureCases :: [(String, [((Integer, Integer), Integer)])] +closureCases = + [ -- A callable created inside a callable. + ("auto outer = \\(int v) -> { auto inner = \\(int w) -> w + 1; return inner(v) * 2; }; return outer(left);", [((3, 0), 8)]) + , -- The inner callable reads a local of the method through the outer one. + ( "int k = left; auto outer = \\(int v) -> { auto inner = \\(int w) -> w + k + v; return inner(v) * 2; }; return outer(right);" + , [((3, 3), 18), ((5, 1), 14)] + ) + , -- The same with the captures written out. + ( "int k = left; auto outer = [k] \\(int v) -> { auto inner = [k, v] \\(int w) -> w + k + v; return inner(v); }; return outer(right);" + , [((3, 3), 9)] + ) + , -- Three levels; every level reads a name of each level around it. + ( "int k = left; auto a = \\(int v) -> { auto b = \\(int w) -> { auto c = \\(int x) -> x + w * 10 + v * 100 + k * 1000; return c(1); }; return b(2); }; return a(3);" + , [((4, 0), 4321)] + ) + , -- A callable outlives the call that created it and keeps its capture. + ( "auto make = \\(int v) -> { auto inner = \\(int w) -> w + v; return inner; }; auto f = make(left); auto g = make(right); return f(2) * 100 + g(2);" + , [((5, 7), 709)] + ) + , -- A capture initializer is evaluated once, where the callable is created. + ( "int k = left; auto held = [kept = k] \\ -> kept; k += 10; return held() * 100 + k;" + , [((1, 0), 111)] + ) + , -- The return type of a callable is inferred through its expressions. + ( "auto pick = \\(int v) -> { int q = if (v > 0) { return 1; } else { 2 }; return q; }; return pick(left) * 10 + pick(0 - left);" + , [((3, 0), 12)] + ) + , + ( "auto scan = \\(int v) -> { int q = while (true) { if (v > 3) { return 7; } break 2; }; return q + v; }; return scan(left) * 100 + scan(left + 3);" + , [((1, 0), 307)] + ) + , -- A callable passed to a method, made by one and returned by one. + ("return Apply(\\(int w) -> w * 3, left);", [((4, 0), 12)]) + , ("auto add = Adder(left); auto ten = Adder(10); return add(right) * 100 + ten(1);", [((5, 2), 711)]) + , ("auto f = \\(int w) -> w + 1; auto g = Pass(f); return g(left) + f(left);", [((3, 0), 8)]) + , + ( "auto a = \\(int w) -> w + 1; auto b = \\(int w) -> w * 2; auto c = Choose(left > right, a, b); return c(left);" + , [((5, 1), 6), ((5, 9), 10)] + ) + , -- A method named where a value is expected. + ("auto f = Plain; return Apply(f, left) + Apply(Plain, right);", [((3, 4), 14)]) + , -- A callable variable assigned again, in a loop and from itself. + ( "auto f = Adder(0); for (int i = 1; i <= left; i += 1) { f = Adder(i); } return f(100);" + , [((3, 0), 103), ((0, 0), 100)] + ) + , ("auto f = Adder(1); f = Compose2(f); return f(left);", [((5, 0), 7)]) + , -- Callables that exist on one branch only. + ( "int r = 0; if (left > right) { auto f = Adder(left); r = f(1); } else { auto g = Adder(right); auto h = Compose2(g); r = h(1); } return r;" + , [((4, 1), 5), ((1, 3), 7)] + ) + , -- A callable result that nothing receives. + ("_ = Adder(left); auto f = Adder(2); _ = Adder(3); return f(left);", [((5, 0), 7)]) + , -- A callable kept alive only by the callable that captured it. + ("auto inner = Adder(left); auto outer = \\(int w) -> inner(w) * 2; return outer(right);", [((5, 1), 12)]) + , -- A callable made in every pass of a loop that returns from its middle. + ( "for (int i = 0; i < 5; i += 1) { auto f = Adder(i); if (f(left) > 6) { return f(100); } } return 0;" + , [((4, 0), 103), ((0, 0), 0)] + ) + , -- Callables alive across the transfers of a loop header. + ( "int n = 0; int t = 0; while (if (n >= left) { break; } else { true }) { auto step = [by = n] \\ -> by + 1; n = step(); t += n; } return t;" + , [((3, 0), 6)] + ) + , + ( "int t = 0; for (int i = 0; i < 10; i += if (i == left) { break; } else { 1 }) { auto add = [by = i] \\(int w) -> w + by; t = add(t); } return t;" + , [((3, 0), 6)] + ) + ] + +-- | Whole sources: an inferred method that only another class declares. +classCases :: [(String, [((Integer, Integer), Integer)])] +classCases = + [ + ( unlines + [ "namespace Test;" + , "class Program {" + , " public static int Evaluate(_ int left, _ int right) { return Later.Sum(left, right) + Later.Flag(left); }" + , "}" + , "class Later {" + , " public static auto Sum(_ int a, _ int b) { return Double(a) + b; }" + , " public static auto Flag(_ int a) { return Positive(a) ? 100 : 200; }" + , " private static auto Double(_ int a) { return a + a; }" + , " private static auto Positive(_ int a) { return a > 0; }" + , "}" + ] + , [((3, 4), 110), ((0, 1), 201)] + ) + ] + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "inferred.vxs" text) + +accepted :: String -> Bool +accepted = either (const False) (const True) . compileSource + +rejected :: String -> Bool +rejected = not . accepted + +rejectedWith :: String -> String -> Bool +rejectedWith code text = case compileSource text of + Left problems -> any ((== code) . diagnosticCode) problems + Right _ -> False + +runs :: (FrontendArtifacts -> CoreModule) -> String -> Integer -> Integer -> Integer -> Bool +runs select source left right expected = case compileSource source of + Right artifacts -> + runFunction (select artifacts) "Evaluate" [IntegerValue left, IntegerValue right] == Just (IntegerValue expected) + Left _ -> False + +verifies :: String -> Bool +verifies source = case compileSource source of + Right artifacts -> + verifyCore (artifactCore artifacts) == Right (artifactCore artifacts) + && verifyCore (artifactOptimizedCore artifacts) == Right (artifactOptimizedCore artifacts) + && verifyCorePrep (artifactCorePrep artifacts) == Right (artifactCorePrep artifacts) + Left _ -> False diff --git a/Compiler/Haskell/Driver/test/IterationTests.hs b/Compiler/Haskell/Driver/test/IterationTests.hs index 4bf6bf83..137f0328 100644 --- a/Compiler/Haskell/Driver/test/IterationTests.hs +++ b/Compiler/Haskell/Driver/test/IterationTests.hs @@ -33,7 +33,7 @@ iterationTests = , ("continue outside a loop is rejected", continueOutsideLoopIsRejected) , ("value-carrying break remains an explicit unsupported feature", valuedBreakIsRejected) , ("enumerable for remains guarded by its missing generator ABI", forEachIsRejected) - , ("Core v8 round-trips all structured loop statement tags", loopCoreRoundTrips) + , ("Core v10 round-trips all structured loop statement tags", loopCoreRoundTrips) , ("loop CorePrep survives its verifier", loopCorePrepVerifies) ] diff --git a/Compiler/Haskell/Driver/test/LazyEvaluationTests.hs b/Compiler/Haskell/Driver/test/LazyEvaluationTests.hs new file mode 100644 index 00000000..ebcdfc97 --- /dev/null +++ b/Compiler/Haskell/Driver/test/LazyEvaluationTests.hs @@ -0,0 +1,281 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Tests for evaluation by need. + +Visual X# computes a value when it is first needed and at most once, and +never computes a value that is not needed. The tests observe that through +the only things that tell a computed value from one that was not: a division +by zero, a call that never returns, and the number of steps a program takes. +The reference evaluator of "CoreInterpreter" gives no result for a division +by zero and for a program that exceeds its step budget, so a value computed +without need shows as a missing result. + +The effects of a program are not lazy. A store into a variable happens where +it is written, whether or not the value around it is ever read, and a value +means what its variables held where it was bound, however late it is +computed. The tests pin both. + +An argument is a value like any other. A method receives it by need: it is +computed when the method first needs it, at most once however often the +method reads it and however many methods it is handed through, and not at +all when no one needs it. The caller and the method share the one +computation when both need the value. + +The expected values are written by hand. +-} +module LazyEvaluationTests (lazyEvaluationTests) where + +import CoreInterpreter +import Data.List (isInfixOf) +import Visual.XSharp.Compiler +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep.Verifier +import Visual.XSharp.Core.Verifier +import Visual.XSharp.Diagnostic + +lazyEvaluationTests :: [(String, Bool)] +lazyEvaluationTests = + concat + [ [ ("unoptimized Core computes " ++ label, runs artifactCore statements arguments == Just expected) + , ("optimized Core computes " ++ label, runs artifactOptimizedCore statements arguments == Just expected) + ] + | (statements, runsOfCase) <- evaluationCases + , (arguments, expected) <- runsOfCase + , let label = show expected ++ " for " ++ show arguments ++ ": " ++ statements + ] + ++ [ ("every lazy program verifies as Core and as CorePrep", all (verifies . fst) evaluationCases) + , -- A value that is needed and cannot be computed is still a failure. + ("a needed division by zero has no result", runs artifactCore "int x = left / right; return x;" (1, 0) == Nothing) + , + ( "a needed division by zero has no result on one path" + , runs artifactCore neededOnOnePath (6, 0) == Just 7 && runs artifactCore neededOnOnePath (6, 3) == Just 2 + ) + , ("a discarded value is evaluated", runs artifactCore "_ = left / right; return 5;" (1, 0) == Nothing) + , -- A value that is never needed leaves nothing behind in Core. + ("a value that is never read is not in the lowered Core", not (mentions "CoreDivide" artifactCore "int x = left / right; return 5;")) + , -- A value that the next statement is certain to need is + -- computed where it is bound, without a flag. + ("a value needed at once is computed in place", not (mentions "$known" artifactCore "int x = Half(left); return x + x;")) + , ("a value needed on one path only keeps its flag", mentions "$known" artifactCore neededOnOnePath) + , -- A value needed fifty times is computed once: the budget that + -- suffices for one computation does not suffice for fifty. + ("a value is computed at most once", runsWithin 2000 neededOften (100, 1) == Just 5000) + , ("the same work fifty times exceeds that budget", runsWithin 2000 computedOften (100, 1) == Nothing) + , -- An argument the method reads fifty times is computed once. + ("an argument is computed at most once", runsWithin 2000 "return Often(right, Step(left));" (100, 1) == Just 5000) + , -- The caller and the method it calls need the same value: + -- whoever needs it first computes it for both. The budget that + -- suffices for one computation does not suffice for two. + ("a value the caller and the method both need is computed once", runsWithin 800 sharedWithCallee (300, 1) == Just 600) + , ("the same work in the caller and in the method exceeds that budget", runsWithin 800 computedTwice (300, 1) == Nothing) + , -- An argument that may fail is handed on as a suspended + -- computation; one that cannot is passed as the value it is. + ("an argument that may fail is suspended", mentions "CoreMemoize" artifactCore "return Pick(right, left / right);") + , ("an argument that cannot fail is passed as a value", not (mentions "CoreMemoize" artifactCore "return Pick(right, left + 1);")) + , -- The method itself is unchanged: it still takes values, for + -- every caller that has them. + ("a method with a parameter by need still takes values", calls "Pick" [1, 5] == Just 5 && calls "Pick" [0, 5] == Just 7) + , -- Pending, and pinned so that a change shows: a callable + -- receives its arguments as values, and an argument that a + -- closure of the method captures is computed when the method + -- is entered. + ("an argument of a callable is still computed at the call", runs artifactCore throughCallable (6, 0) == Nothing) + , ("a captured argument is still computed on entry", runs artifactCore "return Kept(0, left / right);" (1, 0) == Nothing) + , -- A value that reads another value by need twice does not + -- carry the computation of that value twice: the Core of a + -- chain grows with its length, not with a power of it. + ("a chain of values read twice each grows linearly", coreSize (sharedChain 12) < 3 * coreSize (sharedChain 6)) + ] + where + neededOnOnePath = "int x = left / right; if (right > 0) { return x; } return 7;" + neededOften = "int x = Step(left); int t = 0; for (int i = 0; i < 50; i += 1) { if (right > 0) { t += x; } } return t;" + computedOften = "int t = 0; for (int i = 0; i < 50; i += 1) { if (right > 0) { t += Step(left); } } return t;" + sharedWithCallee = "int x = Step(left); int y = Pick(right, x); return right > 0 ? x + y : y;" + computedTwice = "return Step(left) + Pick(right, Step(left));" + throughCallable = "auto f = \\(int a, int b) -> a > 0 ? b : 7; return f(right, left / right);" + calls name arguments = case compileSource (program "return 0;") of + Right artifacts -> case runFunction (artifactCore artifacts) name (map IntegerValue arguments) of + Just (IntegerValue value) -> Just value + _ -> Nothing + Left _ -> Nothing + +-- | A program around the given body of @Evaluate@. +program :: String -> String +program statements = + unlines + [ "namespace Test;" + , "class Program {" + , " public static int Never(_ int v) { return Never(v + 1); }" + , " public static int Half(_ int v) { return v / 2; }" + , " public static int Step(_ int n) { return n > 0 ? 1 + Step(n - 1) : 0; }" + , " public static int Pick(_ int flag, _ int value) { if (flag > 0) { return value; } return 7; }" + , " public static int Pass(_ int flag, _ int value) { return Pick(flag, value); }" + , " public static int Both(_ int flag, _ int first, _ int second) { if (flag > 0) { return first; } return second; }" + , " public static int Kept(_ int flag, _ int value) { auto read = \\(int more) -> value + more; if (flag > 0) { return read(1); } return 3; }" + , " public static int Often(_ int flag, _ int value) { int total = 0; for (int i = 0; i < 50; i += 1) { if (flag > 0) { total += value; } } return total; }" + , " public static int Down(_ int n, _ int spare) { if (n <= 0) { return 0; } return 1 + Down(n - 1, spare / 0); }" + , " public static int Evaluate(_ int left, _ int right) {" + , " " ++ statements + , " }" + , "}" + ] + +-- | Bodies of @Evaluate@, the arguments to run them on, and the value each run must return. +evaluationCases :: [(String, [((Integer, Integer), Integer)])] +evaluationCases = + [ -- A value that is never needed is never computed: not a division by + -- zero, and not a call that never returns. + ("int x = left / right; return 5;", [((1, 0), 5)]) + , ("int x = Never(left); return 5;", [((1, 0), 5)]) + , -- A value is computed on the path that needs it and not on the other. + ("int x = left / right; if (right > 0) { return x; } return 7;", [((6, 0), 7), ((6, 3), 2)]) + , ("int x = left / right; return match (right) { 0 -> 1, _ -> x };", [((6, 0), 1), ((6, 2), 3)]) + , ("int x = left / right; return right > 0 && x > 1 ? 1 : 2;", [((6, 0), 2), ((6, 3), 1), ((1, 3), 2)]) + , ("int x = left / right; int r = if (right > 0) { x + 1 } else { 0 }; return r;", [((6, 0), 0), ((6, 2), 4)]) + , -- A value computed from a value that is not needed is not needed either. + ("int x = left / right; int y = x + 1; return right > 0 ? y : 9;", [((6, 0), 9), ((6, 3), 3)]) + , ("int x = Never(left); int y = x + 1; int z = y * 2; return right > 0 ? 1 : 2;", [((0, 1), 1), ((0, 0), 2)]) + , -- In a loop, each pass has its own value, computed if that pass needs it. + ( "int t = 0; for (int i = 0; i <= 3; i += 1) { int x = left / i; if (i > 1) { t += x; } } return t;" + , [((12, 0), 10)] + ) + , ("int limit = left / right; int n = 0; while (right > 0 && n < limit) { n += 1; } return n;", [((6, 0), 0), ((6, 2), 3)]) + , -- A value means what its variables held where it was bound. + ("int a = left; int x = Half(a) + a; a = 100; return x * 1000 + a;", [((8, 0), 12100)]) + , + ( "int a = left; int x = Half(a); a = a + 100; if (right > 0) { return x; } return a;" + , [((8, 1), 4), ((8, 0), 108)] + ) + , ("int t = 0; for (int i = 1; i <= 3; i += 1) { int x = Half(i * 10); if (i == 2) { t = x; } } return t;", [((0, 0), 10)]) + , -- A store happens where it is written, whether or not the value + -- around it is ever read. + ("int n = 0; int x = (n += 1) + left; return n * 10;", [((1, 0), 10)]) + , ("int n = left; int x = n++ + Half(n); return n;", [((4, 0), 5)]) + , -- The body of a callable evaluates by need as the body of a method + -- does, also when the callable stands inside another. + ( "auto f = \\(int v, int d) -> { int q = v / d; if (d > 0) { return q; } return 7; }; return f(left, right);" + , [((6, 0), 7), ((6, 3), 2)] + ) + , ("auto f = \\(int v) -> { int q = Never(v); return 5; }; return f(left);", [((1, 0), 5)]) + , + ( "auto f = \\(int v, int d) -> { auto g = \\(int w) -> { int q = w / d; return d > 0 ? q : 9; }; return g(v); }; return f(left, right);" + , [((6, 0), 9), ((6, 2), 3)] + ) + , ("auto f = \\(int v) -> { int a = v; int q = Half(a + 2); a = 100; return q * 1000 + a; }; return f(left);", [((8, 0), 5100)]) + , -- A value that needs another value twice computes it once, also + -- through a long chain of such values. + (sharedChain 40, [((6, 0), 7), ((6, 3), 1)]) + , ("int x = left / right; int y = right > 0 ? x / 1 + x / 2 : 0; return right > 1 ? y : 9;", [((6, 0), 9), ((6, 1), 9), ((6, 3), 3)]) + , -- A value by need is computed where a loop header or a guard reads it. + ( "int x = left / right; int t = 0; for (int i = 0; right > 0 && i < 3; i += x) { t += 1; } return t;" + , [((6, 0), 0), ((6, 3), 2)] + ) + , ("int x = left / right; int n = 0; do { n += 1; } while (right > 0 && n < x); return n;", [((6, 0), 1), ((6, 2), 3)]) + , ("int x = left / right; return match (right) { 0 -> 1, _ if x > 1 -> 2, _ -> 3 };", [((6, 0), 1), ((6, 3), 2), ((6, 6), 3)]) + , -- An argument that the method never needs is never computed: not a + -- division by zero, and not a call that never returns. + ("return Pick(right, left / right);", [((6, 0), 7), ((6, 3), 2)]) + , ("return Pick(right, Never(left));", [((1, 0), 7)]) + , ("return Pick(0, Never(left)) + Pick(1, Half(left));", [((8, 0), 11)]) + , ("return Both(right, Never(left), left + 1);", [((5, 0), 6)]) + , ("return Both(right, left / right, Never(left));", [((6, 3), 2)]) + , -- A value handed through one method to another is computed by the + -- one that needs it, if any does. + ("return Pass(right, left / right);", [((6, 0), 7), ((6, 3), 2)]) + , ("return Pick(right, Pick(right, left / right) + 1);", [((6, 0), 7), ((6, 2), 4)]) + , -- A method that calls itself hands a value on at every level; none + -- of them is ever needed. + ("return Down(left, right);", [((5, 0), 5), ((0, 0), 0)]) + , -- A local by need is handed on by need, alone and inside a larger + -- argument, and to several methods. + ("int x = Never(left); int y = Pick(right, x); return y;", [((1, 0), 7)]) + , ("int x = left / right; return Pick(right, x) + Pick(right, x + 1);", [((6, 0), 14), ((6, 3), 5)]) + , ("int x = Step(left); int y = Pick(right, x); return right > 0 ? x + y : y;", [((30, 1), 60), ((30, 0), 7)]) + , ("int x = left / right; int y = x + 1; return Pass(right, y * 2);", [((6, 0), 7), ((6, 3), 6)]) + , -- An argument means what its variables held at the call. + ("int a = left; int r = Pick(right, Half(a) + a); a = 100; return r * 1000 + a;", [((8, 1), 12100), ((8, 0), 7100)]) + , -- A store in an argument happens at the call, whether or not the + -- method ever reads the argument. + ("int n = 0; int r = Pick(0, (n += 1) + left); return n * 10 + r;", [((1, 0), 17)]) + , -- A method reads an argument often and it is one value. + ("return Often(right, Half(left));", [((8, 1), 200), ((8, 0), 0)]) + , -- An argument that a closure of the method captures is a value of + -- the closure. + ("return Kept(1, left / right);", [((6, 3), 3)]) + , ("return Kept(0, left / right);", [((6, 3), 3)]) + , -- In a loop every pass hands on a value of its own. + ( "int t = 0; for (int i = 0; i <= 3; i += 1) { t += Pick(i, left / i); } return t;" + , [((12, 0), 29)] + ) + , -- An expression written as a statement is evaluated: that is all a + -- statement can be for. + ("int n = 0; _ = (n += 1) + left; return n;", [((1, 0), 1)]) + , -- Needed values give what they always gave. + ("int x = Half(left); return x + x + x;", [((8, 0), 12)]) + , ("int x = Step(left); int y = Step(right); return x * 10 + y;", [((3, 4), 34)]) + ] + +{- | A chain of values of which each reads the one before it twice, and of +which only the last is read, on one path. +-} +sharedChain :: Int -> String +sharedChain links = + "int z0 = left / right; " + ++ concat + [ "int z" ++ show link ++ " = z" ++ show (link - 1) ++ " / 1 - z" ++ show (link - 1) ++ " / 2; " + | link <- [1 .. links] + ] + ++ "if (right > 0) { return z" + ++ show links + ++ "; } return 7;" + +-- | The size of the unoptimized Core of @Evaluate@ for the given body. +coreSize :: String -> Int +coreSize statements = case compileSource (program statements) of + Right artifacts -> + sum + [ length (show (coreFunctionBody function)) + | function <- coreModuleFunctions (artifactCore artifacts) + , "Evaluate" `isInfixOf` show (coreFunctionName function) + ] + Left _ -> 0 + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "lazy.vxs" text) + +runs :: (FrontendArtifacts -> CoreModule) -> String -> (Integer, Integer) -> Maybe Integer +runs select statements (left, right) = case compileSource (program statements) of + Right artifacts -> case runFunction (select artifacts) "Evaluate" [IntegerValue left, IntegerValue right] of + Just (IntegerValue value) -> Just value + _ -> Nothing + Left _ -> Nothing + +-- | The result of the unoptimized program within the given number of steps. +runsWithin :: Int -> String -> (Integer, Integer) -> Maybe Integer +runsWithin budget statements (left, right) = case compileSource (program statements) of + Right artifacts -> + case runFunctionWithBudget budget (artifactCore artifacts) "Evaluate" [IntegerValue left, IntegerValue right] of + Just (IntegerValue value) -> Just value + _ -> Nothing + Left _ -> Nothing + +-- | Whether the Core of @Evaluate@ mentions the given constructor or generated name. +mentions :: String -> (FrontendArtifacts -> CoreModule) -> String -> Bool +mentions needle select statements = case compileSource (program statements) of + Right artifacts -> + any + ((needle `isInfixOf`) . show . coreFunctionBody) + [ function + | function <- coreModuleFunctions (select artifacts) + , "Evaluate" `isInfixOf` show (coreFunctionName function) + ] + Left _ -> False + +verifies :: String -> Bool +verifies statements = case compileSource (program statements) of + Right artifacts -> + verifyCore (artifactCore artifacts) == Right (artifactCore artifacts) + && verifyCore (artifactOptimizedCore artifacts) == Right (artifactOptimizedCore artifacts) + && verifyCorePrep (artifactCorePrep artifacts) == Right (artifactCorePrep artifacts) + Left _ -> False diff --git a/Compiler/Haskell/Driver/test/LoopExpressionTests.hs b/Compiler/Haskell/Driver/test/LoopExpressionTests.hs index 338b3ef7..3418b022 100644 --- a/Compiler/Haskell/Driver/test/LoopExpressionTests.hs +++ b/Compiler/Haskell/Driver/test/LoopExpressionTests.hs @@ -216,8 +216,16 @@ typeTests = , rejectedWith "VXT0044" (body "int n = 0; auto text = while (true) { break \"done\"; }; return n;") ) , - ( "a return inside a loop expression is rejected" - , rejectedWith "VXT0045" (body "int r = while (true) { if (flag) { return 1; } break 2; }; return r;") + ( "a return inside a loop expression leaves the method" + , accepted (body "int r = while (true) { if (flag) { return 1; } break 2; }; return r;") + ) + , + ( "a return inside a loop expression carries the return type of the method" + , rejectedWith "VXT0005" (body "int r = while (true) { if (flag) { return true; } break 2; }; return r;") + ) + , + ( "a loop expression that only returns never yields a value" + , accepted (body "int r = while (true) { return left; }; return r;") ) , ( "a loop value of the wrong type is rejected by its receiver" diff --git a/Compiler/Haskell/Driver/test/Main.hs b/Compiler/Haskell/Driver/test/Main.hs index 9112d971..dc292013 100644 --- a/Compiler/Haskell/Driver/test/Main.hs +++ b/Compiler/Haskell/Driver/test/Main.hs @@ -19,18 +19,29 @@ import Data.List (isInfixOf) import Data.Word (Word8) import DiagnosticProtocolTests (diagnosticProtocolTests) import DiagnosticSideChannelTests (diagnosticSideChannelTests) +import EnumTests (enumTests) import FloatingOptimizerTests (floatingOptimizerTests) import IntegerEvaluationTests (integerEvaluationTests) +import InferredReturnTests (inferredReturnTests) import IntegerFlowTests (integerFlowTests) import IterationTests (iterationTests) +import LazyEvaluationTests (lazyEvaluationTests) +import ConsoleTests (consoleTests) +import EffectTests (effectTests) +import FormatSweepTests (formatSweepTests) +import RuntimeCallTests (runtimeCallTests) +import FallThroughTests (fallThroughTests) +import MemoizeTests (memoizeTests) import LoopExpressionTests (loopExpressionTests) import LoopFlowOracleTests (loopFlowOracleTests) import LoopFlowTests (loopFlowTests) import MonomorphizationTests (monomorphizationTests) +import NestingLimitTests (nestingLimitTests) import Numeric (readHex) import NumericTests (numericTests) import ParserContractTests (parserContractTests) import PatternTests (patternTests) +import QualifiedNameTests (qualifiedNameTests) import ScalarWireTests (scalarWireTests) import ShortCircuitTests (shortCircuitTests) import SourceSetTests (sourceSetTests) @@ -121,14 +132,14 @@ main = do check "Core wire rejects trailing bytes" coreWireRejectsTrailingInput check "Core wire rejects unresolved types" coreWireRejectsUnresolvedType check "Core wire preserves Unicode scalar values" coreWirePreservesUnicode - check "Core wire v8 provenance fields remain stable" coreWireGoldenDocument - check "Core wire v8 preserves non-empty source-owner field order" coreWireProjectSourceGolden + check "Core wire v10 provenance fields remain stable" coreWireGoldenDocument + check "Core wire v10 preserves non-empty source-owner field order" coreWireProjectSourceGolden check "CorePrep wire codec round-trips the frontend result" wireRoundTrip check "CorePrep wire codec rejects truncated input" wireRejectsTruncation check "CorePrep wire codec rejects trailing input" wireRejectsTrailingInput check "CorePrep wire codec rejects unsupported types" wireRejectsUnsupportedType check "CorePrep wire codec preserves Unicode scalar values" wirePreservesUnicode - check "CorePrep wire v6 provenance fields remain stable" wireGoldenDocument + check "CorePrep wire v7 provenance fields remain stable" wireGoldenDocument checkIO "real Core artifact round-trips through .core I/O" coreArtifactRoundTrip checkIO "Core artifact rejects an invalid Core module" coreArtifactRejectsInvalidModule checkIO "Core artifact rejects a non-.core path" coreArtifactRejectsExtension @@ -175,8 +186,19 @@ main = do mapM_ (uncurry check) assignmentExpressionTests mapM_ (uncurry check) loopExpressionTests mapM_ (uncurry check) branchingTests + mapM_ (uncurry check) inferredReturnTests + mapM_ (uncurry check) enumTests + mapM_ (uncurry check) lazyEvaluationTests + mapM_ (uncurry check) memoizeTests + mapM_ (uncurry check) fallThroughTests + mapM_ (uncurry check) runtimeCallTests + mapM_ (uncurry check) consoleTests + mapM_ (uncurry check) effectTests + mapM_ (uncurry check) qualifiedNameTests + mapM_ (uncurry check) formatSweepTests mapM_ (uncurry check) branchingOracleTests mapM_ (uncurry check) branchingDiagnosticTests + mapM_ (uncurry check) nestingLimitTests mapM_ (uncurry check) specializationTests mapM_ (uncurry check) voidTests @@ -544,7 +566,7 @@ coreWireGoldenDocument = moduleValue = CoreModuleWithSources (QualifiedName [Identifier "Demo"]) [mainFunction] [] [] bytes = upgradeSimpleV5 - 0x08 + 0x0a [ 0x56 , 0x58 , 0x43 @@ -639,7 +661,7 @@ coreWireProjectSourceGolden = [(symbolIdValue (resolvedSymbol mainName), source)] goldenHex = unwords - [ "56 58 43 52 08 00 00 00 01 00 00 00" + [ "56 58 43 52 0A 00 00 00 01 00 00 00" , "04 00 00 00 44 00 00 00 65 00 00 00" , "6d 00 00 00 6f 00 00 00 01 00 00 00" , "10 00 00 00 53 00 00 00 6f 00 00 00" @@ -670,7 +692,7 @@ coreWireProjectSourceGolden = _ -> Nothing -- Adding the empty source catalog and function owner to the compact v5 golden --- shape gives an independent byte-level v8 expectation. The owner follows the +-- shape gives an independent byte-level v10 expectation. The owner follows the -- function symbol, before its parameter vector, matching the wire contract. upgradeSimpleV5 :: Word8 -> [Word8] -> [Word8] upgradeSimpleV5 currentVersion v5Bytes = @@ -740,7 +762,7 @@ goldenModule = goldenBytes :: [Word8] goldenBytes = upgradeSimpleV5 - 0x06 + 0x08 [ 0x56 , 0x58 , 0x43 @@ -842,7 +864,7 @@ coreArtifactRoundTrip = case compile sample of Left _ -> pure False Right artifacts -> do temporary <- getTemporaryDirectory - let path = temporary "visual-xsharp-core-wire-v8.core" + let path = temporary "visual-xsharp-core-wire-v10.core" cleanup = doesFileExist path >>= \exists -> if exists then removeFile path else pure () value = artifactOptimizedCore artifacts ( do diff --git a/Compiler/Haskell/Driver/test/MemoizeTests.hs b/Compiler/Haskell/Driver/test/MemoizeTests.hs new file mode 100644 index 00000000..9b27d5ed --- /dev/null +++ b/Compiler/Haskell/Driver/test/MemoizeTests.hs @@ -0,0 +1,365 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Tests for the callable that remembers its result, 'CoreMemoize'. + +It is the suspended computation of evaluation by need: a callable without +parameters that calls its operand the first time it is called, keeps what +the operand returned, and returns that again from then on, to every holder +of it. The tests here build Core by hand, so that they state what the +primitive is without the lowering that produces it: what the verifiers +accept, that both wire formats carry it, that the optimizer neither removes +a needed computation nor repeats one, and what the reference evaluator +computes and how often. + +"LazyEvaluationTests" checks the lowering that uses the primitive. +-} +module MemoizeTests (memoizeTests) where + +import CoreInterpreter +import Data.List (isInfixOf) +import Visual.XSharp.AST +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep +import Visual.XSharp.Core.CorePrep.Verifier +import Visual.XSharp.Core.CorePrep.Wire +import Visual.XSharp.Core.Optimizer +import Visual.XSharp.Core.Verifier +import Visual.XSharp.Core.Wire +import Visual.XSharp.Diagnostic + +memoizeTests :: [(String, Bool)] +memoizeTests = + [ -- What the Core verifier accepts. + ("a remembered callable without parameters verifies", accepted (returning (callOf remembered))) + , ("a remembered Boolean result verifies", accepted booleanModule) + , ("remembering a callable with a parameter is refused", rejectedWith "VXC1073" (holding (memoize parameterized))) + , ("remembering a callable that returns nothing is refused", rejectedWith "VXC1073" (holding (memoize unitCallable))) + , ("remembering a callable that returns a String is refused", rejectedWith "VXC1073" (holding (memoize stringCallable))) + , ("remembering a number is refused", rejectedWith "VXC1073" (holding (CorePrimitive CoreMemoize [integer 1] intType))) + , ("remembering takes one operand", rejectedWith "VXC1026" (holding (CorePrimitive CoreMemoize [stepClosure, stepClosure] intCallable))) + , ("a remembered callable has the type of its operand", rejectedWith "VXC1028" (holding (CorePrimitive CoreMemoize [stepClosure] boolCallable))) + , -- Both wire formats carry it, and CorePrep accepts what Core does. + ("the Core wire format carries a remembered callable", coreRoundTrip sharedModule) + , ("CorePrep accepts a remembered callable", preparedAndVerified sharedModule) + , ("the CorePrep wire format carries a remembered callable", corePrepRoundTrip sharedModule) + , ("CorePrep refuses to remember a number", corePrepRejects "VXC0025" (malformedCorePrep intType)) + , ("CorePrep refuses to remember a callable with a parameter", corePrepRejects "VXC0025" (malformedCorePrep (FunctionType [intType] intType))) + , -- What it computes, and how often. The budgets are those of one + -- computation of Step(100): a second one does not fit. + ("a remembered callable returns what its operand returns", run sharedModule 5 == Just 15) + , ("three calls of one remembered callable compute once", runWithin 500 sharedModule 100 == Just 300) + , ("three calls of the callable itself compute three times", runWithin 500 repeatedModule 100 == Nothing) + , ("the callable itself, called three times, gives the same value", run repeatedModule 5 == Just 15) + , ("a copy of a remembered callable shares its result", runWithin 500 copiedModule 100 == Just 200) + , ("two remembered callables each compute their own result", runWithin 500 separateModule 100 == Nothing) + , ("two remembered callables give two results", run separateModule 5 == Just 10) + , ("a remembered callable handed to a function is computed once for both", runWithin 500 handedModule 100 == Just 200) + , ("a remembered callable that is never called computes nothing", run neverCalledModule 5 == Just 7) + , ("a remembered callable keeps the values it was created with", run capturedModule 5 == Just 506) + , -- The optimizer keeps the meaning and the count. + ("the optimizer keeps a remembered callable verified", all optimizedVerifies allModules) + , ("the optimizer does not repeat a remembered computation", runOptimizedWithin 500 sharedModule 100 == Just 300) + , ("the optimizer does not compute what is never called", runOptimized neverCalledModule 5 == Just 7) + , ("the optimizer keeps sharing through a copy", runOptimizedWithin 500 copiedModule 100 == Just 200) + , ("the optimizer keeps sharing across a call", runOptimizedWithin 500 handedModule 100 == Just 200) + , ("the optimizer does not turn one remembered callable into two", all (not . memoizesMore) allModules) + ] + where + allModules = [sharedModule, repeatedModule, copiedModule, separateModule, handedModule, neverCalledModule, capturedModule, booleanModule] + +-- ---------------------------------------------------------------- fixtures + +intCallable :: Type +intCallable = FunctionType [] intType + +boolCallable :: Type +boolCallable = FunctionType [] boolType + +named :: Int -> String -> ResolvedName +named identifier spelling = ResolvedName (SymbolId identifier) (Identifier spelling) + +runName, stepName, takeName, inputName, heldName, otherName, countName, callableParameter :: ResolvedName +runName = named 1 "Run" +stepName = named 2 "Step" +takeName = named 3 "Take" +inputName = named 10 "input" +heldName = named 11 "held" +otherName = named 12 "other" +countName = named 13 "count" +callableParameter = named 14 "computation" + +integer :: Integer -> CoreExpression +integer value = CoreLiteral (CoreInteger value) intType + +variable :: ResolvedName -> Type -> CoreExpression +variable = CoreVariable + +input :: CoreExpression +input = variable inputName intType + +-- | @Step(n)@ counts down to zero and returns how far it counted: n steps of work. +stepFunction :: CoreFunction +stepFunction = + CoreFunction + stepName + [(countName, intType)] + intType + [ CoreIf + (CorePrimitive CoreGreaterThan [count, integer 0] boolType) + [ CoreReturn + ( CorePrimitive + CoreAdd + [integer 1, CoreApply stepReference [CorePrimitive CoreSubtract [count, integer 1] intType] intType] + intType + ) + ] + [] + , CoreReturn (integer 0) + ] + where + count = variable countName intType + +stepReference :: CoreExpression +stepReference = variable stepName (FunctionType [intType] intType) + +-- | A callable that computes @Step(input)@ from the input it was created with. +stepClosure :: CoreExpression +stepClosure = + CoreClosure + [CoreCapture StrongCapture inputName intType input] + [] + intType + [CoreReturn (CoreApply stepReference [input] intType)] + intCallable + +memoize :: CoreExpression -> CoreExpression +memoize computation = CorePrimitive CoreMemoize [computation] (expressionType computation) + +remembered :: CoreExpression +remembered = memoize stepClosure + +callOf :: CoreExpression -> CoreExpression +callOf callable = CoreApply callable [] intType + +held, other :: CoreExpression +held = variable heldName intCallable +other = variable otherName intCallable + +sumOf :: [CoreExpression] -> CoreExpression +sumOf = foldr1 (\left right -> CorePrimitive CoreAdd [left, right] intType) + +-- | A module whose @Run(input)@ has the given body, beside @Step@. +moduleOf :: [CoreFunction] -> [CoreStatement] -> CoreModule +moduleOf others body = + CoreModule + (QualifiedName [Identifier "Memo"]) + (CoreFunction runName [(inputName, intType)] intType body : stepFunction : others) + +returning :: CoreExpression -> CoreModule +returning value = moduleOf [] [CoreReturn value] + +-- | @Run@ binds the value and returns zero; for the cases the verifier refuses. +holding :: CoreExpression -> CoreModule +holding value = + moduleOf [] [CoreBind (CoreBinding heldName (expressionType value) False value), CoreReturn (integer 0)] + +bindHeld :: CoreExpression -> CoreStatement +bindHeld value = CoreBind (CoreBinding heldName intCallable False value) + +-- | One remembered callable, called three times. +sharedModule :: CoreModule +sharedModule = moduleOf [] [bindHeld remembered, CoreReturn (sumOf [callOf held, callOf held, callOf held])] + +-- | The same callable without remembering, called three times. +repeatedModule :: CoreModule +repeatedModule = moduleOf [] [bindHeld stepClosure, CoreReturn (sumOf [callOf held, callOf held, callOf held])] + +-- | A second name for one remembered callable: both names call the same one. +copiedModule :: CoreModule +copiedModule = + moduleOf + [] + [ bindHeld remembered + , CoreBind (CoreBinding otherName intCallable False held) + , CoreReturn (sumOf [callOf held, callOf other]) + ] + +-- | Two remembered callables of the same computation. +separateModule :: CoreModule +separateModule = + moduleOf + [] + [ bindHeld remembered + , CoreBind (CoreBinding otherName intCallable False remembered) + , CoreReturn (sumOf [callOf held, callOf other]) + ] + +-- | @Take(computation)@ calls the callable it is given. +takeFunction :: CoreFunction +takeFunction = + CoreFunction + takeName + [(callableParameter, intCallable)] + intType + [CoreReturn (callOf (variable callableParameter intCallable))] + +-- | A remembered callable that the function it is handed to and the caller both call. +handedModule :: CoreModule +handedModule = + moduleOf + [takeFunction] + [ bindHeld remembered + , CoreReturn + (sumOf [CoreApply (variable takeName (FunctionType [intCallable] intType)) [held] intType, callOf held]) + ] + +-- | A remembered callable of a computation that never returns, never called. +neverCalledModule :: CoreModule +neverCalledModule = + moduleOf + [] + [ bindHeld + ( memoize + ( CoreClosure + [CoreCapture StrongCapture inputName intType input] + [] + intType + [CoreWhile (CoreLiteral (CoreBoolean True) boolType) [], CoreReturn input] + intCallable + ) + ) + , CoreReturn (integer 7) + ] + +-- | The callable keeps the input it was created with; a later store does not reach it. +capturedModule :: CoreModule +capturedModule = + moduleOf + [] + [ CoreBind (CoreBinding countName intType True input) + , bindHeld + ( memoize + ( CoreClosure + [CoreCapture StrongCapture countName intType (variable countName intType)] + [] + intType + [CoreReturn (CorePrimitive CoreAdd [variable countName intType, integer 1] intType)] + intCallable + ) + ) + , CoreAssign countName (integer 500) + , CoreReturn (sumOf [callOf held, variable countName intType]) + ] + +-- | A remembered Boolean. +booleanModule :: CoreModule +booleanModule = + moduleOf + [] + [ CoreBind + ( CoreBinding + heldName + boolCallable + False + ( memoize + ( CoreClosure + [CoreCapture StrongCapture inputName intType input] + [] + boolType + [CoreReturn (CorePrimitive CoreGreaterThan [input, integer 0] boolType)] + boolCallable + ) + ) + ) + , CoreIf (CoreApply (variable heldName boolCallable) [] boolType) [CoreReturn (integer 1)] [] + , CoreReturn (integer 0) + ] + +parameterized, unitCallable, stringCallable :: CoreExpression +parameterized = + CoreClosure [] [(countName, intType)] intType [CoreReturn (variable countName intType)] (FunctionType [intType] intType) +unitCallable = CoreClosure [] [] unitType [CoreReturn (CoreLiteral CoreUnit unitType)] (FunctionType [] unitType) +stringCallable = + CoreClosure [] [] stringType [CoreReturn (CoreLiteral (CoreString "text") stringType)] (FunctionType [] stringType) + +-- | A CorePrep function that remembers an atom of the given type. +malformedCorePrep :: Type -> CorePrepModule +malformedCorePrep operandType = + CorePrepModule + (QualifiedName [Identifier "Memo"]) + [ CorePrepFunction + runName + "" + [(inputName, operandType)] + intType + 0 + [ CorePrepBlock + 0 + [CorePrepBind heldName operandType False (CorePrepPrimitive CoreMemoize [CorePrepVariable inputName operandType])] + (CorePrepReturn (CorePrepLiteral (CoreInteger 0) intType)) + ] + ] + [] + +-- ----------------------------------------------------------------- checks + +accepted :: CoreModule -> Bool +accepted moduleValue = verifyCore moduleValue == Right moduleValue + +rejectedWith :: String -> CoreModule -> Bool +rejectedWith code moduleValue = case verifyCore moduleValue of + Left problems -> any ((== code) . diagnosticCode) problems + Right _ -> False + +coreRoundTrip :: CoreModule -> Bool +coreRoundTrip moduleValue = + (encodeCore defaultCoreWireLimits moduleValue >>= decodeCore defaultCoreWireLimits) == Right moduleValue + +preparedAndVerified :: CoreModule -> Bool +preparedAndVerified moduleValue = case prepareCore moduleValue of + Right prepared -> verifyCorePrep prepared == Right prepared && remembers prepared + Left _ -> False + where + remembers prepared = "CoreMemoize" `isInfixOf` show prepared + +corePrepRoundTrip :: CoreModule -> Bool +corePrepRoundTrip moduleValue = case prepareCore moduleValue of + Right prepared -> (encodeCorePrep prepared >>= decodeCorePrep) == Right prepared + Left _ -> False + +corePrepRejects :: String -> CorePrepModule -> Bool +corePrepRejects code moduleValue = case verifyCorePrep moduleValue of + Left problems -> any ((== code) . diagnosticCode) problems + Right _ -> False + +run :: CoreModule -> Integer -> Maybe Integer +run = runWithin 200000 + +runWithin :: Int -> CoreModule -> Integer -> Maybe Integer +runWithin budget moduleValue argument = case runFunctionWithBudget budget moduleValue "Run" [IntegerValue argument] of + Just (IntegerValue value) -> Just value + _ -> Nothing + +optimized :: CoreModule -> Maybe CoreModule +optimized moduleValue = either (const Nothing) Just (runCoreOptimizer defaultCoreOptimizer moduleValue) + +optimizedVerifies :: CoreModule -> Bool +optimizedVerifies moduleValue = case optimized moduleValue of + Just result -> accepted result + Nothing -> False + +runOptimized :: CoreModule -> Integer -> Maybe Integer +runOptimized = runOptimizedWithin 200000 + +runOptimizedWithin :: Int -> CoreModule -> Integer -> Maybe Integer +runOptimizedWithin budget moduleValue argument = optimized moduleValue >>= \result -> runWithin budget result argument + +-- | Whether the optimized module holds more remembering than the module had. +memoizesMore :: CoreModule -> Bool +memoizesMore moduleValue = case optimized moduleValue of + Just result -> occurrences (show result) > occurrences (show moduleValue) + Nothing -> True + where + occurrences text = length [() | index <- [0 .. length text - 1], "CoreMemoize" `isPrefixOfAt` (index, text)] + isPrefixOfAt needle (index, text) = take (length needle) (drop index text) == needle diff --git a/Compiler/Haskell/Driver/test/NestingLimitTests.hs b/Compiler/Haskell/Driver/test/NestingLimitTests.hs new file mode 100644 index 00000000..7b252bd2 --- /dev/null +++ b/Compiler/Haskell/Driver/test/NestingLimitTests.hs @@ -0,0 +1,353 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Tests for the nesting limits of "Visual.XSharp.NestingLimits". + +Every construct that holds statements or operands is nested to exactly the +limit, which must be accepted, and to one level more, which must be rejected +with the diagnostic of that limit at the first node that is too deep. The +accepted programs are also lowered and run, on the unoptimized and the +optimized Core, so the limit is known to be a depth the frontend handles and +not only one it does not reject. Chains that are not nesting, an @else if@ +chain above all, are far longer than either limit and are accepted. +-} +module NestingLimitTests (nestingLimitTests) where + +import CoreInterpreter +import Data.List (intercalate) +import Visual.XSharp.AST +import Visual.XSharp.Compiler +import Visual.XSharp.Diagnostic +import Visual.XSharp.NestingLimits + +nestingLimitTests :: [(String, Bool)] +nestingLimitTests = limitTests ++ statementTests ++ expressionTests ++ combinedTests ++ chainTests + +-- ---------------------------------------------------------------- sources + +-- | A method whose body is the given text on line 3, starting at column 1. +method :: String -> String +method statements = + unlines + [ "class Program {" + , "public static int Evaluate(_ int value) {" + , statements + , "}" + , "}" + ] + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "nesting-limit.vxs" text) + +accepted :: String -> Bool +accepted text = either (const False) (const True) (compileSource text) + +-- | The codes a source is rejected with, in order, or none when it is accepted. +codesOf :: String -> [String] +codesOf text = either (map diagnosticCode) (const []) (compileSource text) + +-- | The one-based columns on line 3 at which the code is reported. +columnsOf :: String -> String -> [Int] +columnsOf code text = case compileSource text of + Right _ -> [] + Left problems -> + [ sourceColumn (sourceStart spanValue) + | problem <- problems + , diagnosticCode problem == code + , Just spanValue <- [diagnosticSpan problem] + , sourceLine (sourceStart spanValue) == 3 + ] + +-- | The value the method returns for the argument, on both forms of Core. +valuesOf :: String -> Integer -> [Maybe Integer] +valuesOf text argument = case compileSource text of + Left _ -> [Nothing] + Right artifacts -> + [ case runFunction (select artifacts) "Evaluate" [IntegerValue argument] of + Just (IntegerValue result) -> Just result + _ -> Nothing + | select <- [artifactCore, artifactOptimizedCore] + ] + +-- ------------------------------------------------------------------ limits + +limitTests :: [(String, Bool)] +limitTests = + [ ("the statement limit is 256 levels", maximumStatementNesting == 256) + , ("the expression limit is 1024 levels", maximumExpressionNesting == 1024) + ] + +-- -------------------------------------------------------------- statements + +{- | A statement nest: the opening text repeated, the innermost statement, +and the closing text repeated. With @count@ openings the innermost statement +is at level @count + 1@. +-} +nest :: Int -> String -> String -> String +nest count open close = "int total = 0; " ++ concat (replicate count open) ++ "total += 1; " ++ concat (replicate count close) + +-- | The column of the innermost statement of 'nest'. +innermostColumn :: Int -> String -> Int +innermostColumn count open = 1 + length "int total = 0; " + count * length open + +-- | Constructs that put a statement one level below themselves. +nestingConstructs :: [(String, String, String)] +nestingConstructs = + [ ("if", "if (value > 0) { ", "} ") + , ("while", "while (total < 1) { ", "} ") + , ("do/while", "do { ", "} while (total < 1); ") + , ("for", "for (; total < 1; total += 1) { ", "} ") + , ("guard", "guard (total > 0) else { ", "return 0; } ") + , ("block", "{ ", "} ") + , ("match arm", "match (value) { _ -> { ", "} } ") + ] + +statementTests :: [(String, Bool)] +statementTests = + concat + [ [ ( "a statement at level 256 inside nested " ++ name ++ " statements is accepted" + , accepted (method (nest atLimit open close ++ "return total;")) + ) + , ( "a statement at level 257 inside nested " ++ name ++ " statements is rejected at that statement" + , columnsOf "VXP0039" (method (nest beyondLimit open close ++ "return total;")) + == [innermostColumn beyondLimit open] + ) + , ( "nothing else is reported for nested " ++ name ++ " statements" + , codesOf (method (nest beyondLimit open close ++ "return total;")) == ["VXP0039"] + ) + ] + | (name, open, close) <- nestingConstructs + ] + ++ [ ("256 levels of if statements compute their value", valuesOf (method (nest atLimit ifOpen "} " ++ "return total;")) 5 == [Just 1, Just 1]) + , ("256 levels of if statements skip the innermost one", valuesOf (method (nest atLimit ifOpen "} " ++ "return total;")) 0 == [Just 0, Just 0]) + , ("256 levels of blocks compute their value", valuesOf (method (nest atLimit "{ " "} " ++ "return total;")) 0 == [Just 1, Just 1]) + , -- Each level of nested loops once multiplied the time of the + -- loop analysis, so that 50 levels did not finish. These nests + -- are compiled, optimized and run. + ("255 nested while loops compute their value", valuesOf (method (nest atLimit "while (total < 1) { " "} " ++ "return total;")) 0 == [Just 1, Just 1]) + , ("255 nested for loops compute their value", valuesOf (method (nest atLimit "for (; total < 1; total += 1) { " "} " ++ "return total;")) 0 == [Just 256, Just 256]) + , ("40 nested counting loops compute their value", valuesOf (method (countingLoops 40)) 1 == [Just 1, Just 1]) + , ("3 nested counting loops compute their value", valuesOf (method (countingLoops 3)) 4 == [Just 64, Just 64]) + , ("a statement far beyond the limit is still reported once", codesOf (method (nest 2000 ifOpen "} " ++ "return total;")) == ["VXP0039"]) + , ("every function reports its own excess", codesOf twoDeepFunctions == ["VXP0039", "VXP0039"]) + , ("a deep function does not hide an accepted one", codesOf oneDeepFunction == ["VXP0039"]) + ] + where + -- The innermost statement of `count` constructs is at level count + 1. + atLimit = maximumStatementNesting - 1 + beyondLimit = maximumStatementNesting + ifOpen = "if (value > 0) { " + -- Loops that each count to the argument with a counter of their + -- own, around one statement: the innermost statement runs + -- value ^ depth times. + countingLoops :: Int -> String + countingLoops depth = + "int total = 0; " + ++ concat ["for (int c" ++ show level ++ " = 0; c" ++ show level ++ " < value; c" ++ show level ++ " += 1) { " | level <- [1 .. depth]] + ++ "total += 1; " + ++ concat (replicate depth "} ") + ++ "return total;" + twoDeepFunctions = + unlines + [ "class Program {" + , "public static int First(_ int value) { " ++ nest beyondLimit ifOpen "} " ++ "return total; }" + , "public static int Second(_ int value) { " ++ nest beyondLimit ifOpen "} " ++ "return total; }" + , "}" + ] + oneDeepFunction = + unlines + [ "class Program {" + , "public static int First(_ int value) { return value; }" + , "public static int Second(_ int value) { " ++ nest beyondLimit ifOpen "} " ++ "return total; }" + , "}" + ] + +-- ------------------------------------------------------------- expressions + +-- | @value + value + ...@ with the given number of operands. +sumOf :: Int -> String +sumOf operands = intercalate " + " (replicate operands "value") + +-- | @value + (value + (... value))@ with the given number of additions. +rightNested :: Int -> String +rightNested additions = concat (replicate additions "value + (") ++ "value" ++ replicate additions ')' + +-- | @(((value + 1) + 1) ...)@ with the given number of additions. +parenthesized :: Int -> String +parenthesized additions = replicate additions '(' ++ "value" ++ concat (replicate additions " + 1)") + +-- | @value == 0 ? 0 : value == 1 ? 1 : ... : 7@ with the given number of tests. +conditionalChain :: Int -> String +conditionalChain tests = concat ["value == " ++ show index ++ " ? " ++ show index ++ " : " | index <- [0 .. tests - 1]] ++ "7" + +-- | @Twice(Twice(...(value)))@ with the given number of calls. +calls :: Int -> String +calls count = concat (replicate count "Twice(") ++ "value" ++ replicate count ')' + +withTwice :: String -> String +withTwice statements = + unlines + [ "class Program {" + , "public static int Evaluate(_ int value) {" + , statements + , "}" + , "public static int Twice(_ int value) { return value + value; }" + , "}" + ] + +expressionTests :: [(String, Bool)] +expressionTests = + [ -- A chain of a left-associative operator is not nesting: the left + -- operand of a binary operator is at the level of the operator, and + -- every stage walks the chain in a loop. + ("a sum of 1024 operands is accepted", accepted (method ("return " ++ sumOf limit ++ ";"))) + , ("a sum of 1024 operands computes its value", valuesOf (method ("return " ++ sumOf limit ++ ";")) 3 == [Just 3072, Just 3072]) + , ("a sum of 1025 operands is not nesting", codesOf (method ("return " ++ sumOf (limit + 1) ++ ";")) == []) + , ("a sum of 20000 operands computes its value", valuesOf (method ("return " ++ sumOf 20000 ++ ";")) 3 == [Just 60000, Just 60000]) + , -- Parentheses around a left operand do not change the tree. + ( "5000 parentheses around left operands compute their value" + , valuesOf (method ("return " ++ parenthesized 5000 ++ ";")) 1 == [Just 5001, Just 5001] + ) + , ("a chain of comparisons joined by && is not nesting", accepted (method ("return (" ++ intercalate " && " (replicate 3000 "value > 0") ++ ") ? 1 : 0;"))) + , -- A right operand is one level below its operator: n additions that + -- nest to the right put the last `value` at level n + 1. + ("1023 additions nested to the right are accepted", accepted (method ("return " ++ rightNested (limit - 1) ++ ";"))) + , + ( "1023 additions nested to the right compute their value" + , valuesOf (method ("return " ++ rightNested (limit - 1) ++ ";")) 2 == [Just 2048, Just 2048] + ) + , + ( "1024 additions nested to the right are rejected at the innermost operand" + , columnsOf "VXP0040" (method ("return " ++ rightNested limit ++ ";")) == [1 + length "return " + limit * length "value + ("] + ) + , ("nothing else is reported for the deep right operand", codesOf (method ("return " ++ rightNested limit ++ ";")) == ["VXP0040"]) + , -- A long chain as a right operand costs one level, not its length. + ( "a long sum as a right operand is one level" + , accepted (method ("return " ++ rightNested 1000 ++ " + (" ++ sumOf 5000 ++ ");")) + ) + , -- Each test of a conditional chain puts the rest of the chain one level deeper. + -- The operands of the last test are two levels below its conditional. + ("a conditional chain of 1022 tests is accepted", accepted (method ("return " ++ conditionalChain (limit - 2) ++ ";"))) + , ("a conditional chain of 1022 tests selects its last result", valuesOf (method ("return " ++ conditionalChain (limit - 2) ++ ";")) 5000 == [Just 7, Just 7]) + , ("a conditional chain of 1022 tests selects a late test", valuesOf (method ("return " ++ conditionalChain (limit - 2) ++ ";")) 1000 == [Just 1000, Just 1000]) + , ("a conditional chain of 1023 tests is rejected", codesOf (method ("return " ++ conditionalChain (limit - 1) ++ ";")) == ["VXP0040"]) + , -- n calls put `value` at level n + 1. + ("1023 nested calls are accepted", accepted (withTwice ("return " ++ calls (limit - 1) ++ ";"))) + , ("1024 nested calls are rejected", codesOf (withTwice ("return " ++ calls limit ++ ";")) == ["VXP0040"]) + , ("an expression far beyond the limit is reported once", codesOf (method ("return " ++ rightNested 5000 ++ ";")) == ["VXP0040"]) + , + ( "many separate expressions at the limit are accepted" + , accepted (method ("int a = " ++ sumOf limit ++ "; int b = " ++ sumOf limit ++ "; return a + b;")) + ) + ] + where + limit = maximumExpressionNesting + +-- ---------------------------------------------------------------- combined + +combinedTests :: [(String, Bool)] +combinedTests = + [ -- A value block is one statement level below the statement that holds + -- its expression: 255 if expressions put the innermost block at 256. + ( "value blocks nested to level 256 are accepted" + , accepted (method ("return " ++ valueBlocks (maximumStatementNesting - 1) ++ ";")) + ) + , + ( "value blocks nested to level 256 compute their value" + , valuesOf (method ("return " ++ valueBlocks (maximumStatementNesting - 1) ++ ";")) 1 == [Just 1, Just 1] + ) + , + ( "if expressions nested in last position need no parentheses" + , valuesOf (method ("return " ++ concat (replicate 100 "if (value > 0) { ") ++ "value" ++ concat (replicate 100 " } else { 0 }") ++ ";")) 4 + == [Just 4, Just 4] + ) + , + ( "value blocks nested to level 257 are rejected as statements" + , codesOf (method ("return " ++ valueBlocks maximumStatementNesting ++ ";")) == ["VXP0039"] + ) + , -- Expressions inside nested statements keep counting from the + -- expression around them, so the two limits bound the total depth. + ( "an expression split by value blocks is still limited" + , codesOf (method ("return " ++ nestedThroughBlocks 4 300 ++ ";")) == ["VXP0040"] + ) + , + ( "an expression split by value blocks below the limit is accepted" + , accepted (method ("return " ++ nestedThroughBlocks 4 200 ++ ";")) + ) + , + ( "a statement excess and an expression excess are both reported" + , codesOf (method (nest maximumStatementNesting "if (value > 0) { " "} " ++ "return " ++ rightNested maximumExpressionNesting ++ ";")) + == ["VXP0039", "VXP0040"] + ) + , + ( "a closure body is one statement level below its expression" + , codesOf (method ("auto f = " ++ concat (replicate maximumStatementNesting "\\() -> { return ") ++ "1" ++ concat (replicate maximumStatementNesting "; }") ++ "; return 0;")) + == ["VXP0039"] + ) + ] + where + -- `if (value > 0) { () } else { 0 }`, nested the given number + -- of times around `value`. The parentheses keep the inner `if` in + -- operand position: at the start of a statement it would be the + -- `if` statement. + valueBlocks :: Int -> String + valueBlocks count = concat (replicate count "if (value > 0) { (") ++ "value" ++ concat (replicate count ") } else { 0 }") + -- Nested if expressions whose blocks each hold additions that + -- nest to the right, the innermost operand of which is the next + -- if expression: the deepest operand is below every addition and + -- every block around it. + nestedThroughBlocks :: Int -> Int -> String + nestedThroughBlocks blocks additions = + concat (replicate blocks ("if (value > 0) { " ++ concat (replicate additions "value + ("))) + ++ "value" + ++ concat (replicate blocks (replicate additions ')' ++ " } else { 0 }")) + +-- ------------------------------------------------------------------ chains + +-- | An else-if chain: link @index@ returns @index * 3 + 1@. +chain :: Int -> String +chain links = + concat + [ (if index == 0 then "if" else " else if") ++ " (value == " ++ show index ++ ") { return " ++ show (index * 3 + 1) ++ "; }" + | index <- [0 .. links - 1] + ] + ++ " return 0;" + +chainTests :: [(String, Bool)] +chainTests = + [ -- The links of an else-if chain are all at the level of the first if. + ("an else-if chain of 1000 links is not nesting", accepted (method (chain 1000))) + , ("an else-if chain of 1000 links selects its last link", valuesOf (method (chain 1000)) 999 == [Just 2998, Just 2998]) + , ("an else-if chain of 1000 links falls through", valuesOf (method (chain 1000)) 1000 == [Just 0, Just 0]) + , -- The body of every link is one level below the chain, whichever link it belongs to. + ( "the bodies of a chain at level 255 are accepted in every link" + , accepted (method (around (maximumStatementNesting - 2) threeLinks)) + ) + , + ( "the bodies of a chain at level 256 are rejected once" + , codesOf (method (around (maximumStatementNesting - 1) threeLinks)) == ["VXP0039"] + ) + , -- An else block that holds exactly one if is the same tree as `else if`. + ("else blocks that hold only an if form a chain", accepted (method (nest 600 "if (value < 0) { } else { " "} " ++ "return total;"))) + , -- The body of an arm is one level below its match, however many arms it has. + ( "the arms of a wide match are one level below it" + , accepted (method (around (maximumStatementNesting - 2) wideArms)) + && codesOf (method (around (maximumStatementNesting - 1) wideArms)) == ["VXP0039"] + ) + , + ( "the arms of a narrow match are one level below it" + , accepted (method (around (maximumStatementNesting - 2) narrowArms)) + && codesOf (method (around (maximumStatementNesting - 1) narrowArms)) == ["VXP0039"] + ) + , ("a statement match of 300 arms is not nesting", accepted (method ("match (value) { " ++ concat [show index ++ " -> { return " ++ show index ++ "; }, " | index <- [0 .. 298 :: Int]] ++ "_ -> { return 0; } } return 1;"))) + ] + where + -- Twenty arms with block bodies. + wideArms = "match (value) { " ++ concat [show index ++ " -> { total = " ++ show index ++ "; }, " | index <- [0 .. 18 :: Int]] ++ "_ -> { total = 0; } } " + -- Three arms with block bodies. + narrowArms = "match (value) { 1 -> { total = 1; }, 2 -> { total = 2; }, _ -> { total = 0; } } " + threeLinks = "if (value == 1) { total = 1; } else if (value == 2) { total = 2; } else if (value == 3) { total = 3; } " + -- The statement inside the given number of if statements, at level count + 1. + around :: Int -> String -> String + around count statement = + "int total = 0; " ++ concat (replicate count "if (value >= 0) { ") ++ statement ++ concat (replicate count "} ") ++ "return total;" diff --git a/Compiler/Haskell/Driver/test/QualifiedNameTests.hs b/Compiler/Haskell/Driver/test/QualifiedNameTests.hs new file mode 100644 index 00000000..f1921e96 --- /dev/null +++ b/Compiler/Haskell/Driver/test/QualifiedNameTests.hs @@ -0,0 +1,136 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Names written through what declares them. + +A declaration of a namespace may be named through the namespace: +@Demo.Program@ is @Program@ inside the namespace @Demo@. A static method may +be referred to through its type without being called: @Program::Log@ is the +method as a callable value, like the bare @Log@ inside @Program@. A selector +with a dot is not a reference: @Program.Log@ without a call is rejected. + +Each program is run in the reference evaluator, before and after +optimization, and what it writes is compared with text written by hand. The +programs that must be rejected are held with the code of the diagnostic. +-} +module QualifiedNameTests (qualifiedNameTests) where + +import CoreInterpreter +import Visual.XSharp.Compiler +import Visual.XSharp.Core +import Visual.XSharp.Diagnostic + +qualifiedNameTests :: [(String, Bool)] +qualifiedNameTests = + concat + [ [ ("unoptimized: " ++ label, written artifactCore (source body) == Just expected) + , ("optimized: " ++ label, written artifactOptimizedCore (source body) == Just expected) + ] + | (label, source, body, expected) <- outputs + ] + ++ [ (label ++ " (" ++ code ++ ")", rejectedWith code (source body)) + | (label, source, body, code) <- rejections + ] + +-- | The label, the program around the body of @Main@, the body and the text. +outputs :: [(String, String -> String, String, String)] +outputs = + [ ("a method is called through its namespace", demo, "Console.Println(Demo.Program.Half(8));", "4\n") + , ("a method called through its namespace writes where it stands", demo, "int x = Demo.Program.Log(1); Console.Println(9);", "1\n9\n") + , ("a method of another class is called through the namespace", demo, "Console.Println(Demo.Other.Triple(3));", "9\n") + , ("an enum member is named through the namespace", demo, "Color c = Demo.Color.Green; Console.Println(c == Color.Green && c \\= Demo.Color.Red);", "true\n") + , ("the namespace and the bare name are the same declaration", demo, "Console.Println(Demo.Program.Half(8) == Program.Half(8) && Program.Half(8) == Half(8));", "true\n") + , ("a namespace of several parts is written whole", nested, "Console.Println(Outer.Inner.Program.Half(10));", "5\n") + , ("a method named through its type is a value", demo, "auto f = Program::Log; int x = f(1); Console.Println(9);", "1\n9\n") + , ("a method named through its type is handed to a method", demo, "int x = Run(Program::Log); Console.Println(9);", "3\n9\n") + , ("a method of another class is a value", demo, "Console.Println(Run(Other::Triple));", "9\n") + , ("a method named through namespace and type is a value", demo, "Console.Println(Run(Demo.Other::Triple));", "9\n") + , ("a quiet method named through its type is called by need", demo, "int x = Run(Program::Half) / Zero(); Console.Println(9);", "9\n") + , ("the place selects among the overloads of a method value", demo, "Console.Println(Run(Other::Pick));", "13\n") + , ("a method value is called as often as its call is reached", demo, "auto f = Other::Triple; int t = 0; for (int i = 0; i < 4; i += 1) { t += f(i); } Console.Println(t);", "18\n") + , -- A class of the program named like the namespace is the class. + ("a class named like the namespace hides the namespace", shadowed, "Console.Println(Demo.Seven());", "7\n") + ] + +-- | The label, the program, the body of @Main@ and the code it is rejected with. +rejections :: [(String, String -> String, String, String)] +rejections = + [ ("a type the namespace does not declare", demo, "Console.Println(Demo.Missing.Half(8));", "VXN0001") + , ("another namespace is not known", demo, "Console.Println(Elsewhere.Program.Half(8));", "VXN0001") + , ("a part of a namespace is not the namespace", nested, "Console.Println(Outer.Program.Half(10));", "VXN0001") + , ("the last part of a namespace is not the namespace", nested, "Console.Println(Inner.Program.Half(10));", "VXN0001") + , ("a local name hides the namespace", demo, "int Demo = 1; Console.Println(Demo.Program.Half(8));", "VXT0032") + , ("a selector with a dot is not a method reference", demo, "auto f = Program.Log;", "VXT0034") + , ("a selector with a dot is not an argument", demo, "Console.Println(Run(Program.Log));", "VXT0009") + , ("a method reference through a value", demo, "int v = 1; auto f = v::Half;", "VXT0081") + , ("a method reference through a name that is not declared", demo, "auto f = Missing::Half;", "VXN0001") + , ("a method value of a name the type does not declare", demo, "auto f = Program::Missing;", "VXT0029") + , ("a method value of a private method of another class", demo, "auto f = Other::Hidden;", "VXT0033") + , ("a method value with overloads and no expected type", demo, "auto f = Other::Pick;", "VXT0080") + , ("a call that expects none of the overloads of a method value", demo, "Console.Println(Wide(Other::Pick));", "VXT0009") + ] + +demo :: String -> String +demo body = + unlines + [ "namespace Demo;" + , "enum Color { Red, Green }" + , "public class Other {" + , " public static int Triple(_ int v) { return v * 3; }" + , " public static int Pick(_ int v) { return v + 10; }" + , " public static int Pick(_ int v, _ int w) { return v + w; }" + , " private static int Hidden(_ int v) { return v; }" + , "}" + , "public class Program {" + , " public static int Zero() { return 0; }" + , " public static int Half(_ int v) { return v / 2; }" + , " public static int Log(_ int v) { Console.Println(v); return v; }" + , " public static int Run(_ (int) -> int f) { return f(3); }" + , " public static int Wide(_ (int, int, int) -> int f) { return f(1, 2, 3); }" + , " public static void Main() {" + , " " ++ body + , " }" + , "}" + ] + +nested :: String -> String +nested body = + unlines + [ "namespace Outer.Inner;" + , "public class Program {" + , " public static int Half(_ int v) { return v / 2; }" + , " public static void Main() {" + , " " ++ body + , " }" + , "}" + ] + +shadowed :: String -> String +shadowed body = + unlines + [ "namespace Demo;" + , "public class Demo {" + , " public static int Seven() { return 7; }" + , "}" + , "public class Program {" + , " public static void Main() {" + , " " ++ body + , " }" + , "}" + ] + +compileSource :: String -> Either [Diagnostic] FrontendArtifacts +compileSource text = compileToCorePrep (CompilerInput "qualified.vxs" text) + +-- | What @Main@ writes to standard output. +written :: (FrontendArtifacts -> CoreModule) -> String -> Maybe String +written select text = case compileSource text of + Right artifacts -> case runFunctionWriting 200000 (select artifacts) "Main" [] of + Just (_, Written output "") -> Just output + _ -> Nothing + Left _ -> Nothing + +rejectedWith :: String -> String -> Bool +rejectedWith code text = case compileSource text of + Left diagnostics -> any ((== code) . diagnosticCode) diagnostics + Right _ -> False diff --git a/Compiler/Haskell/Driver/test/RuntimeCallTests.hs b/Compiler/Haskell/Driver/test/RuntimeCallTests.hs new file mode 100644 index 00000000..e5ce7725 --- /dev/null +++ b/Compiler/Haskell/Driver/test/RuntimeCallTests.hs @@ -0,0 +1,564 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | The runtime call in Core, its catalog, and the reading of formats. + +A runtime call is one Core primitive whose first operand is a literal that +names a function of the runtime catalog. These cases hold the three things +the rest of the compiler builds on: + +* the catalog, row by row. An identity is written into artifacts and a + symbol is linked against, so a row that changed would silently change what + compiled programs call; the native stages pin the same rows in + @RuntimeCallPipelineTests.cpp@; +* the rule each verifier applies to a call, broken in each way it can be + broken, in Core and in CorePrep, and both wire formats; +* the reading of a format into pieces, which is the one place the output + format grammar is decided. + +The programs that use all of it together are in "ConsoleTests". +-} +module RuntimeCallTests (runtimeCallTests) where + +import CoreInterpreter +import Data.List (nub) +import Visual.XSharp.AST +import Visual.XSharp.Core +import Visual.XSharp.Core.CorePrep +import Visual.XSharp.Core.CorePrep.Verifier +import Visual.XSharp.Core.CorePrep.Wire +import Visual.XSharp.Core.Optimizer +import Visual.XSharp.Core.Verifier +import Visual.XSharp.Core.Wire +import Visual.XSharp.Diagnostic +import Visual.XSharp.RuntimeCall +import Visual.XSharp.TypeChecker.Format + +runtimeCallTests :: [(String, Bool)] +runtimeCallTests = + catalogTests ++ acceptanceTests ++ coreTests ++ corePrepTests ++ optimizerTests ++ formatTests + +-- ------------------------------------------------------------------ catalog + +-- | A row as the tests state it: identity, symbol, parameters, result, observable. +type Row = (RuntimeFunction, Integer, String, [RuntimeParameter], RuntimeResult, Bool) + +rows :: [Row] +rows = + [ (TextConcat, 1, "vxs_text_concat", [TextParameter, TextParameter], TextResult, False) + , (TextFromSigned, 2, "vxs_text_from_signed", [SignedParameter], TextResult, False) + , (TextFromUnsigned, 3, "vxs_text_from_unsigned", [UnsignedParameter], TextResult, False) + , (TextFromBool, 4, "vxs_text_from_bool", [BoolParameter], TextResult, False) + , (TextFromChar, 5, "vxs_text_from_char", [CharParameter], TextResult, False) + , (TextFormatSigned, 6, "vxs_text_format_signed", conversion SignedParameter, TextResult, False) + , (TextFormatUnsigned, 7, "vxs_text_format_unsigned", conversion UnsignedParameter, TextResult, False) + , (TextFormatFloating, 8, "vxs_text_format_floating", conversion FloatingParameter, TextResult, False) + , (TextFormatString, 9, "vxs_text_format_string", conversion TextParameter, TextResult, False) + , (TextFormatChar, 10, "vxs_text_format_char", conversion CharParameter, TextResult, False) + , (TextNewline, 11, "vxs_text_newline", [], TextResult, False) + , (ConsoleWrite, 12, "vxs_console_write", [TextParameter, CountParameter], NoResult, True) + , (TextEquals, 13, "vxs_text_equals", [TextParameter, TextParameter], TruthResult, False) + ] + where + conversion value = [CountParameter, CountParameter, CountParameter, value] + +catalogTests :: [(String, Bool)] +catalogTests = + [ ("the runtime catalog has thirteen functions", length runtimeFunctions == 13) + , ("every runtime function has a row in the tests", map (\(function, _, _, _, _, _) -> function) rows == runtimeFunctions) + , ("runtime identities are distinct", length (nub (map runtimeFunctionIdentity runtimeFunctions)) == length runtimeFunctions) + , ("runtime symbols are distinct", length (nub (map runtimeFunctionSymbol runtimeFunctions)) == length runtimeFunctions) + , ("runtime identities are the numbers from one", map runtimeFunctionIdentity runtimeFunctions == [1 .. 13]) + , ("zero names no runtime function", runtimeFunctionOf 0 == Nothing) + , ("a negative number names no runtime function", runtimeFunctionOf (-1) == Nothing) + , ("the number after the last names no runtime function", runtimeFunctionOf 14 == Nothing) + , ("only a console write is observable", filter runtimeObservable runtimeFunctions == [ConsoleWrite]) + , -- The name of a runtime call in the typed tree: reserved, and + -- distinct for every function. + ("runtime names have negative symbols", all ((< 0) . symbolIdValue . resolvedSymbol . runtimeName) runtimeFunctions) + , ("runtime names are distinct", length (nub (map runtimeName runtimeFunctions)) == length runtimeFunctions) + , ("a runtime name is read back as its function", all (\function -> runtimeFunctionOfName (runtimeName function) == Just function) runtimeFunctions) + , ("a source name is no runtime name", runtimeFunctionOfName (name 5 "Print") == Nothing) + , ("the names the language declares are no runtime names", all ((== Nothing) . runtimeFunctionOfName . declared) [builtinSystemSymbol, builtinConsoleSymbol]) + , ("System and Console have symbols of their own", builtinSystemSymbol /= builtinConsoleSymbol) + , ("a runtime name begins with a character no identifier has", all ((== "$") . take 1 . identifierText . resolvedSpelling . runtimeName) runtimeFunctions) + , -- The values the runtime header gives the flags and the targets. + ("the conversion flags are distinct bits", [flagLeft, flagZero, flagPlus, flagSpace, flagAlternate, flagGroup, flagHexadecimal] == [1, 2, 4, 8, 16, 32, 64]) + , ("an absent width is minus one", absent == -1) + , ("the console targets are the four the runtime knows", [consoleOutput, consoleOutputLine, consoleError, consoleErrorLine] == [0, 1, 2, 3]) + ] + ++ concat + [ [ ("runtime identity of " ++ show function, runtimeFunctionIdentity function == identity) + , ("runtime function of identity " ++ show identity, runtimeFunctionOf identity == Just function) + , ("runtime symbol of " ++ show function, runtimeFunctionSymbol function == symbol) + , ("runtime parameters of " ++ show function, runtimeParameters function == parameters) + , ("runtime result of " ++ show function, runtimeResult function == result) + , ("runtime observability of " ++ show function, runtimeObservable function == observable) + ] + | (function, identity, symbol, parameters, result, observable) <- rows + ] + where + declared symbol = ResolvedName symbol (Identifier "Console") + +{- | Which types each kind of parameter takes. The runtime functions are +written for 64 bits, so the wider types are not taken. +-} +acceptanceTests :: [(String, Bool)] +acceptanceTests = + [ ("a signed parameter takes the signed integers up to 64 bits", accepted SignedParameter == ["byte", "short", "long", "int"]) + , ("an unsigned parameter takes the unsigned integers up to 64 bits", accepted UnsignedParameter == ["ubyte", "ushort", "ulong", "uint"]) + , ("a floating parameter takes the floating types up to 64 bits", accepted FloatingParameter == ["sfloat", "lfloat", "float"]) + , ("a Boolean parameter takes bool", accepted BoolParameter == ["bool"]) + , ("a character parameter takes char", accepted CharParameter == ["char"]) + , ("a text parameter takes String", accepted TextParameter == ["String"]) + , ("a count is an int", accepted CountParameter == ["int"]) + , ("no parameter takes a callable", not (any (`runtimeAccepts` FunctionType [] intType) parameterKinds)) + , ("no parameter takes a generic type", not (any (`runtimeAccepts` NamedType (QualifiedName [Identifier "int"]) [TypeTemplateArgument intType]) parameterKinds)) + , ("no parameter takes a qualified name", not (any (`runtimeAccepts` NamedType (QualifiedName [Identifier "System", Identifier "int"]) []) parameterKinds)) + ] + where + parameterKinds = + [SignedParameter, UnsignedParameter, FloatingParameter, BoolParameter, CharParameter, TextParameter, CountParameter] + spellings = + [ "bool" + , "char" + , "byte" + , "short" + , "long" + , "int" + , "longint" + , "ubyte" + , "ushort" + , "ulong" + , "uint" + , "ulongint" + , "sfloat" + , "lfloat" + , "float" + , "double" + , "String" + , "unit" + ] + accepted parameter = [spelling | spelling <- spellings, runtimeAccepts parameter (namedType spelling)] + +-- --------------------------------------------------------------------- Core + +name :: Int -> String -> ResolvedName +name symbol spelling = ResolvedName (SymbolId symbol) (Identifier spelling) + +runName, countName, lineName :: ResolvedName +runName = name 1 "Run" +countName = name 2 "count" +lineName = name 3 "line" + +count :: CoreExpression +count = CoreVariable countName intType + +integer :: Integer -> CoreExpression +integer value = CoreLiteral (CoreInteger value) intType + +text :: String -> CoreExpression +text value = CoreLiteral (CoreString value) stringType + +-- | A runtime call of the given function with the given result type. +runtime :: RuntimeFunction -> [CoreExpression] -> Type -> CoreExpression +runtime function arguments = CorePrimitive CoreRuntimeCall (integer (runtimeFunctionIdentity function) : arguments) + +-- | @Run(count)@ with the given body. +moduleOf :: [CoreStatement] -> CoreModule +moduleOf body = + CoreModuleWithSources + (QualifiedName [Identifier "Calls"]) + [CoreFunction runName [(countName, intType)] intType body] + [] + [] + +{- | @line = "Count: " + format(count); write(line); return count@, the shape +a call of @Console.Printfn("Count: %5d", count)@ lowers to. +-} +writingModule :: CoreModule +writingModule = + moduleOf + [ CoreBind + ( CoreBinding + lineName + stringType + False + ( runtime + TextConcat + [text "Count: ", runtime TextFormatSigned [integer 0, integer 5, integer absent, count] stringType] + stringType + ) + ) + , CoreEvaluate (runtime ConsoleWrite [CoreVariable lineName stringType, integer consoleOutputLine] unitType) + , CoreReturn count + ] + +-- | A module whose one statement before the return evaluates the expression. +evaluating :: CoreExpression -> CoreModule +evaluating expression = moduleOf [CoreEvaluate expression, CoreReturn count] + +-- | A module that binds the expression at the given type. +binding :: Type -> CoreExpression -> CoreModule +binding valueType expression = moduleOf [CoreBind (CoreBinding lineName valueType False expression), CoreReturn count] + +verifiesCore :: CoreModule -> Bool +verifiesCore moduleValue = verifyCore moduleValue == Right moduleValue + +rejectsCore :: CoreModule -> Bool +rejectsCore moduleValue = case verifyCore moduleValue of + Left problems -> any ((== "VXC1075") . diagnosticCode) problems + Right _ -> False + +coreTests :: [(String, Bool)] +coreTests = + [ ("Core accepts well-formed runtime calls", verifiesCore writingModule) + , ("Core wire carries runtime calls", (encodeCore defaultCoreWireLimits writingModule >>= decodeCore defaultCoreWireLimits) == Right writingModule) + , ("the reference evaluator runs the calls", runFunctionWriting 1000 writingModule "Run" [IntegerValue 42] == Just (IntegerValue 42, Written "Count: 42\n" "")) + , -- Every function of the catalog with arguments of its types. + ("Core accepts a call of every runtime function", all (verifiesCore . wellFormed) runtimeFunctions) + , -- What the first operand must be. + ("Core rejects an identity the catalog does not have", rejectsCore (evaluating (CorePrimitive CoreRuntimeCall [integer 999] unitType))) + , ("Core rejects the identity zero", rejectsCore (evaluating (CorePrimitive CoreRuntimeCall [integer 0] unitType))) + , ("Core rejects a negative identity", rejectsCore (evaluating (CorePrimitive CoreRuntimeCall [integer (-12)] unitType))) + , ("Core rejects a runtime call without operands", rejectsCore (evaluating (CorePrimitive CoreRuntimeCall [] unitType))) + , -- The function is fixed when the program is compiled. + ("Core rejects an identity that is computed", rejectsCore (binding stringType (CorePrimitive CoreRuntimeCall [count] stringType))) + , ( "Core rejects an identity of another integer type" + , rejectsCore (binding stringType (CorePrimitive CoreRuntimeCall [CoreLiteral (CoreInteger 11) (namedType "uint")] stringType)) + ) + , ("Core rejects a Boolean as an identity", rejectsCore (evaluating (CorePrimitive CoreRuntimeCall [CoreLiteral (CoreBoolean True) boolType] unitType))) + , -- The arguments are the ones the function takes. + ("Core rejects a missing argument", rejectsCore (binding stringType (runtime TextConcat [text "a"] stringType))) + , ("Core rejects an extra argument", rejectsCore (binding stringType (runtime TextConcat [text "a", text "b", text "c"] stringType))) + , ("Core rejects an argument to a function that takes none", rejectsCore (binding stringType (runtime TextNewline [count] stringType))) + , ("Core rejects a number where a string is taken", rejectsCore (binding stringType (runtime TextConcat [text "a", count] stringType))) + , ("Core rejects a string where a number is taken", rejectsCore (binding stringType (runtime TextFromSigned [text "a"] stringType))) + , ("Core rejects a signed integer where an unsigned one is taken", rejectsCore (binding stringType (runtime TextFromUnsigned [count] stringType))) + , ( "Core rejects a width that is not an int" + , rejectsCore + ( binding + stringType + (runtime TextFormatSigned [integer 0, CoreLiteral (CoreInteger 5) (namedType "long"), integer absent, count] stringType) + ) + ) + , ("Core rejects a write target that is a Boolean", rejectsCore (evaluating (runtime ConsoleWrite [text "a", CoreLiteral (CoreBoolean True) boolType] unitType))) + , -- The call has the type the function returns. + ("Core rejects a string result typed as a number", rejectsCore (binding intType (runtime TextConcat [text "a", text "b"] intType))) + , ("Core rejects a write that yields a string", rejectsCore (binding stringType (runtime ConsoleWrite [text "a", integer 0] stringType))) + , ("Core rejects an equality that yields a string", rejectsCore (binding stringType (runtime TextEquals [text "a", text "b"] stringType))) + , ("Core accepts an equality that yields a Boolean", verifiesCore (binding boolType (runtime TextEquals [text "a", text "b"] boolType))) + ] + +-- | A call of the function with an argument of a type each parameter takes. +wellFormed :: RuntimeFunction -> CoreModule +wellFormed function = case runtimeResult function of + NoResult -> evaluating call + _ -> binding (runtimeResultType function) call + where + call = runtime function (map argument (runtimeParameters function)) (runtimeResultType function) + argument parameter = case parameter of + SignedParameter -> count + UnsignedParameter -> CoreLiteral (CoreInteger 7) (namedType "uint") + FloatingParameter -> CoreLiteral (CoreFloating "1.5") (namedType "float") + BoolParameter -> CoreLiteral (CoreBoolean True) boolType + CharParameter -> CoreLiteral (CoreInteger 65) (namedType "char") + TextParameter -> text "t" + CountParameter -> integer 0 + +-- ----------------------------------------------------------------- CorePrep + +corePrepTests :: [(String, Bool)] +corePrepTests = + [ ("CorePrep accepts well-formed runtime calls", preparedAndVerified writingModule) + , ("CorePrep accepts a call of every runtime function", all (preparedAndVerified . wellFormed) runtimeFunctions) + , ( "CorePrep wire carries runtime calls" + , case prepareCore writingModule of + Right prepared -> (encodeCorePrep prepared >>= decodeCorePrep) == Right prepared + Left _ -> False + ) + , -- The identity stays a literal: it is not bound to a temporary. + ("CorePrep keeps the identity a literal", identityStaysLiteral) + , ("CorePrep rejects an unknown identity", rejectsCorePrep stringType [literal 999]) + , ("CorePrep rejects a call without operands", rejectsCorePrep stringType []) + , ("CorePrep rejects an identity that is a variable", rejectsCorePrep stringType [CorePrepVariable countName intType]) + , ("CorePrep rejects a missing argument", rejectsCorePrep stringType [identity TextConcat, string "a"]) + , ("CorePrep rejects an extra argument", rejectsCorePrep stringType [identity TextNewline, string "a"]) + , ("CorePrep rejects an argument of another type", rejectsCorePrep stringType [identity TextConcat, string "a", CorePrepVariable countName intType]) + , ("CorePrep rejects a result of another type", rejectsCorePrep intType [identity TextConcat, string "a", string "b"]) + , ("CorePrep accepts the same call with its own types", acceptsCorePrep stringType [identity TextConcat, string "a", string "b"]) + ] + where + literal value = CorePrepLiteral (CoreInteger value) intType + identity = literal . runtimeFunctionIdentity + string value = CorePrepLiteral (CoreString value) stringType + +preparedAndVerified :: CoreModule -> Bool +preparedAndVerified moduleValue = case prepareCore moduleValue of + Right prepared -> verifyCorePrep prepared == Right prepared + Left _ -> False + +identityStaysLiteral :: Bool +identityStaysLiteral = case prepareCore writingModule of + Right prepared -> + and + [ case atoms of + CorePrepLiteral (CoreInteger _) valueType : _ -> valueType == intType + _ -> False + | function <- corePrepModuleFunctions prepared + , block <- corePrepFunctionBlocks function + , CorePrepPrimitive CoreRuntimeCall atoms <- map operationOf (corePrepBlockInstructions block) + ] + Left _ -> False + where + operationOf instruction = case instruction of + CorePrepBind _ _ _ operation -> operation + CorePrepEvaluate operation -> operation + CorePrepAssign _ atom -> CorePrepCopy atom + +-- | A CorePrep function that binds a runtime call with the given atoms. +handmade :: Type -> [CorePrepAtom] -> CorePrepModule +handmade valueType atoms = + CorePrepModule + (QualifiedName [Identifier "Calls"]) + [ CorePrepFunction + runName + "" + [(countName, intType)] + intType + 0 + [ CorePrepBlock + 0 + [CorePrepBind lineName valueType False (CorePrepPrimitive CoreRuntimeCall atoms)] + (CorePrepReturn (CorePrepVariable countName intType)) + ] + ] + [] + +rejectsCorePrep :: Type -> [CorePrepAtom] -> Bool +rejectsCorePrep valueType atoms = case verifyCorePrep (handmade valueType atoms) of + Left problems -> any ((== "VXC0026") . diagnosticCode) problems + Right _ -> False + +acceptsCorePrep :: Type -> [CorePrepAtom] -> Bool +acceptsCorePrep valueType atoms = let moduleValue = handmade valueType atoms in verifyCorePrep moduleValue == Right moduleValue + +-- ---------------------------------------------------------------- optimizer + +optimizerTests :: [(String, Bool)] +optimizerTests = + [ -- A write whose result nothing reads stays: the write is the point. + ("the optimizer keeps a console write", writes (optimized writingModule) == Just "Count: 42\n") + , ("the optimizer keeps the order of writes", writes (optimized twoWrites) == Just "first\nsecond\n") + , ("the optimizer does not repeat a write", writes (optimized boundWrite) == Just "once\n") + , -- A string that nothing uses is only computed. Whether the optimizer + -- removes it is its own business; it must not come to be written. + ("a string that nothing uses is not written", writes (optimized unusedText) == Just "") + , ("optimized runtime calls still verify", all (maybe False verifiesCore . optimized) [writingModule, twoWrites, boundWrite, unusedText]) + ] + where + write value = CoreEvaluate (runtime ConsoleWrite [text value, integer consoleOutputLine] unitType) + twoWrites = moduleOf [write "first", write "second", CoreReturn count] + -- The same line is written once however often its string is read. + boundWrite = + moduleOf + [ CoreBind (CoreBinding lineName stringType False (text "once")) + , CoreEvaluate (runtime ConsoleWrite [CoreVariable lineName stringType, integer consoleOutputLine] unitType) + , CoreReturn count + ] + unusedText = + moduleOf + [ CoreBind (CoreBinding lineName stringType False (runtime TextConcat [text "a", text "b"] stringType)) + , CoreReturn count + ] + writes moduleValue = do + value <- moduleValue + (_, Written output _) <- runFunctionWriting 1000 value "Run" [IntegerValue 42] + Just output + +optimized :: CoreModule -> Maybe CoreModule +optimized moduleValue = either (const Nothing) Just (runCoreOptimizer defaultCoreOptimizer moduleValue) + +-- ------------------------------------------------------------------ formats + +formatTests :: [(String, Bool)] +formatTests = + [ ("a format without a percent sign is one literal", parseFormat "plain text" == Right [LiteralPiece "plain text"]) + , ("an empty format has no pieces", parseFormat "" == Right []) + , ("a conversion alone is one piece", parseFormat "%d" == Right [ConversionPiece (plain SignedDecimal 1)]) + , ( "text and conversions keep their order" + , parseFormat "a%db%sc" + == Right + [ LiteralPiece "a" + , ConversionPiece (plain SignedDecimal 2) + , LiteralPiece "b" + , ConversionPiece (plain Text 5) + , LiteralPiece "c" + ] + ) + , ("every conversion letter is read", map kindOf "duxfscb" == map Just [SignedDecimal, UnsignedDecimal, Hexadecimal, FixedPoint, Text, Character, Truth]) + , ("a conversion is written with the letter it was read from", all (\letter -> fmap conversionLetter (kindOf letter) == Just letter) "duxfscb") + , -- %% and %n are not conversions of an argument. + ("a doubled percent sign is a percent sign", parseFormat "100%%" == Right [LiteralPiece "100%"]) + , ("percent signs join the text around them", parseFormat "a%%b%%c" == Right [LiteralPiece "a%b%c"]) + , ("%n is the line terminator", parseFormat "a%nb" == Right [LiteralPiece "a", NewlinePiece, LiteralPiece "b"]) + , ("neither takes an argument", argumentCount "%%%n%%" == Just 0) + , -- Flags, width and precision. + ("flags are read in the order written", flagsOf "%-'d" == Just [LeftFlag, GroupFlag]) + , ("a leading zero is a flag and not part of the width", parseFormat "%08d" == Right [ConversionPiece (Conversion SignedDecimal [ZeroFlag] (FixedSize 8) NoSize 1)]) + , ("a width may have several digits", widthOf "%120s" == Just (FixedSize 120)) + , ("a zero inside a width is a digit", widthOf "%10d" == Just (FixedSize 10)) + , ("a precision follows a point", parseFormat "%.3f" == Right [ConversionPiece (Conversion FixedPoint [] NoSize (FixedSize 3) 1)]) + , ("a precision of zero is a precision", precisionOf "%.0f" == Just (FixedSize 0)) + , ("width and precision may both be written", parseFormat "%10.3f" == Right [ConversionPiece (Conversion FixedPoint [] (FixedSize 10) (FixedSize 3) 1)]) + , ("a star is a width from an argument", widthOf "%*d" == Just ArgumentSize) + , ("a star after a point is a precision from an argument", precisionOf "%.*f" == Just ArgumentSize) + , ("a conversion takes one argument", argumentCount "%d" == Just 1) + , ("a star takes one more", argumentCount "%*d" == Just 2) + , ("two stars take two more", argumentCount "%*.*f" == Just 3) + , ("the arguments of a format are those of its conversions", argumentCount "%s: %*d (%.2f)%n" == Just 4) + , -- What the runtime is given for the flags. + ("no flag is no bit", bitsOf "%d" == Just 0) + , ("each flag is its bit", map bitsOf ["%-d", "%0d", "%+d", "% d", "%#x", "%'d"] == map Just [1, 2, 4, 8, 16 + 64, 32]) + , ("flags are added", bitsOf "%+08d" == Just (4 + 2)) + , ("%x is the hexadecimal flag of the integer conversions", bitsOf "%x" == Just 64) + , -- The position a problem is reported at counts from one. + ("a problem names the position of its conversion", fmap formatProblemOffset (problemOf "abc%q") == Just 4) + , ("the position counts the conversions before it", fmap formatProblemOffset (problemOf "%d%5d%q") == Just 6) + , ("a doubled percent sign counts as two characters", fmap formatProblemOffset (problemOf "%%%q") == Just 3) + ] + ++ [ ("the format " ++ show format ++ " is rejected with " ++ code, fmap formatProblemCode (problemOf format) == Just code) + | (format, code) <- rejected + ] + ++ [ ("the format " ++ show format ++ " is accepted", either (const False) (const True) (parseFormat format)) + | format <- acceptedFormats + ] + where + plain kind offset = Conversion kind [] NoSize NoSize offset + single format = case parseFormat format of + Right [ConversionPiece value] -> Just value + _ -> Nothing + kindOf letter = conversionKind <$> single ['%', letter] + flagsOf = fmap conversionFlags . single + widthOf = fmap conversionWidth . single + precisionOf = fmap conversionPrecision . single + bitsOf = fmap conversionFlagBits . single + argumentCount format = case parseFormat format of + Right found -> Just (sum [conversionArgumentCount value | ConversionPiece value <- found]) + Left _ -> Nothing + problemOf format = either Just (const Nothing) (parseFormat format) + rejected = + [ -- Not a conversion, or not finished. + ("%", "VXT0075") + , ("abc%", "VXT0075") + , ("%q", "VXT0075") + , ("%D", "VXT0075") + , ("%i", "VXT0075") + , ("%e", "VXT0075") + , ("%g", "VXT0075") + , ("%X", "VXT0075") + , ("%o", "VXT0075") + , ("%5", "VXT0075") + , ("%-", "VXT0075") + , ("%.", "VXT0075") + , ("%.f", "VXT0075") + , ("%5.d", "VXT0075") + , ("% ", "VXT0075") + , ("%1234567890d", "VXT0075") + , ("%.1234567890f", "VXT0075") + , -- A flag the conversion does not take. + ("%+s", "VXT0076") + , ("%0s", "VXT0076") + , ("% s", "VXT0076") + , ("%#s", "VXT0076") + , ("%'s", "VXT0076") + , ("%#d", "VXT0076") + , ("%'x", "VXT0076") + , ("%+x", "VXT0076") + , ("%+u", "VXT0076") + , ("% u", "VXT0076") + , ("%#u", "VXT0076") + , ("%#f", "VXT0076") + , ("%0c", "VXT0076") + , ("%+c", "VXT0076") + , ("%'c", "VXT0076") + , ("%0b", "VXT0076") + , ("%+b", "VXT0076") + , -- Flags that exclude each other, and a flag written twice. + ("%-08d", "VXT0076") + , ("%0-8d", "VXT0076") + , ("%+ d", "VXT0076") + , ("% +d", "VXT0076") + , ("%++d", "VXT0076") + , ("%--d", "VXT0076") + , ("%00d", "VXT0076") + , -- A precision the conversion does not take. + ("%.2d", "VXT0076") + , ("%.2u", "VXT0076") + , ("%.2x", "VXT0076") + , ("%.1c", "VXT0076") + , ("%.1b", "VXT0076") + , ("%.*d", "VXT0076") + , -- %% and %n take nothing. + ("%5%", "VXT0076") + , ("%-%", "VXT0076") + , ("%.2%", "VXT0076") + , ("%5n", "VXT0076") + , ("%-n", "VXT0076") + , ("%*n", "VXT0076") + , -- Specified, and waiting for the object model. + ("%A", "VXT0079") + , ("%O", "VXT0079") + , ("%.4A", "VXT0079") + , ("%#A", "VXT0079") + ] + acceptedFormats = + [ "%d" + , "%5d" + , "%-5d" + , "%05d" + , "%+d" + , "% d" + , "%'d" + , "%+'010d" + , "%-+'12d" + , "%*d" + , "%-*d" + , "%u" + , "%08u" + , "%'u" + , "%-'12u" + , "%x" + , "%#x" + , "%08x" + , "%#010x" + , "%-#12x" + , "%f" + , "%.2f" + , "%10.3f" + , "%-10.3f" + , "%010.3f" + , "%+.1f" + , "% .1f" + , "%'.2f" + , "%*.*f" + , "%.*f" + , "%s" + , "%10s" + , "%-10s" + , "%.5s" + , "%10.5s" + , "%*s" + , "%.*s" + , "%c" + , "%3c" + , "%-3c" + , "%b" + , "%6b" + , "%-6b" + , "%n" + , "%%" + , "Name: %s, Age: %d" + , "%d%%" + , "%s%s%s" + ] diff --git a/Compiler/Haskell/Driver/test/RuntimeText.hs b/Compiler/Haskell/Driver/test/RuntimeText.hs new file mode 100644 index 00000000..df410793 --- /dev/null +++ b/Compiler/Haskell/Driver/test/RuntimeText.hs @@ -0,0 +1,119 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | A reference for the text functions of the runtime, used only by tests. + +The runtime library that programs call is written in C++ and works on +machine integers, a buffer of scalars and a bounded natural number. This +module states what each of its functions returns in the most direct way the +language of the tests allows: unbounded integers, exact rationals and lists. +It shares nothing with the library, so a test that runs a program in the +reference evaluator and through the native pipeline compares two +independent answers. + +A floating-point value is an exact rational here: the value of the binary +number, not of the decimal literal it was written as. @%f@ rounds that exact +value to the digits asked for, a tie to the even digit. +-} +module RuntimeText + ( Conversion (..) + , plain + , textOfSigned + , textOfBool + , formatInteger + , formatFloating + , formatText + , lineTerminator + ) where + +import Data.Bits ((.&.)) +import Numeric (showHex) +import Visual.XSharp.RuntimeCall + +-- | The flags, the width and the precision of one conversion, as the runtime takes them. +data Conversion = Conversion + { conversionFlags :: Integer + , conversionWidth :: Integer + , conversionPrecision :: Integer + } + deriving (Eq, Show) + +-- | A conversion with nothing written in it. +plain :: Conversion +plain = Conversion 0 absent absent + +has :: Conversion -> Integer -> Bool +has conversion flag = conversionFlags conversion .&. flag /= 0 + +{- | The line terminator the reference writes. The native runtime writes the +terminator of its platform; a test that compares the two reads a carriage +return and line feed as this. +-} +lineTerminator :: String +lineTerminator = "\n" + +-- | An integer in decimal, with a minus sign when negative. +textOfSigned :: Integer -> String +textOfSigned = show + +textOfBool :: Bool -> String +textOfBool flag = if flag then "true" else "false" + +{- | The field of a conversion: what stands before the body, the body, and the +padding that brings the two to the width. Zeros stand between the two. +-} +field :: Conversion -> Bool -> String -> String -> String +field conversion zerosAllowed prefix body + | has conversion flagLeft = prefix ++ body ++ replicate padding ' ' + | zerosAllowed && has conversion flagZero = prefix ++ replicate padding '0' ++ body + | otherwise = replicate padding ' ' ++ prefix ++ body + where + padding = max 0 (fromInteger (conversionWidth conversion) - length prefix - length body) + +sign :: Conversion -> Bool -> String +sign conversion negative + | negative = "-" + | has conversion flagPlus = "+" + | has conversion flagSpace = " " + | otherwise = "" + +-- | Apostrophes between groups of three digits, counted from the right. +grouped :: String -> String +grouped digits = reverse (go (reverse digits)) + where + go (a : b : c : rest@(_ : _)) = a : b : c : '\'' : go rest + go remaining = remaining + +{- | @%d@, @%u@ and @%x@. A negative number is its sign and its magnitude in +either base. +-} +formatInteger :: Conversion -> Integer -> String +formatInteger conversion value = field conversion True prefix body + where + hexadecimal = has conversion flagHexadecimal + magnitude = abs value + digits = if hexadecimal then showHex magnitude "" else show magnitude + body = if not hexadecimal && has conversion flagGroup then grouped digits else digits + prefix = sign conversion (value < 0) ++ (if hexadecimal && has conversion flagAlternate then "0x" else "") + +{- | @%f@ of an exact value. The flag says whether the value is negative zero, +which a rational cannot say for itself. +-} +formatFloating :: Conversion -> Bool -> Rational -> String +formatFloating conversion negativeZero value = field conversion True (sign conversion negative) body + where + negative = value < 0 || negativeZero + precision = if conversionPrecision conversion < 0 then 6 else conversionPrecision conversion + -- 'round' on a rational rounds a tie to the even integer. + scaled = round (abs value * 10 ^ precision) :: Integer + digits = show scaled + padded = replicate (fromInteger precision + 1 - length digits) '0' ++ digits + (whole, fraction) = splitAt (length padded - fromInteger precision) padded + integer = if has conversion flagGroup then grouped whole else whole + body = if precision > 0 then integer ++ "." ++ fraction else integer + +-- | @%s@ and @%c@: at most the precision's number of characters, when there is one. +formatText :: Conversion -> String -> String +formatText conversion value = field conversion False "" kept + where + kept = if conversionPrecision conversion >= 0 then take (fromInteger (conversionPrecision conversion)) value else value diff --git a/Compiler/Haskell/Driver/test/ScalarWireTests.hs b/Compiler/Haskell/Driver/test/ScalarWireTests.hs index 0cd1f19d..d306f563 100644 --- a/Compiler/Haskell/Driver/test/ScalarWireTests.hs +++ b/Compiler/Haskell/Driver/test/ScalarWireTests.hs @@ -24,8 +24,8 @@ scalarWireTests = versionTests :: [(String, Bool)] versionTests = - [ ("Core wire current version is 8", currentCoreWireVersion == CoreWireVersion 8) - , ("CorePrep wire current version is 6", currentWireVersion == WireVersion 6) + [ ("Core wire current version is 10", currentCoreWireVersion == CoreWireVersion 10) + , ("CorePrep wire current version is 8", currentWireVersion == WireVersion 8) , ("Core numeric payload default is bounded", maximumCoreNumericBytes defaultCoreWireLimits == 4096) , ("CorePrep numeric payload default is bounded", maximumNumericBytes defaultWireLimits == 4096) ] @@ -286,10 +286,11 @@ discardedFloorDivideModule leftType left rightType right = malformedWireTests :: [(String, Bool)] malformedWireTests = [ ("Core wire rejects v2 input", rejectsCoreVersion 2) - , ("Core wire rejects the previous v7 schema", rejectsCoreVersion 7) - , ("Core wire rejects future input", rejectsCoreVersion 9) + , ("Core wire rejects the previous v9 schema", rejectsCoreVersion 9) + , ("Core wire rejects future input", rejectsCoreVersion 11) , ("CorePrep wire rejects v2 input", rejectsCorePrepVersion 2) - , ("CorePrep wire rejects future input", rejectsCorePrepVersion 7) + , ("CorePrep wire rejects the previous v7 schema", rejectsCorePrepVersion 7) + , ("CorePrep wire rejects future input", rejectsCorePrepVersion 9) , ("Core wire enforces numeric byte limit", coreNumericLimit) , ("CorePrep wire enforces numeric byte limit", corePrepNumericLimit) ] diff --git a/Compiler/Haskell/Driver/test/StaticMemberOverloadTests.hs b/Compiler/Haskell/Driver/test/StaticMemberOverloadTests.hs index f6ec754b..96bcaf6e 100644 --- a/Compiler/Haskell/Driver/test/StaticMemberOverloadTests.hs +++ b/Compiler/Haskell/Driver/test/StaticMemberOverloadTests.hs @@ -265,6 +265,7 @@ methodDeclarations (TypedAST (SyntaxTree _ declarations)) = concatMap membersOf membersOf declaration@TypeDeclaration {} = declaration : concatMap membersOf (typeMembers declaration) membersOf declaration@TemplateTypeDeclaration {} = declaration : concatMap membersOf (typeMembers declaration) membersOf declaration@FunctionDeclaration {} = [declaration] + membersOf EnumDeclaration {} = [] findCall :: TypedAST -> Maybe (Expression ResolvedName Type) findCall (TypedAST (SyntaxTree _ declarations)) = find isSelected (concatMap callsInDeclaration declarations) @@ -277,6 +278,7 @@ callsInDeclaration declaration = case declaration of TypeDeclaration {typeMembers = members} -> concatMap callsInDeclaration members TemplateTypeDeclaration {typeMembers = members} -> concatMap callsInDeclaration members FunctionDeclaration {declarationBody = body} -> callsInBlock body + EnumDeclaration {} -> [] callsInBlock :: Block ResolvedName Type -> [Expression ResolvedName Type] callsInBlock (Block statements) = concatMap callsInStatement statements @@ -309,6 +311,7 @@ callsInExpression expression = case expression of NameExpression {} -> [] LiteralExpression {} -> [] MemberAccessExpression _ receiver _ _ -> callsInExpression receiver + MethodReferenceExpression _ receiver _ _ -> callsInExpression receiver CallExpression _ callee arguments _ -> [expression | isSelectedCall expression] ++ callsInExpression callee diff --git a/Compiler/Haskell/Driver/test/StaticMemberParserTests.hs b/Compiler/Haskell/Driver/test/StaticMemberParserTests.hs index 639bdaac..7a4198eb 100644 --- a/Compiler/Haskell/Driver/test/StaticMemberParserTests.hs +++ b/Compiler/Haskell/Driver/test/StaticMemberParserTests.hs @@ -34,7 +34,7 @@ staticMemberParserTests = , ("call span includes its closing parenthesis", callSpanIncludesClose) , ("selector after a parenthesized receiver parses", parenthesizedReceiver) , ("a trailing dot without a member is rejected", trailingDotRejected) - , ("a dot without a receiver is rejected", leadingDotRejected) + , ("a dot without a receiver is a target-typed member selection", leadingDotIsTargetTyped) , ("two dots without a component are rejected", repeatedDotRejected) , ("a numeric token cannot be a member name", numericMemberRejected) , ("a keyword cannot be a member name", keywordMemberRejected) @@ -186,8 +186,13 @@ parenthesizedReceiver = trailingDotRejected :: Bool trailingDotRejected = parseSource "class Program { void Run() { Counter.; return; } }" `isLeft` True -leadingDotRejected :: Bool -leadingDotRejected = parseSource "class Program { void Run() { .Current(); return; } }" `isLeft` True +-- `.Member` selects from the enum that the context expects. The parser keeps +-- the missing receiver as the unit literal, which no source can write; whether +-- there is an enum to select from is the type checker's question. +leadingDotIsTargetTyped :: Bool +leadingDotIsTargetTyped = case firstExpression "class Program { void Run() { .Current(); return; } }" of + Just (CallExpression _ (MemberAccessExpression _ (LiteralExpression _ UnitLiteral _) (Identifier "Current") _) [] _) -> True + _ -> False repeatedDotRejected :: Bool repeatedDotRejected = parseSource "class Program { void Run() { Counter..Current(); return; } }" `isLeft` True @@ -345,6 +350,7 @@ firstCallIn :: Expression Identifier () -> Maybe (Expression Identifier ()) firstCallIn expression = case expression of call@CallExpression {} -> Just call MemberAccessExpression _ receiver _ _ -> firstCallIn receiver + MethodReferenceExpression _ receiver _ _ -> firstCallIn receiver UnaryExpression _ _ value _ -> firstCallIn value BinaryExpression _ _ left right _ -> firstJust [firstCallIn left, firstCallIn right] _ -> Nothing @@ -408,6 +414,7 @@ selectorPath :: Expression Identifier annotation -> [Identifier] selectorPath expression = case expression of NameExpression _ name _ -> [name] MemberAccessExpression _ receiver member _ -> selectorPath receiver ++ [member] + MethodReferenceExpression _ receiver member _ -> selectorPath receiver ++ [member] CallExpression _ callee _ _ -> selectorPath callee _ -> [] @@ -423,6 +430,7 @@ spanOf expression = case expression of NameExpression spanValue _ _ -> spanValue LiteralExpression spanValue _ _ -> spanValue MemberAccessExpression spanValue _ _ _ -> spanValue + MethodReferenceExpression spanValue _ _ _ -> spanValue CallExpression spanValue _ _ _ -> spanValue UnaryExpression spanValue _ _ _ -> spanValue BinaryExpression spanValue _ _ _ _ -> spanValue diff --git a/Compiler/Haskell/Driver/test/StaticMemberSemanticTests.hs b/Compiler/Haskell/Driver/test/StaticMemberSemanticTests.hs index 7c874892..bf870886 100644 --- a/Compiler/Haskell/Driver/test/StaticMemberSemanticTests.hs +++ b/Compiler/Haskell/Driver/test/StaticMemberSemanticTests.hs @@ -32,6 +32,8 @@ staticMemberSemanticTests = , ("a missing member reports the member-not-found diagnostic", missingMethod) , ("a case-mismatched member reports the member-not-found diagnostic", caseSensitiveMember) , ("a bare member selector is rejected until it denotes a method call", bareMethodSelector) + , ("a method reference is the method and not its result", methodReferenceAsResult) + , ("a member selected on a value is rejected until members of values exist", valueMemberSelector) , ("an overload call with too few arguments reports arity", tooFewArguments) , ("an overload call with too many arguments reports arity", tooManyArguments) , ("an overload call rejects a mismatched argument type", mismatchedArgumentType) @@ -136,6 +138,20 @@ bareMethodSelector = "VXT0034" "class Catalog { public static int Read() { return 7; } } class Caller { int Use() { return Catalog.Read; } }" +-- The reference is the method itself, so returning it as an int is a +-- mismatch of types. +methodReferenceAsResult :: Bool +methodReferenceAsResult = + hasCode + "VXT0005" + "class Catalog { public static int Read() { return 7; } } class Caller { int Use() { return Catalog::Read; } }" + +valueMemberSelector :: Bool +valueMemberSelector = + hasCode + "VXT0034" + "class Catalog { public static int Read() { return 7; } } class Caller { int Use() { int value = 1; return value.Read; } }" + tooFewArguments :: Bool tooFewArguments = hasCode @@ -517,6 +533,7 @@ declarationCalls declaration = case declaration of TypeDeclaration {typeMembers = members} -> concatMap declarationCalls members TemplateTypeDeclaration {typeMembers = members} -> concatMap declarationCalls members FunctionDeclaration {declarationBody = body} -> blockCalls body + EnumDeclaration {} -> [] blockCalls :: Block ResolvedName Type -> [Expression ResolvedName Type] blockCalls (Block statements) = concatMap statementCalls statements @@ -549,6 +566,7 @@ expressionCalls expression = case expression of NameExpression {} -> [] LiteralExpression {} -> [] MemberAccessExpression _ receiver _ _ -> expressionCalls receiver + MethodReferenceExpression _ receiver _ _ -> expressionCalls receiver CallExpression _ callee arguments _ -> [expression | isCall expression] ++ expressionCalls callee ++ concatMap expressionCalls arguments UnaryExpression _ _ value _ -> expressionCalls value diff --git a/Compiler/Haskell/Driver/test/TemplateDiscoveryTests.hs b/Compiler/Haskell/Driver/test/TemplateDiscoveryTests.hs index 67118bc1..9c588eb3 100644 --- a/Compiler/Haskell/Driver/test/TemplateDiscoveryTests.hs +++ b/Compiler/Haskell/Driver/test/TemplateDiscoveryTests.hs @@ -244,6 +244,7 @@ renameBoxApplication replacement (TypedAST tree) = TypedAST tree {syntaxDeclarat isStatic access TemplateTypeDeclaration {} -> declaration + EnumDeclaration {} -> declaration rewriteParameter parameter = parameter {parameterAnnotation = rewriteType (parameterAnnotation parameter)} rewriteBlock (Block statements) = Block (map rewriteStatement statements) rewriteStatement statement = case statement of @@ -290,6 +291,8 @@ renameBoxApplication replacement (TypedAST tree) = TypedAST tree {syntaxDeclarat LiteralExpression spanValue literal annotation -> LiteralExpression spanValue literal (rewriteType annotation) MemberAccessExpression spanValue receiver member annotation -> MemberAccessExpression spanValue (rewriteExpression receiver) member (rewriteType annotation) + MethodReferenceExpression spanValue receiver member annotation -> + MethodReferenceExpression spanValue (rewriteExpression receiver) member (rewriteType annotation) CallExpression spanValue callee arguments annotation -> CallExpression spanValue (rewriteExpression callee) (map rewriteExpression arguments) (rewriteType annotation) UnaryExpression spanValue operator value annotation -> diff --git a/Compiler/Haskell/Driver/test/TemplateTests.hs b/Compiler/Haskell/Driver/test/TemplateTests.hs index a8f230c9..b0ff6ef2 100644 --- a/Compiler/Haskell/Driver/test/TemplateTests.hs +++ b/Compiler/Haskell/Driver/test/TemplateTests.hs @@ -339,12 +339,12 @@ identityTests = wireTests :: [(String, Bool)] wireTests = - [ ("Core v8 round-trips fixed array type", coreTypeRoundTrip (fixed intType 4096)) - , ("Core v8 round-trips Boolean template value", coreTypeRoundTrip (applied "Flag" [boolean True])) - , ("Core v8 round-trips character template value", coreTypeRoundTrip (applied "Code" [character 0x10ffff])) - , ("Core v8 round-trips template value parameter", coreTypeRoundTrip valueParameter) + [ ("Core v10 round-trips fixed array type", coreTypeRoundTrip (fixed intType 4096)) + , ("Core v10 round-trips Boolean template value", coreTypeRoundTrip (applied "Flag" [boolean True])) + , ("Core v10 round-trips character template value", coreTypeRoundTrip (applied "Code" [character 0x10ffff])) + , ("Core v10 round-trips template value parameter", coreTypeRoundTrip valueParameter) , - ( "Core v8 round-trips mixed arguments" + ( "Core v10 round-trips mixed arguments" , coreTypeRoundTrip (applied "Mix" [value (-3), typeArg stringType, boolean False]) ) , ("CorePrep v6 round-trips fixed array type", corePrepTypeRoundTrip (fixed intType 4096)) diff --git a/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal b/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal index c256b819..e972cc2c 100644 --- a/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal +++ b/Compiler/Haskell/Driver/visual-xsharp-compiler.cabal @@ -2,7 +2,7 @@ cabal-version: 3.12 -- SPDX-FileCopyrightText: 2026 Progmasoft -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 name: visual-xsharp-compiler -version: 0.4.1 +version: 0.5.0 synopsis: Target-independent Visual X# compiler driver license: MPL-2.0 author: Progmasoft @@ -22,8 +22,10 @@ library visual-xsharp-syntax:Visual.XSharp.Diagnostic.Protocol, visual-xsharp-syntax:Visual.XSharp.FloatingLiteral, visual-xsharp-syntax:Visual.XSharp.Lexer, + visual-xsharp-syntax:Visual.XSharp.NestingLimits, visual-xsharp-syntax:Visual.XSharp.NumericLiteral, visual-xsharp-syntax:Visual.XSharp.Parser, + visual-xsharp-syntax:Visual.XSharp.RuntimeCall, visual-xsharp-syntax:Visual.XSharp.SourceText, visual-xsharp-syntax:Visual.XSharp.SourceWarnings, visual-xsharp-frontend:Visual.XSharp.Frontend, @@ -133,16 +135,28 @@ test-suite visual-xsharp-compiler-tests CoreOptimizerSourceTests CoreVerifierTests CompileTimeParityTests + EnumTests FloatingOptimizerTests IntegerEvaluationTests + InferredReturnTests IntegerFlowTests AssignmentExpressionTests ConditionalExpressionTests CoreInterpreter IterationTests BranchingDiagnosticTests + BranchingEvaluationCases BranchingOracleTests BranchingTests + LazyEvaluationTests + ConsoleTests + EffectTests + QualifiedNameTests + FormatSweepTests + FallThroughTests + RuntimeCallTests + RuntimeText + MemoizeTests LoopExpressionTests LoopFlowTests LoopFlowOracleTests @@ -151,6 +165,7 @@ test-suite visual-xsharp-compiler-tests NumericTests PatternTests MonomorphizationTests + NestingLimitTests ParserContractTests ScalarWireTests ShortCircuitTests diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Closure/Analysis.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Closure/Analysis.hs index 3e4dc034..58da04d4 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Closure/Analysis.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Closure/Analysis.hs @@ -133,6 +133,7 @@ walkDeclaration state declaration = case declaration of -- relationships before and after concrete declaration instantiation. TemplateTypeDeclaration {typeMembers = members} -> foldl walkDeclaration state members FunctionDeclaration {declarationBody = body} -> walkBlock Nothing state body + EnumDeclaration {} -> state walkBlock :: Maybe ClosureId -> WalkState -> Block ResolvedName Type -> WalkState walkBlock parent state block = foldStatements parent (blockStatements block) state @@ -174,6 +175,7 @@ walkExpression parent state expression = case expression of NameExpression {} -> state LiteralExpression {} -> state MemberAccessExpression _ receiver _ _ -> walkExpression parent state receiver + MethodReferenceExpression _ receiver _ _ -> walkExpression parent state receiver CallExpression _ callee arguments _ -> foldl (walkExpression parent) (walkExpression parent state callee) arguments UnaryExpression _ _ value _ -> walkExpression parent state value @@ -301,6 +303,7 @@ expressionFacts expression = case expression of NameExpression _ name valueType -> BodyFacts [(name, valueType)] [] [] LiteralExpression {} -> emptyFacts MemberAccessExpression _ receiver _ _ -> expressionFacts receiver + MethodReferenceExpression _ receiver _ _ -> expressionFacts receiver CallExpression _ callee arguments _ -> foldl appendFacts (expressionFacts callee) (map expressionFacts arguments) UnaryExpression _ _ value _ -> expressionFacts value @@ -435,6 +438,7 @@ expressionContainsCall :: Expression name annotation -> Bool expressionContainsCall expression = case expression of CallExpression {} -> True MemberAccessExpression _ receiver _ _ -> expressionContainsCall receiver + MethodReferenceExpression _ receiver _ _ -> expressionContainsCall receiver UnaryExpression _ _ value _ -> expressionContainsCall value BinaryExpression _ _ left right _ -> expressionContainsCall left || expressionContainsCall right IsPatternExpression _ subject _ _ -> expressionContainsCall subject diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Frontend.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Frontend.hs index 4c53fd60..ce3266f8 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Frontend.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Frontend.hs @@ -20,6 +20,7 @@ module Visual.XSharp.Frontend import Visual.XSharp.AST import Visual.XSharp.Diagnostic import Visual.XSharp.Lexer +import Visual.XSharp.NestingLimits import Visual.XSharp.Parser import Visual.XSharp.Resolver.NameResolution import Visual.XSharp.Resolver.Renamer @@ -70,6 +71,12 @@ analyzeSemantics input = syntaxParsedAST <$> analyzeSyntax input >>= analyzePars -- | Run semantic passes on an existing parsed AST without lexing or parsing. analyzeParsedSemantics :: ParsedAST -> Either [Diagnostic] SemanticArtifacts analyzeParsedSemantics parsed = do + -- Every later pass, and every stage after the frontend, recurses along + -- the nesting of the tree. A tree that nests beyond the limits is + -- rejected here, with the place where it becomes too deep. + case nestingProblems parsed of + [] -> pure () + problems -> Left problems renamed <- runRenamer defaultRenamer parsed resolved <- runNameResolution defaultNameResolution renamed typed <- runTypeChecker defaultTypeChecker resolved diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/NameResolution.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/NameResolution.hs index ddcdae26..9723ede8 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/NameResolution.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/NameResolution.hs @@ -8,6 +8,7 @@ module Visual.XSharp.Resolver.NameResolution (NameResolution (..), defaultNameRe import Visual.XSharp.AST import Visual.XSharp.Diagnostic +import Visual.XSharp.RuntimeCall (builtinConsoleSymbol, builtinSystemSymbol) -- | Pluggable name-resolution pass over a fully renamed syntax tree. newtype NameResolution = NameResolution @@ -49,6 +50,9 @@ resolveDeclaration declaration = case declaration of (map fst resolvedMembers) , snd name ++ concatMap snd parameters ++ concatMap snd resolvedMembers ) + EnumDeclaration spanValue sourceName _ underlying cases -> + let name = resolveName spanValue sourceName + in (EnumDeclaration spanValue (fst name) () underlying cases, snd name) FunctionDeclaration spanValue sourceName _ returnSyntax sourceParameters sourceBody isStatic access -> let name = resolveName spanValue sourceName parameters = map resolveParameter sourceParameters @@ -152,6 +156,9 @@ resolveExpression expression = case expression of MemberAccessExpression spanValue receiver member _ -> let (resolvedReceiver, problems) = resolveExpression receiver in (MemberAccessExpression spanValue resolvedReceiver member (), problems) + MethodReferenceExpression spanValue receiver member _ -> + let (resolvedReceiver, problems) = resolveExpression receiver + in (MethodReferenceExpression spanValue resolvedReceiver member (), problems) CallExpression spanValue callee arguments _ -> let (resolvedCallee, firstProblems) = resolveExpression callee; values = map resolveExpression arguments in (CallExpression spanValue resolvedCallee (map fst values) (), firstProblems ++ concatMap snd values) @@ -267,6 +274,9 @@ resolveCallableBody body = case body of resolveName :: SourceSpan -> RenamedName -> (ResolvedName, [Diagnostic]) resolveName spanValue name | renamedUnique name > 0 = (ResolvedName (SymbolId (renamedUnique name)) (renamedSpelling name), []) + -- The names the language declares for every program. + | SymbolId (renamedUnique name) `elem` [builtinSystemSymbol, builtinConsoleSymbol] = + (ResolvedName (SymbolId (renamedUnique name)) (renamedSpelling name), []) | renamedUnique name == 0 = ( ResolvedName (SymbolId 0) (renamedSpelling name) , diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/Renamer.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/Renamer.hs index 8463ea6e..cd8c6042 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/Renamer.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Resolver/Renamer.hs @@ -6,8 +6,11 @@ duplicate declarations while preserving overload-family semantics. -} module Visual.XSharp.Resolver.Renamer (Renamer (..), defaultRenamer, runRenamer) where +import Data.List (intercalate) +import Data.Map.Strict qualified as Map import Visual.XSharp.AST import Visual.XSharp.Diagnostic +import Visual.XSharp.RuntimeCall (builtinConsoleSymbol, builtinSystemSymbol) -- | Pluggable pass that assigns declaration and local-binding identities. newtype Renamer = Renamer @@ -23,7 +26,23 @@ runRenamer = renameParsedAST defaultRenamer :: Renamer defaultRenamer = Renamer renameTree -type Environment = [(Identifier, RenamedName)] +{- | The names in scope, each with the binding a use of it refers to. A +binding made later replaces an earlier one of the same spelling, which is +how an inner scope shadows an outer one. + +It is a map and not a list of pairs: a type with thousands of members made +every lookup and every duplicate check walk all of them, and renaming grew +with the square of the member count. +-} +type Environment = Map.Map Identifier RenamedName + +-- | The scope with one more binding, which shadows an earlier one. +bind :: Identifier -> RenamedName -> Environment -> Environment +bind = Map.insert + +-- | The inner scope laid over the outer one; inner bindings shadow outer ones. +over :: Environment -> Environment -> Environment +over = Map.union renameTree :: ParsedAST -> Either [Diagnostic] RenamedAST renameTree (ParsedAST (SyntaxTree namespace declarations)) = @@ -32,21 +51,63 @@ renameTree (ParsedAST (SyntaxTree namespace declarations)) = RenamerStage "VXR0001" 1 - [] + Map.empty [(declarationName declaration, declarationSpan declaration) | declaration <- declarations] - (renamed, _, problems) = renameDeclarations globals next declarations + -- What the language declares for every program stands outside + -- the program's own names, which shadow it. + (renamed, _, problems) = + renameDeclarations (globals `over` qualified namespace globals `over` predeclared) next declarations allProblems = duplicateProblems ++ problems in if null allProblems then Right (RenamedAST (SyntaxTree namespace renamed)) else Left allProblems +{- | The names every program may use without declaring them: @System@, and +@Console@, because @System@ is imported implicitly. Each has a reserved +symbol, which no declaration of a program can have. +-} +predeclared :: Environment +predeclared = + Map.fromList + [ (Identifier "System", RenamedName (Identifier "System") (symbolIdValue builtinSystemSymbol)) + , (Identifier "Console", RenamedName (Identifier "Console") (symbolIdValue builtinConsoleSymbol)) + ] + +{- | The declarations of a namespace under their qualified names: +@Demo.Program@ beside @Program@ in the namespace @Demo@. A qualified name +is kept as one spelling with its dots, which no identifier of a program +has, so nothing a program declares can replace it. +-} +qualified :: Maybe QualifiedName -> Environment -> Environment +qualified namespace globals = case namespace of + Just (QualifiedName parts@(_ : _)) -> + Map.mapKeys (\name -> Identifier (intercalate "." (map identifierText parts ++ [identifierText name]))) globals + _ -> Map.empty + +{- | The qualified name a selector spells, when it spells one that is +declared: @Demo.Program@ for the selector of @Program@ on @Demo@. A name of +the program that is spelled like the first part hides the namespace, as an +inner name hides an outer one. +-} +qualifiedSelector :: Environment -> Expression Identifier () -> Maybe RenamedName +qualifiedSelector environment expression = do + parts@(first : _ : _) <- path expression + case Map.lookup first environment of + Just _ -> Nothing + Nothing -> Map.lookup (Identifier (intercalate "." (map identifierText parts))) environment + where + path value = case value of + NameExpression _ name _ -> Just [name] + MemberAccessExpression _ receiver member _ -> (++ [member]) <$> path receiver + _ -> Nothing + declareMany :: DiagnosticStage -> String -> Int -> Environment -> [(Identifier, SourceSpan)] -> (Environment, Int, [Diagnostic]) declareMany _ _ next environment [] = (environment, next, []) declareMany stage code next environment ((name, spanValue) : remaining) = - let duplicate = case lookup name environment of + let duplicate = case Map.lookup name environment of Just _ -> [Diagnostic stage Error code (Just spanValue) ("duplicate declaration " ++ identifierText name)] Nothing -> [] renamed = RenamedName name next - (final, after, problems) = declareMany stage code (next + 1) ((name, renamed) : environment) remaining + (final, after, problems) = declareMany stage code (next + 1) (bind name renamed environment) remaining in (final, after, duplicate ++ problems) -- Methods with one spelling are a single overload family. Other declarations @@ -54,13 +115,13 @@ declareMany stage code next environment ((name, spanValue) : remaining) = -- a field-like or nested declaration of the same name. Each overload receives -- its own SymbolId; the type checker later validates signature uniqueness. declareMembers :: Int -> [Declaration Identifier ()] -> (Environment, [RenamedName], Int, [Diagnostic]) -declareMembers next declarations = go [] [] next declarations +declareMembers next declarations = go Map.empty Map.empty next declarations where go environment _ current [] = (environment, [], current, []) go environment seen current (declaration : remaining) = let name = declarationName declaration isMethod = case declaration of FunctionDeclaration {} -> True; _ -> False - collision = case lookup name seen of + collision = case Map.lookup name seen of Nothing -> [] Just previousWasMethod | previousWasMethod && isMethod -> [] @@ -74,7 +135,7 @@ declareMembers next declarations = go [] [] next declarations ] renamed = RenamedName name current (finalEnvironment, laterNames, finalNext, laterProblems) = - go ((name, renamed) : environment) ((name, isMethod) : seen) (current + 1) remaining + go (bind name renamed environment) (Map.insert name isMethod seen) (current + 1) remaining in (finalEnvironment, renamed : laterNames, finalNext, collision ++ laterProblems) renameDeclarations :: @@ -102,7 +163,7 @@ renameDeclarationsWithBindings globals next (declaration : remaining) (assignedN let name = assignedName (declaredMembers, memberNames, afterMembers, duplicateProblems) = declareMembers next members (renamedMembers, afterBody, memberProblems) = - renameDeclarationsWithBindings (declaredMembers ++ globals) afterMembers members memberNames + renameDeclarationsWithBindings (declaredMembers `over` globals) afterMembers members memberNames renamed = TypeDeclaration spanValue name () renamedMembers (rest, final, restProblems) = renameDeclarationsWithBindings globals afterBody remaining assignedRemaining in (renamed : rest, final, duplicateProblems ++ memberProblems ++ restProblems) @@ -111,7 +172,7 @@ renameDeclarationsWithBindings globals next (declaration : remaining) (assignedN (templateParameters, templateEnvironment, afterTemplateParameters, templateProblems) = renameTemplateParameters globals next sourceTemplateParameters (declaredMembers, memberNames, afterMembers, duplicateProblems) = declareMembers afterTemplateParameters members - memberEnvironment = declaredMembers ++ templateEnvironment + memberEnvironment = declaredMembers `over` templateEnvironment (renamedMembers, afterBody, memberProblems) = renameDeclarationsWithBindings memberEnvironment afterMembers members memberNames renamed = TemplateTypeDeclaration spanValue name () templateParameters renamedMembers @@ -120,6 +181,11 @@ renameDeclarationsWithBindings globals next (declaration : remaining) (assignedN , final , templateProblems ++ duplicateProblems ++ memberProblems ++ restProblems ) + -- The members of an enum are named by spelling under their enum + -- and take no symbols of their own. + EnumDeclaration spanValue _ _ underlying cases -> + let (rest, final, restProblems) = renameDeclarationsWithBindings globals next remaining assignedRemaining + in (EnumDeclaration spanValue assignedName () underlying cases : rest, final, restProblems) FunctionDeclaration spanValue _ _ returnSyntax sourceParameters sourceBody isStatic access -> let name = assignedName (parameters, parameterEnvironment, afterParameters, parameterProblems) = renameParameters globals next sourceParameters @@ -171,11 +237,11 @@ renameParameters environment next parameters = go environment next parameters [] go env current [] output problems = (reverse output, env, current, reverse problems) go env current (Parameter spanValue name _ syntax : rest) output problems = let duplicate = - if any ((== name) . fst) (take (length output) env) + if any (\(Parameter _ earlier _ _) -> renamedSpelling earlier == name) output then Diagnostic RenamerStage Error "VXR0002" (Just spanValue) ("duplicate parameter " ++ identifierText name) : problems else problems renamed = RenamedName name current - in go ((name, renamed) : env) (current + 1) rest (Parameter spanValue renamed () syntax : output) duplicate + in go (bind name renamed env) (current + 1) rest (Parameter spanValue renamed () syntax : output) duplicate renameBlock :: Environment -> Int -> Block Identifier () -> (Block RenamedName (), Int, [Diagnostic]) renameBlock environment next (Block statements) = let (values, _, final, problems) = go environment next statements in (Block values, final, problems) @@ -191,14 +257,14 @@ renameStatement :: renameStatement environment next statement = case statement of BindingStatement spanValue kind syntax name _ value -> let (renamedValue, afterValue, problems) = renameExpression environment next value - duplicate = any ((== name) . fst) environment + duplicate = Map.member name environment renamed = RenamedName name afterValue duplicateProblems = if duplicate then [Diagnostic RenamerStage Error "VXR0003" (Just spanValue) ("duplicate local " ++ identifierText name)] else [] in ( BindingStatement spanValue kind syntax renamed () renamedValue - , (name, renamed) : environment + , bind name renamed environment , afterValue + 1 , problems ++ duplicateProblems ) @@ -251,14 +317,14 @@ renameStatement environment next statement = case statement of ) ForEachStatement spanValue kind syntax sourceName _ source body -> let (renamedSource, afterSource, sourceProblems) = renameExpression environment next source - duplicate = any ((== sourceName) . fst) environment + duplicate = Map.member sourceName environment renamedName = RenamedName sourceName afterSource duplicateProblems = if duplicate then [Diagnostic RenamerStage Error "VXR0003" (Just spanValue) ("duplicate loop binding " ++ identifierText sourceName)] else [] (renamedBody, afterBody, bodyProblems) = - renameBlock ((sourceName, renamedName) : environment) (afterSource + 1) body + renameBlock (bind sourceName renamedName environment) (afterSource + 1) body in ( ForEachStatement spanValue kind syntax renamedName () renamedSource renamedBody , environment , afterBody @@ -309,9 +375,15 @@ renameExpression :: Environment -> Int -> Expression Identifier () -> (Expressio renameExpression environment next expression = case expression of NameExpression spanValue name _ -> (NameExpression spanValue (valueOrMissing name environment) (), next, []) LiteralExpression spanValue literal _ -> (LiteralExpression spanValue literal (), next, []) + -- A declaration named through its namespace is that declaration. + MemberAccessExpression spanValue _ _ _ + | Just name <- qualifiedSelector environment expression -> (NameExpression spanValue name (), next, []) MemberAccessExpression spanValue receiver member _ -> let (renamedReceiver, afterReceiver, problems) = renameExpression environment next receiver in (MemberAccessExpression spanValue renamedReceiver member (), afterReceiver, problems) + MethodReferenceExpression spanValue receiver member _ -> + let (renamedReceiver, afterReceiver, problems) = renameExpression environment next receiver + in (MethodReferenceExpression spanValue renamedReceiver member (), afterReceiver, problems) CallExpression spanValue callee arguments _ -> let (renamedCallee, afterCallee, firstProblems) = renameExpression environment next callee (renamedArguments, after, problems) = renameExpressions environment afterCallee arguments @@ -368,7 +440,7 @@ renameExpression environment next expression = case expression of bodyOuterEnvironment = if explicit then captureEnvironment - else captureEnvironment ++ environment + else captureEnvironment `over` environment (parameters, parameterEnvironment, afterParameters, parameterProblems) = renameParameters bodyOuterEnvironment afterCaptures sourceParameters (body, afterBody, bodyProblems) = @@ -424,10 +496,10 @@ renameMatchPattern environment next patternValue = case patternValue of let renamed = RenamedName name next duplicateProblems = [ Diagnostic RenamerStage Error "VXR0008" (Just spanValue) ("duplicate match binding " ++ identifierText name) - | any ((== name) . fst) environment + | Map.member name environment ] in ( MatchTypePattern spanValue syntax (Just renamed) () - , (name, renamed) : environment + , bind name renamed environment , next + 1 , duplicateProblems ) @@ -459,7 +531,7 @@ renameCaptures :: Int -> [Capture Identifier ()] -> ([Capture RenamedName ()], Environment, Int, [Diagnostic]) -renameCaptures environment next captures = go next captures [] [] [] +renameCaptures environment next captures = go next captures [] Map.empty [] where go current [] output localEnvironment problems = (reverse output, localEnvironment, current, reverse problems) @@ -469,7 +541,7 @@ renameCaptures environment next captures = go next captures [] [] [] Nothing -> NameExpression spanValue name () (renamedInitializer, afterInitializer, initializerProblems) = renameExpression environment current sourceExpression - duplicate = any ((== name) . fst) localEnvironment + duplicate = Map.member name localEnvironment duplicateProblems = if duplicate then @@ -487,7 +559,7 @@ renameCaptures environment next captures = go next captures [] [] [] (afterInitializer + 1) remaining (renamedCapture : output) - ((name, renamedName) : localEnvironment) + (bind name renamedName localEnvironment) (reverse initializerProblems ++ duplicateProblems) renameCallableBody :: @@ -512,4 +584,4 @@ renameExpressions environment next (value : rest) = in (renamed : remaining, final, problems ++ restProblems) valueOrMissing :: Identifier -> Environment -> RenamedName -valueOrMissing name environment = maybe (RenamedName name (-1)) id (lookup name environment) +valueOrMissing name environment = maybe (RenamedName name (-1)) id (Map.lookup name environment) diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Discovery.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Discovery.hs index b72f7a42..48bb6cb1 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Discovery.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Discovery.hs @@ -136,6 +136,8 @@ discoverTop catalog namespace state declaration = case declaration of -- grammar, but keeping this traversal total makes hand-built TypedAST -- fixtures and future declaration categories deterministic. discoverFunction catalog namespace (declarationName declaration) Nothing declaration state + -- An enum uses no template: its underlying type is a scalar. + EnumDeclaration {} -> state discoverMember :: TemplateCatalog -> @@ -151,6 +153,7 @@ discoverMember catalog namespace owner state member = case member of afterType = discoverType catalog namespace origin annotation state in foldl' (discoverMember catalog namespace owner) afterType members TemplateTypeDeclaration {} -> state {stateSkippedTemplates = stateSkippedTemplates state + 1} + EnumDeclaration {} -> state discoverFunction :: TemplateCatalog -> @@ -277,6 +280,8 @@ discoverExpression catalog namespace origin expression state = case expression o LiteralExpression _ _ annotation -> discoverType catalog namespace origin annotation state MemberAccessExpression _ receiver _ annotation -> discoverExpression catalog namespace origin receiver (discoverType catalog namespace origin annotation state) + MethodReferenceExpression _ receiver _ annotation -> + discoverExpression catalog namespace origin receiver (discoverType catalog namespace origin annotation state) CallExpression _ callee arguments annotation -> let afterType = discoverType catalog namespace origin annotation state afterCallee = discoverExpression catalog namespace origin callee afterType diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Freshen.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Freshen.hs index bc641971..da53108e 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Freshen.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Freshen.hs @@ -120,6 +120,8 @@ freshDeclaration declaration = case declaration of -- A nested template introduces a separate substitution/freshening -- environment. It will be selected independently when demanded. pure declaration + -- An enum has no body and no template parameter in it. + EnumDeclaration {} -> pure declaration reserveDeclarationName :: Declaration ResolvedName Type -> Fresh () reserveDeclarationName declaration = case declaration of @@ -129,6 +131,7 @@ reserveDeclarationName declaration = case declaration of -- Nested templates own a separate specialization environment and must -- not leak definitions into the enclosing concrete type's map. pure () + EnumDeclaration {} -> pure () freshParameterDefinition :: Parameter ResolvedName Type -> Fresh (Parameter ResolvedName Type) freshParameterDefinition parameter = do @@ -206,6 +209,8 @@ freshExpression expression = case expression of LiteralExpression spanValue literal <$> freshType annotation MemberAccessExpression spanValue receiver member annotation -> MemberAccessExpression spanValue <$> freshExpression receiver <*> pure member <*> freshType annotation + MethodReferenceExpression spanValue receiver member annotation -> + MethodReferenceExpression spanValue <$> freshExpression receiver <*> pure member <*> freshType annotation CallExpression spanValue callee arguments annotation -> CallExpression spanValue <$> freshExpression callee @@ -321,6 +326,7 @@ declarationSymbols declaration = case declaration of : typeSymbols annotation ++ concatMap templateParameterSymbols parameters ++ concatMap declarationSymbols members + EnumDeclaration _ name annotation _ _ -> nameSymbol name : typeSymbols annotation templateParameterSymbols :: TemplateParameter ResolvedName Type -> [Int] templateParameterSymbols parameter = @@ -363,6 +369,7 @@ expressionSymbols expression = case expression of NameExpression _ name annotation -> nameSymbol name : typeSymbols annotation LiteralExpression _ _ annotation -> typeSymbols annotation MemberAccessExpression _ receiver _ annotation -> expressionSymbols receiver ++ typeSymbols annotation + MethodReferenceExpression _ receiver _ annotation -> expressionSymbols receiver ++ typeSymbols annotation CallExpression _ callee arguments annotation -> expressionSymbols callee ++ concatMap expressionSymbols arguments ++ typeSymbols annotation UnaryExpression _ _ value annotation -> expressionSymbols value ++ typeSymbols annotation diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Instantiation.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Instantiation.hs index 051390b1..41c79074 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Instantiation.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Instantiation.hs @@ -85,6 +85,7 @@ instantiateMember binding declaration = case declaration of -- outer binding blindly would capture same-spelled inner parameters. -- Nested instantiation will be selected independently by the planner. pure declaration + EnumDeclaration {} -> pure declaration -- | Instantiate one parameter annotation while preserving its source syntax. instantiateParameter :: @@ -179,6 +180,10 @@ instantiateExpression binding expression = case expression of closedReceiver <- instantiateExpression binding receiver closedAnnotation <- instantiateType binding annotation pure (MemberAccessExpression spanValue closedReceiver member closedAnnotation) + MethodReferenceExpression spanValue receiver member annotation -> do + closedReceiver <- instantiateExpression binding receiver + closedAnnotation <- instantiateType binding annotation + pure (MethodReferenceExpression spanValue closedReceiver member closedAnnotation) CallExpression spanValue callee arguments annotation -> do closedCallee <- instantiateExpression binding callee closedArguments <- traverse (instantiateExpression binding) arguments diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Mangling.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Mangling.hs index c8b2101a..ad90dd3e 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Mangling.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Mangling.hs @@ -127,6 +127,7 @@ mangleTemplateMember limits owner member = do ) TypeDeclaration {} -> nestedTypeMember ownerName member TemplateTypeDeclaration {} -> Left ExpectedMangleableMemberDeclaration + EnumDeclaration {} -> Left ExpectedMangleableMemberDeclaration where nestedTypeMember ownerName declaration = do name <- encodeIdentifier (resolvedSpelling (declarationName declaration)) diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/MemberReachability.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/MemberReachability.hs index b27a2b5b..8b411e2a 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/MemberReachability.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/MemberReachability.hs @@ -335,6 +335,7 @@ memberCalls owner declaration = case declaration of -- not accidentally pull siblings into the enclosing template closure. TypeDeclaration {} -> [] TemplateTypeDeclaration {} -> [] + EnumDeclaration {} -> [] blockCalls :: SymbolId -> Block ResolvedName Type -> [TemplateMemberCall] blockCalls owner (Block statements) = concatMap (statementCalls owner) statements @@ -370,6 +371,7 @@ expressionCalls owner expression = case expression of NameExpression {} -> [] LiteralExpression {} -> [] MemberAccessExpression _ receiver _ _ -> expressionCalls owner receiver + MethodReferenceExpression _ receiver _ _ -> expressionCalls owner receiver CallExpression spanValue callee arguments _ -> directCall spanValue callee ++ expressionCalls owner callee diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization.hs index 8714a9db..2d3f910e 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization.hs @@ -422,6 +422,7 @@ declarationSources = Map.fromList . concatMap collect (resolvedSymbol (declarationName declaration), declaration) : concatMap collect (typeMembers declaration) collect TypeDeclaration {typeMembers = members} = concatMap collect members collect FunctionDeclaration {} = [] + collect EnumDeclaration {} = [] attachDependencies :: [TemplateSpecialization] -> [TemplateSpecialization] attachDependencies specializations = map attach specializations @@ -515,6 +516,7 @@ declarationTypes declaration = case declaration of FunctionDeclaration _ _ annotation _ parameters body _ _ -> annotation : concatMap parameterTypes parameters ++ blockTypes body TemplateTypeDeclaration {} -> [] + EnumDeclaration {} -> [] parameterTypes :: Parameter ResolvedName Type -> [Type] parameterTypes parameter = [parameterAnnotation parameter] @@ -551,6 +553,7 @@ expressionTypes expression = case expression of NameExpression _ _ annotation -> [annotation] LiteralExpression _ _ annotation -> [annotation] MemberAccessExpression _ receiver _ annotation -> annotation : expressionTypes receiver + MethodReferenceExpression _ receiver _ annotation -> annotation : expressionTypes receiver CallExpression _ callee arguments annotation -> annotation : expressionTypes callee ++ concatMap expressionTypes arguments UnaryExpression _ _ value annotation -> annotation : expressionTypes value diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization/Verifier.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization/Verifier.hs index ad5602b2..f64ffa49 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization/Verifier.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/Template/Specialization/Verifier.hs @@ -274,6 +274,7 @@ declarationDefinitionSymbols declaration = case declaration of : map (resolvedSymbol . parameterName) parameters ++ blockDefinitionSymbols body TemplateTypeDeclaration {} -> [] + EnumDeclaration _ name _ _ _ -> [resolvedSymbol name] blockDefinitionSymbols :: Block ResolvedName Type -> [SymbolId] blockDefinitionSymbols (Block statements) = concatMap statementDefinitionSymbols statements @@ -308,6 +309,7 @@ statementDefinitionSymbols statement = case statement of expressionDefinitionSymbols :: Expression ResolvedName Type -> [SymbolId] expressionDefinitionSymbols expression = case expression of MemberAccessExpression _ receiver _ _ -> expressionDefinitionSymbols receiver + MethodReferenceExpression _ receiver _ _ -> expressionDefinitionSymbols receiver CallExpression _ callee arguments _ -> expressionDefinitionSymbols callee ++ concatMap expressionDefinitionSymbols arguments UnaryExpression _ _ value _ -> expressionDefinitionSymbols value @@ -352,6 +354,7 @@ declarationTypes declaration = case declaration of annotation : map parameterAnnotation parameters ++ blockTypes body TemplateTypeDeclaration _ _ annotation parameters members -> annotation : map templateParameterAnnotation parameters ++ concatMap declarationTypes members + EnumDeclaration _ _ annotation _ _ -> [annotation] blockTypes :: Block ResolvedName Type -> [Type] blockTypes (Block statements) = concatMap statementTypes statements @@ -385,6 +388,7 @@ expressionTypes expression = case expression of NameExpression _ _ annotation -> [annotation] LiteralExpression _ _ annotation -> [annotation] MemberAccessExpression _ receiver _ annotation -> annotation : expressionTypes receiver + MethodReferenceExpression _ receiver _ annotation -> annotation : expressionTypes receiver CallExpression _ callee arguments annotation -> annotation : expressionTypes callee ++ concatMap expressionTypes arguments UnaryExpression _ _ value annotation -> annotation : expressionTypes value diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker.hs index 8f0a9696..f30b9357 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker.hs @@ -9,13 +9,18 @@ used as a substitute for symbol identity. -} module Visual.XSharp.TypeChecker (TypeChecker (..), defaultTypeChecker, runTypeChecker) where +import Data.Map.Strict qualified as Map import Visual.XSharp.AST import Visual.XSharp.BuiltinTypes -import Visual.XSharp.ConstantEvaluation import Visual.XSharp.Diagnostic import Visual.XSharp.NumericSemantics -import Visual.XSharp.TemplateValue import Visual.XSharp.TypeChecker.Branching +import Visual.XSharp.TypeChecker.Console +import Visual.XSharp.TypeChecker.Context +import Visual.XSharp.TypeChecker.Enums +import Visual.XSharp.TypeChecker.Literals +import Visual.XSharp.TypeChecker.Loops +import Visual.XSharp.TypeChecker.Returns -- | A resolved-tree checker that produces a typed tree only when checking succeeds. newtype TypeChecker = TypeChecker {checkResolvedAST :: ResolvedAST -> Either [Diagnostic] TypedAST} @@ -28,41 +33,12 @@ runTypeChecker = checkResolvedAST defaultTypeChecker :: TypeChecker defaultTypeChecker = TypeChecker checkTree -type TypeEnvironment = [(SymbolId, (Type, Bool))] - --- The type checker owns a whole source-set catalog before it checks any body. --- That permits calls to later-declared classes without making parsing depend --- on declaration order or mutating the Renamer's lexical environment. -data MethodCandidate = MethodCandidate - { candidateOwner :: SymbolId - , candidateDeclaration :: Declaration ResolvedName () - } - -data TypeCatalog = TypeCatalog - { catalogTypes :: [(SymbolId, ResolvedName)] - , catalogMethods :: [MethodCandidate] - } - --- Type syntax deliberately keeps source spellings. This side environment is --- the bridge from those spellings to the SymbolIds assigned by the renamer. --- Type and value parameters are separate because `T` in a type position and --- `N` in `[T; N]` have different semantic representations. -data TemplateContext = TemplateContext - { templateTypeNames :: [(Identifier, ResolvedName)] - , templateValueNames :: [(Identifier, ResolvedName)] - , templateCatalog :: TypeCatalog - , templateCurrentType :: Maybe SymbolId - } - -emptyTemplateContext :: TypeCatalog -> Maybe SymbolId -> TemplateContext -emptyTemplateContext catalog owner = TemplateContext [] [] catalog owner - {- | Type-check every top-level declaration and collect independent diagnostics. No partially typed AST escapes when any declaration has an error. -} checkTree :: ResolvedAST -> Either [Diagnostic] TypedAST checkTree (ResolvedAST (SyntaxTree namespace declarations)) = - let catalog = catalogDeclarations declarations + let catalog = inferAutoReturns declarations (catalogDeclarations declarations) checked = map (checkTopDeclaration catalog) declarations problems = concatMap snd checked in if null problems then Right (TypedAST (SyntaxTree namespace (map fst checked))) else Left problems @@ -80,6 +56,8 @@ catalogDeclarations declarations = , member <- typeMembersOf owner , case member of FunctionDeclaration {} -> True; _ -> False ] + (enumInfos declarations) + [] where isTypeDeclaration TypeDeclaration {} = True isTypeDeclaration TemplateTypeDeclaration {} = True @@ -88,39 +66,107 @@ catalogDeclarations declarations = typeMembersOf TemplateTypeDeclaration {typeMembers = members} = members typeMembersOf _ = [] +{- | Infer the return types of the methods declared with @auto@ before any +caller is checked against them. + +A method's return type comes from its own returns, and those may be calls of +other such methods, in any class and declared later. The bodies are therefore +checked in rounds. In a round, a call of a method whose type is not known yet +has no type and takes no part in the inference, so a method is inferred as +soon as one of its returns is independent of the methods still unknown: a +recursive method from its base case, and a chain of methods from its end +towards its start. A round that learns nothing ends the inference; every +round before it resolves at least one method, so there are at most as many +rounds as methods. A method that is still unknown then has no independent +result, which is reported when its declaration is checked. The diagnostics +of the rounds are dropped: every body is checked once more against the +final catalog. +-} +inferAutoReturns :: [Declaration ResolvedName ()] -> TypeCatalog -> TypeCatalog +inferAutoReturns declarations = rounds (length inferable) + where + inferable = + [ (owner, member) + | owner <- declarations + , member <- membersOf owner + , FunctionDeclaration {declarationReturnSyntax = AutoType} <- [member] + ] + membersOf declaration = case declaration of + TypeDeclaration {typeMembers = members} -> members + TemplateTypeDeclaration {typeMembers = members} -> members + _ -> [] + rounds :: Int -> TypeCatalog -> TypeCatalog + rounds remaining catalog + | remaining <= 0 || learned == catalogInferredReturns catalog = catalog + | otherwise = rounds (remaining - 1) catalog {catalogInferredReturns = learned} + where + learned = + [ (inferredReturnKey member, result) + | (owner, member) <- inferable + , let (context, globals) = memberScope catalog owner + , FunctionDeclaration {declarationAnnotation = FunctionType _ result} <- + [fst (checkDeclarationWith context globals member)] + , result /= ErrorType + ] + +-- | What identifies a method declaration among its overloads. +inferredReturnKey :: Declaration ResolvedName annotation -> (SymbolId, SourceSpan) +inferredReturnKey declaration = (resolvedSymbol (declarationName declaration), declarationSpan declaration) + +{- | The context and the names in scope of the members of a type: the +signatures of its members and, for a template, its value parameters. +-} +memberScope :: TypeCatalog -> Declaration ResolvedName () -> (TemplateContext, TypeEnvironment) +memberScope catalog declaration = case declaration of + TemplateTypeDeclaration _ name _ parameters members -> + let context = templateContext catalog (Just (resolvedSymbol name)) parameters + templateValues = + [ (resolvedSymbol (templateParameterName parameter), (templateParameterAnnotation parameter, False)) + | parameter <- map (typeTemplateParameter context) parameters + , case templateParameterKind parameter of TemplateValueParameterKind _ -> True; _ -> False + ] + in (context, templateValues ++ signaturesOf context members) + TypeDeclaration _ name _ members -> + let context = emptyTemplateContext catalog (Just (resolvedSymbol name)) + in (context, signaturesOf context members) + _ -> (emptyTemplateContext catalog Nothing, []) + where + signaturesOf context members = + [(resolvedSymbol (declarationName member), (signature context member, False)) | member <- members] + signature :: TemplateContext -> Declaration ResolvedName () -> Type signature context declaration = case declaration of FunctionDeclaration _ _ _ returnSyntax parameters _ _ _ -> FunctionType (map (syntaxTypeIn context . parameterTypeSyntax) parameters) - (syntaxTypeIn context returnSyntax) + ( case returnSyntax of + AutoType -> + maybe ErrorType id (lookup (inferredReturnKey declaration) (catalogInferredReturns (templateCatalog context))) + _ -> syntaxTypeIn context returnSyntax + ) TypeDeclaration _ name _ _ -> NamedType (QualifiedName [resolvedSpelling name]) [] TemplateTypeDeclaration _ name _ parameters _ -> NamedType (QualifiedName [resolvedSpelling name]) (map templateParameterAsArgument parameters) + EnumDeclaration _ name _ _ _ -> enumTypeOf (templateCatalog context) name + +-- | The type of the values of a declared enum. +enumTypeOf :: TypeCatalog -> ResolvedName -> Type +enumTypeOf catalog name = maybe ErrorType enumInfoType (enumBySymbol (catalogEnums catalog) (resolvedSymbol name)) checkTopDeclaration :: TypeCatalog -> Declaration ResolvedName () -> (Declaration ResolvedName Type, [Diagnostic]) checkTopDeclaration catalog declaration = case declaration of TypeDeclaration spanValue name _ members -> - let owner = resolvedSymbol name - context = emptyTemplateContext catalog (Just owner) - signatures = [(resolvedSymbol (declarationName member), (signature context member, False)) | member <- members] + let (context, signatures) = memberScope catalog declaration checked = map (checkDeclarationWith context signatures) members overloadProblems = duplicateOverloadProblems context members valueType = NamedType (QualifiedName [resolvedSpelling name]) [] in (TypeDeclaration spanValue name valueType (map fst checked), overloadProblems ++ concatMap snd checked) TemplateTypeDeclaration spanValue name _ parameters members -> - let owner = resolvedSymbol name - context = (templateContext catalog (Just owner) parameters) + let (context, scope) = memberScope catalog declaration typedTemplateParameters = map (typeTemplateParameter context) parameters - templateValues = - [ (resolvedSymbol (templateParameterName parameter), (templateParameterAnnotation parameter, False)) - | parameter <- typedTemplateParameters - , case templateParameterKind parameter of TemplateValueParameterKind _ -> True; _ -> False - ] - signatures = [(resolvedSymbol (declarationName member), (signature context member, False)) | member <- members] - checked = map (checkDeclarationWith context (templateValues ++ signatures)) members + checked = map (checkDeclarationWith context scope) members parameterProblems = validateTemplateParameters context parameters overloadProblems = duplicateOverloadProblems context members valueType = @@ -131,157 +177,10 @@ checkTopDeclaration catalog declaration = case declaration of , parameterProblems ++ overloadProblems ++ concatMap snd checked ) FunctionDeclaration {} -> checkDeclarationWith (emptyTemplateContext catalog Nothing) [] declaration - -{- | Keep type and value parameters in separate lookup tables. -Identical source spelling in the two categories must not collapse their roles. --} -templateContext :: TypeCatalog -> Maybe SymbolId -> [TemplateParameter ResolvedName annotation] -> TemplateContext -templateContext catalog owner parameters = - TemplateContext - [ (resolvedSpelling name, name) - | parameter <- parameters - , case templateParameterKind parameter of - TemplateTypeParameter -> True - TemplateTemplateParameter _ -> True - _ -> False - , let name = templateParameterName parameter - ] - [ (resolvedSpelling name, name) - | parameter <- parameters - , case templateParameterKind parameter of TemplateValueParameterKind _ -> True; _ -> False - , let name = templateParameterName parameter - ] - catalog - owner - -templateParameterAsArgument :: TemplateParameter ResolvedName annotation -> TemplateArgument -templateParameterAsArgument parameter = case templateParameterKind parameter of - TemplateValueParameterKind _ -> ValueTemplateArgument (TemplateValueParameter (templateParameterName parameter)) - _ -> TypeTemplateArgument (TypeVariable (templateParameterName parameter)) - -syntaxTypeIn :: TemplateContext -> TypeSyntax -> Type -syntaxTypeIn _ AutoType = ErrorType -syntaxTypeIn context (QualifiedTypeSyntax name arguments) = - case name of - QualifiedName [identifier] | Just resolved <- lookup identifier (templateTypeNames context) -> TypeVariable resolved - _ -> NamedType name (map (syntaxTemplateArgumentIn context) arguments) -syntaxTypeIn context (BuiltinArrayTypeSyntax element) = - -- `[]T` is a language type, not a public class invented by the compiler. - -- Its structural spelling keeps that distinction visible through Core - -- until ownership-aware lowering assigns the final runtime layout. - NamedType (QualifiedName [Identifier "[]"]) [TypeTemplateArgument (syntaxTypeIn context element)] -syntaxTypeIn context (ArrayTypeSyntax element) = - NamedType - (QualifiedName [Identifier "System", Identifier "Array"]) - [TypeTemplateArgument (syntaxTypeIn context element)] -syntaxTypeIn context (FixedArrayTypeSyntax element size) = - NamedType - (QualifiedName [Identifier "System", Identifier "Array"]) - [TypeTemplateArgument (syntaxTypeIn context element), ValueTemplateArgument (syntaxTemplateValueIn context size)] -syntaxTypeIn context (DictionaryTypeSyntax key value) = - NamedType - (QualifiedName [Identifier "System", Identifier "Dictionary"]) - [TypeTemplateArgument (syntaxTypeIn context key), TypeTemplateArgument (syntaxTypeIn context value)] -syntaxTypeIn context (CallableTypeSyntax parameters result) = FunctionType (map (syntaxTypeIn context) parameters) (syntaxTypeIn context result) -syntaxTypeIn context (ExplicitType identifier@(Identifier name)) = case lookup identifier (templateTypeNames context) of - Just resolved -> TypeVariable resolved - Nothing -> case name of - "String" -> stringType - "unit" -> unitType - "void" -> voidType - _ -> maybe (NamedType (QualifiedName [Identifier name]) []) scalarTypeToType (lookupScalar name) - where - lookupScalar spelling = lookup spelling [(scalarTypeName scalar, scalar) | scalar <- scalarTypes] - -syntaxTemplateArgumentIn :: TemplateContext -> TemplateArgumentSyntax -> TemplateArgument -syntaxTemplateArgumentIn context argument = case argument of - TemplateTypeSyntax valueType -> TypeTemplateArgument (syntaxTypeIn context valueType) - TemplateValueArgumentSyntax value -> ValueTemplateArgument (syntaxTemplateValueIn context value) - --- Parser construction guarantees the expression tree is side-effect free. --- Exact evaluation here gives every concrete specialization one canonical --- identity. Invalid arithmetic becomes a sentinel and is diagnosed by the --- type-syntax validation pass before Core can be emitted. - -{- | Canonicalize a template argument before it contributes to specialization identity. -The fallback value is only a recovery sentinel; validation reports the original -invalid expression before a specialization plan is emitted. --} -syntaxTemplateValueIn :: TemplateContext -> TemplateValueSyntax -> TemplateValue -syntaxTemplateValueIn context value = case value of - TemplateNameSyntax _ (QualifiedName [identifier]) - | Just resolved <- lookup identifier (templateValueNames context) -> TemplateValueParameter resolved - _ -> case evaluateTemplateValue value of - Right result -> result - _ -> IntegerTemplateValue 0 - -typeTemplateParameter :: TemplateContext -> TemplateParameter ResolvedName () -> TemplateParameter ResolvedName Type -typeTemplateParameter context parameter = - TemplateParameter - (templateParameterSpan parameter) - (templateParameterName parameter) - annotation - (templateParameterKind parameter) - (templateParameterIsPack parameter) - (templateParameterDefault parameter) - where - annotation = case templateParameterKind parameter of - TemplateValueParameterKind valueType -> syntaxTypeIn context valueType - _ -> TypeVariable (templateParameterName parameter) - -validateTemplateParameters :: TemplateContext -> [TemplateParameter ResolvedName ()] -> [Diagnostic] -validateTemplateParameters context = concatMap validate - where - validate parameter = - kindProblems parameter - ++ defaultProblems parameter - ++ packDefaultProblems parameter - kindProblems parameter = case templateParameterKind parameter of - TemplateTypeParameter -> [] - TemplateValueParameterKind valueType -> typeSyntaxProblemsIn context valueType - TemplateTemplateParameter shapes -> concatMap shapeProblems shapes - shapeProblems shape = case templateParameterShapeKind shape of - TemplateTypeParameterShape -> [] - TemplateValueParameterShape valueType -> typeSyntaxProblemsIn context valueType - TemplateTemplateParameterShape shapes -> concatMap shapeProblems shapes - defaultProblems parameter = case templateParameterDefault parameter of - Nothing -> [] - Just (TemplateTypeDefault valueType) -> typeSyntaxProblemsIn context valueType - Just (TemplateValueDefault value) -> templateValueProblemsIn context "VXT0020" value - packDefaultProblems parameter - | templateParameterIsPack parameter - , Just _ <- templateParameterDefault parameter = - [problem (templateParameterSpan parameter) "VXT0019" "a template parameter pack cannot have a default"] - | otherwise = [] - -typeSyntaxProblemsIn :: TemplateContext -> TypeSyntax -> [Diagnostic] -typeSyntaxProblemsIn context syntax = case syntax of - ExplicitType _ -> [] - AutoType -> [] - BuiltinArrayTypeSyntax element -> typeSyntaxProblemsIn context element - ArrayTypeSyntax element -> typeSyntaxProblemsIn context element - DictionaryTypeSyntax key value -> typeSyntaxProblemsIn context key ++ typeSyntaxProblemsIn context value - CallableTypeSyntax parameters result -> concatMap (typeSyntaxProblemsIn context) parameters ++ typeSyntaxProblemsIn context result - QualifiedTypeSyntax _ arguments -> concatMap (templateArgumentProblemsIn context) arguments - FixedArrayTypeSyntax element size -> - typeSyntaxProblemsIn context element - ++ case syntaxTemplateValueIn context size of - TemplateValueParameter _ -> [] - _ -> case evaluateFixedArraySize size of - Left issue -> [problem (templateValueSyntaxSpan size) "VXT0016" (renderTemplateValueError issue)] - Right _ -> [] - -templateArgumentProblemsIn :: TemplateContext -> TemplateArgumentSyntax -> [Diagnostic] -templateArgumentProblemsIn context argument = case argument of - TemplateTypeSyntax valueType -> typeSyntaxProblemsIn context valueType - TemplateValueArgumentSyntax value -> templateValueProblemsIn context "VXT0017" value - -templateValueProblemsIn :: TemplateContext -> String -> TemplateValueSyntax -> [Diagnostic] -templateValueProblemsIn context code value = case syntaxTemplateValueIn context value of - TemplateValueParameter _ -> [] - _ -> case evaluateTemplateValue value of - Left issue -> [problem (templateValueSyntaxSpan value) code (renderTemplateValueError issue)] - Right _ -> [] + EnumDeclaration spanValue name _ underlying cases -> + ( EnumDeclaration spanValue name (enumTypeOf catalog name) underlying cases + , enumDeclarationProblems declaration + ) checkDeclarationWith :: TemplateContext -> TypeEnvironment -> Declaration ResolvedName () -> (Declaration ResolvedName Type, [Diagnostic]) @@ -291,21 +190,37 @@ checkDeclarationWith context globals declaration@FunctionDeclaration {} = | parameter <- declarationParameters declaration ] expected = syntaxTypeIn context (declarationReturnSyntax declaration) - (body, _, explicitReturns, problems) = checkBlockWith context (parameters ++ globals) expected outsideLoops (declarationBody declaration) + (body, _, _, problems) = checkBlockWith context (parameters ++ globals) expected outsideLoops (declarationBody declaration) finalReturn = finalExpressionType body - returns = explicitReturns ++ maybe [] (: []) finalReturn + -- Every return of the body counts, also one reached through an + -- expression; the returns of a nested callable are its own. + returns = blockReturnTypes body ++ maybe [] (: []) finalReturn inferred = inferReturn expected returns - returnProblems = - if expected /= ErrorType && any (not . compatible expected) returns - then - [ Diagnostic - TypeCheckerStage - Error - "VXT0001" - (Just (declarationSpan declaration)) - "return expression does not match the declared function type" - ] - else [] + isInferred = case declarationReturnSyntax declaration of AutoType -> True; _ -> False + returnProblems + | expected /= ErrorType && any (not . compatible expected) returns = + [ Diagnostic + TypeCheckerStage + Error + "VXT0001" + (Just (declarationSpan declaration)) + "return expression does not match the declared function type" + ] + | isInferred && any (not . compatible inferred) returns = + [ problem + (declarationSpan declaration) + "VXT0062" + "the return statements of this method carry values of different types" + ] + -- Every result of the method is a call that depends on the + -- method itself: nothing gives it a type. + | isInferred && inferred == ErrorType && null problems = + [ problem + (declarationSpan declaration) + "VXT0063" + "the return type of this method cannot be inferred: no result is independent of the method itself" + ] + | otherwise = [] typedParameters = map (typeParameterWith context) (declarationParameters declaration) signatureProblems = typeSyntaxProblemsIn context (declarationReturnSyntax declaration) @@ -324,6 +239,7 @@ checkDeclarationWith context globals declaration@FunctionDeclaration {} = ) checkDeclarationWith context _ declaration@TypeDeclaration {} = checkTopDeclaration (templateCatalog context) declaration checkDeclarationWith context _ declaration@TemplateTypeDeclaration {} = checkTopDeclaration (templateCatalog context) declaration +checkDeclarationWith context _ declaration@EnumDeclaration {} = checkTopDeclaration (templateCatalog context) declaration -- A method overload is distinguished only by its ordered parameter types. -- Access, return type, and static-ness intentionally do not rescue duplicate @@ -331,28 +247,29 @@ checkDeclarationWith context _ declaration@TemplateTypeDeclaration {} = checkTop duplicateOverloadProblems :: TemplateContext -> [Declaration ResolvedName ()] -> [Diagnostic] duplicateOverloadProblems context members = reverse problems where - (_, problems) = foldl inspect ([], []) members + -- The signatures seen so far are kept by method name, so that a + -- method is compared with its own overloads only. Comparing it with + -- every earlier member made a type with thousands of methods take + -- time with the square of their number. + (_, problems) = foldl inspect (Map.empty, []) members inspect (seen, diagnostics) declaration@FunctionDeclaration {} = - let duplicate = any (sameSignature declaration) seen + let spelling = resolvedSpelling (declarationName declaration) + parameterTypes = methodParameterTypes declaration + duplicate = parameterTypes `elem` Map.findWithDefault [] spelling seen currentDiagnostics = if duplicate then [ problem (declarationSpan declaration) "VXT0028" - ( "method overload has a duplicate parameter signature: " - ++ identifierText (resolvedSpelling (declarationName declaration)) - ) + ("method overload has a duplicate parameter signature: " ++ identifierText spelling) ] else [] - in (declaration : seen, reverse currentDiagnostics ++ diagnostics) + in (Map.insertWith (++) spelling [parameterTypes] seen, reverse currentDiagnostics ++ diagnostics) inspect state _ = state - sameSignature current previous = - resolvedSpelling (declarationName previous) == resolvedSpelling (declarationName current) - && methodParameterTypes context previous == methodParameterTypes context current - methodParameterTypes valueContext FunctionDeclaration {declarationParameters = parameters} = - map (syntaxTypeIn valueContext . parameterTypeSyntax) parameters - methodParameterTypes _ _ = [] + methodParameterTypes FunctionDeclaration {declarationParameters = parameters} = + map (syntaxTypeIn context . parameterTypeSyntax) parameters + methodParameterTypes _ = [] typeParameterWith :: TemplateContext -> Parameter ResolvedName () -> Parameter ResolvedName Type typeParameterWith context parameter = @@ -372,39 +289,15 @@ compatible ErrorType _ = True compatible _ ErrorType = True compatible left right = left == right --- | How a loop is used, which decides what its @break@ statements carry. -data LoopKind - = -- | A loop statement: @break@ carries no value. - StatementLoop - | {- | A loop used as an expression: every @break@ carries the loop's - value, typed in the context that receives it. - -} - ExpressionLoop (Maybe Type) - | {- | Not a loop: the edge of a block used as a value. A @break@ or - @continue@ inside it would have to leave the block before its value - exists, so neither may cross this edge. - -} - ValueBlockEdge - -{- | The loops around a statement, innermost first, and the kind of the loop -statement that is about to be checked. A loop expression checks its loop -statement with the pending kind set; every loop moves the pending kind onto -the stack for its own body. --} -data LoopContext = LoopContext - { pendingLoop :: LoopKind - , enclosingLoops :: [LoopKind] - } +-- | The context inside a block used as a value at the given place. +insideValueBlock :: TemplateContext -> LoopContext +insideValueBlock context = LoopContext StatementLoop (ValueBlockEdge : enclosingLoops (contextLoops context)) -outsideLoops :: LoopContext -outsideLoops = LoopContext StatementLoop [] - -enterLoop :: LoopContext -> LoopContext -enterLoop loops = LoopContext StatementLoop (pendingLoop loops : enclosingLoops loops) - --- | The context inside a block used as a value, whatever surrounds it. -insideValueBlock :: LoopContext -insideValueBlock = LoopContext StatementLoop [ValueBlockEdge] +{- | The context of the condition of the loop that is about to be checked in +the given context: a transfer there reaches that loop. +-} +inLoopCondition :: LoopContext -> TemplateContext -> TemplateContext +inLoopCondition loops context = context {contextLoops = loopCondition loops} {- | The checker for expressions and statements as the branching rules of "Visual.XSharp.TypeChecker.Branching" receive it: applied to the template @@ -419,8 +312,9 @@ branchChecker context expected = in (typed, final, returns, problems) , branchType = \syntax -> (syntaxTypeIn context syntax, typeSyntaxProblemsIn context syntax) , branchLiteral = literalTypeInContext + , branchEnumMember = enumMemberValue (catalogEnums (templateCatalog context)) , branchHasEffect = effectCapable - , branchValueLoops = insideValueBlock + , branchValueLoops = insideValueBlock context } -- | The context of a statement that belongs to a loop header, not its body. @@ -450,7 +344,18 @@ checkStatementWith :: LoopContext -> Statement ResolvedName () -> (Statement ResolvedName Type, TypeEnvironment, [Type], [Diagnostic]) -checkStatementWith context environment expected loops statement = case statement of +checkStatementWith outer environment expected loops = + checkStatementIn (outer {contextReturn = expected, contextLoops = loops}) environment expected loops + +-- The context already names the return type and the loops of the statement. +checkStatementIn :: + TemplateContext -> + TypeEnvironment -> + Type -> + LoopContext -> + Statement ResolvedName () -> + (Statement ResolvedName Type, TypeEnvironment, [Type], [Diagnostic]) +checkStatementIn context environment expected loops statement = case statement of BindingStatement spanValue kind syntax name _ value -> let declared = syntaxTypeIn context syntax target = if declared == ErrorType then Nothing else Just declared @@ -497,7 +402,7 @@ checkStatementWith context environment expected loops statement = case statement , conditionProblems ++ conditionMismatch ++ trueProblems ++ falseProblems ) WhileStatement spanValue condition body -> - let (typedCondition, conditionType, conditionProblems) = checkExpressionWith context environment condition + let (typedCondition, conditionType, conditionProblems) = checkExpressionWith (inLoopCondition loops context) environment condition conditionProblems' = if booleanContextType conditionType then [] @@ -510,7 +415,7 @@ checkStatementWith context environment expected loops statement = case statement ) DoWhileStatement spanValue body condition -> let (typedBody, _, returns, bodyProblems) = checkBlockWith context environment expected (enterLoop loops) body - (typedCondition, conditionType, conditionProblems) = checkExpressionWith context environment condition + (typedCondition, conditionType, conditionProblems) = checkExpressionWith (inLoopCondition loops context) environment condition conditionProblems' = if booleanContextType conditionType then [] @@ -529,14 +434,14 @@ checkStatementWith context environment expected loops statement = case statement (typedCondition, conditionType, conditionProblems) = case condition of Nothing -> (Nothing, boolType, []) Just value -> - let (typed, valueType, problems) = checkExpressionWith context loopEnvironment value + let (typed, valueType, problems) = checkExpressionWith (inLoopCondition loops context) loopEnvironment value in (Just typed, valueType, problems) conditionProblems' = if booleanContextType conditionType then [] else [problem spanValue "VXT0020" "for condition must be bool or numeric"] (typedBody, _, returns, bodyProblems) = checkBlockWith context loopEnvironment expected (enterLoop loops) body - (typedUpdates, updateProblems) = checkStatementsWith context loopEnvironment expected (enterLoop loops) updates + (typedUpdates, updateProblems) = checkStatementsWith context loopEnvironment expected (loopUpdate loops) updates in ( ForStatement spanValue typedInitializer typedCondition typedUpdates typedBody , environment , returns @@ -570,6 +475,20 @@ checkStatementWith context environment expected loops statement = case statement then [] else [problem spanValue "VXT0024" "increment target must have a numeric type"] in (IncrementStatement spanValue name targetType, environment, [], writableProblems ++ numericProblems) + CompoundAssignmentStatement spanValue Add name _ value + | Just (targetType, writable) <- lookup (resolvedSymbol name) environment + , targetType == stringType -> + -- `text += value` is `text = text + value`: the value is + -- written as text and joined to what the target holds. + let (typedValue, valueType, problems) = checkExpressionWith context environment value + (joined, _, joinProblems) = + concatenation spanValue (NameExpression spanValue name stringType, stringType) (typedValue, valueType) + immutable = [problem spanValue "VXT0003" "cannot assign to an immutable binding" | not writable] + in ( AssignmentStatement spanValue name stringType joined + , environment + , [] + , problems ++ immutable ++ joinProblems + ) CompoundAssignmentStatement spanValue operator name _ value -> -- `target op= value` has the typing of `target = target op value`: -- the operator rule is applied to the target type and the result @@ -600,28 +519,33 @@ checkStatementWith context environment expected loops statement = case statement -- A break leaves the innermost loop. Whether it may, or must, carry -- a value is decided by how that loop is used; the value takes its -- context from the place that receives the loop's value. - let innermost = case enclosingLoops loops of - kind : _ -> Just kind - [] -> Nothing - valueExpected = case innermost of - Just (ExpressionLoop target) -> target + let target = transferTarget loops + valueExpected = case target of + Just (ExpressionLoop expectedValue) -> expectedValue + Just (LoopCondition (ExpressionLoop expectedValue)) -> expectedValue + Just (LoopUpdate (ExpressionLoop expectedValue)) -> expectedValue _ -> Nothing (typedValue, _, valueProblems) = checkOptionalExpectedWith context environment valueExpected value - placementProblems = case (innermost, value) of - (Nothing, _) -> [problem spanValue "VXT0025" "break is only valid inside a loop"] - (Just StatementLoop, Just _) -> + placement kind = case (kind, value) of + (StatementLoop, Just _) -> [problem spanValue "VXT0026" "a value-carrying break is only valid in a loop used as an expression"] - (Just (ExpressionLoop _), Nothing) -> + (ExpressionLoop _, Nothing) -> [problem spanValue "VXT0040" "a loop used as an expression must be left by a break that carries a value"] - (Just ValueBlockEdge, _) -> - [problem spanValue "VXT0059" "break cannot leave a block that is used as a value"] + -- A break in the condition or the update clause of a loop + -- leaves that loop. + (LoopCondition loop, _) -> placement loop + (LoopUpdate loop, _) -> placement loop _ -> [] + placementProblems = case target of + Nothing -> [problem spanValue "VXT0025" "break is only valid inside a loop"] + Just kind -> placement kind in (BreakStatement spanValue typedValue, environment, [], placementProblems ++ valueProblems) ContinueStatement spanValue -> - let problems = case enclosingLoops loops of - [] -> [problem spanValue "VXT0027" "continue is only valid inside a loop"] - ValueBlockEdge : _ -> - [problem spanValue "VXT0059" "continue cannot leave a block that is used as a value"] + -- A continue in the condition of a loop evaluates that condition + -- again, and one in the update clause ends the update: the + -- condition of the loop is tested next. + let problems = case transferTarget loops of + Nothing -> [problem spanValue "VXT0027" "continue is only valid inside a loop"] _ -> [] in (ContinueStatement spanValue, environment, [], problems) GuardStatement spanValue condition block -> @@ -677,24 +601,6 @@ finalExpressionType (Block statements) = case reverse statements of ExpressionStatement _ expression False : _ -> Just (typedExpressionType expression) _ -> Nothing -typedExpressionType :: Expression name Type -> Type -typedExpressionType expression = case expression of - NameExpression _ _ valueType -> valueType - LiteralExpression _ _ valueType -> valueType - MemberAccessExpression _ _ _ valueType -> valueType - CallExpression _ _ _ valueType -> valueType - UnaryExpression _ _ _ valueType -> valueType - BinaryExpression _ _ _ _ valueType -> valueType - IsPatternExpression _ _ _ valueType -> valueType - ConditionalExpression _ _ _ _ valueType -> valueType - CoalesceExpression _ _ _ valueType -> valueType - AssignmentExpression _ _ _ _ valueType -> valueType - IncrementExpression _ _ _ valueType -> valueType - LoopExpression _ _ valueType -> valueType - BlockExpression _ _ valueType -> valueType - MatchExpression _ _ _ valueType -> valueType - CallableExpression _ _ _ _ _ valueType -> valueType - effectCapable :: Expression name annotation -> Bool effectCapable CallExpression {} = True effectCapable (IsPatternExpression _ subject _ _) = effectCapable subject @@ -754,10 +660,68 @@ checkExpressionExpectedWith context environment expected expression = case expre LiteralExpression spanValue literal _ -> let (valueType, problems) = literalTypeInContext spanValue expected literal in (LiteralExpression spanValue literal valueType, valueType, problems) + -- @Enum.Member@ is the value of that member. It is a constant of the + -- enum's type, and it is kept as the integer literal it stands for. + MemberAccessExpression spanValue (NameExpression _ name _) member _ + | Just info <- enumBySymbol (catalogEnums (templateCatalog context)) (resolvedSymbol name) -> + let valueType = enumInfoType info + in case lookup member (enumInfoMembers info) of + Just value -> (LiteralExpression spanValue (IntegerLiteral value) valueType, valueType, []) + Nothing -> + ( LiteralExpression spanValue (IntegerLiteral 0) valueType + , ErrorType + , + [ problem + spanValue + "VXT0064" + ( "the enum " + ++ identifierText (resolvedSpelling name) + ++ " has no member named " + ++ identifierText member + ) + ] + ) + -- @.Member@ is the member of that name of the enum the place expects. + -- Without an expected enum type there is nothing to select from. + MemberAccessExpression spanValue (LiteralExpression _ UnitLiteral _) member _ -> + let enums = catalogEnums (templateCatalog context) + in case [valueType | Just valueType <- [expected], isEnumType valueType] of + valueType : _ -> case enumMemberValue enums valueType member of + Just value -> (LiteralExpression spanValue (IntegerLiteral value) valueType, valueType, []) + Nothing -> + ( LiteralExpression spanValue (IntegerLiteral 0) valueType + , ErrorType + , [problem spanValue "VXT0064" ("the expected enum has no member named " ++ identifierText member)] + ) + [] -> + ( LiteralExpression spanValue (IntegerLiteral 0) ErrorType + , ErrorType + , + [ problem + spanValue + "VXT0069" + "a target-typed .Member needs a place whose type is known to be an enum" + ] + ) MemberAccessExpression spanValue receiver member _ -> let (typedReceiver, _, receiverProblems) = checkExpressionWith context environment receiver memberProblems = [problem spanValue "VXT0034" "member selection is currently supported only as a type-qualified method call"] in (MemberAccessExpression spanValue typedReceiver member ErrorType, ErrorType, receiverProblems ++ memberProblems) + -- @Type::Method@ is the static method as a callable value. + MethodReferenceExpression spanValue (NameExpression _ name _) member _ + | resolvedSymbol name `elem` map fst (catalogTypes (templateCatalog context)) -> + methodValue context expected spanValue (resolvedSymbol name) member + -- A reference through a value is a method bound to that value, which + -- needs the methods of values. + MethodReferenceExpression spanValue receiver member _ -> + let (typedReceiver, _, receiverProblems) = checkExpressionWith context environment receiver + memberProblems = + [ problem + spanValue + "VXT0081" + "a method reference is currently supported only for a static method named through its type" + ] + in (MethodReferenceExpression spanValue typedReceiver member ErrorType, ErrorType, receiverProblems ++ memberProblems) CallExpression spanValue callee arguments _ -> case callee of MemberAccessExpression _ receiver member _ -> @@ -779,8 +743,13 @@ checkExpressionExpectedWith context environment expected expression = case expre let operandExpected = if operator == LogicalNot then Nothing else expected (typedValue, valueType, problems) = checkExpressionExpectedWith context environment operandExpected value rule = unaryNumericRule operator valueType - mismatch = ruleProblems spanValue "VXT0011" rule - in (UnaryExpression spanValue operator typedValue (numericRuleType rule), numericRuleType rule, problems ++ mismatch) + -- An operand without a type was reported already, or never + -- yields a value; either way the operator has nothing to check. + (resultType, mismatch) = + if valueType == ErrorType + then (ErrorType, []) + else (numericRuleType rule, ruleProblems spanValue "VXT0011" rule) + in (UnaryExpression spanValue operator typedValue resultType, resultType, problems ++ mismatch) BinaryExpression spanValue operator left right _ -> -- A Boolean result does not imply Boolean operands: pushing the return -- context into 1 == 2 would convert both literals to true. Comparisons @@ -796,15 +765,48 @@ checkExpressionExpectedWith context environment expected expression = case expre (FloorDivide, _) -> checkExpressionWith context environment left _ -> checkExpressionExpectedWith context environment (if booleanResult operator then Nothing else expected) left (typedLeft, leftType, leftProblems) = leftResult - rightExpected = if operator `elem` [LogicalAnd, LogicalOr] then Nothing else Just leftType + -- The other side of a string is whatever it is by itself: `+` + -- writes it as text, so it does not borrow the string's type. + rightExpected = + if operator `elem` [LogicalAnd, LogicalOr] || leftType == stringType then Nothing else Just leftType (typedRight, rightType, rightProblems) = checkExpressionExpectedWith context environment rightExpected right rule = binaryNumericRule operator leftType rightType - resultType = numericRuleType rule - mismatch = ruleProblems spanValue "VXT0012" rule - in ( BinaryExpression spanValue operator typedLeft typedRight resultType - , resultType - , leftProblems ++ rightProblems ++ mismatch - ) + typed = leftType /= ErrorType && rightType /= ErrorType + -- `+` with a string on either side joins the two as text, and + -- `==` and `\=` on two strings compare their characters. + joins = typed && operator == Add && (leftType == stringType || rightType == stringType) + comparesText = typed && operator `elem` [Equal, NotEqual] && leftType == stringType && rightType == stringType + -- An operand without a type was reported already, or never + -- yields a value; either way the operator has nothing to check. + -- Values of an enum are compared for equality with values of + -- the same enum, and take part in no other operation. + (resultType, mismatch) + | leftType == ErrorType || rightType == ErrorType = + (if booleanResult operator then boolType else ErrorType, []) + | isEnumType leftType || isEnumType rightType = + if operator `elem` [Equal, NotEqual] && leftType == rightType + then (boolType, []) + else + ( if booleanResult operator then boolType else ErrorType + , + [ problem + spanValue + "VXT0065" + "values of an enum are only compared, with == and \\=, with values of the same enum" + ] + ) + | otherwise = (numericRuleType rule, ruleProblems spanValue "VXT0012" rule) + withOperands (value, valueType, problems) = (value, valueType, leftProblems ++ rightProblems ++ problems) + in if joins + then withOperands (concatenation spanValue (typedLeft, leftType) (typedRight, rightType)) + else + if comparesText + then withOperands (textEquality spanValue (operator == Equal) typedLeft typedRight) + else + ( BinaryExpression spanValue operator typedLeft typedRight resultType + , resultType + , leftProblems ++ rightProblems ++ mismatch + ) IsPatternExpression spanValue subject patternValue _ -> let (typedSubject, subjectType, subjectProblems) = checkExpressionWith context environment subject (typedPattern, patternProblems) = checkPatternWith context subjectType patternValue @@ -821,9 +823,24 @@ checkExpressionExpectedWith context environment expected expression = case expre ] ((typedFirst, firstType, firstProblems), (typedSecond, secondType, secondProblems)) = checkOperandPair context environment expected first second - (resultType, resultProblems) = - selectedValueType spanValue "VXT0037" "conditional results must have the same type" firstType secondType - in ( ConditionalExpression spanValue typedCondition typedFirst typedSecond resultType + -- A block that leaves instead of completing has no value, so + -- the other block alone gives the expression its type. + -- When neither completes, the expression never yields a value: + -- it is annotated void, and what receives it is not held to a + -- type, because that place is never reached. + (annotation, resultType, resultProblems) = case (doesNotComplete typedFirst, doesNotComplete typedSecond) of + (True, True) -> (voidType, ErrorType, []) + (True, False) -> (secondType, secondType, []) + (False, True) -> (firstType, firstType, []) + -- A string is selected like a scalar: one of the two is + -- the value, and the other is never made. + (False, False) + | firstType == stringType && secondType == stringType -> (stringType, stringType, []) + | otherwise -> + let (valueType, problems) = + selectedValueType spanValue "VXT0037" "conditional results must have the same type" firstType secondType + in (valueType, valueType, problems) + in ( ConditionalExpression spanValue typedCondition typedFirst typedSecond annotation , resultType , conditionProblems ++ conditionMismatch ++ firstProblems ++ secondProblems ++ resultProblems ) @@ -857,6 +874,16 @@ checkExpressionExpectedWith context environment expected expression = case expre , targetType , problems ++ immutable ++ mismatch ) + AssignmentExpression spanValue (Just Add) name value _ + | target@(Just (targetType, _)) <- lookup (resolvedSymbol name) environment + , targetType == stringType -> + let (typedValue, valueType, problems) = checkExpressionWith context environment value + (joined, _, joinProblems) = + concatenation spanValue (NameExpression spanValue name stringType, stringType) (typedValue, valueType) + in ( AssignmentExpression spanValue Nothing name joined stringType + , stringType + , problems ++ immutableTargetProblems spanValue target ++ joinProblems + ) AssignmentExpression spanValue (Just operator) name value _ -> let target = lookup (resolvedSymbol name) environment targetType = maybe ErrorType fst target @@ -897,8 +924,21 @@ checkExpressionExpectedWith context environment expected expression = case expre -- leaves it carries a value. LoopExpression spanValue loop _ -> let loops = LoopContext (ExpressionLoop expected) [] - (typedLoop, _, _, loopProblems) = checkStatementWith context environment ErrorType loops loop - (resultType, resultProblems) = loopValueType spanValue (loopBreakTypes typedLoop) + (typedLoop, _, _, loopProblems) = checkStatementWith context environment (contextReturn context) loops loop + breakTypes = loopBreakTypes typedLoop + -- A loop that no break leaves never yields a value; it leaves + -- through a return or does not end. That is a fact about its + -- control flow, so it is annotated void and what receives it is + -- not held to a type. + -- A loop that only runs forever is still reported as lacking + -- a value: it is far more likely a forgotten break. + neverYields = + null breakTypes + && not (null (blockReturnTypes (Block [typedLoop]))) + && doesNotComplete (LoopExpression spanValue typedLoop voidType) + (annotation, resultType, resultProblems) + | neverYields = (voidType, ErrorType, []) + | otherwise = let (valueType, problems) = loopValueType spanValue breakTypes in (valueType, valueType, problems) endProblems = [ problem spanValue @@ -906,19 +946,15 @@ checkExpressionExpectedWith context environment expected expression = case expre "a loop used as an expression can end without a value; its condition must be the constant true" | loopMayEndWithoutValue typedLoop ] - returnProblems = - [ problem spanValue "VXT0045" "return inside a loop used as an expression is not supported" - | statementReturns typedLoop - ] - in ( LoopExpression spanValue typedLoop resultType + in ( LoopExpression spanValue typedLoop annotation , resultType - , loopProblems ++ endProblems ++ returnProblems ++ resultProblems + , loopProblems ++ endProblems ++ resultProblems ) BlockExpression spanValue block _ -> - checkValueBlock (branchChecker context ErrorType) environment expected spanValue block + checkValueBlock (branchChecker context (contextReturn context)) environment expected spanValue block MatchExpression spanValue subjects arms _ -> let (typedMatch, resultType, _, problems) = - checkMatch (branchChecker context ErrorType) MatchValue environment expected spanValue subjects arms + checkMatch (branchChecker context (contextReturn context)) MatchValue environment expected spanValue subjects arms in (typedMatch, resultType, problems) CallableExpression spanValue explicit captures parameters body _ -> let checkedCaptures = checkCapturesWith context environment captures @@ -934,6 +970,18 @@ checkExpressionExpectedWith context environment expected expression = case expre callableEnvironment = parameterEnvironment ++ captureEnvironment ++ environment (typedBody, resultType, bodyProblems) = checkCallableBodyWith context callableEnvironment body callableType = FunctionType (map parameterAnnotation typedParameters) resultType + -- The result type is inferred from the returns of the body, + -- which are read through expressions as well; they must agree, + -- or the callable has no one type to return. + returnProblems = case typedBody of + CallableBlockBody block -> + [ problem + spanValue + "VXT0062" + "the return statements of this callable carry values of different types" + | any (not . compatible resultType) (blockReturnTypes block) + ] + CallableExpressionBody _ -> [] captureProblems = concatMap captureDiagnostics checkedCaptures parameterProblems = concatMap (typeSyntaxProblemsIn context . parameterTypeSyntax) parameters in ( CallableExpression @@ -944,28 +992,9 @@ checkExpressionExpectedWith context environment expected expression = case expre typedBody callableType , callableType - , captureProblems ++ parameterProblems ++ bodyProblems + , captureProblems ++ parameterProblems ++ bodyProblems ++ returnProblems ) -{- | Types of the values carried by the breaks that leave this loop itself. -Breaks of nested loops leave those loops, so nested loops are not entered. --} -loopBreakTypes :: Statement ResolvedName Type -> [Type] -loopBreakTypes loop = case loop of - WhileStatement _ _ body -> blockBreaks body - ForStatement _ _ _ _ body -> blockBreaks body - _ -> [] - where - blockBreaks (Block statements) = concatMap statementBreaks statements - statementBreaks statement = case statement of - BreakStatement _ (Just value) -> [typedExpressionType value] - IfStatement _ _ trueBlock falseBlock -> blockBreaks trueBlock ++ maybe [] blockBreaks falseBlock - GuardStatement _ _ block -> blockBreaks block - BlockStatement _ block -> blockBreaks block - ExpressionStatement _ (MatchExpression _ _ arms _) _ -> - concat [blockBreaks block | BlockExpression _ block _ <- map matchArmBody arms] - _ -> [] - {- | Result type of a loop expression from the types of its break values. The value is materialized in one storage slot, like the result of a @@ -981,8 +1010,9 @@ loopValueType spanValue breakTypes = case filter (/= ErrorType) breakTypes of first : remaining | any (/= first) remaining -> (first, [problem spanValue "VXT0043" "the break values of a loop used as an expression must have the same type"]) - | not (booleanContextType first) -> - (first, [problem spanValue "VXT0044" "loop expressions currently support only bool and numeric results"]) + | not (booleanContextType first) + , not (isEnumType first) -> + (first, [problem spanValue "VXT0044" "loop expressions currently support only bool, numeric and enum results"]) | otherwise -> (first, []) -- | Whether the loop can finish by its condition becoming false. @@ -996,24 +1026,6 @@ loopMayEndWithoutValue loop = case loop of LiteralExpression _ (BooleanLiteral True) _ -> True _ -> False --- | Whether a statement contains a return outside any closure. -statementReturns :: Statement name annotation -> Bool -statementReturns statement = case statement of - ReturnStatement {} -> True - IfStatement _ _ trueBlock falseBlock -> blockReturns trueBlock || maybe False blockReturns falseBlock - WhileStatement _ _ body -> blockReturns body - DoWhileStatement _ body _ -> blockReturns body - ForStatement _ initializer _ updates body -> - maybe False statementReturns initializer || any statementReturns updates || blockReturns body - ForEachStatement _ _ _ _ _ _ body -> blockReturns body - GuardStatement _ _ block -> blockReturns block - BlockStatement _ block -> blockReturns block - ExpressionStatement _ (MatchExpression _ _ arms _) _ -> - or [blockReturns block | BlockExpression _ block _ <- map matchArmBody arms] - _ -> False - where - blockReturns (Block statements) = any statementReturns statements - immutableTargetProblems :: SourceSpan -> Maybe (Type, Bool) -> [Diagnostic] immutableTargetProblems spanValue target = case target of Just (_, False) -> [problem spanValue "VXT0003" "cannot assign to an immutable binding"] @@ -1068,9 +1080,10 @@ selectedValueType spanValue mismatchCode mismatchMessage firstType secondType | firstType == ErrorType = (secondType, []) | secondType == ErrorType = (firstType, []) | firstType /= secondType = (firstType, [problem spanValue mismatchCode mismatchMessage]) - | not (booleanContextType firstType) = + | not (booleanContextType firstType) + , not (isEnumType firstType) = ( firstType - , [problem spanValue "VXT0039" "conditional expressions currently support only bool and numeric results"] + , [problem spanValue "VXT0039" "conditional expressions currently support only bool, numeric and enum results"] ) | otherwise = (firstType, []) @@ -1103,6 +1116,37 @@ checkOrdinaryCall context environment spanValue callee arguments = , calleeProblems ++ concatMap (\(_, _, ps) -> ps) checkedArguments ++ callProblems ) +{- | A static method referred to through its type, @Type::Method@: the +method as a callable value, like its bare name inside the type. + +A name with one static method that is accessible is that method. A name +with several is the one whose signature the place expects; where the place +expects none of them, or nothing, the reference does not say which is meant. +-} +methodValue :: + TemplateContext -> Maybe Type -> SourceSpan -> SymbolId -> Identifier -> (Expression ResolvedName Type, Type, [Diagnostic]) +methodValue context expected spanValue owner member = case chosen of + [declaration] -> + let valueType = signature context declaration + in (NameExpression spanValue (declarationName declaration) valueType, valueType, []) + _ -> (NameExpression spanValue (ResolvedName (SymbolId (-1)) member) ErrorType, ErrorType, [failure]) + where + candidates = overloadsFor context owner member + static = [candidate | candidate@(MethodCandidate _ FunctionDeclaration {declarationIsStatic = True}) <- candidates] + visible = map candidateDeclaration (filter (candidateVisibleFrom context) static) + chosen = case visible of + [_] -> visible + _ -> [declaration | declaration <- visible, Just (signature context declaration) == expected] + failure + | null candidates = problem spanValue "VXT0029" "no method with this name is declared on the selected type" + | null static = problem spanValue "VXT0031" "an instance method cannot be referred to through a type name" + | null visible = problem spanValue "VXT0033" "the selected method is not accessible from this declaration" + | otherwise = + problem + spanValue + "VXT0080" + "the method name has several overloads and the place does not expect the type of one of them" + typeQualifiedReceiver :: Expression ResolvedName () -> Maybe ResolvedName typeQualifiedReceiver (NameExpression _ name _) = Just name typeQualifiedReceiver _ = Nothing @@ -1134,6 +1178,9 @@ checkTypeQualifiedCall :: Identifier -> [Expression ResolvedName ()] -> (Expression ResolvedName Type, Type, [Diagnostic]) +checkTypeQualifiedCall context environment _ callSpan receiver member arguments + | consoleReceiver receiver = + checkConsoleCall (checkExpressionExpectedWith context environment) callSpan member arguments checkTypeQualifiedCall context environment expected callSpan receiver member arguments = case typeQualifiedReceiver receiver of Just typeName @@ -1170,6 +1217,7 @@ sourceSpanOf expression = case expression of NameExpression spanValue _ _ -> spanValue LiteralExpression spanValue _ _ -> spanValue MemberAccessExpression spanValue _ _ _ -> spanValue + MethodReferenceExpression spanValue _ _ _ -> spanValue CallExpression spanValue _ _ _ -> spanValue UnaryExpression spanValue _ _ _ -> spanValue BinaryExpression spanValue _ _ _ _ -> spanValue @@ -1374,67 +1422,24 @@ checkCallableBodyWith :: TypeEnvironment -> CallableBody ResolvedName () -> (CallableBody ResolvedName Type, Type, [Diagnostic]) -checkCallableBodyWith context environment body = case body of +checkCallableBodyWith outer environment body = case body of CallableExpressionBody expression -> - let (typed, valueType, problems) = checkExpressionWith context environment expression + -- A callable is a function of its own: nothing in its body returns + -- from, or leaves a loop of, the function that creates it. + let context = outer {contextReturn = ErrorType, contextLoops = outsideLoops} + (typed, valueType, problems) = checkExpressionWith context environment expression in (CallableExpressionBody typed, valueType, problems) CallableBlockBody block -> - let (typed, _, returns, problems) = checkBlockWith context environment ErrorType outsideLoops block - finalType = maybe (inferReturn ErrorType returns) id (finalExpressionType typed) + let (typed, _, _, problems) = checkBlockWith outer environment ErrorType outsideLoops block + finalType = maybe (inferReturn ErrorType (blockReturnTypes typed)) id (finalExpressionType typed) in (CallableBlockBody typed, finalType, problems) +-- An expression without a type was reported already, or never yields a +-- value; a condition of either kind has nothing to check. booleanContextType :: Type -> Bool -booleanContextType = acceptsBooleanContext - -literalTypeInContext :: SourceSpan -> Maybe Type -> Literal -> (Type, [Diagnostic]) -literalTypeInContext spanValue expected literal = case literal of - IntegerLiteral value -> integerLiteralType spanValue expected value - FloatingLiteral _ -> floatingLiteralType expected - CharacterLiteral _ -> (scalarTypeToType CharacterScalar, []) - BooleanLiteral _ -> (boolType, []) - StringLiteral _ -> (stringType, []) - UnitLiteral -> (unitType, []) - -integerLiteralType :: SourceSpan -> Maybe Type -> Integer -> (Type, [Diagnostic]) -integerLiteralType spanValue expected value = - let context = maybe NoNumericContext targetContext expected - rule = integerLiteralRule context value - code = case numericRuleError rule of Just (UntargetedIntegerOutsideInt _) -> "VXT0017"; _ -> "VXT0016" - in (numericRuleType rule, ruleProblems spanValue code rule) - where - targetContext target | target == boolType = BooleanNumericContext - targetContext target = TargetNumericType target - -floatingLiteralType :: Maybe Type -> (Type, [Diagnostic]) -floatingLiteralType expected = - let context = maybe NoNumericContext TargetNumericType expected - rule = floatingLiteralRule context - in (numericRuleType rule, []) - -ruleProblems :: SourceSpan -> String -> NumericRuleResult -> [Diagnostic] -ruleProblems spanValue code rule = case numericRuleError rule of - Nothing -> [] - Just issue -> [problem spanValue code (renderNumericRuleError issue)] - -constantRangeProblems :: SourceSpan -> Type -> Expression ResolvedName Type -> [Diagnostic] -constantRangeProblems spanValue target expression = case evaluateConstantInteger expression of - Left issue -> [problem spanValue "VXT0019" (renderConstantIntegerError issue)] - Right (Just value) -> case typeToScalarType target of - Just scalar - | scalarTypeFamily scalar `elem` [SignedIntegerFamily, UnsignedIntegerFamily] - , not (integerFits scalar value) -> - [ problem - spanValue - "VXT0018" - ("constant expression result " ++ show value ++ " does not fit " ++ scalarTypeName scalar) - ] - _ -> [] - Right Nothing -> [] +booleanContextType valueType = valueType == ErrorType || acceptsBooleanContext valueType safeIndex :: [a] -> Int -> Maybe a safeIndex values index | index < 0 = Nothing | otherwise = case drop index values of value : _ -> Just value; [] -> Nothing - -problem :: SourceSpan -> String -> String -> Diagnostic -problem spanValue code message = Diagnostic TypeCheckerStage Error code (Just spanValue) message diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Branching.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Branching.hs index 06f7c753..c3270e1e 100644 --- a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Branching.hs +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Branching.hs @@ -17,9 +17,12 @@ module Visual.XSharp.TypeChecker.Branching , checkMatch , guardBlockProblems , matchArmsAlwaysAccept + , blockCannotComplete + , doesNotComplete ) where import Visual.XSharp.AST +import Visual.XSharp.Completion import Visual.XSharp.Diagnostic import Visual.XSharp.NumericSemantics @@ -46,11 +49,13 @@ data BranchChecker loops = BranchChecker -- ^ Resolve a type written in source. , branchLiteral :: SourceSpan -> Maybe Type -> Literal -> (Type, [Diagnostic]) -- ^ Type a literal in the context of an expected type. + , branchEnumMember :: Type -> Identifier -> Maybe Integer + -- ^ The value of a member of the enum with the given type. , branchHasEffect :: Expression ResolvedName () -> Bool -- ^ Whether evaluating an expression can do more than produce a value. , branchValueLoops :: loops - {- ^ The loop context inside a block used as a value: @break@ and - @continue@ cannot leave such a block. + {- ^ The loop context inside a block used as a value: the loops around + the expression the block belongs to, seen across the edge of the block. -} } @@ -71,9 +76,21 @@ problem spanValue code message = Diagnostic TypeCheckerStage Error code (Just sp {- | Check a block used as a value. Its value is its final expression, written without a semicolon, and that -expression is typed in the context that receives the value. A @return@ would -leave the function from the middle of an expression; that is not supported, -and neither is leaving the block with @break@ or @continue@. +expression is typed in the context that receives the value. A block may +instead leave: with @return@ it leaves the enclosing function, with @break@ +or @continue@ the enclosing loop, under the rules those statements have +anywhere else. A block that cannot complete normally produces no value and +needs no final expression, and a block whose final expression never +completes produces none either. Whether a block completes is a fact about +its control flow, kept apart from its type: 'doesNotComplete' answers it from +the typed tree, the annotation of such a block is @void@, and the form that +holds it takes its type from the blocks that do complete. A block that can +complete normally must end with its value. + +A @return@ in such a block is checked against the declared return type of +the enclosing function like any other. Where that type is inferred, the +returns of the body are read from its typed tree afterwards, through +expressions as well ("Visual.XSharp.TypeChecker.Returns"). -} checkValueBlock :: BranchChecker loops -> @@ -86,28 +103,33 @@ checkValueBlock checker environment expected spanValue (Block statements) = let (leading, final) = case reverse statements of ExpressionStatement finalSpan value False : before -> (reverse before, Just (finalSpan, value)) _ -> (statements, Nothing) - (typedLeading, inner, returns, leadingProblems) = + (typedLeading, inner, _, leadingProblems) = branchStatements checker environment (branchValueLoops checker) leading - returnProblems = - [ problem spanValue "VXT0047" "return inside a block used as a value is not supported" - | not (null returns) - ] in case final of - Nothing -> - ( BlockExpression spanValue (Block typedLeading) ErrorType - , ErrorType - , leadingProblems - ++ returnProblems - ++ [problem spanValue "VXT0046" "a block used as a value must end with an expression that has no semicolon"] - ) + Nothing + | blockCannotComplete (Block typedLeading) -> + (BlockExpression spanValue (Block typedLeading) voidType, voidType, leadingProblems) + | otherwise -> + ( BlockExpression spanValue (Block typedLeading) ErrorType + , ErrorType + , leadingProblems + ++ [ problem + spanValue + "VXT0046" + "a block used as a value must end with an expression that has no semicolon, or leave on every path" + ] + ) Just (finalSpan, value) -> let (typedValue, valueType, valueProblems) = branchExpression checker inner expected value + -- A final expression that never completes gives the + -- block no value and therefore no type. + blockType = if doesNotComplete typedValue then voidType else valueType in ( BlockExpression spanValue (Block (typedLeading ++ [ExpressionStatement finalSpan typedValue False])) - valueType + blockType , valueType - , leadingProblems ++ returnProblems ++ valueProblems + , leadingProblems ++ valueProblems ) {- | Check a @match@. @@ -135,15 +157,25 @@ checkMatch checker use environment expected spanValue subjects arms = ++ [ problem (expressionSpanOf typed) "VXT0058" - "match subjects currently support only bool and numeric values" + "match subjects currently support only bool, numeric and enum values" | (typed, valueType) <- zip typedSubjects subjectTypes , valueType /= ErrorType , not (acceptsBooleanContext valueType) + , Nothing <- [enumUnderlyingType valueType] ] (typedArms, armTypes, returns, armProblems) = checkArms subjectTypes expected arms - (resultType, resultProblems) = case use of - MatchStatement _ -> (voidType, []) - MatchValue -> matchValueType spanValue armTypes + -- An arm that does not complete produces no value; the arms that + -- complete give the match its type. When none completes, the match + -- never yields a value: it is annotated void, like the statement + -- form it is lowered as, and what receives it is not held to a + -- type, because that place is never reached. + valueTypes = [armType | (arm, armType) <- zip typedArms armTypes, not (doesNotComplete (matchArmBody arm))] + noArmCompletes = null valueTypes && not (null typedArms) + (annotation, resultType, resultProblems) = case use of + MatchStatement _ -> (voidType, voidType, []) + MatchValue + | noArmCompletes -> (voidType, ErrorType, []) + | otherwise -> let (valueType, problems) = matchValueType spanValue valueTypes in (valueType, valueType, problems) coverageProblems = case use of MatchStatement _ -> [] MatchValue -> @@ -153,7 +185,7 @@ checkMatch checker use environment expected spanValue subjects arms = "a match used as an expression must accept every value of its subjects; add a '_' arm" | not (matchArmsAlwaysAccept subjectTypes typedArms) ] - in ( MatchExpression spanValue typedSubjects typedArms resultType + in ( MatchExpression spanValue typedSubjects typedArms annotation , resultType , returns , subjectProblems ++ armProblems ++ unreachableArmProblems typedArms ++ resultProblems ++ coverageProblems @@ -195,8 +227,10 @@ checkArm checker use environment expected subjectTypes (MatchArm spanValue patte checkedPatterns = zipWith (checkMatchPattern checker) (subjectTypes ++ repeat ErrorType) patterns typedPatterns = map fst checkedPatterns patternProblems = concatMap snd checkedPatterns + -- A pattern binding is an ordinary local: it may be assigned, like + -- every binding that is not declared final. armEnvironment = - [ (resolvedSymbol name, (matchPatternAnnotation patternValue, False)) + [ (resolvedSymbol name, (matchPatternAnnotation patternValue, True)) | patternValue <- reverse typedPatterns , Just name <- [matchPatternBinding patternValue] ] @@ -212,10 +246,12 @@ checkArm checker use environment expected subjectTypes (MatchArm spanValue patte ] in (Just typed, problems ++ mismatch) (typedBody, bodyType, returns, bodyProblems) = case (use, body) of - -- The block of a statement arm is an ordinary statement block. + -- The block of a statement arm is an ordinary statement block. A + -- match in its last position was parsed as the value of the + -- block; here nothing takes that value, so it is a statement. (MatchStatement loops, BlockExpression blockSpan (Block statements) _) -> let (typedStatements, _, blockReturns, problems) = - branchStatements checker armEnvironment loops statements + branchStatements checker armEnvironment loops (lastMatchAsStatement statements) in (BlockExpression blockSpan (Block typedStatements) voidType, voidType, blockReturns, problems) (MatchStatement _, expression) -> let (typed, _, problems) = branchExpression checker armEnvironment Nothing expression @@ -233,6 +269,12 @@ checkArm checker use environment expected subjectTypes (MatchArm spanValue patte , arityProblems ++ patternProblems ++ guardProblems ++ bodyProblems ) +lastMatchAsStatement :: [Statement name annotation] -> [Statement name annotation] +lastMatchAsStatement statements = case reverse statements of + ExpressionStatement spanValue value@MatchExpression {} False : before -> + reverse (ExpressionStatement spanValue value True : before) + _ -> statements + {- | Check one pattern against the type of its subject. A literal is typed as the other operand of a comparison with the subject, so @@ -256,10 +298,23 @@ checkMatchPattern checker subjectType patternValue = case patternValue of ( MatchNullPattern spanValue subjectType , [problem spanValue "VXT0055" "a null pattern requires a reference subject, which match does not support yet"] ) - MatchCasePattern spanValue name _ -> - ( MatchCasePattern spanValue name subjectType - , [problem spanValue "VXT0056" "enum case patterns require enum declarations, which are not implemented"] - ) + -- @.Member@ names a member of the subject's enum. It accepts the value + -- of that member, so it is kept as the literal pattern of that value: + -- two members with one value are then one pattern, and the second of + -- them can never be selected. + MatchCasePattern spanValue name _ -> case (enumUnderlyingType subjectType, branchEnumMember checker subjectType name) of + (Just _, Just value) -> (MatchLiteralPattern spanValue (IntegerLiteral value) subjectType, []) + (Just _, Nothing) -> + ( MatchCasePattern spanValue name subjectType + , [problem spanValue "VXT0064" ("the enum of the subject has no member named " ++ identifierText name)] + ) + (Nothing, _) -> + ( MatchCasePattern spanValue name subjectType + , + [ problem spanValue "VXT0056" "an enum case pattern requires a subject of an enum type" + | subjectType /= ErrorType + ] + ) MatchTypePattern spanValue syntax name _ -> let (namedTypeValue, syntaxProblems) = branchType checker syntax relationProblems = @@ -287,55 +342,11 @@ matchValueType spanValue armTypes = case filter (/= ErrorType) armTypes of first : remaining | any (/= first) remaining -> (first, [problem spanValue "VXT0050" "the arms of a match used as an expression must have the same type"]) - | not (acceptsBooleanContext first) -> - (first, [problem spanValue "VXT0051" "match expressions currently support only bool and numeric results"]) + | not (acceptsBooleanContext first) + , Nothing <- enumUnderlyingType first -> + (first, [problem spanValue "VXT0051" "match expressions currently support only bool, numeric and enum results"]) | otherwise -> (first, []) --- | Whether a pattern accepts every value of its subject. -acceptsEveryValue :: MatchPattern name annotation -> Bool -acceptsEveryValue patternValue = case patternValue of - MatchWildcardPattern {} -> True - MatchTypePattern {} -> True - _ -> False - -{- | Whether some arm is certain to accept, whatever the subjects are. - -That is the case when an arm without a guard has only patterns that accept -every value. It is also the case when every subject is a @bool@ and the arms -without guards accept each combination of @true@ and @false@ between them; -the combinations are enumerated, which is bounded by 'maximumBoolSubjects'. -Nothing else is recognized: a guard may be false, and the literals of a wider -type are never listed in full. --} -matchArmsAlwaysAccept :: [Type] -> [MatchArm name annotation] -> Bool -matchArmsAlwaysAccept subjectTypes arms = any catchAll unguarded || coversBooleans - where - unguarded = filter isUnguarded arms - catchAll arm = all acceptsEveryValue (matchArmPatterns arm) - coversBooleans = - not (null subjectTypes) - && length subjectTypes <= maximumBoolSubjects - && all (== boolType) subjectTypes - && all (\values -> any (`acceptsBooleans` values) unguarded) (combinations (length subjectTypes)) - combinations :: Int -> [[Bool]] - combinations count = sequence (replicate count [True, False]) - -{- | The most @bool@ subjects whose combinations are enumerated to decide -whether a match accepts every value. A match with more is complete only -through a catch-all arm; the bound keeps the check linear in practice. --} -maximumBoolSubjects :: Int -maximumBoolSubjects = 8 - --- | Whether an arm's patterns accept the given values of @bool@ subjects. -acceptsBooleans :: MatchArm name annotation -> [Bool] -> Bool -acceptsBooleans arm values = - length (matchArmPatterns arm) == length values && and (zipWith accepts (matchArmPatterns arm) values) - where - accepts patternValue value = case patternValue of - MatchLiteralPattern _ (BooleanLiteral literal) _ -> literal == value - _ -> acceptsEveryValue patternValue - {- | Arms that can never be selected because an earlier arm without a guard accepts everything they accept: pattern by pattern, the earlier one accepts every value or names the same literal. @@ -360,42 +371,27 @@ unreachableArmProblems = go [] (MatchLiteralPattern _ left _, MatchLiteralPattern _ right _) -> left == right _ -> acceptsEveryValue first -isUnguarded :: MatchArm name annotation -> Bool -isUnguarded arm = case matchArmGuard arm of - Nothing -> True - Just _ -> False - {- | Problems of the else block of a @guard@. The statements after a guard rely on its condition, so the block must not -complete normally: its last statement returns, leaves a loop, continues one, -or is an @if@ whose two branches both do, or a nested block that does. +complete normally on any path. That is decided from its control flow by +'blockCannotComplete', not from the spelling of its last statement. -} -guardBlockProblems :: SourceSpan -> Block name annotation -> [Diagnostic] +guardBlockProblems :: SourceSpan -> Block name Type -> [Diagnostic] guardBlockProblems spanValue block = [ problem spanValue "VXT0061" - "the else block of a guard must end by leaving the enclosing scope with return, break, or continue" - | not (blockLeaves block) + "the else block of a guard can complete normally; every path through it must leave the enclosing scope" + | not (blockCannotComplete block) ] - where - blockLeaves (Block statements) = case reverse statements of - final : _ -> statementLeaves final - [] -> False - statementLeaves statement = case statement of - ReturnStatement {} -> True - BreakStatement {} -> True - ContinueStatement {} -> True - IfStatement _ _ whenTrue (Just whenFalse) -> blockLeaves whenTrue && blockLeaves whenFalse - BlockStatement _ nested -> blockLeaves nested - _ -> False expressionSpanOf :: Expression name annotation -> SourceSpan expressionSpanOf expression = case expression of NameExpression spanValue _ _ -> spanValue LiteralExpression spanValue _ _ -> spanValue MemberAccessExpression spanValue _ _ _ -> spanValue + MethodReferenceExpression spanValue _ _ _ -> spanValue CallExpression spanValue _ _ _ -> spanValue UnaryExpression spanValue _ _ _ -> spanValue BinaryExpression spanValue _ _ _ _ -> spanValue diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Console.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Console.hs new file mode 100644 index 00000000..dddf1034 --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Console.hs @@ -0,0 +1,357 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Text and the console in the type checker. + +@System.Console@ is not declared in any source file. Its methods are known +to the compiler, and a call of one is checked here and rewritten into calls +of the runtime: the typed tree that leaves the type checker has no +@Console.Printf@ in it, only the conversions of its format, joined, and one +write. The stages after the type checker therefore need to know nothing of +formats, and a format that is wrong never reaches them. + +The same holds for the two operations of the language on strings that need +the runtime: @+@, which joins two strings and writes a value that is not a +string as text first, and @==@ and @\\=@, which compare the characters of +two strings and not the places they are kept. + +A runtime call is written in the typed tree as a call of a name whose +symbol is reserved for one function of the runtime catalog; see +'runtimeName'. No source name has such a symbol. +-} +module Visual.XSharp.TypeChecker.Console + ( CheckExpression + , consoleReceiver + , checkConsoleCall + , concatenation + , textEquality + , runtimeCall + ) where + +import Visual.XSharp.AST +import Visual.XSharp.Diagnostic +import Visual.XSharp.RuntimeCall +import Visual.XSharp.TypeChecker.Format +import Visual.XSharp.TypeChecker.Returns (typedExpressionType) + +{- | The type checker for an expression, given the type its place expects. +This module is called by the checker and calls it back for the arguments. +-} +type CheckExpression = + Maybe Type -> Expression ResolvedName () -> (Expression ResolvedName Type, Type, [Diagnostic]) + +type Checked = (Expression ResolvedName Type, Type, [Diagnostic]) + +{- | Whether the receiver of a member call is the console: @Console@ or +@System.Console@, where neither name has been declared by the program. A +declaration of the program with one of those names is found by the renamer +first and has an ordinary symbol. +-} +consoleReceiver :: Expression ResolvedName () -> Bool +consoleReceiver receiver = case receiver of + NameExpression _ name _ -> resolvedSymbol name == builtinConsoleSymbol + MemberAccessExpression _ (NameExpression _ name _) member _ -> + resolvedSymbol name == builtinSystemSymbol && identifierText member == "Console" + _ -> False + +-- | A call of a runtime function, with the type its row of the catalog gives it. +runtimeCall :: SourceSpan -> RuntimeFunction -> [Expression ResolvedName Type] -> Expression ResolvedName Type +runtimeCall spanValue function arguments = + CallExpression + spanValue + (NameExpression spanValue (runtimeName function) (FunctionType (map typedExpressionType arguments) result)) + arguments + result + where + result = case runtimeResult function of + NoResult -> voidType + TextResult -> stringType + TruthResult -> boolType + +integer :: SourceSpan -> Integer -> Expression ResolvedName Type +integer spanValue value = LiteralExpression spanValue (IntegerLiteral value) intType + +text :: SourceSpan -> String -> Expression ResolvedName Type +text spanValue value = LiteralExpression spanValue (StringLiteral value) stringType + +problem :: SourceSpan -> String -> String -> Diagnostic +problem spanValue code message = Diagnostic TypeCheckerStage Error code (Just spanValue) message + +-- | What a failed check leaves in the tree; the diagnostics stop compilation. +failed :: SourceSpan -> [Diagnostic] -> Checked +failed spanValue problems = (LiteralExpression spanValue UnitLiteral ErrorType, ErrorType, problems) + +typeName :: Type -> String +typeName valueType = case valueType of + NamedType (QualifiedName parts) [] | not (null parts) -> identifierText (last parts) + FunctionType _ _ -> "callable" + _ -> "value of this type" + +-- | The scalar types the runtime does not take yet, with what they wait for. +pendingScalar :: Type -> Maybe String +pendingScalar valueType = case valueType of + NamedType (QualifiedName [Identifier name]) [] + | name `elem` ["longint", "ulongint"] -> Just "128-bit integers are not written as text yet" + | name == "double" -> Just "128-bit floating-point numbers are not written as text yet" + _ -> Nothing + +isFloating :: Type -> Bool +isFloating = runtimeAccepts FloatingParameter + +{- | A value as text, the way @Console.Print@ and @+@ write it: a string is +itself, an integer is written in decimal, a Boolean as @true@ or @false@ and +a character as itself. A floating-point number has no form of its own here: +how many digits to write is a decision @%f@ makes with its precision. +-} +asText :: SourceSpan -> Expression ResolvedName Type -> Type -> Either [Diagnostic] (Expression ResolvedName Type) +asText spanValue value valueType + | valueType == ErrorType = Left [] + | valueType == stringType = Right value + | runtimeAccepts SignedParameter valueType = Right (runtimeCall spanValue TextFromSigned [value]) + | runtimeAccepts UnsignedParameter valueType = Right (runtimeCall spanValue TextFromUnsigned [value]) + | runtimeAccepts BoolParameter valueType = Right (runtimeCall spanValue TextFromBool [value]) + | runtimeAccepts CharParameter valueType = Right (runtimeCall spanValue TextFromChar [value]) + | Just reason <- pendingScalar valueType = Left [problem spanValue "VXT0079" reason] + | isFloating valueType = + Left + [ problem + spanValue + "VXT0079" + "a floating-point number has no text form of its own yet; write it with Console.Printf or Console.Format and %f" + ] + | otherwise = + Left [problem spanValue "VXT0073" ("a " ++ typeName valueType ++ " cannot be written as text")] + +{- | @left + right@ where at least one side is a string: the two as text, +joined. The operands are already checked. +-} +concatenation :: SourceSpan -> (Expression ResolvedName Type, Type) -> (Expression ResolvedName Type, Type) -> Checked +concatenation spanValue (left, leftType) (right, rightType) = + case (asText (expressionSourceSpan left) left leftType, asText (expressionSourceSpan right) right rightType) of + (Right first, Right second) -> (runtimeCall spanValue TextConcat [first, second], stringType, []) + (first, second) -> failed spanValue (either id (const []) first ++ either id (const []) second) + +{- | @left == right@ or @left \\= right@ on two strings: whether they hold +the same characters. +-} +textEquality :: SourceSpan -> Bool -> Expression ResolvedName Type -> Expression ResolvedName Type -> Checked +textEquality spanValue equal left right = + ( if equal then same else UnaryExpression spanValue LogicalNot same boolType + , boolType + , [] + ) + where + same = runtimeCall spanValue TextEquals [left, right] + +{- | Check a call of a method of the console and rewrite it into runtime +calls. +-} +checkConsoleCall :: CheckExpression -> SourceSpan -> Identifier -> [Expression ResolvedName ()] -> Checked +checkConsoleCall check callSpan member arguments = case identifierText member of + "Print" -> write consoleOutput + "Println" -> write consoleOutputLine + "Error" -> write consoleError + "Errorln" -> write consoleErrorLine + "Printf" -> formatted (Just consoleOutput) + "Printfn" -> formatted (Just consoleOutputLine) + "Errorf" -> formatted (Just consoleError) + "Errorfn" -> formatted (Just consoleErrorLine) + "Format" -> formatted Nothing + name + | name `elem` ["Stdin", "Stdout", "Stderr"] -> + failed + callSpan + ( argumentProblems + ++ [problem callSpan "VXT0079" ("Console." ++ name ++ " needs stream objects and is not implemented yet")] + ) + | otherwise -> + failed callSpan (argumentProblems ++ [problem callSpan "VXT0071" ("Console has no method named " ++ name)]) + where + method = "Console." ++ identifierText member + argumentProblems = concat [problems | (_, _, problems) <- map (check Nothing) arguments] + + -- One value, written as text. + write target = case arguments of + [argument] -> + let (value, valueType, problems) = check Nothing argument + in case asText (expressionSourceSpan value) value valueType of + Right written -> + (runtimeCall callSpan ConsoleWrite [written, integer callSpan target], voidType, problems) + Left more -> failed callSpan (problems ++ more) + _ -> + failed + callSpan + ( argumentProblems + ++ [problem callSpan "VXT0072" (method ++ " takes one argument; " ++ given (length arguments))] + ) + + -- A format and its arguments: written when there is a target, and + -- the string itself when there is none. + formatted target = case arguments of + [] -> failed callSpan [problem callSpan "VXT0072" (method ++ " takes a format; " ++ given 0)] + formatArgument : values -> case formatArgument of + LiteralExpression formatSpan (StringLiteral format) _ -> case parseFormat format of + Left wrong -> + failed + callSpan + ( concat [problems | (_, _, problems) <- map (check Nothing) values] + ++ [ problem + formatSpan + (formatProblemCode wrong) + ( formatProblemMessage wrong + ++ " (character " + ++ show (formatProblemOffset wrong) + ++ " of the format)" + ) + ] + ) + Right parsed -> + let needed = sum [conversionArgumentCount value | ConversionPiece value <- parsed] + in if needed /= length values + then + failed + callSpan + ( concat [problems | (_, _, problems) <- map (check Nothing) values] + ++ [ problem + callSpan + "VXT0077" + ( "the format of " + ++ method + ++ " takes " + ++ counted needed + ++ " after it; " + ++ given (length values) + ) + ] + ) + else finish target (convert formatSpan parsed values) + _ -> + failed + callSpan + ( argumentProblems + ++ [ problem + (expressionSourceSpan formatArgument) + "VXT0074" + ("the format of " ++ method ++ " must be a string literal, so that it can be checked when the program is compiled") + ] + ) + + finish target (pieces, problems) + | not (null problems) = failed callSpan problems + | otherwise = + let joined = case pieces of + [] -> text callSpan "" + first : rest -> foldl (\whole part -> runtimeCall callSpan TextConcat [whole, part]) first rest + in case target of + Just stream -> (runtimeCall callSpan ConsoleWrite [joined, integer callSpan stream], voidType, []) + Nothing -> (joined, stringType, []) + + -- Each piece of the format as a string expression, taking the + -- arguments in the order the conversions name them. + convert _ [] _ = ([], []) + convert formatSpan (piece : later) values = case piece of + LiteralPiece literal -> prepend (text formatSpan literal, []) (convert formatSpan later values) + NewlinePiece -> prepend (runtimeCall formatSpan TextNewline [], []) (convert formatSpan later values) + ConversionPiece value -> + let (width, afterWidth, widthProblems) = size (conversionWidth value) values + (precision, afterPrecision, precisionProblems) = size (conversionPrecision value) afterWidth + in case afterPrecision of + argument : remaining -> + let (converted, valueProblems) = converting value width precision argument + in prepend + (converted, widthProblems ++ precisionProblems ++ valueProblems) + (convert formatSpan later remaining) + -- The count was checked before; nothing is missing. + [] -> ([], widthProblems ++ precisionProblems) + prepend (piece, problems) (pieces, more) = (piece : pieces, problems ++ more) + + -- A width or a precision: absent, written in the format, or the + -- next argument, which is an int. + size written values = case written of + NoSize -> (integer callSpan absent, values, []) + FixedSize amount -> (integer callSpan amount, values, []) + ArgumentSize -> case values of + argument : remaining -> + let (value, valueType, problems) = check (Just intType) argument + mismatch = + [ problem + (expressionSourceSpan value) + "VXT0078" + ("a width or a precision written as * takes an int, and the argument is a " ++ typeName valueType) + | valueType /= ErrorType + , valueType /= intType + ] + in (value, remaining, problems ++ mismatch) + [] -> (integer callSpan absent, values, []) + + converting value width precision argument = + let kind = conversionKind value + (typed, valueType, problems) = check (expectedFor kind) argument + spanValue = expressionSourceSpan typed + flags = integer spanValue (conversionFlagBits value) + call function operand = runtimeCall spanValue function [flags, width, precision, operand] + plain = null (conversionFlags value) && conversionWidth value == NoSize && conversionPrecision value == NoSize + mismatch wanted = + ( typed + , problems + ++ [ problem + spanValue + "VXT0078" + ( "the conversion %" + ++ [conversionLetter kind] + ++ " takes " + ++ wanted + ++ ", and the argument is a " + ++ typeName valueType + ) + | valueType /= ErrorType + ] + ) + accepted + | valueType == ErrorType = (typed, problems) + | Just reason <- pendingScalar valueType = (typed, problems ++ [problem spanValue "VXT0079" reason]) + | otherwise = case kind of + SignedDecimal + | runtimeAccepts SignedParameter valueType -> (call TextFormatSigned typed, problems) + | otherwise -> mismatch "a signed integer" + UnsignedDecimal + | runtimeAccepts UnsignedParameter valueType -> (call TextFormatUnsigned typed, problems) + | otherwise -> mismatch "an unsigned integer" + Hexadecimal + | runtimeAccepts SignedParameter valueType -> (call TextFormatSigned typed, problems) + | runtimeAccepts UnsignedParameter valueType -> (call TextFormatUnsigned typed, problems) + | otherwise -> mismatch "an integer" + FixedPoint + | isFloating valueType -> (call TextFormatFloating typed, problems) + | otherwise -> mismatch "a floating-point number" + Text + | valueType /= stringType -> mismatch "a String" + | plain -> (typed, problems) + | otherwise -> (call TextFormatString typed, problems) + Character + | runtimeAccepts CharParameter valueType -> (call TextFormatChar typed, problems) + | otherwise -> mismatch "a char" + Truth + | not (runtimeAccepts BoolParameter valueType) -> mismatch "a bool" + | plain -> (runtimeCall spanValue TextFromBool [typed], problems) + | otherwise -> (call TextFormatString (runtimeCall spanValue TextFromBool [typed]), problems) + in accepted + + -- A literal has the type its conversion takes; a value that already + -- has a type keeps it and is then held to the conversion. + expectedFor kind = case kind of + UnsignedDecimal -> Just (namedType "uint") + FixedPoint -> Just (namedType "float") + _ -> Nothing + + counted :: Int -> String + counted amount = case amount of + 0 -> "no argument" + 1 -> "one argument" + _ -> show amount ++ " arguments" + + given :: Int -> String + given amount = case amount of + 0 -> "none was given" + 1 -> "one was given" + _ -> show amount ++ " were given" diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Context.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Context.hs new file mode 100644 index 00000000..faa31939 --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Context.hs @@ -0,0 +1,255 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | What the type checker carries from place to place, and how a type +written in source becomes a type. + +The catalog is read from the whole source set before any body is checked, +so that a use may precede its declaration. The template context is the +catalog seen from one declaration: its template parameters, the type that +owns it, the type a @return@ must carry and the loops around the place +being checked. A type written in source is resolved in that context, +because the same spelling names a template parameter in one declaration and +a declared type in another. +-} +module Visual.XSharp.TypeChecker.Context + ( TypeEnvironment + , MethodCandidate (..) + , TypeCatalog (..) + , TemplateContext (..) + , emptyTemplateContext + , templateContext + , templateParameterAsArgument + , syntaxTypeIn + , syntaxTemplateArgumentIn + , syntaxTemplateValueIn + , typeTemplateParameter + , validateTemplateParameters + , typeSyntaxProblemsIn + , templateArgumentProblemsIn + , templateValueProblemsIn + ) where + +import Visual.XSharp.AST +import Visual.XSharp.BuiltinTypes +import Visual.XSharp.Diagnostic +import Visual.XSharp.TemplateValue +import Visual.XSharp.TypeChecker.Enums +import Visual.XSharp.TypeChecker.Literals +import Visual.XSharp.TypeChecker.Loops + +-- | The names in scope: each symbol with its type and whether it may be assigned. +type TypeEnvironment = [(SymbolId, (Type, Bool))] + +{- | A method that a call may select, with the type that declares it. + +The type checker owns a whole source-set catalog before it checks any body. +That permits calls to later-declared classes without making parsing depend +on declaration order or mutating the Renamer's lexical environment. +-} +data MethodCandidate = MethodCandidate + { candidateOwner :: SymbolId + , candidateDeclaration :: Declaration ResolvedName () + } + +-- | What the checker knows of a whole source set before it checks a body. +data TypeCatalog = TypeCatalog + { catalogTypes :: [(SymbolId, ResolvedName)] + , catalogMethods :: [MethodCandidate] + , catalogEnums :: [EnumInfo] + -- ^ The classic enums of the source set. + , catalogInferredReturns :: [((SymbolId, SourceSpan), Type)] + {- ^ The return types inferred for the methods declared with @auto@, by + the symbol and the place of the declaration. A method that is absent has + no return type that is known yet. + -} + } + +{- | The catalog as one declaration sees it. + +Type syntax deliberately keeps source spellings. This side environment is +the bridge from those spellings to the SymbolIds assigned by the renamer. +Type and value parameters are separate because @T@ in a type position and +@N@ in @[T; N]@ have different semantic representations. +-} +data TemplateContext = TemplateContext + { templateTypeNames :: [(Identifier, ResolvedName)] + , templateValueNames :: [(Identifier, ResolvedName)] + , templateCatalog :: TypeCatalog + , templateCurrentType :: Maybe SymbolId + , contextReturn :: Type + {- ^ The type a @return@ must carry at the place being checked, or the + error type where it is not declared. Statements know it already; it is + kept here for the statements of a block used as a value, which are + reached through an expression. + -} + , contextLoops :: LoopContext + -- ^ The loops around the place being checked, for the same reason. + } + +-- | The context of a declaration that has no template parameters. +emptyTemplateContext :: TypeCatalog -> Maybe SymbolId -> TemplateContext +emptyTemplateContext catalog owner = TemplateContext [] [] catalog owner ErrorType outsideLoops + +{- | Keep type and value parameters in separate lookup tables. +Identical source spelling in the two categories must not collapse their roles. +-} +templateContext :: TypeCatalog -> Maybe SymbolId -> [TemplateParameter ResolvedName annotation] -> TemplateContext +templateContext catalog owner parameters = + TemplateContext + [ (resolvedSpelling name, name) + | parameter <- parameters + , case templateParameterKind parameter of + TemplateTypeParameter -> True + TemplateTemplateParameter _ -> True + _ -> False + , let name = templateParameterName parameter + ] + [ (resolvedSpelling name, name) + | parameter <- parameters + , case templateParameterKind parameter of TemplateValueParameterKind _ -> True; _ -> False + , let name = templateParameterName parameter + ] + catalog + owner + ErrorType + outsideLoops + +-- | A template parameter as the argument that stands for itself. +templateParameterAsArgument :: TemplateParameter ResolvedName annotation -> TemplateArgument +templateParameterAsArgument parameter = case templateParameterKind parameter of + TemplateValueParameterKind _ -> ValueTemplateArgument (TemplateValueParameter (templateParameterName parameter)) + _ -> TypeTemplateArgument (TypeVariable (templateParameterName parameter)) + +-- | The type a type written in source denotes in a context. +syntaxTypeIn :: TemplateContext -> TypeSyntax -> Type +syntaxTypeIn _ AutoType = ErrorType +syntaxTypeIn context (QualifiedTypeSyntax name arguments) = + case name of + QualifiedName [identifier] | Just resolved <- lookup identifier (templateTypeNames context) -> TypeVariable resolved + _ -> NamedType name (map (syntaxTemplateArgumentIn context) arguments) +syntaxTypeIn context (BuiltinArrayTypeSyntax element) = + -- `[]T` is a language type, not a public class invented by the compiler. + -- Its structural spelling keeps that distinction visible through Core + -- until ownership-aware lowering assigns the final runtime layout. + NamedType (QualifiedName [Identifier "[]"]) [TypeTemplateArgument (syntaxTypeIn context element)] +syntaxTypeIn context (ArrayTypeSyntax element) = + NamedType + (QualifiedName [Identifier "System", Identifier "Array"]) + [TypeTemplateArgument (syntaxTypeIn context element)] +syntaxTypeIn context (FixedArrayTypeSyntax element size) = + NamedType + (QualifiedName [Identifier "System", Identifier "Array"]) + [TypeTemplateArgument (syntaxTypeIn context element), ValueTemplateArgument (syntaxTemplateValueIn context size)] +syntaxTypeIn context (DictionaryTypeSyntax key value) = + NamedType + (QualifiedName [Identifier "System", Identifier "Dictionary"]) + [TypeTemplateArgument (syntaxTypeIn context key), TypeTemplateArgument (syntaxTypeIn context value)] +syntaxTypeIn context (CallableTypeSyntax parameters result) = FunctionType (map (syntaxTypeIn context) parameters) (syntaxTypeIn context result) +syntaxTypeIn context (ExplicitType identifier@(Identifier name)) = case lookup identifier (templateTypeNames context) of + Just resolved -> TypeVariable resolved + Nothing -> case name of + "String" -> stringType + "unit" -> unitType + "void" -> voidType + _ + | Just info <- enumBySpelling (catalogEnums (templateCatalog context)) identifier -> enumInfoType info + | otherwise -> maybe (NamedType (QualifiedName [Identifier name]) []) scalarTypeToType (lookupScalar name) + where + lookupScalar spelling = lookup spelling [(scalarTypeName scalar, scalar) | scalar <- scalarTypes] + +-- | The template argument a written argument denotes in a context. +syntaxTemplateArgumentIn :: TemplateContext -> TemplateArgumentSyntax -> TemplateArgument +syntaxTemplateArgumentIn context argument = case argument of + TemplateTypeSyntax valueType -> TypeTemplateArgument (syntaxTypeIn context valueType) + TemplateValueArgumentSyntax value -> ValueTemplateArgument (syntaxTemplateValueIn context value) + +-- Parser construction guarantees the expression tree is side-effect free. +-- Exact evaluation here gives every concrete specialization one canonical +-- identity. Invalid arithmetic becomes a sentinel and is diagnosed by the +-- type-syntax validation pass before Core can be emitted. + +{- | Canonicalize a template argument before it contributes to specialization identity. +The fallback value is only a recovery sentinel; validation reports the original +invalid expression before a specialization plan is emitted. +-} +syntaxTemplateValueIn :: TemplateContext -> TemplateValueSyntax -> TemplateValue +syntaxTemplateValueIn context value = case value of + TemplateNameSyntax _ (QualifiedName [identifier]) + | Just resolved <- lookup identifier (templateValueNames context) -> TemplateValueParameter resolved + _ -> case evaluateTemplateValue value of + Right result -> result + _ -> IntegerTemplateValue 0 + +-- | A template parameter with the type of its values. +typeTemplateParameter :: TemplateContext -> TemplateParameter ResolvedName () -> TemplateParameter ResolvedName Type +typeTemplateParameter context parameter = + TemplateParameter + (templateParameterSpan parameter) + (templateParameterName parameter) + annotation + (templateParameterKind parameter) + (templateParameterIsPack parameter) + (templateParameterDefault parameter) + where + annotation = case templateParameterKind parameter of + TemplateValueParameterKind valueType -> syntaxTypeIn context valueType + _ -> TypeVariable (templateParameterName parameter) + +-- | The problems of the template parameters of a declaration. +validateTemplateParameters :: TemplateContext -> [TemplateParameter ResolvedName ()] -> [Diagnostic] +validateTemplateParameters context = concatMap validate + where + validate parameter = + kindProblems parameter + ++ defaultProblems parameter + ++ packDefaultProblems parameter + kindProblems parameter = case templateParameterKind parameter of + TemplateTypeParameter -> [] + TemplateValueParameterKind valueType -> typeSyntaxProblemsIn context valueType + TemplateTemplateParameter shapes -> concatMap shapeProblems shapes + shapeProblems shape = case templateParameterShapeKind shape of + TemplateTypeParameterShape -> [] + TemplateValueParameterShape valueType -> typeSyntaxProblemsIn context valueType + TemplateTemplateParameterShape shapes -> concatMap shapeProblems shapes + defaultProblems parameter = case templateParameterDefault parameter of + Nothing -> [] + Just (TemplateTypeDefault valueType) -> typeSyntaxProblemsIn context valueType + Just (TemplateValueDefault value) -> templateValueProblemsIn context "VXT0020" value + packDefaultProblems parameter + | templateParameterIsPack parameter + , Just _ <- templateParameterDefault parameter = + [problem (templateParameterSpan parameter) "VXT0019" "a template parameter pack cannot have a default"] + | otherwise = [] + +-- | The problems of a type written in source. +typeSyntaxProblemsIn :: TemplateContext -> TypeSyntax -> [Diagnostic] +typeSyntaxProblemsIn context syntax = case syntax of + ExplicitType _ -> [] + AutoType -> [] + BuiltinArrayTypeSyntax element -> typeSyntaxProblemsIn context element + ArrayTypeSyntax element -> typeSyntaxProblemsIn context element + DictionaryTypeSyntax key value -> typeSyntaxProblemsIn context key ++ typeSyntaxProblemsIn context value + CallableTypeSyntax parameters result -> concatMap (typeSyntaxProblemsIn context) parameters ++ typeSyntaxProblemsIn context result + QualifiedTypeSyntax _ arguments -> concatMap (templateArgumentProblemsIn context) arguments + FixedArrayTypeSyntax element size -> + typeSyntaxProblemsIn context element + ++ case syntaxTemplateValueIn context size of + TemplateValueParameter _ -> [] + _ -> case evaluateFixedArraySize size of + Left issue -> [problem (templateValueSyntaxSpan size) "VXT0016" (renderTemplateValueError issue)] + Right _ -> [] + +-- | The problems of a template argument written in source. +templateArgumentProblemsIn :: TemplateContext -> TemplateArgumentSyntax -> [Diagnostic] +templateArgumentProblemsIn context argument = case argument of + TemplateTypeSyntax valueType -> typeSyntaxProblemsIn context valueType + TemplateValueArgumentSyntax value -> templateValueProblemsIn context "VXT0017" value + +-- | The problems of a template value written in source, reported with the given code. +templateValueProblemsIn :: TemplateContext -> String -> TemplateValueSyntax -> [Diagnostic] +templateValueProblemsIn context code value = case syntaxTemplateValueIn context value of + TemplateValueParameter _ -> [] + _ -> case evaluateTemplateValue value of + Left issue -> [problem (templateValueSyntaxSpan value) code (renderTemplateValueError issue)] + Right _ -> [] diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Enums.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Enums.hs new file mode 100644 index 00000000..f2cb979d --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Enums.hs @@ -0,0 +1,204 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Classic enums: their members, their values and their rules. + +A classic enum is a value type whose members are named integers of one +underlying type. Members are numbered from zero, a member without a written +value takes the value after that of the member before it, and two members +may have the same value. A written value is a constant integer expression: +integer literals and earlier members of the same enum, joined by the +arithmetic, shift and bitwise operators. Inside its own declaration the name +of a member stands for its number; that is where the numbers behind the names +are defined, and it is the only place where a member is a number. An enum is +a type of its own: its values are +compared with @==@ and @\\=@ and with nothing else, and there is no +conversion between an enum and an integer in either direction. + +The checker reads the enums of a source set into a table before it checks +any body, so that a use may precede the declaration. +-} +module Visual.XSharp.TypeChecker.Enums + ( EnumInfo (..) + , enumInfos + , enumBySymbol + , enumBySpelling + , enumMemberValue + , enumDeclarationProblems + , isEnumType + ) where + +import Data.List (nub, sort) +import Data.Map.Strict qualified as Map +import Visual.XSharp.AST +import Visual.XSharp.BuiltinTypes +import Visual.XSharp.ConstantEvaluation +import Visual.XSharp.Diagnostic +import Visual.XSharp.TypeChecker.Literals (problem) + +-- | What the checker knows of one enum. +data EnumInfo = EnumInfo + { enumInfoName :: ResolvedName + -- ^ The declared name and its symbol. + , enumInfoType :: Type + -- ^ The type of its values, as 'enumType' encodes it. + , enumInfoMembers :: [(Identifier, Integer)] + -- ^ Its members in source order, each with its value. + } + +-- | The enums among the declarations of a source set. +enumInfos :: [Declaration ResolvedName ()] -> [EnumInfo] +enumInfos declarations = + [ EnumInfo name (enumType (resolvedSpelling name) valueType (sort (nub (map snd members)))) members + | EnumDeclaration _ name _ underlying cases <- declarations + , let valueType = underlyingType underlying + , let members = [(enumCaseName member, value) | (member, value, _) <- numbered valueType cases] + ] + +{- | Each member with its value: the written one, or one more than the last, +and what is wrong with the written one when it has no value. A member whose +written value has none is numbered as if nothing had been written, so that +one mistake is reported once and the members after it keep their numbers. +-} +numbered :: Type -> [EnumCase] -> [(EnumCase, Integer, Maybe String)] +numbered valueType = go 0 Map.empty + where + go _ _ [] = [] + go next earlier (member : remaining) = + let (value, issue) = case enumCaseValue member of + Nothing -> (next, Nothing) + Just written -> case memberConstant valueType earlier written of + Right computed -> (computed, Nothing) + Left reason -> (next, Just reason) + in (member, value, issue) : go (value + 1) (Map.insert (enumCaseName member) value earlier) remaining + +{- | The value of a constant integer expression written for a member, given +the values of the members before it. + +The expression is given the underlying type of the enum and the earlier +members their numbers, and the constant evaluator of the language computes +it: the same arithmetic, the same rounding of @//@, the same complement of +@!@ for the width and signedness of the type, and the same refusal of a +division by zero as anywhere else in a program. +-} +memberConstant :: Type -> Map.Map Identifier Integer -> Expression Identifier () -> Either String Integer +memberConstant valueType earlier written = do + typed <- typedConstant written + case evaluateConstantInteger typed of + Right (Just value) -> Right value + Right Nothing -> Left notConstant + Left issue -> Left (renderConstantIntegerError issue) + where + typedConstant :: Expression Identifier () -> Either String (Expression Identifier Type) + typedConstant expression = case expression of + LiteralExpression spanValue (IntegerLiteral value) _ -> + Right (LiteralExpression spanValue (IntegerLiteral value) valueType) + NameExpression spanValue name _ -> case Map.lookup name earlier of + Just value -> Right (LiteralExpression spanValue (IntegerLiteral value) valueType) + Nothing -> Left (identifierText name ++ " is not an earlier member of this enum") + UnaryExpression spanValue operator value _ + | operator `elem` [UnaryPlus, UnaryNegate, BitwiseNot] -> + (\operand -> UnaryExpression spanValue operator operand valueType) <$> typedConstant value + BinaryExpression spanValue operator left right _ + | operator `elem` constantOperators -> + (\first second -> BinaryExpression spanValue operator first second valueType) + <$> typedConstant left + <*> typedConstant right + _ -> Left notConstant + constantOperators = + [ Add + , Subtract + , Multiply + , Divide + , FloorDivide + , Remainder + , Power + , ShiftLeft + , ShiftRight + , BitwiseAnd + , BitwiseXor + , BitwiseOr + ] + notConstant = + "the value of an enum member is a constant integer expression: integer literals and earlier members of " + ++ "the enum, joined by arithmetic, shift and bitwise operators" + +{- | The underlying type an enum declares, @int@ when it declares none. A +type that is not an integer type is reported by 'enumDeclarationProblems'; +the enum is then checked as if it had declared none. +-} +underlyingType :: Maybe TypeSyntax -> Type +underlyingType syntax = maybe intType scalarTypeToType (syntax >>= integerScalar) + +integerScalar :: TypeSyntax -> Maybe ScalarType +integerScalar syntax = case syntax of + ExplicitType (Identifier spelling) -> + case [scalar | scalar <- scalarTypes, scalarTypeName scalar == spelling, isInteger scalar] of + scalar : _ -> Just scalar + [] -> Nothing + _ -> Nothing + where + isInteger scalar = scalarTypeFamily scalar `elem` [SignedIntegerFamily, UnsignedIntegerFamily] + +-- | The enum a name in an expression refers to. +enumBySymbol :: [EnumInfo] -> SymbolId -> Maybe EnumInfo +enumBySymbol enums symbol = case [info | info <- enums, resolvedSymbol (enumInfoName info) == symbol] of + info : _ -> Just info + [] -> Nothing + +-- | The enum a name in a type refers to. +enumBySpelling :: [EnumInfo] -> Identifier -> Maybe EnumInfo +enumBySpelling enums spelling = case [info | info <- enums, resolvedSpelling (enumInfoName info) == spelling] of + info : _ -> Just info + [] -> Nothing + +-- | The value of a member of the enum with the given type. +enumMemberValue :: [EnumInfo] -> Type -> Identifier -> Maybe Integer +enumMemberValue enums valueType member = + case [value | info <- enums, enumInfoType info == valueType, (name, value) <- enumInfoMembers info, name == member] of + value : _ -> Just value + [] -> Nothing + +-- | Whether a type is the type of an enum. +isEnumType :: Type -> Bool +isEnumType valueType = case enumUnderlyingType valueType of + Just _ -> True + Nothing -> False + +{- | The problems of an enum declaration by itself: an underlying type that +is not an integer type, a member named twice, a written value that is not a +constant integer expression or has no value, and a value its underlying type +cannot hold. +-} +enumDeclarationProblems :: Declaration ResolvedName () -> [Diagnostic] +enumDeclarationProblems declaration = case declaration of + EnumDeclaration spanValue _ _ underlying cases -> + let scalar = maybe (Just defaultScalar) integerScalar underlying + members = numbered (underlyingType underlying) cases + in [ problem spanValue "VXT0066" "the underlying type of an enum must be an integer type" + | Nothing <- [scalar] + ] + ++ [ problem + (enumCaseSpan member) + "VXT0067" + ("the enum already has a member named " ++ identifierText (enumCaseName member)) + | (index, member) <- zip [0 :: Int ..] cases + , enumCaseName member `elem` map enumCaseName (take index cases) + ] + ++ [ problem (enumCaseSpan member) "VXT0070" reason + | (member, _, Just reason) <- members + ] + ++ [ problem + (enumCaseSpan member) + "VXT0068" + ( "the value " + ++ show value + ++ " of this enum member does not fit " + ++ scalarTypeName (maybe defaultScalar id scalar) + ) + | (member, value, Nothing) <- members + , not (integerFits (maybe defaultScalar id scalar) value) + ] + _ -> [] + where + defaultScalar = defaultIntegerScalar diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Format.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Format.hs new file mode 100644 index 00000000..caed0249 --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Format.hs @@ -0,0 +1,275 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | The output format grammar of @Console.Printf@ and @Console.Format@. + +A format is a compile-time string. It is read here, once, into literal text +and conversions, and everything about it that does not depend on the types +of the arguments is decided here: which conversions exist, which flags each +one takes, which flags exclude each other, and where a width or a precision +may stand. A format that is wrong is an error of the program, found when it +is compiled; nothing about a format is left to be discovered while the +program runs. + +A conversion is written + +> % flags width .precision letter + +where the flags are any of @-@, @0@, @+@, a space, @#@ and @'@; the width is +a decimal number or @*@; and the precision is a decimal number or @*@ after a +point. A @*@ takes the width or the precision from an @int@ argument that +stands before the value. + +The conversions: + +[@%d@] a signed integer in decimal +[@%u@] an unsigned integer in decimal +[@%x@] an integer in hexadecimal; a negative number is a minus sign and its +magnitude, never the bit pattern of its representation +[@%f@] a floating-point number with a fixed number of digits after the point +[@%s@] a string +[@%c@] a character +[@%b@] a Boolean, as @true@ or @false@ +[@%n@] the line terminator of the platform; takes no argument +[@%%@] a percent sign; takes no argument + +@%A@ and @%O@, the debug and display forms of an object, are part of the +grammar and need the object model; they are recognized and reported as not +implemented. +-} +module Visual.XSharp.TypeChecker.Format + ( FormatPiece (..) + , Conversion (..) + , ConversionKind (..) + , Flag (..) + , Size (..) + , FormatProblem (..) + , parseFormat + , conversionLetter + , conversionFlagBits + , conversionArgumentCount + ) where + +import Data.Char (isDigit) +import Data.List (nub) +import Visual.XSharp.RuntimeCall + +-- | One part of a format, in the order it is written. +data FormatPiece + = -- | Text that is written as it stands. + LiteralPiece String + | -- | A conversion of one argument. + ConversionPiece Conversion + | -- | @%n@. + NewlinePiece + deriving (Eq, Ord, Read, Show) + +-- | What a conversion writes. +data ConversionKind + = SignedDecimal + | UnsignedDecimal + | Hexadecimal + | FixedPoint + | Text + | Character + | Truth + deriving (Eq, Ord, Read, Show) + +-- | A flag written between @%@ and the width. +data Flag + = -- | @-@: the value stands at the left of its field. + LeftFlag + | -- | @0@: a number is padded with zeros after its sign. + ZeroFlag + | -- | @+@: a number that is not negative is written with a plus sign. + PlusFlag + | -- | space: a number that is not negative is written with a leading space. + SpaceFlag + | -- | @#@: a hexadecimal number is written with the prefix @0x@. + AlternateFlag + | -- | @'@: integer digits are grouped in threes with apostrophes. + GroupFlag + deriving (Eq, Ord, Read, Show) + +-- | A width or a precision. +data Size + = -- | Not written. + NoSize + | -- | Written as a number in the format. + FixedSize Integer + | -- | Written as @*@: taken from an @int@ argument. + ArgumentSize + deriving (Eq, Ord, Read, Show) + +-- | One conversion with everything written in it. +data Conversion = Conversion + { conversionKind :: ConversionKind + , conversionFlags :: [Flag] + , conversionWidth :: Size + , conversionPrecision :: Size + , conversionOffset :: Int + -- ^ The position of its @%@ in the format, counted from one. + } + deriving (Eq, Ord, Read, Show) + +{- | What is wrong with a format: the diagnostic code, the position of the +conversion in the format counted from one, and what to say. +-} +data FormatProblem = FormatProblem + { formatProblemCode :: String + , formatProblemOffset :: Int + , formatProblemMessage :: String + } + deriving (Eq, Ord, Read, Show) + +-- | The letter a conversion is written with. +conversionLetter :: ConversionKind -> Char +conversionLetter kind = case kind of + SignedDecimal -> 'd' + UnsignedDecimal -> 'u' + Hexadecimal -> 'x' + FixedPoint -> 'f' + Text -> 's' + Character -> 'c' + Truth -> 'b' + +{- | The flags of a conversion as the runtime takes them: the sum of the +@VXS_TEXT_FLAG_*@ values of the flags written, and the hexadecimal flag for +@%x@, which the runtime takes as a flag of the integer conversions. +-} +conversionFlagBits :: Conversion -> Integer +conversionFlagBits conversion = + sum (map bit (conversionFlags conversion)) + (if conversionKind conversion == Hexadecimal then flagHexadecimal else 0) + where + bit flag = case flag of + LeftFlag -> flagLeft + ZeroFlag -> flagZero + PlusFlag -> flagPlus + SpaceFlag -> flagSpace + AlternateFlag -> flagAlternate + GroupFlag -> flagGroup + +{- | How many arguments a conversion takes: its value, and one more for each +of a width and a precision written as @*@. +-} +conversionArgumentCount :: Conversion -> Int +conversionArgumentCount conversion = + 1 + length (filter (== ArgumentSize) [conversionWidth conversion, conversionPrecision conversion]) + +{- | Read a format. The result is its pieces in order, with neighbouring +literal text joined, or the first thing wrong with it. +-} +parseFormat :: String -> Either FormatProblem [FormatPiece] +parseFormat = fmap joined . pieces 1 + where + joined (LiteralPiece first : LiteralPiece second : rest) = joined (LiteralPiece (first ++ second) : rest) + joined (piece : rest) = piece : joined rest + joined [] = [] + +pieces :: Int -> String -> Either FormatProblem [FormatPiece] +pieces _ [] = Right [] +pieces offset ('%' : rest) = do + (piece, used) <- readConversion offset rest + (piece :) <$> pieces (offset + 1 + used) (drop used rest) +pieces offset text = + let (literal, rest) = break (== '%') text + in (LiteralPiece literal :) <$> pieces (offset + length literal) rest + +{- | Read what follows a @%@. The result is the piece and the number of +characters after the @%@ that belong to it. +-} +readConversion :: Int -> String -> Either FormatProblem (FormatPiece, Int) +readConversion offset text = do + let (flagText, afterFlags) = span (`elem` "-0+ #'") text + flags <- mapM flagOf flagText + (width, afterWidth) <- sizeOf afterFlags + (precision, afterPrecision) <- case afterWidth of + '.' : afterPoint -> do + (size, rest) <- sizeOf afterPoint + case size of + NoSize -> failure "VXT0075" "a precision needs a number or * after its point" + _ -> Right (size, rest) + _ -> Right (NoSize, afterWidth) + let used = length text - length afterPrecision + 1 + plain = null flags && width == NoSize && precision == NoSize + case afterPrecision of + [] -> failure "VXT0075" "the format ends inside a conversion" + letter : _ -> case letter of + '%' + | plain -> Right (LiteralPiece "%", used) + | otherwise -> failure "VXT0076" "%% takes no flag, width or precision" + 'n' + | plain -> Right (NewlinePiece, used) + | otherwise -> failure "VXT0076" "%n takes no flag, width or precision" + _ -> case lookup letter kinds of + Just kind -> do + checked <- validated (Conversion kind flags width precision offset) + Right (ConversionPiece checked, used) + Nothing + | letter `elem` "AO" -> + failure "VXT0079" ("the conversion %" ++ [letter] ++ " needs the object model and is not implemented yet") + | otherwise -> failure "VXT0075" ("%" ++ [letter] ++ " is not a conversion") + where + failure :: String -> String -> Either FormatProblem a + failure code message = Left (FormatProblem code offset message) + kinds = + [ ('d', SignedDecimal) + , ('u', UnsignedDecimal) + , ('x', Hexadecimal) + , ('f', FixedPoint) + , ('s', Text) + , ('c', Character) + , ('b', Truth) + ] + flagOf character = case character of + '-' -> Right LeftFlag + '0' -> Right ZeroFlag + '+' -> Right PlusFlag + ' ' -> Right SpaceFlag + '#' -> Right AlternateFlag + _ -> Right GroupFlag + -- A number, a star, or nothing. A number is written without a sign. + sizeOf value = case value of + '*' : rest -> Right (ArgumentSize, rest) + _ -> + let (digits, rest) = span isDigit value + in if null digits + then Right (NoSize, rest) + else + if length digits > 9 + then failure "VXT0075" "a width or a precision written in a format has at most nine digits" + else Right (FixedSize (read digits), rest) + validated value + | flags' /= nub flags' = flagFailure "repeats a flag" + | LeftFlag `elem` flags' && ZeroFlag `elem` flags' = flagFailure "combines - and 0, which exclude each other" + | PlusFlag `elem` flags' && SpaceFlag `elem` flags' = flagFailure "combines + and a space, which exclude each other" + | unwanted : _ <- filter (`notElem` allowedFlags kind) flags' = + flagFailure ("takes no " ++ flagName unwanted ++ " flag") + | conversionPrecision value /= NoSize && kind `notElem` [FixedPoint, Text] = + flagFailure "takes no precision" + | otherwise = Right value + where + flags' = conversionFlags value + kind = conversionKind value + flagFailure :: String -> Either FormatProblem a + flagFailure text' = failure "VXT0076" ("the conversion %" ++ [conversionLetter kind] ++ " " ++ text') + +-- | The flags each conversion has a meaning for. +allowedFlags :: ConversionKind -> [Flag] +allowedFlags kind = case kind of + SignedDecimal -> [LeftFlag, ZeroFlag, PlusFlag, SpaceFlag, GroupFlag] + UnsignedDecimal -> [LeftFlag, ZeroFlag, GroupFlag] + Hexadecimal -> [LeftFlag, ZeroFlag, AlternateFlag] + FixedPoint -> [LeftFlag, ZeroFlag, PlusFlag, SpaceFlag, GroupFlag] + Text -> [LeftFlag] + Character -> [LeftFlag] + Truth -> [LeftFlag] + +flagName :: Flag -> String +flagName flag = case flag of + LeftFlag -> "-" + ZeroFlag -> "0" + PlusFlag -> "+" + SpaceFlag -> "space" + AlternateFlag -> "#" + GroupFlag -> "'" diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Literals.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Literals.hs new file mode 100644 index 00000000..28788e40 --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Literals.hs @@ -0,0 +1,78 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | The types of literals and the range of constant expressions. + +A literal takes its type from the place that receives it, and a constant +expression must fit the type it is stored in. Both are rules of +"Visual.XSharp.NumericSemantics" applied at a source position; the checker +of expressions calls them and adds what they report. +-} +module Visual.XSharp.TypeChecker.Literals + ( literalTypeInContext + , integerLiteralType + , floatingLiteralType + , ruleProblems + , constantRangeProblems + , problem + ) where + +import Visual.XSharp.AST +import Visual.XSharp.BuiltinTypes +import Visual.XSharp.ConstantEvaluation +import Visual.XSharp.Diagnostic +import Visual.XSharp.NumericSemantics + +-- | The type of a literal at a place that expects the given type, if any. +literalTypeInContext :: SourceSpan -> Maybe Type -> Literal -> (Type, [Diagnostic]) +literalTypeInContext spanValue expected literal = case literal of + IntegerLiteral value -> integerLiteralType spanValue expected value + FloatingLiteral _ -> floatingLiteralType expected + CharacterLiteral _ -> (scalarTypeToType CharacterScalar, []) + BooleanLiteral _ -> (boolType, []) + StringLiteral _ -> (stringType, []) + UnitLiteral -> (unitType, []) + +-- | The type of an integer literal, and the problems of a value its type cannot hold. +integerLiteralType :: SourceSpan -> Maybe Type -> Integer -> (Type, [Diagnostic]) +integerLiteralType spanValue expected value = + let context = maybe NoNumericContext targetContext expected + rule = integerLiteralRule context value + code = case numericRuleError rule of Just (UntargetedIntegerOutsideInt _) -> "VXT0017"; _ -> "VXT0016" + in (numericRuleType rule, ruleProblems spanValue code rule) + where + targetContext target | target == boolType = BooleanNumericContext + targetContext target = TargetNumericType target + +-- | The type of a floating literal at a place that expects the given type, if any. +floatingLiteralType :: Maybe Type -> (Type, [Diagnostic]) +floatingLiteralType expected = + let context = maybe NoNumericContext TargetNumericType expected + rule = floatingLiteralRule context + in (numericRuleType rule, []) + +-- | The diagnostic of a numeric rule that failed, with the given code. +ruleProblems :: SourceSpan -> String -> NumericRuleResult -> [Diagnostic] +ruleProblems spanValue code rule = case numericRuleError rule of + Nothing -> [] + Just issue -> [problem spanValue code (renderNumericRuleError issue)] + +-- | The problems of a constant expression whose value its target type cannot hold. +constantRangeProblems :: SourceSpan -> Type -> Expression ResolvedName Type -> [Diagnostic] +constantRangeProblems spanValue target expression = case evaluateConstantInteger expression of + Left issue -> [problem spanValue "VXT0019" (renderConstantIntegerError issue)] + Right (Just value) -> case typeToScalarType target of + Just scalar + | scalarTypeFamily scalar `elem` [SignedIntegerFamily, UnsignedIntegerFamily] + , not (integerFits scalar value) -> + [ problem + spanValue + "VXT0018" + ("constant expression result " ++ show value ++ " does not fit " ++ scalarTypeName scalar) + ] + _ -> [] + Right Nothing -> [] + +-- | An error of the type checker at a source position. +problem :: SourceSpan -> String -> String -> Diagnostic +problem spanValue code message = Diagnostic TypeCheckerStage Error code (Just spanValue) message diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Loops.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Loops.hs new file mode 100644 index 00000000..22d6c9a1 --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Loops.hs @@ -0,0 +1,84 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | The loops around the place the type checker is at. + +A @break@ or a @continue@ is checked against the innermost loop, and what a +@break@ may carry depends on how that loop is used. Blocks used as values and +loop headers are not loops, but a transfer has to cross them to reach one, so +they are entries of the same stack. +-} +module Visual.XSharp.TypeChecker.Loops + ( LoopKind (..) + , LoopContext (..) + , outsideLoops + , enterLoop + , loopCondition + , loopUpdate + , transferTarget + ) where + +import Visual.XSharp.AST (Type) + +-- | How a loop is used, which decides what its @break@ statements carry. +data LoopKind + = -- | A loop statement: @break@ carries no value. + StatementLoop + | {- | A loop used as an expression: every @break@ carries the loop's + value, typed in the context that receives it. + -} + ExpressionLoop (Maybe Type) + | {- | Not a loop: the edge of a block used as a value. A @break@ or + @continue@ inside it leaves the block without a value and targets the + loop around the expression the block belongs to. + -} + ValueBlockEdge + | {- | The condition of a loop of the given kind. A @break@ there leaves + that loop, like one in its body. A @continue@ there abandons the rest + of the condition and evaluates the condition of that loop again. + -} + LoopCondition LoopKind + | {- | The update clause of a loop of the given kind. A @continue@ there + ends the update, and the condition of the loop is tested next. A + @break@ there leaves that loop. + -} + LoopUpdate LoopKind + +{- | The loops around a statement, innermost first, and the kind of the loop +statement that is about to be checked. A loop expression checks its loop +statement with the pending kind set; every loop moves the pending kind onto +the stack for its own body. +-} +data LoopContext = LoopContext + { pendingLoop :: LoopKind + , enclosingLoops :: [LoopKind] + } + +-- | The context of a statement that no loop encloses. +outsideLoops :: LoopContext +outsideLoops = LoopContext StatementLoop [] + +-- | The context of the body of the loop statement being checked. +enterLoop :: LoopContext -> LoopContext +enterLoop loops = LoopContext StatementLoop (pendingLoop loops : enclosingLoops loops) + +-- | The context of the condition of the loop statement about to be checked. +loopCondition :: LoopContext -> LoopContext +loopCondition loops = LoopContext StatementLoop (LoopCondition (pendingLoop loops) : enclosingLoops loops) + +-- | The context of the update clause of the loop statement about to be checked. +loopUpdate :: LoopContext -> LoopContext +loopUpdate loops = LoopContext StatementLoop (LoopUpdate (pendingLoop loops) : enclosingLoops loops) + +{- | The loop a @break@ or @continue@ targets: the innermost entry that is +not the edge of a block used as a value. A transfer crosses such an edge +freely; the block it leaves simply yields no value. +-} +transferTarget :: LoopContext -> Maybe LoopKind +transferTarget loops = case dropWhile isValueBlockEdge (enclosingLoops loops) of + kind : _ -> Just kind + [] -> Nothing + where + isValueBlockEdge kind = case kind of + ValueBlockEdge -> True + _ -> False diff --git a/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Returns.hs b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Returns.hs new file mode 100644 index 00000000..42d3162c --- /dev/null +++ b/Compiler/Haskell/Frontend/src/Visual/XSharp/TypeChecker/Returns.hs @@ -0,0 +1,110 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | What the typed tree of a body returns and what leaves its loops. + +The checker for statements reports the returns it meets, but a @return@ may +also stand in a block used as a value, which is reached through an +expression. The types a body returns are therefore read from its typed tree +here, statements and expressions alike. A nested callable is a function of +its own: its returns are not the returns of the body that creates it. +-} +module Visual.XSharp.TypeChecker.Returns + ( typedExpressionType + , blockReturnTypes + , loopBreakTypes + ) where + +import Visual.XSharp.AST +import Visual.XSharp.Completion + +-- | The type the checker gave an expression. +typedExpressionType :: Expression name Type -> Type +typedExpressionType expression = case expression of + NameExpression _ _ valueType -> valueType + LiteralExpression _ _ valueType -> valueType + MemberAccessExpression _ _ _ valueType -> valueType + MethodReferenceExpression _ _ _ valueType -> valueType + CallExpression _ _ _ valueType -> valueType + UnaryExpression _ _ _ valueType -> valueType + BinaryExpression _ _ _ _ valueType -> valueType + IsPatternExpression _ _ _ valueType -> valueType + ConditionalExpression _ _ _ _ valueType -> valueType + CoalesceExpression _ _ _ valueType -> valueType + AssignmentExpression _ _ _ _ valueType -> valueType + IncrementExpression _ _ _ valueType -> valueType + LoopExpression _ _ valueType -> valueType + BlockExpression _ _ valueType -> valueType + MatchExpression _ _ _ valueType -> valueType + CallableExpression _ _ _ _ _ valueType -> valueType + +{- | The types of the values every @return@ of a block carries, in source +order. A @return@ without a value carries @void@. A returned expression that +never yields a value carries none; the returns inside it are listed in its +place. +-} +blockReturnTypes :: Block name Type -> [Type] +blockReturnTypes (Block statements) = concatMap statementReturnTypes statements + +statementReturnTypes :: Statement name Type -> [Type] +statementReturnTypes statement = case statement of + ReturnStatement _ Nothing -> [voidType] + ReturnStatement _ (Just value) + | doesNotComplete value -> expressionReturnTypes value + | otherwise -> typedExpressionType value : expressionReturnTypes value + BindingStatement _ _ _ _ _ value -> expressionReturnTypes value + AssignmentStatement _ _ _ value -> expressionReturnTypes value + IfStatement _ condition whenTrue whenFalse -> + expressionReturnTypes condition ++ blockReturnTypes whenTrue ++ maybe [] blockReturnTypes whenFalse + WhileStatement _ condition body -> expressionReturnTypes condition ++ blockReturnTypes body + DoWhileStatement _ body condition -> blockReturnTypes body ++ expressionReturnTypes condition + ForStatement _ initializer condition updates body -> + maybe [] statementReturnTypes initializer + ++ maybe [] expressionReturnTypes condition + ++ blockReturnTypes body + ++ concatMap statementReturnTypes updates + ForEachStatement _ _ _ _ _ source body -> expressionReturnTypes source ++ blockReturnTypes body + IncrementStatement {} -> [] + CompoundAssignmentStatement _ _ _ _ value -> expressionReturnTypes value + DiscardStatement _ value -> expressionReturnTypes value + BreakStatement _ value -> maybe [] expressionReturnTypes value + ContinueStatement {} -> [] + GuardStatement _ condition block -> expressionReturnTypes condition ++ blockReturnTypes block + BlockStatement _ block -> blockReturnTypes block + ExpressionStatement _ value _ -> expressionReturnTypes value + +expressionReturnTypes :: Expression name Type -> [Type] +expressionReturnTypes expression = case expression of + NameExpression {} -> [] + LiteralExpression {} -> [] + MemberAccessExpression _ receiver _ _ -> expressionReturnTypes receiver + MethodReferenceExpression _ receiver _ _ -> expressionReturnTypes receiver + CallExpression _ callee arguments _ -> concatMap expressionReturnTypes (callee : arguments) + UnaryExpression _ _ value _ -> expressionReturnTypes value + BinaryExpression _ _ left right _ -> expressionReturnTypes left ++ expressionReturnTypes right + IsPatternExpression _ subject _ _ -> expressionReturnTypes subject + ConditionalExpression _ condition first second _ -> concatMap expressionReturnTypes [condition, first, second] + CoalesceExpression _ left fallback _ -> expressionReturnTypes left ++ expressionReturnTypes fallback + AssignmentExpression _ _ _ value _ -> expressionReturnTypes value + IncrementExpression {} -> [] + LoopExpression _ loop _ -> statementReturnTypes loop + BlockExpression _ block _ -> blockReturnTypes block + MatchExpression _ subjects arms _ -> + concatMap expressionReturnTypes subjects ++ concatMap armReturnTypes arms + -- A callable returns from itself. + CallableExpression {} -> [] + where + armReturnTypes arm = maybe [] expressionReturnTypes (matchArmGuard arm) ++ expressionReturnTypes (matchArmBody arm) + +{- | The types of the values carried by the breaks that leave this loop +itself, from its body, its condition and its update clause, also out of +blocks used as values. Breaks of nested loops leave those loops. +-} +loopBreakTypes :: Statement name Type -> [Type] +loopBreakTypes loop = [typedExpressionType value | BreakStatement _ (Just value) <- transfers] + where + transfers = case loop of + WhileStatement _ condition body -> expressionTransfers condition ++ blockTransfers body + ForStatement _ _ condition updates body -> + maybe [] expressionTransfers condition ++ blockTransfers body ++ concatMap statementTransfers updates + _ -> [] diff --git a/Compiler/Haskell/Frontend/visual-xsharp-frontend.cabal b/Compiler/Haskell/Frontend/visual-xsharp-frontend.cabal index f76d4286..9418a3b4 100644 --- a/Compiler/Haskell/Frontend/visual-xsharp-frontend.cabal +++ b/Compiler/Haskell/Frontend/visual-xsharp-frontend.cabal @@ -2,7 +2,7 @@ cabal-version: 3.12 -- SPDX-FileCopyrightText: 2026 Progmasoft -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 name: visual-xsharp-frontend -version: 0.4.1 +version: 0.5.0 synopsis: Visual X# resolution and type-checking frontend license: MPL-2.0 author: Progmasoft @@ -33,6 +33,13 @@ library Visual.XSharp.TypeClassification Visual.XSharp.TypeChecker Visual.XSharp.TypeChecker.Branching + Visual.XSharp.TypeChecker.Console + Visual.XSharp.TypeChecker.Context + Visual.XSharp.TypeChecker.Enums + Visual.XSharp.TypeChecker.Format + Visual.XSharp.TypeChecker.Literals + Visual.XSharp.TypeChecker.Loops + Visual.XSharp.TypeChecker.Returns hs-source-dirs: src build-depends: base >=4.20 && <4.23, diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/AST.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/AST.hs index 3050cc61..6754449c 100644 --- a/Compiler/Haskell/Syntax/src/Visual/XSharp/AST.hs +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/AST.hs @@ -13,6 +13,10 @@ module Visual.XSharp.AST , SourceSpan (..) , SyntaxTree (..) , Declaration (..) + , EnumCase (..) + , enumType + , enumUnderlyingType + , enumMemberValues , TemplateParameter (..) , TemplateParameterKind (..) , TemplateParameterShape (..) @@ -30,6 +34,8 @@ module Visual.XSharp.AST , matchPatternSpan , matchPatternAnnotation , matchPatternBinding + , statementSourceSpan + , expressionSourceSpan , traverseMatchArm , traverseMatchPattern , CallableBody (..) @@ -147,6 +153,31 @@ data Declaration name annotation , declarationTemplateParameters :: [TemplateParameter name annotation] , typeMembers :: [Declaration name annotation] } + | {- | A classic enum: a value type whose members are named integers. + The underlying type is absent when the source does not write one. + -} + EnumDeclaration + { declarationSpan :: SourceSpan + , declarationName :: name + , declarationAnnotation :: annotation + , enumUnderlying :: Maybe TypeSyntax + , enumCases :: [EnumCase] + } + deriving stock (Eq, Ord, Read, Show) + +{- | A member of a classic enum. A member without a written value takes the +value after that of the member before it, and the first takes zero. + +A written value is a constant integer expression. It is kept as it was +parsed in every stage: the names in it are earlier members of the same enum, +which are named by their spelling and take no symbols, so no later stage has +anything to add to it. The type checker computes its value. +-} +data EnumCase = EnumCase + { enumCaseSpan :: SourceSpan + , enumCaseName :: Identifier + , enumCaseValue :: Maybe (Expression Identifier ()) + } deriving stock (Eq, Ord, Read, Show) {- | Generic parameter with its declaration span, category, pack flag, and default. @@ -291,6 +322,10 @@ data Expression name annotation -- the checker knows whether the receiver denotes a type or a value. The -- current executable subset accepts only type-qualified method calls. MemberAccessExpression SourceSpan (Expression name annotation) Identifier annotation + | -- @Receiver::Member@: the method of that name as a callable value, + -- without a call of it. The checker replaces a reference it can + -- resolve with the name of the method, so no later stage sees one. + MethodReferenceExpression SourceSpan (Expression name annotation) Identifier annotation | CallExpression SourceSpan (Expression name annotation) [Expression name annotation] annotation | UnaryExpression SourceSpan UnaryOperator (Expression name annotation) annotation | BinaryExpression SourceSpan BinaryOperator (Expression name annotation) (Expression name annotation) annotation @@ -377,6 +412,46 @@ data MatchPattern name annotation MatchCasePattern SourceSpan Identifier annotation deriving stock (Eq, Ord, Read, Show) +-- | Source range of a statement. +statementSourceSpan :: Statement name annotation -> SourceSpan +statementSourceSpan statement = case statement of + BindingStatement value _ _ _ _ _ -> value + AssignmentStatement value _ _ _ -> value + ReturnStatement value _ -> value + IfStatement value _ _ _ -> value + WhileStatement value _ _ -> value + DoWhileStatement value _ _ -> value + ForStatement value _ _ _ _ -> value + ForEachStatement value _ _ _ _ _ _ -> value + IncrementStatement value _ _ -> value + CompoundAssignmentStatement value _ _ _ _ -> value + DiscardStatement value _ -> value + BreakStatement value _ -> value + ContinueStatement value -> value + GuardStatement value _ _ -> value + BlockStatement value _ -> value + ExpressionStatement value _ _ -> value + +-- | Source range of an expression. +expressionSourceSpan :: Expression name annotation -> SourceSpan +expressionSourceSpan expression = case expression of + NameExpression value _ _ -> value + LiteralExpression value _ _ -> value + MemberAccessExpression value _ _ _ -> value + MethodReferenceExpression value _ _ _ -> value + CallExpression value _ _ _ -> value + UnaryExpression value _ _ _ -> value + BinaryExpression value _ _ _ _ -> value + IsPatternExpression value _ _ _ -> value + ConditionalExpression value _ _ _ _ -> value + CoalesceExpression value _ _ _ -> value + AssignmentExpression value _ _ _ _ -> value + IncrementExpression value _ _ _ -> value + LoopExpression value _ _ -> value + BlockExpression value _ _ -> value + MatchExpression value _ _ _ -> value + CallableExpression value _ _ _ _ _ -> value + -- | The guard, when present, and the body of an arm, in evaluation order. matchArmExpressions :: MatchArm name annotation -> [Expression name annotation] matchArmExpressions arm = maybe [] (: []) (matchArmGuard arm) ++ [matchArmBody arm] @@ -567,6 +642,35 @@ namedType value = NamedType (QualifiedName [Identifier value]) [] boolType :: Type boolType = namedType "bool" +{- | The type of a classic enum. + +An enum is a type of its own, and it is a value of its underlying integer +type with a closed set of values. Both facts are needed after type checking: +the lowering stores an enum as its underlying type, and a @match@ over an +enum is complete when its arms name every value. The type therefore carries +them: its name under the reserved root @enum@, which no source can spell +because @enum@ is a keyword, then the underlying type, then the distinct +values of its members in ascending order. +-} +enumType :: Identifier -> Type -> [Integer] -> Type +enumType name underlying values = + NamedType + (QualifiedName [Identifier "enum", name]) + (TypeTemplateArgument underlying : map (ValueTemplateArgument . IntegerTemplateValue) values) + +-- | The underlying integer type of an enum type, and nothing for any other type. +enumUnderlyingType :: Type -> Maybe Type +enumUnderlyingType valueType = case valueType of + NamedType (QualifiedName [Identifier "enum", _]) (TypeTemplateArgument underlying : _) -> Just underlying + _ -> Nothing + +-- | The distinct member values of an enum type, and nothing for any other type. +enumMemberValues :: Type -> Maybe [Integer] +enumMemberValues valueType = case valueType of + NamedType (QualifiedName [Identifier "enum", _]) (TypeTemplateArgument _ : values) -> + Just [value | ValueTemplateArgument (IntegerTemplateValue value) <- values] + _ -> Nothing + -- | Canonical built-in signed integer type used by the current frontend. intType :: Type intType = namedType "int" diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/Completion.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/Completion.hs new file mode 100644 index 00000000..62a83789 --- /dev/null +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/Completion.hs @@ -0,0 +1,276 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Whether control can reach the end of a statement or an expression. + +This is a fact about control flow and is kept apart from types. A block that +ends with @return@, an @if@ expression whose two blocks both leave, and a +call one of whose arguments is such an expression never produce a value, but +no type says so: the question is answered from the typed tree by the +functions of this module. The type checker uses them to decide what must +have a value, and the desugarer uses the same functions to lower what never +completes as statements alone, so that nothing is stored for a value that +does not exist. + +The answers err on the side of completing: a call is assumed to return, and +a condition is assumed to be able to take either value unless it is the +literal @true@. Whether a @break@ or @continue@ has a loop to leave is a +separate rule of the type checker. +-} +module Visual.XSharp.Completion + ( blockCannotComplete + , statementCannotComplete + , doesNotComplete + , neverCompletingOperand + , matchArmsAlwaysAccept + , acceptsEveryValue + , isUnguarded + , blockTransfers + , statementTransfers + , expressionTransfers + , isBreak + , isContinue + ) where + +import Visual.XSharp.AST + +{- | Whether control can never reach the end of a block. + +A statement cannot complete normally when it is a @return@, a @break@ or a +@continue@; an @if@ with an @else@ whose two blocks both cannot; a nested +block that cannot; a loop whose condition is the constant true, or absent in +a @for@, and that no @break@ leaves; a statement @match@ some arm of which +always accepts and all of whose arms cannot; or any statement an expression +of which is always evaluated and never completes. A block cannot complete +when one of its statements cannot, because the statements after that one are +never reached. +-} +blockCannotComplete :: Block name Type -> Bool +blockCannotComplete (Block statements) = any statementCannotComplete statements + +-- | Whether control can never reach the point after a statement. +statementCannotComplete :: Statement name Type -> Bool +statementCannotComplete statement = case statement of + ReturnStatement {} -> True + BreakStatement {} -> True + ContinueStatement {} -> True + BindingStatement _ _ _ _ _ value -> doesNotComplete value + AssignmentStatement _ _ _ value -> doesNotComplete value + CompoundAssignmentStatement _ _ _ _ value -> doesNotComplete value + DiscardStatement _ value -> doesNotComplete value + IfStatement _ condition whenTrue whenFalse -> + doesNotComplete condition + || maybe False (\block -> blockCannotComplete whenTrue && blockCannotComplete block) whenFalse + GuardStatement _ condition _ -> doesNotComplete condition + BlockStatement _ nested -> blockCannotComplete nested + -- A loop ends normally when a break leaves it, from its body or from + -- its condition. Without one it does not end when its condition is the + -- constant true or never completes; a do/while also when its body + -- cannot complete, because its body runs first. + WhileStatement _ condition body -> + not (leftByBreak (expressionTransfers condition ++ blockTransfers body)) + && (doesNotComplete condition || isConstantTrue condition) + DoWhileStatement _ body condition -> + not (leftByBreak (expressionTransfers condition ++ blockTransfers body)) + && (doesNotComplete condition || isConstantTrue condition) + ForStatement _ initializer condition updates body -> + maybe False statementCannotComplete initializer + || ( not + ( leftByBreak + ( maybe [] expressionTransfers condition + ++ blockTransfers body + ++ concatMap statementTransfers updates + ) + ) + && maybe True (\test -> doesNotComplete test || isConstantTrue test) condition + ) + ExpressionStatement _ value _ -> doesNotComplete value + _ -> False + where + isConstantTrue expression = case expression of + LiteralExpression _ (BooleanLiteral True) _ -> True + _ -> False + leftByBreak = any isBreak + +{- | Whether evaluating an expression never yields a value. + +That is so when a part of it that is always evaluated never does, or when it +is a choice every branch of which leaves: an @if@ expression whose two +blocks do not complete, a block used as a value that cannot complete or +whose final expression does not, or a @match@ one arm of which always +accepts and no arm of which completes. An operand that is evaluated only +sometimes, such as the right operand of @&&@, decides nothing. +-} +doesNotComplete :: Expression name Type -> Bool +doesNotComplete expression = case expression of + -- The final expression of a block is its last statement, so one walk + -- of the statements covers it; a second look at it here would double + -- the work at every level of nested blocks. + BlockExpression _ block _ -> blockCannotComplete block + ConditionalExpression _ condition first second _ -> + doesNotComplete condition || (doesNotComplete first && doesNotComplete second) + MatchExpression _ subjects arms _ -> + any doesNotComplete subjects + || ( not (null arms) + && matchArmsAlwaysAccept (subjectTypesOf arms) arms + && all (doesNotComplete . matchArmBody) arms + ) + -- A loop used as an expression yields the value of the break that + -- leaves it; without such a break it never yields one. + LoopExpression _ loop _ -> statementCannotComplete loop + _ -> case neverCompletingOperand expression of + Just _ -> True + Nothing -> False + where + -- Every typed pattern carries the type of the value it accepts. + subjectTypesOf arms = case arms of + first : _ -> map matchPatternAnnotation (matchArmPatterns first) + [] -> [] + +{- | The operands of an expression that are evaluated before its first +operand that never completes, and that operand; nothing when every operand +that is always evaluated completes. Choices between branches are not +operands in this sense and are answered by 'doesNotComplete' itself. +-} +neverCompletingOperand :: Expression name Type -> Maybe ([Expression name Type], Expression name Type) +neverCompletingOperand expression = case break doesNotComplete (alwaysEvaluated expression) of + (before, operand : _) -> Just (before, operand) + (_, []) -> Nothing + where + alwaysEvaluated value = case value of + MemberAccessExpression _ receiver _ _ -> [receiver] + MethodReferenceExpression _ receiver _ _ -> [receiver] + CallExpression _ callee arguments _ -> callee : arguments + UnaryExpression _ _ operand _ -> [operand] + BinaryExpression _ operator left right _ + | operator `elem` [LogicalAnd, LogicalOr] -> [left] + | otherwise -> [left, right] + IsPatternExpression _ subject _ _ -> [subject] + CoalesceExpression _ left _ _ -> [left] + AssignmentExpression _ _ _ operand _ -> [operand] + MatchExpression _ subjects _ _ -> subjects + _ -> [] + +-- | Whether a pattern accepts every value of its subject. +acceptsEveryValue :: MatchPattern name annotation -> Bool +acceptsEveryValue patternValue = case patternValue of + MatchWildcardPattern {} -> True + MatchTypePattern {} -> True + _ -> False + +{- | Whether some arm is certain to accept, whatever the subjects are. + +That is the case when an arm without a guard has only patterns that accept +every value. It is also the case when every subject has a closed set of +values, a @bool@ or an enum, and the arms without guards accept each +combination of those values between them; the combinations are enumerated, +up to 'maximumEnumeratedCombinations' of them. Nothing else is recognized: a +guard may be false, and the literals of a wider type are never listed in +full. +-} +matchArmsAlwaysAccept :: [Type] -> [MatchArm name annotation] -> Bool +matchArmsAlwaysAccept subjectTypes arms = any catchAll unguarded || coversClosedValues + where + unguarded = filter isUnguarded arms + catchAll arm = all acceptsEveryValue (matchArmPatterns arm) + coversClosedValues = case traverse closedValues subjectTypes of + Just domains + | not (null domains) + , product (map length domains) <= maximumEnumeratedCombinations -> + all (\values -> any (`acceptsLiterals` values) unguarded) (sequence domains) + _ -> False + +-- | The values of a type that has a closed set of them, as the literals a pattern names. +closedValues :: Type -> Maybe [Literal] +closedValues valueType + | valueType == boolType = Just [BooleanLiteral True, BooleanLiteral False] + | otherwise = map IntegerLiteral <$> enumMemberValues valueType + +{- | The most combinations of subject values that are enumerated to decide +whether a match accepts every value. A match with more is complete only +through a catch-all arm; the bound keeps the check linear in practice. +-} +maximumEnumeratedCombinations :: Int +maximumEnumeratedCombinations = 256 + +-- | Whether an arm's patterns accept the given values of its subjects. +acceptsLiterals :: MatchArm name annotation -> [Literal] -> Bool +acceptsLiterals arm values = + length (matchArmPatterns arm) == length values && and (zipWith accepts (matchArmPatterns arm) values) + where + accepts patternValue value = case patternValue of + MatchLiteralPattern _ literal _ -> literal == value + _ -> acceptsEveryValue patternValue + +-- | Whether an arm has no guard. +isUnguarded :: MatchArm name annotation -> Bool +isUnguarded arm = case matchArmGuard arm of + Nothing -> True + Just _ -> False + +{- | The @break@ and @continue@ statements in a block that target the loop +the block belongs to. + +Loops nested in the block keep their own transfers, in their bodies and in +their conditions and update clauses. A transfer in a block used as a value +inside an expression targets the same loop as one in a statement, so +expressions are searched as well; closures and loops used as expressions +are not, because a transfer in them cannot reach this loop. +-} +blockTransfers :: Block name annotation -> [Statement name annotation] +blockTransfers (Block statements) = concatMap statementTransfers statements + +-- | The transfers of one statement that target the loop around it. +statementTransfers :: Statement name annotation -> [Statement name annotation] +statementTransfers statement = case statement of + BreakStatement _ value -> statement : maybe [] expressionTransfers value + ContinueStatement {} -> [statement] + BindingStatement _ _ _ _ _ value -> expressionTransfers value + AssignmentStatement _ _ _ value -> expressionTransfers value + ReturnStatement _ value -> maybe [] expressionTransfers value + IfStatement _ condition whenTrue whenFalse -> + expressionTransfers condition ++ blockTransfers whenTrue ++ maybe [] blockTransfers whenFalse + WhileStatement {} -> [] + DoWhileStatement {} -> [] + ForStatement _ initializer _ _ _ -> maybe [] statementTransfers initializer + ForEachStatement _ _ _ _ _ source _ -> expressionTransfers source + IncrementStatement {} -> [] + CompoundAssignmentStatement _ _ _ _ value -> expressionTransfers value + DiscardStatement _ value -> expressionTransfers value + GuardStatement _ condition block -> expressionTransfers condition ++ blockTransfers block + BlockStatement _ block -> blockTransfers block + ExpressionStatement _ value _ -> expressionTransfers value + +-- | The transfers in an expression that target the loop around it. +expressionTransfers :: Expression name annotation -> [Statement name annotation] +expressionTransfers expression = case expression of + NameExpression {} -> [] + LiteralExpression {} -> [] + MemberAccessExpression _ receiver _ _ -> expressionTransfers receiver + MethodReferenceExpression _ receiver _ _ -> expressionTransfers receiver + CallExpression _ callee arguments _ -> concatMap expressionTransfers (callee : arguments) + UnaryExpression _ _ value _ -> expressionTransfers value + BinaryExpression _ _ left right _ -> expressionTransfers left ++ expressionTransfers right + IsPatternExpression _ subject _ _ -> expressionTransfers subject + ConditionalExpression _ condition first second _ -> concatMap expressionTransfers [condition, first, second] + CoalesceExpression _ left fallback _ -> expressionTransfers left ++ expressionTransfers fallback + AssignmentExpression _ _ _ value _ -> expressionTransfers value + IncrementExpression {} -> [] + LoopExpression {} -> [] + BlockExpression _ block _ -> blockTransfers block + MatchExpression _ subjects arms _ -> + concatMap expressionTransfers subjects + ++ concatMap (\arm -> maybe [] expressionTransfers (matchArmGuard arm) ++ expressionTransfers (matchArmBody arm)) arms + CallableExpression {} -> [] + +-- | Whether a statement is a @break@. +isBreak :: Statement name annotation -> Bool +isBreak statement = case statement of + BreakStatement {} -> True + _ -> False + +-- | Whether a statement is a @continue@. +isContinue :: Statement name annotation -> Bool +isContinue statement = case statement of + ContinueStatement {} -> True + _ -> False diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/Lexer.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/Lexer.hs index 9cf5398d..56775961 100644 --- a/Compiler/Haskell/Syntax/src/Visual/XSharp/Lexer.hs +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/Lexer.hs @@ -244,6 +244,7 @@ keywords = , "do" , "for" , "else" + , "enum" , "false" , "final" , "guard" @@ -299,6 +300,7 @@ longestSymbol source = , "^=" , "??" , "?:" + , "::" , "==" , "\\=" , "<=" diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/NestingLimits.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/NestingLimits.hs new file mode 100644 index 00000000..fb884a60 --- /dev/null +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/NestingLimits.hs @@ -0,0 +1,175 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | Limits on how deeply a program may nest, checked on the parsed tree. + +The stages after Core walk a function body recursively: one level of +recursion for every statement nested in another and for every expression +that is an operand of another. Their stack is finite, so a program nested +deeply enough would end the compiler with a stack overflow instead of a +diagnostic. This module rejects such a program first, at the place where the +nesting becomes too deep. + +The limits bound real nesting only. Three shapes nest in the tree as deep as +they are long and are walked in a loop by every stage, so they do not count: +an @else if@ chain, whose links are all at the level of the first @if@; the +arms of a @match@, which lower to such a chain, so that the body of an arm +is one level below its match however many arms the match has; and a chain of +a binary operator, whose left operand is at the level of the operator. + +What a level costs is measured, not estimated: the stack every native stage +needs per level of nesting, in an ordinary and in a sanitizer build, is +recorded in @Benchmarks/2026-10-04-Nesting-And-Chains.md@ and can be measured +again with @//Compiler/Support/Tests:stack_probe@. At both limits the whole +native pipeline needs a few megabytes of the stack the compiler runs on +(@Visual/XSharp/Support/CompilerStack.hpp@). The native Core reader and +writer bound both depths at 4096 for Core that does not come from this +frontend. +-} +module Visual.XSharp.NestingLimits + ( maximumStatementNesting + , maximumExpressionNesting + , nestingProblems + ) where + +import Visual.XSharp.AST +import Visual.XSharp.Diagnostic + +{- | The deepest a statement may be nested in other statements of one +function body. The statements of the body itself are at level 1. +-} +maximumStatementNesting :: Int +maximumStatementNesting = 256 + +{- | The deepest an expression may be nested in other expressions of one +function body. An expression that is not an operand is at level 1. +-} +maximumExpressionNesting :: Int +maximumExpressionNesting = 1024 + +-- | A place where one of the limits is exceeded. +data Excess + = StatementExcess SourceSpan + | ExpressionExcess SourceSpan + +{- | Diagnostics for every function body that nests too deeply. + +Each function reports at most one statement and one expression: the first +of each, in source order, that is one level beyond the limit. Nothing below +a reported node is visited, so the work stays proportional to the size of +the accepted part of the program. +-} +nestingProblems :: ParsedAST -> [Diagnostic] +nestingProblems (ParsedAST (SyntaxTree _ declarations)) = concatMap declarationProblems declarations + +declarationProblems :: Declaration name annotation -> [Diagnostic] +declarationProblems declaration = case declaration of + FunctionDeclaration {declarationBody = body} -> report (blockExcess 1 0 body) + TypeDeclaration {typeMembers = members} -> concatMap declarationProblems members + TemplateTypeDeclaration {typeMembers = members} -> concatMap declarationProblems members + -- The value of a member is an expression of its own, at level 1. + EnumDeclaration {enumCases = cases} -> + concat [report (expressionExcess 1 1 value) | EnumCase {enumCaseValue = Just value} <- cases] + where + report found = + take 1 [statementProblem spanValue | StatementExcess spanValue <- found] + ++ take 1 [expressionProblem spanValue | ExpressionExcess spanValue <- found] + +statementProblem :: SourceSpan -> Diagnostic +statementProblem spanValue = + Diagnostic + ParserStage + Error + "VXP0039" + (Just spanValue) + ( "statements are nested more than " + ++ show maximumStatementNesting + ++ " levels deep here; move the inner statements into a method of their own" + ) + +expressionProblem :: SourceSpan -> Diagnostic +expressionProblem spanValue = + Diagnostic + ParserStage + Error + "VXP0040" + (Just spanValue) + ( "this expression is nested more than " + ++ show maximumExpressionNesting + ++ " levels deep; compute part of it in a statement of its own" + ) + +{- | The excesses in a block whose statements are at the given statement +level, inside the given number of enclosing expressions. +-} +blockExcess :: Int -> Int -> Block name annotation -> [Excess] +blockExcess level enclosing (Block statements) = concatMap (statementExcess level enclosing) statements + +statementExcess :: Int -> Int -> Statement name annotation -> [Excess] +statementExcess level enclosing statement + | level > maximumStatementNesting = [StatementExcess (statementSourceSpan statement)] + | otherwise = case statement of + BindingStatement _ _ _ _ _ value -> operand value + AssignmentStatement _ _ _ value -> operand value + ReturnStatement _ value -> maybe [] operand value + IfStatement _ condition whenTrue whenFalse -> + operand condition ++ inner whenTrue ++ case whenFalse of + Nothing -> [] + -- The next link of an else-if chain stays at this level. + Just (Block [next@IfStatement {}]) -> statementExcess level enclosing next + Just block -> inner block + WhileStatement _ condition body -> operand condition ++ inner body + DoWhileStatement _ body condition -> inner body ++ operand condition + ForStatement _ initializer condition updates body -> + maybe [] (statementExcess level enclosing) initializer + ++ maybe [] operand condition + ++ concatMap (statementExcess level enclosing) updates + ++ inner body + ForEachStatement _ _ _ _ _ source body -> operand source ++ inner body + IncrementStatement {} -> [] + CompoundAssignmentStatement _ _ _ _ value -> operand value + DiscardStatement _ value -> operand value + BreakStatement _ value -> maybe [] operand value + ContinueStatement {} -> [] + GuardStatement _ condition block -> operand condition ++ inner block + BlockStatement _ block -> inner block + ExpressionStatement _ value _ -> operand value + where + inner = blockExcess (level + 1) enclosing + operand = expressionExcess level (enclosing + 1) + +{- | The excesses in an expression at the given expression level. Statements +inside it, in a value block, a loop, a match arm or a closure body, are one +statement level below the statement that holds the expression, and +expressions inside those statements keep counting from this one, so the two +levels together bound how deep any walk of the body can recurse. +-} +expressionExcess :: Int -> Int -> Expression name annotation -> [Excess] +expressionExcess level depth expression + | depth > maximumExpressionNesting = [ExpressionExcess (expressionSourceSpan expression)] + | otherwise = case expression of + NameExpression {} -> [] + LiteralExpression {} -> [] + MemberAccessExpression _ receiver _ _ -> operand receiver + MethodReferenceExpression _ receiver _ _ -> operand receiver + CallExpression _ callee arguments _ -> concatMap operand (callee : arguments) + UnaryExpression _ _ value _ -> operand value + -- The left operand of a binary operator is at the level of the + -- operator: `a + b + c` nests in it as deep as the chain is long, + -- and every stage walks that chain in a loop. + BinaryExpression _ _ left right _ -> expressionExcess level depth left ++ operand right + IsPatternExpression _ subject _ _ -> operand subject + ConditionalExpression _ condition first second _ -> concatMap operand [condition, first, second] + CoalesceExpression _ left fallback _ -> operand left ++ operand fallback + AssignmentExpression _ _ _ value _ -> operand value + IncrementExpression {} -> [] + LoopExpression _ loop _ -> statementExcess (level + 1) depth loop + BlockExpression _ block _ -> blockExcess (level + 1) depth block + MatchExpression _ subjects arms _ -> + concatMap operand subjects ++ concatMap (concatMap operand . matchArmExpressions) arms + CallableExpression _ _ captures _ body _ -> + concatMap (maybe [] operand . captureInitializer) captures ++ case body of + CallableExpressionBody value -> operand value + CallableBlockBody block -> blockExcess (level + 1) depth block + where + operand = expressionExcess level (depth + 1) diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser.hs index ff57a133..05ff40d7 100644 --- a/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser.hs +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser.hs @@ -71,7 +71,41 @@ manyUntilEof parser = do parseDeclaration :: P (Declaration Identifier ()) parseDeclaration = do isTemplate <- peekText "template" - if isTemplate then parseTemplateDeclaration else parseOrdinaryTypeDeclaration + isEnum <- peekText "enum" + if isTemplate + then parseTemplateDeclaration + else if isEnum then parseEnumDeclaration else parseOrdinaryTypeDeclaration + +{- | Parse a classic enum: @enum Name { A, B = 2, C }@, with an optional +underlying type written as @enum Name = byte { ... }@. A comma separates the +members and may follow the last one. The value of a member is read as an +integer literal with an optional minus sign; other constant expressions are +not implemented yet. +-} +parseEnumDeclaration :: P (Declaration Identifier ()) +parseEnumDeclaration = do + start <- keyword "enum" + (name, _) <- identifier + hasUnderlying <- optionalSymbol "=" + underlying <- if hasUnderlying then Just <$> parseTypeSyntax else pure Nothing + _ <- symbol "{" + cases <- parseCases + close <- symbol "}" + pure (EnumDeclaration (mergeSpan (tokenSpan start) (tokenSpan close)) name () underlying cases) + where + parseCases = do + done <- peekText "}" + if done + then pure [] + else do + (caseName, caseSpan) <- identifier + hasValue <- optionalSymbol "=" + -- A value is an expression without assignment; which + -- expressions are constant is the type checker's rule. + value <- if hasValue then Just <$> parseConditional else pure Nothing + comma <- optionalSymbol "," + remaining <- if comma then parseCases else pure [] + pure (EnumCase caseSpan caseName value : remaining) parseOrdinaryTypeDeclaration :: P (Declaration Identifier ()) parseOrdinaryTypeDeclaration = do @@ -378,6 +412,7 @@ requireTemplateValue :: Expression Identifier () -> P TemplateValueSyntax requireTemplateValue expression = case expression of NameExpression spanValue name _ -> pure (TemplateNameSyntax spanValue (QualifiedName [name])) MemberAccessExpression spanValue _ _ _ -> unsupported spanValue + MethodReferenceExpression spanValue _ _ _ -> unsupported spanValue LiteralExpression spanValue literal _ -> case literal of IntegerLiteral value -> pure (TemplateIntegerSyntax spanValue value) CharacterLiteral value -> pure (TemplateCharacterSyntax spanValue value) @@ -401,8 +436,22 @@ requireTemplateValue expression = case expression of unsupported spanValue = failAt spanValue "VXP0018" "template value arguments must be compile-time scalar expressions" +{- | What the last item of a block may be. + +Every block may end with a statement. A callable body may also end with an +expression that has no semicolon, which is its result. A value block, the +block of an @if@ expression or of a @match@ arm, is one whose last item is +its value: there an @if@ with an @else@ and a @match@ in last position are +expressions as well, as the grammar's @final-expression@ allows. +-} +data BlockMode = StatementBlock | CallableBlock | ValueBlock + deriving stock (Eq) + parseBlock :: Bool -> P (Block Identifier ()) -parseBlock allowFinalExpression = do _ <- symbol "{"; statements <- go; _ <- symbol "}"; pure (Block statements) +parseBlock allowFinalExpression = parseBlockIn (if allowFinalExpression then CallableBlock else StatementBlock) + +parseBlockIn :: BlockMode -> P (Block Identifier ()) +parseBlockIn mode = do _ <- symbol "{"; statements <- go; _ <- symbol "}"; pure (Block statements) where go = do done <- peekText "}" @@ -412,7 +461,7 @@ parseBlock allowFinalExpression = do _ <- symbol "{"; statements <- go; _ <- sym else if eof then failCurrent "VXP0002" "unterminated block" - else (:) <$> parseStatement allowFinalExpression <*> go + else (:) <$> parseStatementIn mode <*> go blockSpan :: Block name annotation -> SourceSpan -> SourceSpan blockSpan (Block []) fallback = fallback @@ -440,18 +489,21 @@ statementSpan statement = case statement of -- The branching forms of "Visual.XSharp.Parser.Match" are built from the -- expression, block, and type grammar of this module. branchGrammar :: BranchGrammar -branchGrammar = BranchGrammar parseExpression (parseBlock True) (parseBlock False) parseTypeSyntax parsePatternLiteral +branchGrammar = BranchGrammar parseExpression (parseBlockIn ValueBlock) (parseBlock False) parseTypeSyntax parsePatternLiteral -parseStatement :: Bool -> P (Statement Identifier ()) -parseStatement allowFinalExpression = +parseStatementIn :: BlockMode -> P (Statement Identifier ()) +parseStatementIn mode = do + let allowFinalExpression = mode /= StatementBlock tokens <- peekTokens 3 -- Compound types make a fixed token-count heuristic incorrect. Probe -- the complete declaration prefix without consuming it, then commit to -- that grammar branch so a later initializer error remains precise. case tokens of first : _ | tokenKind first == KeywordToken && tokenText first == "return" -> parseReturn - first : _ | tokenKind first == KeywordToken && tokenText first == "if" -> parseIf + first : _ + | tokenKind first == KeywordToken && tokenText first == "if" -> + if mode == ValueBlock then parseIfInValueBlock else parseIf first : _ | tokenKind first == KeywordToken && tokenText first == "while" -> parseWhile first : _ | tokenKind first == KeywordToken && tokenText first == "do" -> parseDoWhile first : _ | tokenKind first == KeywordToken && tokenText first == "for" -> parseFor @@ -465,9 +517,11 @@ parseStatement allowFinalExpression = pure (BlockStatement spanValue block) -- A `match` that starts a statement is the statement form: it -- needs no terminator and its arms need not cover every value. + -- As the last item of a value block it is the value of the block. first : _ | tokenKind first == KeywordToken && tokenText first == "match" -> do value <- parseMatchExpression branchGrammar - pure (ExpressionStatement (expressionSpan value) value True) + closesBlock <- peekText "}" + pure (ExpressionStatement (expressionSpan value) value (not (mode == ValueBlock && closesBlock))) _ | startsIncrement tokens -> parseIncrementStatement _ | startsCompoundAssignment tokens -> parseCompoundAssignmentStatement _ | startsDiscard tokens -> parseDiscard @@ -506,6 +560,86 @@ parseIf = do pure (condition, trueBlock, falseBlock) pure (IfStatement spanValue condition trueBlock falseBlock) +{- | An @if@ at the start of an item of a value block. + +Whether it is the statement or the expression is known only after it: it is +the value of the block when it is the last item, has an @else@ block, and +one of its blocks ends with a value; the other block then either ends with a +value as well or leaves the block, which the type checker decides. Its blocks are therefore parsed once, +as value blocks, and the statement form is recovered from them when the @if@ +turns out to be a statement. Parsing it twice instead would double the work +at every level of nested value blocks. +-} +parseIfInValueBlock :: P (Statement Identifier ()) +parseIfInValueBlock = do + ((condition, first, second, chained), spanValue) <- withSpan $ do + _ <- keyword "if" + _ <- symbol "(" + condition <- parseCondition branchGrammar "an if" + _ <- symbol ")" + first <- withSpan (parseBlockIn ValueBlock) + hasElse <- peekText "else" + if not hasElse + then pure (condition, first, Nothing, Nothing) + else do + _ <- keyword "else" + chain <- peekText "if" + if chain + then do + next <- parseIf + pure (condition, first, Nothing, Just next) + else do + second <- withSpan (parseBlockIn ValueBlock) + pure (condition, first, Just second, Nothing) + closesBlock <- peekText "}" + case (second, chained) of + (Just (secondBlock, secondSpan), Nothing) + | closesBlock && (blockEndsWithValue (fst first) || blockEndsWithValue secondBlock) -> + pure + ( ExpressionStatement + spanValue + ( ConditionalExpression + spanValue + condition + (BlockExpression (snd first) (fst first) ()) + (BlockExpression secondSpan secondBlock ()) + () + ) + False + ) + _ -> do + whenTrue <- statementBlock (fst first) + whenFalse <- case (second, chained) of + (Just (secondBlock, _), _) -> Just <$> statementBlock secondBlock + (Nothing, Just next) -> pure (Just (Block [next])) + (Nothing, Nothing) -> pure Nothing + pure (IfStatement spanValue condition whenTrue whenFalse) + +-- | Whether the last item of a block is an expression without a semicolon. +blockEndsWithValue :: Block name annotation -> Bool +blockEndsWithValue (Block statements) = case reverse statements of + ExpressionStatement _ _ False : _ -> True + _ -> False + +{- | The statement form of a block that was parsed as a value block. + +An @if@ or a @match@ in last position was read as the value of the block; +as statements they need no terminator, so they become statements again. Any +other expression in last position is missing its semicolon. +-} +statementBlock :: Block Identifier () -> P (Block Identifier ()) +statementBlock (Block statements) = case reverse statements of + ExpressionStatement spanValue value False : before -> do + final <- case value of + MatchExpression {} -> pure (ExpressionStatement spanValue value True) + ConditionalExpression _ condition (BlockExpression _ first _) (BlockExpression _ second _) _ -> + IfStatement spanValue condition <$> statementBlock first <*> (Just <$> statementBlock second) + _ -> failAt (endOf spanValue) "VXP0006" "expected \";\"" + pure (Block (reverse (final : before))) + _ -> pure (Block statements) + where + endOf spanValue = spanValue {sourceStart = sourceEnd spanValue} + -- Loops are retained as structured syntax until the Desugarer. Keeping their -- delimiters and header clauses explicit lets the type checker validate loop -- scope before CorePrep assigns basic-block identities. @@ -1033,6 +1167,11 @@ parsePostfix = parsePrimary >>= calls (member, memberSpan) <- identifier let selected = MemberAccessExpression (mergeSpan (expressionSpan callee) memberSpan) callee member () calls selected + -- `::` names a method without calling it. + Just token | tokenText token == "::" -> do + _ <- symbol "::" + (member, memberSpan) <- identifier + calls (MethodReferenceExpression (mergeSpan (expressionSpan callee) memberSpan) callee member ()) Just token | tokenText token == "(" -> do _ <- symbol "(" arguments <- separated "," parseExpression @@ -1053,6 +1192,19 @@ parsePrimary = do next <- peekToken case next of Just token | tokenKind token == SymbolToken && tokenText token `elem` ["\\", "["] -> parseCallable + -- `.Member` names a member of the enum that the context expects. + -- It is a member selection without a receiver; the place of the + -- receiver holds the unit literal, which no source can write. + Just token | tokenKind token == SymbolToken && tokenText token == "." -> do + _ <- takeToken + (member, memberSpan) <- identifier + pure + ( MemberAccessExpression + (mergeSpan (tokenSpan token) memberSpan) + (LiteralExpression (tokenSpan token) UnitLiteral ()) + member + () + ) -- A loop in operand position is a loop expression: its value is -- supplied by `break value;`. The statement parsers are reused, so -- both forms have exactly one grammar. @@ -1263,6 +1415,7 @@ expressionSpan expression = case expression of NameExpression value _ _ -> value LiteralExpression value _ _ -> value MemberAccessExpression value _ _ _ -> value + MethodReferenceExpression value _ _ _ -> value CallExpression value _ _ _ -> value UnaryExpression value _ _ _ -> value BinaryExpression value _ _ _ _ -> value diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser/Match.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser/Match.hs index 13b87c9e..a4d482c4 100644 --- a/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser/Match.hs +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/Parser/Match.hs @@ -125,9 +125,12 @@ parseArms grammar = do {- | One arm: @patterns [if guard] -> body [,]@. -The comma after a block body is optional. After an expression body it is -required unless the arm is the last one: without it, the parenthesized -pattern of the next arm would continue the expression as a call. +The comma is optional after every body, as in the grammar. An expression +body is parsed like any expression, so a parenthesized pattern that follows +it without a comma is read as the argument list of a call of the body. That +reading is the grammar's own: the arrow that then follows cannot continue an +expression, and the diagnostic for it says what happened instead of naming +the arrow as unexpected. -} parseArm :: BranchGrammar -> P (MatchArm Identifier ()) parseArm grammar = do @@ -144,10 +147,13 @@ parseArm grammar = do blockBody <- peekText "{" body <- if blockBody then valueBlock grammar else grammarExpression grammar separated <- optionalSymbol "," - closes <- peekText "}" - if blockBody || separated || closes - then pure () - else failCurrent "VXP0038" "a match arm with an expression body must be followed by ',' or '}'" + swallowed <- peekText "->" + if not blockBody && not separated && swallowed + then + failCurrent + "VXP0038" + "the pattern of this arm was read as part of the previous arm's body; write ',' after that body" + else pure () pure (patterns, guard, body) pure (MatchArm spanValue patterns guard body) @@ -161,6 +167,10 @@ parsePatterns grammar = do A bare name is not a pattern: a binding always states its type, as in @int value@, so a name can never be mistaken for a constant to compare with. +A numeric literal may be preceded by a minus sign, which makes the constant +negative; the range of the constant is checked against the subject's type by +the type checker, so the most negative value of a signed type is a pattern +although its magnitude alone is not a value of that type. -} parseMatchPattern :: BranchGrammar -> P (MatchPattern Identifier ()) parseMatchPattern grammar = do @@ -189,6 +199,17 @@ parseBarePattern grammar = do | isLiteralStart token -> do (literal, _) <- grammarLiteral grammar pure (\spanValue -> MatchLiteralPattern spanValue literal ()) + -- A minus sign belongs to the numeric literal that follows it. + -- It is part of the constant, not an operator: no other + -- expression is a pattern. + | tokenKind token == SymbolToken && tokenText token == "-" -> do + _ <- takeToken + numeric <- peekToken + case numeric of + Just following | tokenKind following `elem` [IntegerToken, FloatingToken] -> do + (literal, _) <- grammarLiteral grammar + pure (\spanValue -> MatchLiteralPattern spanValue (negated literal) ()) + _ -> failCurrent "VXP0036" "a '-' in a match pattern must be followed by a numeric literal" _ -> do typed <- matchesAhead (grammarType grammar >> identifierToken) if not typed @@ -202,6 +223,10 @@ parseBarePattern grammar = do let binding = if isWildcard name then Nothing else Just (Identifier (tokenText name)) pure (\spanValue -> MatchTypePattern spanValue syntax binding ()) where + negated literal = case literal of + IntegerLiteral value -> IntegerLiteral (negate value) + FloatingLiteral spelling -> FloatingLiteral ('-' : spelling) + other -> other isWildcard token = tokenKind token == IdentifierToken && tokenText token == "_" isLiteralStart token = tokenKind token `elem` [IntegerToken, FloatingToken, CharacterToken, StringToken] diff --git a/Compiler/Haskell/Syntax/src/Visual/XSharp/RuntimeCall.hs b/Compiler/Haskell/Syntax/src/Visual/XSharp/RuntimeCall.hs new file mode 100644 index 00000000..dd04bebe --- /dev/null +++ b/Compiler/Haskell/Syntax/src/Visual/XSharp/RuntimeCall.hs @@ -0,0 +1,310 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +{- | The catalog of runtime calls. + +A runtime call is an operation the compiler does not lower to instructions +but to a call of a function of the Visual X# runtime: joining two strings, +converting a number for output, writing to the console. Core and the stages +after it carry it as one operation whose first operand is an integer +literal, the identity of the function, and whose remaining operands are the +arguments. + +This module is what the frontend knows about each function: its identity, +the types it takes and returns, and whether it does something that can be +observed. The native stages have the same table in +@Visual\/XSharp\/Core\/RuntimeCall.hpp@. The identities are part of the +artifact formats: a function is never renumbered, and a new one takes the +next free number. The tests of both sides pin every row. + +The catalog stands in the syntax package because the type checker, which +chooses the function for a source construct, and Core, which verifies the +call, both need it and neither depends on the other. +-} +module Visual.XSharp.RuntimeCall + ( RuntimeFunction (..) + , RuntimeParameter (..) + , RuntimeResult (..) + , runtimeFunctions + , runtimeFunctionIdentity + , runtimeFunctionOf + , runtimeFunctionSymbol + , runtimeParameters + , runtimeResult + , runtimeResultType + , runtimeObservable + , runtimeAccepts + , runtimeName + , runtimeFunctionOfName + , builtinSystemSymbol + , builtinConsoleSymbol + , RuntimeDefect (..) + , runtimeCallDefect + , runtimeDefectText + -- * Conversion flags + , flagLeft + , flagZero + , flagPlus + , flagSpace + , flagAlternate + , flagGroup + , flagHexadecimal + , absent + -- * Console targets + , consoleOutput + , consoleOutputLine + , consoleError + , consoleErrorLine + ) where + +import Visual.XSharp.AST + +-- | A function of the runtime that generated code may call. +data RuntimeFunction + = -- | Two strings one after the other. + TextConcat + | -- | A signed integer in decimal. + TextFromSigned + | -- | An unsigned integer in decimal. + TextFromUnsigned + | -- | @true@ or @false@. + TextFromBool + | -- | The one character. + TextFromChar + | -- | @%d@ and @%x@ of a signed integer. + TextFormatSigned + | -- | @%u@ and @%x@ of an unsigned integer. + TextFormatUnsigned + | -- | @%f@. + TextFormatFloating + | -- | @%s@. + TextFormatString + | -- | @%c@. + TextFormatChar + | -- | @%n@: the line terminator of the platform. + TextNewline + | -- | Write a string to standard output or standard error. + ConsoleWrite + | -- | Whether two strings hold the same characters. + TextEquals + deriving (Bounded, Enum, Eq, Ord, Read, Show) + +{- | What an argument of a runtime function may be. A function takes a family +of types where the language gives the operation to all of them; the backend +widens the argument to the representation the runtime function is written +for, which changes no value. +-} +data RuntimeParameter + = -- | A signed integer of at most 64 bits. + SignedParameter + | -- | An unsigned integer of at most 64 bits. + UnsignedParameter + | -- | A floating-point number of at most 64 bits. + FloatingParameter + | -- | @bool@. + BoolParameter + | -- | @char@. + CharParameter + | -- | @String@. + TextParameter + | -- | @int@: flags, a width or a precision. + CountParameter + deriving (Eq, Ord, Read, Show) + +-- | What a runtime function returns. +data RuntimeResult + = -- | No value. + NoResult + | -- | A @String@ the caller owns. + TextResult + | -- | A @bool@. + TruthResult + deriving (Eq, Ord, Read, Show) + +-- | Every runtime function, in order of identity. +runtimeFunctions :: [RuntimeFunction] +runtimeFunctions = [minBound .. maxBound] + +-- | The identity the first operand of a call holds. +runtimeFunctionIdentity :: RuntimeFunction -> Integer +runtimeFunctionIdentity function = case function of + TextConcat -> 1 + TextFromSigned -> 2 + TextFromUnsigned -> 3 + TextFromBool -> 4 + TextFromChar -> 5 + TextFormatSigned -> 6 + TextFormatUnsigned -> 7 + TextFormatFloating -> 8 + TextFormatString -> 9 + TextFormatChar -> 10 + TextNewline -> 11 + ConsoleWrite -> 12 + TextEquals -> 13 + +-- | The function with the given identity, when there is one. +runtimeFunctionOf :: Integer -> Maybe RuntimeFunction +runtimeFunctionOf identity = lookup identity [(runtimeFunctionIdentity function, function) | function <- runtimeFunctions] + +-- | The symbol of the runtime library that implements the function. +runtimeFunctionSymbol :: RuntimeFunction -> String +runtimeFunctionSymbol function = case function of + TextConcat -> "vxs_text_concat" + TextFromSigned -> "vxs_text_from_signed" + TextFromUnsigned -> "vxs_text_from_unsigned" + TextFromBool -> "vxs_text_from_bool" + TextFromChar -> "vxs_text_from_char" + TextFormatSigned -> "vxs_text_format_signed" + TextFormatUnsigned -> "vxs_text_format_unsigned" + TextFormatFloating -> "vxs_text_format_floating" + TextFormatString -> "vxs_text_format_string" + TextFormatChar -> "vxs_text_format_char" + TextNewline -> "vxs_text_newline" + ConsoleWrite -> "vxs_console_write" + TextEquals -> "vxs_text_equals" + +-- | The arguments of the function, in order. +runtimeParameters :: RuntimeFunction -> [RuntimeParameter] +runtimeParameters function = case function of + TextConcat -> [TextParameter, TextParameter] + TextFromSigned -> [SignedParameter] + TextFromUnsigned -> [UnsignedParameter] + TextFromBool -> [BoolParameter] + TextFromChar -> [CharParameter] + TextFormatSigned -> conversion SignedParameter + TextFormatUnsigned -> conversion UnsignedParameter + TextFormatFloating -> conversion FloatingParameter + TextFormatString -> conversion TextParameter + TextFormatChar -> conversion CharParameter + TextNewline -> [] + ConsoleWrite -> [TextParameter, CountParameter] + TextEquals -> [TextParameter, TextParameter] + where + -- Flags, width and precision, and then the value. The value stands + -- last because that is the order of a format's arguments: a width + -- or a precision written as @*@ is the argument before the value, + -- and the operands of a call are evaluated in order. + conversion value = [CountParameter, CountParameter, CountParameter, value] + +-- | What the function returns. +runtimeResult :: RuntimeFunction -> RuntimeResult +runtimeResult function = case function of + ConsoleWrite -> NoResult + TextEquals -> TruthResult + _ -> TextResult + +-- | The type a call of the function has. +runtimeResultType :: RuntimeFunction -> Type +runtimeResultType function = case runtimeResult function of + NoResult -> unitType + TextResult -> stringType + TruthResult -> boolType + +{- | Whether a call does something that can be observed apart from its +result. Such a call is an effect: it happens where it is written, and it is +never removed, repeated or moved. +-} +runtimeObservable :: RuntimeFunction -> Bool +runtimeObservable function = function == ConsoleWrite + +-- | Whether an argument of the given type may stand where the parameter is. +runtimeAccepts :: RuntimeParameter -> Type -> Bool +runtimeAccepts parameter valueType = case parameter of + SignedParameter -> spelling `elem` ["byte", "short", "long", "int"] + UnsignedParameter -> spelling `elem` ["ubyte", "ushort", "ulong", "uint"] + FloatingParameter -> spelling `elem` ["sfloat", "lfloat", "float"] + BoolParameter -> spelling == "bool" + CharParameter -> spelling == "char" + TextParameter -> spelling == "String" + CountParameter -> spelling == "int" + where + spelling = case valueType of + NamedType (QualifiedName [Identifier name]) [] -> name + _ -> "" + +{- | The name a call of a runtime function has in the typed tree. + +The type checker rewrites a source construct that needs the runtime into a +call of this name, and the lowering to Core turns a call of it into a +runtime call. Its symbol is reserved: the renamer gives every declaration of +a program a positive symbol, so no source name is ever one of these. The +spelling begins with a character no identifier may contain. +-} +runtimeName :: RuntimeFunction -> ResolvedName +runtimeName function = + ResolvedName (SymbolId (negate (runtimeNameBase + fromInteger (runtimeFunctionIdentity function)))) (Identifier ('$' : runtimeFunctionSymbol function)) + +-- | The runtime function a name of the typed tree stands for, if it is one. +runtimeFunctionOfName :: ResolvedName -> Maybe RuntimeFunction +runtimeFunctionOfName name = + let value = negate (symbolIdValue (resolvedSymbol name)) - runtimeNameBase + in if value > 0 then runtimeFunctionOf (toInteger value) else Nothing + +runtimeNameBase :: Int +runtimeNameBase = 1000 + +{- | The symbols of the two names the language declares for every program: +@System@ and, because @System@ is imported implicitly, @Console@. A +declaration of the program with one of these names shadows it. +-} +builtinSystemSymbol, builtinConsoleSymbol :: SymbolId +builtinSystemSymbol = SymbolId (-2) +builtinConsoleSymbol = SymbolId (-3) + +-- | Why a runtime call is malformed. +data RuntimeDefect + = -- | No first operand, or one that names no function of the catalog. + DefectiveIdentity + | -- | The number of arguments is not the number the function takes. + DefectiveArity + | -- | An argument has a type its parameter does not accept. + DefectiveArgument + | -- | The call does not have the type the function returns. + DefectiveResult + deriving (Eq, Ord, Read, Show) + +{- | Check a runtime call: the function its first operand names, if it names +one, the types of the arguments after it, and the type of the call. +-} +runtimeCallDefect :: Maybe RuntimeFunction -> [Type] -> Type -> Maybe RuntimeDefect +runtimeCallDefect named arguments resultType = case named of + Nothing -> Just DefectiveIdentity + Just function + | length arguments /= length (runtimeParameters function) -> Just DefectiveArity + | not (and (zipWith runtimeAccepts (runtimeParameters function) arguments)) -> Just DefectiveArgument + | resultType /= runtimeResultType function -> Just DefectiveResult + | otherwise -> Nothing + +-- | What is wrong with a call, in the words every stage reports. +runtimeDefectText :: RuntimeDefect -> String +runtimeDefectText defect = case defect of + DefectiveIdentity -> "runtime call must begin with an integer literal that names a function of the runtime catalog" + DefectiveArity -> "runtime call has the wrong number of arguments for its function" + DefectiveArgument -> "runtime call argument has a type its function does not take" + DefectiveResult -> "runtime call does not have the type its function returns" + +{- | The flags of a conversion, combined by addition in the @flags@ argument +of the formatting functions. The values are those of @VXS_TEXT_FLAG_*@ in +the runtime header. +-} +flagLeft, flagZero, flagPlus, flagSpace, flagAlternate, flagGroup, flagHexadecimal :: Integer +flagLeft = 1 +flagZero = 2 +flagPlus = 4 +flagSpace = 8 +flagAlternate = 16 +flagGroup = 32 +flagHexadecimal = 64 + +-- | A width or a precision that a conversion does not have. +absent :: Integer +absent = -1 + +{- | Where a console write goes and whether a line ends after it: the second +argument of 'ConsoleWrite'. The values are those of @VXS_CONSOLE_*@. +-} +consoleOutput, consoleOutputLine, consoleError, consoleErrorLine :: Integer +consoleOutput = 0 +consoleOutputLine = 1 +consoleError = 2 +consoleErrorLine = 3 diff --git a/Compiler/Haskell/Syntax/visual-xsharp-syntax.cabal b/Compiler/Haskell/Syntax/visual-xsharp-syntax.cabal index 7c361078..a4ac70ad 100644 --- a/Compiler/Haskell/Syntax/visual-xsharp-syntax.cabal +++ b/Compiler/Haskell/Syntax/visual-xsharp-syntax.cabal @@ -2,7 +2,7 @@ cabal-version: 3.12 -- SPDX-FileCopyrightText: 2026 Progmasoft -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 name: visual-xsharp-syntax -version: 0.4.1 +version: 0.5.0 synopsis: Visual X# source model, lexer, parser, and diagnostics license: MPL-2.0 author: Progmasoft @@ -17,8 +17,11 @@ library Visual.XSharp.Diagnostic.Protocol Visual.XSharp.FloatingLiteral Visual.XSharp.Lexer + Visual.XSharp.Completion + Visual.XSharp.NestingLimits Visual.XSharp.NumericLiteral Visual.XSharp.Parser + Visual.XSharp.RuntimeCall Visual.XSharp.SourceText Visual.XSharp.SourceWarnings other-modules: diff --git a/Compiler/Headers/Visual/XSharp/Core/BUILD.bazel b/Compiler/Headers/Visual/XSharp/Core/BUILD.bazel index 42b3eb1b..82194fd8 100644 --- a/Compiler/Headers/Visual/XSharp/Core/BUILD.bazel +++ b/Compiler/Headers/Visual/XSharp/Core/BUILD.bazel @@ -4,7 +4,10 @@ package(default_visibility = ["//visibility:public"]) cc_library( name = "coreprep_model", - hdrs = ["CorePrep.hpp"], + hdrs = [ + "CorePrep.hpp", + "RuntimeCall.hpp", + ], strip_include_prefix = "/Compiler/Headers", deps = ["//Compiler/Headers/Visual/XSharp:namespace"], ) diff --git a/Compiler/Headers/Visual/XSharp/Core/CorePrep.hpp b/Compiler/Headers/Visual/XSharp/Core/CorePrep.hpp index 8006462f..dbc618b0 100644 --- a/Compiler/Headers/Visual/XSharp/Core/CorePrep.hpp +++ b/Compiler/Headers/Visual/XSharp/Core/CorePrep.hpp @@ -556,7 +556,16 @@ namespace visual_xsharp::core BitwiseXor, ///< Bitwise exclusive disjunction. BitwiseOr, ///< Bitwise inclusive disjunction. BitwiseNot, ///< Bitwise complement. - TypeIs ///< Runtime type test. + TypeIs, ///< Runtime type test. + /// A callable that remembers its result. The operand is a callable + /// without parameters; the result calls it at most once, at its + /// own first call, and returns what it returned from then on. + Memoize, + /// A call of a function of the runtime. The first operand is an + /// integer literal, the identity of the function in the catalog of + /// `Visual/XSharp/Core/RuntimeCall.hpp`; the operands after it are + /// the arguments. + RuntimeCall }; /// One binding, assignment, or value-producing operation in a basic block. diff --git a/Compiler/Headers/Visual/XSharp/Core/CorePrep/Wire/Format.hpp b/Compiler/Headers/Visual/XSharp/Core/CorePrep/Wire/Format.hpp index 61c83254..f66d464e 100644 --- a/Compiler/Headers/Visual/XSharp/Core/CorePrep/Wire/Format.hpp +++ b/Compiler/Headers/Visual/XSharp/Core/CorePrep/Wire/Format.hpp @@ -11,7 +11,7 @@ namespace visual_xsharp::core::wire /// Four-byte identifier at the beginning of each CorePrep wire document. inline constexpr std::uint8_t magic[] = { 'V', 'X', 'C', 'P' }; /// Current CorePrep wire schema version. - inline constexpr std::uint16_t current_version = 6; + inline constexpr std::uint16_t current_version = 8; /// Resource ceilings checked while encoding or decoding CorePrep data. struct Limits final diff --git a/Compiler/Headers/Visual/XSharp/Core/IR.hpp b/Compiler/Headers/Visual/XSharp/Core/IR.hpp index 259c13c7..dbccca57 100644 --- a/Compiler/Headers/Visual/XSharp/Core/IR.hpp +++ b/Compiler/Headers/Visual/XSharp/Core/IR.hpp @@ -50,7 +50,16 @@ namespace Visual::XSharp::Core BitwiseXor, ///< Bitwise exclusive disjunction. BitwiseOr, ///< Bitwise inclusive disjunction. BitwiseNot, ///< Bitwise complement. - TypeIs ///< Runtime type test. + TypeIs, ///< Runtime type test. + /// A callable that remembers its result. The operand is a callable + /// without parameters; the result calls it at most once, at its + /// own first call, and returns what it returned from then on. + Memoize, + /// A call of a function of the runtime. The first operand is an + /// integer literal, the identity of the function in the catalog of + /// `Visual/XSharp/Core/RuntimeCall.hpp`; the operands after it are + /// the arguments. + RuntimeCall }; /// Closure capture ownership contract shared with CorePrep. @@ -133,6 +142,31 @@ namespace Visual::XSharp::Core /// Structured closure body; null outside closure expressions. std::shared_ptr> closureBody; + /// Construct the default value of every member. + Expression() = default; + /// Copy every member; nested values are copied recursively. + Expression(const Expression &) = default; + /// Move every member. + Expression(Expression &&) = default; + /// Copy-assign every member. + /// @return This value. + auto + operator=(const Expression &) -> Expression & = default; + /// Move-assign every member. + /// @return This value. + auto + operator=(Expression &&) -> Expression & = default; + /** + * @brief Release the operands without recursing once per operand + * level. + * + * A chain of operators nests as deep as it is long, in the first + * operand of each primitive. The operands are moved to a list and + * released from there, so the stack this uses does not grow with + * the length of a chain. + */ + ~Expression(); + /// Construct a typed reference to a resolved binding. /// @param name Resolved binding identity. /// @param valueType Static type of the binding. @@ -259,6 +293,31 @@ namespace Visual::XSharp::Core /// Update region executed after each For body iteration. std::vector loopUpdate; + /// Construct the default value of every member. + Statement() = default; + /// Copy every member; nested values are copied recursively. + Statement(const Statement &) = default; + /// Move every member. + Statement(Statement &&) = default; + /// Copy-assign every member. + /// @return This value. + auto + operator=(const Statement &) -> Statement & = default; + /// Move-assign every member. + /// @return This value. + auto + operator=(Statement &&) -> Statement & = default; + /** + * @brief Release the nested statements without recursing once per + * level. + * + * An `else if` chain nests as deep as it is long. The branches and + * loop regions are moved to a list and released from there, so the + * stack this uses does not grow with the length of a chain or the + * depth of nesting. + */ + ~Statement(); + /// Introduce a local binding. /// @param value Binding and initializer to add. /// @return A Bind statement containing value. diff --git a/Compiler/Headers/Visual/XSharp/Core/RuntimeCall.hpp b/Compiler/Headers/Visual/XSharp/Core/RuntimeCall.hpp new file mode 100644 index 00000000..c4cbad91 --- /dev/null +++ b/Compiler/Headers/Visual/XSharp/Core/RuntimeCall.hpp @@ -0,0 +1,354 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep.hpp" + +/// The catalog of runtime calls. +/// +/// A runtime call is an operation that the compiler does not lower to +/// instructions but to a call of a function of the Visual X# runtime: +/// joining two strings, converting a number for output, writing to the +/// console. Core, CorePrep, Xpp and Xmm each carry it as one operation whose +/// first operand is a literal, the identity of the function, and whose +/// remaining operands are the arguments. This header is what every native +/// stage knows about each function: its identity, the symbol of the runtime +/// that implements it, the types it takes and returns, and whether it does +/// something that can be observed. +/// +/// The identities are part of the artifact formats. A function is never +/// renumbered; a new one takes the next free number. The Haskell frontend +/// has the same table in `Visual.XSharp.Core.RuntimeCall`, and the tests of +/// both sides pin each row. +namespace visual_xsharp::core::runtime +{ + /// What an argument of a runtime function may be. + /// + /// A function takes a family of types where the language gives the + /// operation to all of them. The backend widens the argument to the one + /// representation the runtime function is written for, which changes + /// no value. + enum class Parameter : std::uint8_t + { + Signed, ///< A signed integer of at most 64 bits. + Unsigned, ///< An unsigned integer of at most 64 bits. + Floating, ///< A floating-point number of at most 64 bits. + Bool, ///< `bool`. + Char, ///< `char`. + Text, ///< `String`. + Count ///< `int`: flags, a width or a precision. + }; + + /// What a runtime function returns. + enum class Result : std::uint8_t + { + Nothing, ///< No value. + Text, ///< A `String` the caller owns. + Truth ///< A `bool`. + }; + + /// The identity of a runtime function, as the first operand carries it. + enum class Function : std::uint8_t + { + TextConcat = 1, ///< Two strings one after the other. + TextFromSigned = 2, ///< A signed integer in decimal. + TextFromUnsigned = 3, ///< An unsigned integer in decimal. + TextFromBool = 4, ///< `true` or `false`. + TextFromChar = 5, ///< The one character. + TextFormatSigned = 6, ///< `%d` and `%x` of a signed integer. + TextFormatUnsigned = 7, ///< `%u` and `%x` of an unsigned integer. + TextFormatFloating = 8, ///< `%f`. + TextFormatString = 9, ///< `%s`. + TextFormatChar = 10, ///< `%c`. + TextNewline = 11, ///< The line terminator of the platform. + ConsoleWrite = 12, ///< Write a string to a standard stream. + TextEquals = 13 ///< Whether two strings hold the same text. + }; + + /// One row of the catalog. + struct Signature final + { + /// The identity the first operand holds. + Function function{ Function::TextConcat }; + /// The symbol of the runtime library that implements the function. + std::string_view symbol; + /// The arguments, in order. + std::span parameters; + /// What the function returns. + Result result{}; + /// Whether a call does something that can be observed apart from + /// its result. Such a call is never removed, repeated or moved. + bool observable{}; + }; + + namespace detail + { + inline constexpr std::array kTwoTexts{ Parameter::Text, + Parameter::Text }; + inline constexpr std::array kSigned{ Parameter::Signed }; + inline constexpr std::array kUnsigned{ Parameter::Unsigned }; + inline constexpr std::array kBool{ Parameter::Bool }; + inline constexpr std::array kChar{ Parameter::Char }; + // Flags, width and precision, and then the value. The value stands + // last because that is the order of a format's arguments: a width + // or a precision written as `*` is the argument before the value, + // and the operands of a call are evaluated in order. + inline constexpr std::array kFormatSigned{ Parameter::Count, + Parameter::Count, + Parameter::Count, + Parameter::Signed }; + inline constexpr std::array kFormatUnsigned{ Parameter::Count, + Parameter::Count, + Parameter::Count, + Parameter::Unsigned }; + inline constexpr std::array kFormatFloating{ Parameter::Count, + Parameter::Count, + Parameter::Count, + Parameter::Floating }; + inline constexpr std::array kFormatText{ Parameter::Count, + Parameter::Count, + Parameter::Count, + Parameter::Text }; + inline constexpr std::array kFormatChar{ Parameter::Count, + Parameter::Count, + Parameter::Count, + Parameter::Char }; + inline constexpr std::array kWrite{ Parameter::Text, Parameter::Count }; + + inline constexpr std::array kCatalog{ { + { Function::TextConcat, + "vxs_text_concat", + kTwoTexts, + Result::Text, + false }, + { Function::TextFromSigned, + "vxs_text_from_signed", + kSigned, + Result::Text, + false }, + { Function::TextFromUnsigned, + "vxs_text_from_unsigned", + kUnsigned, + Result::Text, + false }, + { Function::TextFromBool, + "vxs_text_from_bool", + kBool, + Result::Text, + false }, + { Function::TextFromChar, + "vxs_text_from_char", + kChar, + Result::Text, + false }, + { Function::TextFormatSigned, + "vxs_text_format_signed", + kFormatSigned, + Result::Text, + false }, + { Function::TextFormatUnsigned, + "vxs_text_format_unsigned", + kFormatUnsigned, + Result::Text, + false }, + { Function::TextFormatFloating, + "vxs_text_format_floating", + kFormatFloating, + Result::Text, + false }, + { Function::TextFormatString, + "vxs_text_format_string", + kFormatText, + Result::Text, + false }, + { Function::TextFormatChar, + "vxs_text_format_char", + kFormatChar, + Result::Text, + false }, + { Function::TextNewline, + "vxs_text_newline", + {}, + Result::Text, + false }, + { Function::ConsoleWrite, + "vxs_console_write", + kWrite, + Result::Nothing, + true }, + { Function::TextEquals, + "vxs_text_equals", + kTwoTexts, + Result::Truth, + false }, + } }; + } // namespace detail + + /// Every runtime function, in order of identity. + /// @return The rows of the catalog. + [[nodiscard]] constexpr auto + Catalog() noexcept -> std::span + { + return detail::kCatalog; + } + + /// The row of the function with the given identity, or null when no + /// function has it. + /// @param identity The number a first operand holds. + /// @return The row, or null. + [[nodiscard]] constexpr auto + Find(const std::uint64_t identity) noexcept -> const Signature * + { + for (const auto &signature : detail::kCatalog) + if (static_cast(signature.function) == identity) + return &signature; + return nullptr; + } + + /// Whether an argument of the given type may stand where the parameter + /// is. + /// @param parameter What the function takes at that position. + /// @param kind The kind of the argument's type. + /// @return true when the function takes an argument of that type there. + [[nodiscard]] constexpr auto + Accepts(const Parameter parameter, const Type::Kind kind) noexcept -> bool + { + switch (parameter) + { + case Parameter::Signed: + return kind == Type::Kind::Int8 || kind == Type::Kind::Int16 + || kind == Type::Kind::Int32 + || kind == Type::Kind::Int64; + case Parameter::Unsigned: + return kind == Type::Kind::UInt8 || kind == Type::Kind::UInt16 + || kind == Type::Kind::UInt32 + || kind == Type::Kind::UInt64; + case Parameter::Floating: + return kind == Type::Kind::Float16 + || kind == Type::Kind::Float32 + || kind == Type::Kind::Float64; + case Parameter::Bool: + return kind == Type::Kind::Bool; + case Parameter::Char: + return kind == Type::Kind::Character; + case Parameter::Text: + return kind == Type::Kind::String; + case Parameter::Count: + return kind == Type::Kind::Int64; + } + return false; + } + + /// The type a call of the function has. + /// @param result What the function returns. + /// @return The kind of the type of a call. + [[nodiscard]] constexpr auto + ResultKind(const Result result) noexcept -> Type::Kind + { + return result == Result::Text ? Type::Kind::String + : result == Result::Truth ? Type::Kind::Bool + : Type::Kind::Unit; + } + + /// The identity a first operand names, or nothing when the operand is + /// not a literal `int` that is not negative and fits sixty-four bits. + /// The result still has to be looked up: not every number is a function. + /// @param literal The payload of the first operand. + /// @param type The type of the first operand. + /// @return The identity, or nothing. + [[nodiscard]] inline auto + IdentityOf(const Literal &literal, const Type &type) noexcept + -> std::optional + { + if (type.kind != Type::Kind::Int64) + return std::nullopt; + if (const auto *fixed = std::get_if(&literal)) + return *fixed < 0 ? std::nullopt + : std::optional( + static_cast(*fixed)); + const auto *integer = std::get_if(&literal); + if (integer == nullptr || integer->negative + || integer->magnitude.size() > 8U) + return std::nullopt; + std::uint64_t value = 0U; + for (std::size_t index = integer->magnitude.size(); index-- > 0U;) + value = (value << 8U) | integer->magnitude[index]; + return value; + } + + /// Why a runtime call is malformed, when it is. + enum class Defect : std::uint8_t + { + /// The call is well formed. + None, + /// There is no first operand, or it is not an integer literal that + /// names a function of the catalog. + Identity, + /// The number of arguments is not the number the function takes. + Arity, + /// An argument has a type its parameter does not accept. + Argument, + /// The call does not have the type the function returns. + Result + }; + + /// Checks the types of a runtime call against its row. + /// + /// Each stage finds the row from the literal its own representation + /// holds and then calls this with the kinds of the arguments, so the + /// rule is written once. + /// @param signature The row the first operand names, or null. + /// @param arguments The kinds of the types of the operands after the + /// first. + /// @param result The kind of the type of the call. + /// @return What is wrong with the call, or Defect::None. + [[nodiscard]] constexpr auto + Check(const Signature *signature, + const std::span arguments, + const Type::Kind result) noexcept -> Defect + { + if (signature == nullptr) + return Defect::Identity; + if (arguments.size() != signature->parameters.size()) + return Defect::Arity; + for (std::size_t index = 0U; index < arguments.size(); ++index) + if (!Accepts(signature->parameters[index], arguments[index])) + return Defect::Argument; + return result == ResultKind(signature->result) ? Defect::None + : Defect::Result; + } + + /// What is wrong with a call, in the words every stage reports. + /// @param defect The defect to describe. + /// @return A sentence without a stage name or a diagnostic code. + [[nodiscard]] constexpr auto + Describe(const Defect defect) noexcept -> std::string_view + { + switch (defect) + { + case Defect::None: + return "runtime call is well formed"; + case Defect::Identity: + return "runtime call must begin with an integer literal that " + "names a function of the runtime catalog"; + case Defect::Arity: + return "runtime call has the wrong number of arguments for " + "its function"; + case Defect::Argument: + return "runtime call argument has a type its function does " + "not take"; + case Defect::Result: + return "runtime call does not have the type its function " + "returns"; + } + return {}; + } +} // namespace visual_xsharp::core::runtime diff --git a/Compiler/Headers/Visual/XSharp/Core/Wire.hpp b/Compiler/Headers/Visual/XSharp/Core/Wire.hpp index 6d455368..d9351200 100644 --- a/Compiler/Headers/Visual/XSharp/Core/Wire.hpp +++ b/Compiler/Headers/Visual/XSharp/Core/Wire.hpp @@ -14,7 +14,7 @@ namespace Visual::XSharp::Core::Wire { /// Current VXCR schema version. Decoders require an exact match. - inline constexpr std::uint16_t kCurrentVersion = 8; + inline constexpr std::uint16_t kCurrentVersion = 10; /// Per-call resource ceilings for encoding and decoding. /// Untrusted input is checked against these bounds before allocation. @@ -36,6 +36,10 @@ namespace Visual::XSharp::Core::Wire std::size_t maximumTypeDepth{ 128U }; /// Maximum recursive expression nesting depth. std::size_t maximumExpressionDepth{ 4096U }; + /// Maximum nesting depth of statement bodies: function, branch, + /// loop and closure bodies. The links of an `else if` chain share + /// one level. The frontend limits source nesting far below this. + std::size_t maximumStatementDepth{ 4096U }; /// Maximum encoded magnitude bytes in one numeric literal. std::size_t maximumNumericBytes{ 4096U }; }; diff --git a/Compiler/Headers/Visual/XSharp/Runtime/AARC.hpp b/Compiler/Headers/Visual/XSharp/Runtime/AARC.hpp index 09eab101..f96d001d 100644 --- a/Compiler/Headers/Visual/XSharp/Runtime/AARC.hpp +++ b/Compiler/Headers/Visual/XSharp/Runtime/AARC.hpp @@ -53,6 +53,17 @@ namespace Visual::XSharp::Runtime::Aarc [[nodiscard]] auto Allocate(const TypeMetadata &metadata) noexcept -> void *; + /** Count the allocations whose storage has not been reclaimed. + * + * An allocation is reclaimed when its last strong, weak and unowned + * handle is gone. A program that balances its references returns this + * count to the value it had before the program ran, which is how tests + * observe a missing release on every platform. The count is not part of + * the C ABI. + */ + [[nodiscard]] auto + LiveAllocations() noexcept -> std::uint64_t; + /** Retain the live object, returning null if its strong lifetime has ended. */ auto @@ -102,4 +113,29 @@ namespace Visual::XSharp::Runtime::Aarc [[nodiscard]] auto IsExactType(const void *object, std::uint64_t typeIdentity) noexcept -> bool; + + /** The scalars of a string object, borrowed for as long as the string + * is alive. */ + struct StringScalars final + { + /** The first scalar, or null for the empty string. */ + const char32_t *scalars{}; + /** How many scalars the string holds. */ + std::size_t count{}; + }; + + /** Read the scalars of a live `System.String`. + * + * A null reference, and an object that is not a string, read as the + * empty string: the text routines of the runtime treat a string that is + * not there as one without characters. + */ + [[nodiscard]] auto + ViewString(const void *string) noexcept -> StringScalars; + + /** Create a `System.String` that holds a copy of the given scalars and + * return its initial strong owner, or null when the scalars are not + * Unicode scalar values or memory is exhausted. */ + [[nodiscard]] auto + MakeString(const char32_t *scalars, std::size_t count) noexcept -> void *; } // namespace Visual::XSharp::Runtime::Aarc diff --git a/Compiler/Headers/Visual/XSharp/Runtime/BUILD.bazel b/Compiler/Headers/Visual/XSharp/Runtime/BUILD.bazel index 1035ec61..f8fdd0c3 100644 --- a/Compiler/Headers/Visual/XSharp/Runtime/BUILD.bazel +++ b/Compiler/Headers/Visual/XSharp/Runtime/BUILD.bazel @@ -14,3 +14,16 @@ cc_library( strip_include_prefix = "/Compiler/Headers", deps = [":aarc_c_api"], ) + +cc_library( + name = "text_c_api", + hdrs = ["Text.h"], + strip_include_prefix = "/Compiler/Headers", +) + +cc_library( + name = "text_api", + hdrs = ["Text.hpp"], + strip_include_prefix = "/Compiler/Headers", + deps = [":text_c_api"], +) diff --git a/Compiler/Headers/Visual/XSharp/Runtime/Text.h b/Compiler/Headers/Visual/XSharp/Runtime/Text.h new file mode 100644 index 00000000..1ab9caa4 --- /dev/null +++ b/Compiler/Headers/Visual/XSharp/Runtime/Text.h @@ -0,0 +1,162 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#ifndef VISUAL_XSHARP_RUNTIME_TEXT_H +#define VISUAL_XSHARP_RUNTIME_TEXT_H + +#include +#include +#include + +/* The text and console entry points of the Visual X# runtime. + * + * Generated code calls these functions for string concatenation, for the + * conversions of `Console.Printf` and `Console.Format`, and for console + * output. Each is one member of the runtime-call catalog of the compiler; + * `Visual/XSharp/Core/RuntimeCall.hpp` gives the member that names it. + * + * Ownership follows the convention of every generated call: a string + * argument is borrowed for the duration of the call, and a string result is + * returned with one strong reference that the caller releases. A null + * string argument is the empty string. A function that cannot allocate its + * result stops the program; none returns null. + * + * A string is a sequence of Unicode scalar values. Width, precision and + * padding count scalar values, not bytes and not display columns. + */ + +#ifdef __cplusplus +# define VXS_TEXT_NOEXCEPT noexcept +extern "C" +{ +#else +# define VXS_TEXT_NOEXCEPT +#endif + +/* A conversion takes its flags, its width and its precision before the value + * it converts. That is the order in which the arguments of a format stand: a + * width or a precision written as `*` is given before the value, and the + * operands of a call are evaluated in order. */ + +/* Flags of a conversion, combined in the `flags` argument of the + * `vxs_text_format_*` functions. The frontend rejects the combinations the + * language does not define, so the runtime gives the remaining ones the + * meaning stated here and ignores a flag that has none for a conversion. */ + +/** `-`: the value stands at the left of its field. */ +#define VXS_TEXT_FLAG_LEFT INT64_C(1) +/** `0`: a number is padded with zeros after its sign and prefix. */ +#define VXS_TEXT_FLAG_ZERO INT64_C(2) +/** `+`: a number that is not negative is written with a plus sign. */ +#define VXS_TEXT_FLAG_PLUS INT64_C(4) +/** space: a number that is not negative is written with a leading space. */ +#define VXS_TEXT_FLAG_SPACE INT64_C(8) +/** `#`: a hexadecimal number is written with the prefix `0x`. */ +#define VXS_TEXT_FLAG_ALTERNATE INT64_C(16) +/** `'`: the integer digits of a decimal number are grouped in threes with + * apostrophes. */ +#define VXS_TEXT_FLAG_GROUP INT64_C(32) +/** `%x`: an integer is written in hexadecimal with lowercase digits. */ +#define VXS_TEXT_FLAG_HEXADECIMAL INT64_C(64) + +/** A width or a precision that the conversion does not have. */ +#define VXS_TEXT_ABSENT INT64_C(-1) + +/* Where `vxs_console_write` writes, and whether a line ends after it. */ + +/** Standard output. */ +#define VXS_CONSOLE_OUTPUT INT64_C(0) +/** Standard output, followed by the line terminator of the platform. */ +#define VXS_CONSOLE_OUTPUT_LINE INT64_C(1) +/** Standard error. */ +#define VXS_CONSOLE_ERROR INT64_C(2) +/** Standard error, followed by the line terminator of the platform. */ +#define VXS_CONSOLE_ERROR_LINE INT64_C(3) + + /** Whether the two strings hold the same scalar values in the same + * order. Strings are compared by what they hold, never by where they are + * kept. */ + bool + vxs_text_equals(void *left, void *right) VXS_TEXT_NOEXCEPT; + + /** The two strings one after the other. */ + void * + vxs_text_concat(void *left, void *right) VXS_TEXT_NOEXCEPT; + + /** A signed integer in decimal, with a minus sign when negative. */ + void * + vxs_text_from_signed(int64_t value) VXS_TEXT_NOEXCEPT; + + /** An unsigned integer in decimal. */ + void * + vxs_text_from_unsigned(uint64_t value) VXS_TEXT_NOEXCEPT; + + /** `true` or `false`. */ + void * + vxs_text_from_bool(bool value) VXS_TEXT_NOEXCEPT; + + /** The one character. A value that is not a Unicode scalar value is + * written as U+FFFD. */ + void * + vxs_text_from_char(uint32_t value) VXS_TEXT_NOEXCEPT; + + /** `%d` and `%x` of a signed integer. A negative number is its minus + * sign and its magnitude in either base; it is never the bit pattern of + * its representation. */ + void * + vxs_text_format_signed(int64_t flags, + int64_t width, + int64_t precision, + int64_t value) VXS_TEXT_NOEXCEPT; + + /** `%u` and `%x` of an unsigned integer. */ + void * + vxs_text_format_unsigned(int64_t flags, + int64_t width, + int64_t precision, + uint64_t value) VXS_TEXT_NOEXCEPT; + + /** `%f`: the decimal expansion of the value, correctly rounded to + * `precision` digits after the point, six when absent. The digits are + * those of the exact binary value; a tie rounds to the even digit. No + * locale changes the point or adds grouping. A value that is not + * finite is `nan`, `inf` or `-inf`. */ + void * + vxs_text_format_floating(int64_t flags, + int64_t width, + int64_t precision, + double value) VXS_TEXT_NOEXCEPT; + + /** `%s`: at most `precision` characters of the string when a precision + * is present. */ + void * + vxs_text_format_string(int64_t flags, + int64_t width, + int64_t precision, + void *value) VXS_TEXT_NOEXCEPT; + + /** `%c`. */ + void * + vxs_text_format_char(int64_t flags, + int64_t width, + int64_t precision, + uint32_t value) VXS_TEXT_NOEXCEPT; + + /** `%n`: the line terminator of the platform, carriage return and line + * feed on Windows and line feed elsewhere. */ + void * + vxs_text_newline(void) VXS_TEXT_NOEXCEPT; + + /** Write a string to a standard stream as UTF-8, or as UTF-16 to a + * Windows console. `target` is one of the `VXS_CONSOLE_*` values. The + * bytes reach the stream before the function returns: the runtime keeps + * no buffer of its own. A stream that cannot be written to is ignored. + */ + void + vxs_console_write(void *text, int64_t target) VXS_TEXT_NOEXCEPT; + +#ifdef __cplusplus +} +#endif + +#undef VXS_TEXT_NOEXCEPT +#endif diff --git a/Compiler/Headers/Visual/XSharp/Runtime/Text.hpp b/Compiler/Headers/Visual/XSharp/Runtime/Text.hpp new file mode 100644 index 00000000..db15575d --- /dev/null +++ b/Compiler/Headers/Visual/XSharp/Runtime/Text.hpp @@ -0,0 +1,35 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include + +#include "Visual/XSharp/Runtime/Text.h" + +namespace Visual::XSharp::Runtime::Console +{ + /** Receives the bytes of one console write. + * + * `stream` is `VXS_CONSOLE_OUTPUT` or `VXS_CONSOLE_ERROR`; a line + * terminator the write asked for is part of the bytes. The bytes are + * UTF-8 and are valid only for the duration of the call. + */ + using Sink = void (*)(std::int64_t stream, + const char *bytes, + std::size_t count, + void *context) noexcept; + + /** Send console output to a function instead of the standard streams. + * + * A process that hosts generated code, such as a test or the + * interactive shell, uses this to read what a program wrote. Passing + * null restores the standard streams. The sink is not part of the C + * ABI, and a native executable never sets one. + * + * @param sink The receiver, or null. + * @param context Passed to every call of the sink unchanged. + */ + void + SetSink(Sink sink, void *context) noexcept; +} // namespace Visual::XSharp::Runtime::Console diff --git a/Compiler/Headers/Visual/XSharp/Support/BUILD.bazel b/Compiler/Headers/Visual/XSharp/Support/BUILD.bazel new file mode 100644 index 00000000..55142089 --- /dev/null +++ b/Compiler/Headers/Visual/XSharp/Support/BUILD.bazel @@ -0,0 +1,12 @@ +load("@rules_cc//cc:cc_library.bzl", "cc_library") + +package(default_visibility = ["//visibility:public"]) + +# The stack the compiler runs on. The frontend's nesting limits are stated +# against its size, so every program that hosts the pipeline uses it. Depend +# on //Compiler/Support:compiler_stack, which carries the implementation. +cc_library( + name = "compiler_stack_api", + hdrs = ["CompilerStack.hpp"], + strip_include_prefix = "/Compiler/Headers", +) diff --git a/Compiler/Headers/Visual/XSharp/Support/CompilerStack.hpp b/Compiler/Headers/Visual/XSharp/Support/CompilerStack.hpp new file mode 100644 index 00000000..13e15bd8 --- /dev/null +++ b/Compiler/Headers/Visual/XSharp/Support/CompilerStack.hpp @@ -0,0 +1,139 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include +#include +#include +#include + +namespace Visual::XSharp::Support +{ + /** + * @brief Stack size, in bytes, of a thread that runs the compiler. + * + * The stages that walk Core recurse once per level of nesting in the + * program, so how deep a program may nest is decided by the stack of the + * thread that compiles it. The stack a process starts with differs by + * platform: one megabyte on Windows, eight on Linux and macOS, and half + * a megabyte for secondary threads on macOS. A limit that is safe on + * one of them would be either a crash on another or needlessly small + * everywhere. The compiler therefore runs on a thread whose stack it + * chooses itself, and the nesting limits of the frontend are stated + * against this size. + * + * Only address space is reserved; pages are committed as they are + * touched, so an ordinary program pays for the stack it uses. + */ + inline constexpr std::size_t kCompilerStackBytes + = std::size_t{ 256U } * 1024U * 1024U; + + /** + * @brief Run a function on a new thread with the compiler stack and + * wait for it. + * + * The stack is reserved, not committed. The process is terminated with + * a report when the thread cannot be created: there is no smaller stack + * on which the caller's limits would still hold. + * + * @param function Called once on the new thread with `argument`. + * @param argument Passed through unchanged. + */ + void + RunOnCompilerStackRaw(void (*function)(void *), void *argument); + + /** + * @brief Run a function on a new thread with a stack of the given size + * and wait for it. + * + * This is the general form of RunOnCompilerStackRaw. Measuring tools use + * it to find how much stack a stage needs; nothing else should choose a + * stack size of its own. + * + * @param bytes Stack size to reserve; the platform may round it up. + * @param function Called once on the new thread with `argument`. + * @param argument Passed through unchanged. + */ + void + RunOnStackRaw(std::size_t bytes, void (*function)(void *), void *argument); + + /** + * @brief The stack the calling thread has committed so far, in bytes. + * + * A stack is reserved as address space and committed page by page as + * it is first touched. The committed part is therefore an upper bound on + * what the thread has used, rounded up to pages and including the guard + * page; it is not the exact number of bytes in use, and it does not + * include frames that a sanitizer keeps on a heap-allocated fake stack. + * Measuring tools call this at the end of the work they measure. + * + * On Windows the figure is the committed part of the stack allocation. + * On Linux and macOS it is the resident pages of the stack mapping, + * which are the pages the thread has touched; there it has no guard + * page in it, and it is zero for the initial thread of a process, whose + * stack is not one mapping. + * + * @return The committed bytes, or zero where this is not determined. + */ + [[nodiscard]] auto + CommittedStackBytes() -> std::size_t; + + /// Run a callable that returns nothing on a thread with a stack of the + /// given size. See RunOnStackRaw. + template + void + RunOnStack(std::size_t bytes, Callable &&callable) + { + RunOnStackRaw( + bytes, + [](void *context) { + (*static_cast *>(context))(); + }, + &callable); + } + + /** + * @brief Run a callable on a thread with the compiler stack and return + * its result. + * + * The caller waits for the thread, so the callable may refer to the + * caller's locals. The callable must not throw; first-party code is + * built without exceptions. + */ + template + auto + RunOnCompilerStack(Callable &&callable) -> std::invoke_result_t + { + using Result = std::invoke_result_t; + if constexpr (std::is_void_v) + { + RunOnCompilerStackRaw( + [](void *context) { + (*static_cast *>( + context))(); + }, + &callable); + } + else + { + struct Call final + { + std::remove_reference_t *callable; + std::optional result; + }; + Call call{ &callable, std::nullopt }; + RunOnCompilerStackRaw( + [](void *context) { + auto *self = static_cast(context); + self->result.emplace((*self->callable)()); + }, + &call); + // The thread was joined, so the callable has run and stored + // its result; anything else is a defect of the raw entry. + if (!call.result) + std::abort(); + return std::move(*call.result); + } + } +} // namespace Visual::XSharp::Support diff --git a/Compiler/Headers/Visual/XSharp/Xmm/IR.hpp b/Compiler/Headers/Visual/XSharp/Xmm/IR.hpp index 7154a202..f1db2f98 100644 --- a/Compiler/Headers/Visual/XSharp/Xmm/IR.hpp +++ b/Compiler/Headers/Visual/XSharp/Xmm/IR.hpp @@ -56,6 +56,15 @@ namespace visual_xsharp::xmm BitwiseOr, ///< Bitwise inclusive disjunction. BitwiseNot, ///< Bitwise complement. TypeIs, ///< Runtime type test. + /// A callable that remembers its result. The operand is a callable + /// without parameters; the result calls it at most once, at its + /// own first call, and returns what it returned from then on. + Memoize, + /// A call of a function of the runtime. The first operand is an + /// integer literal, the identity of the function in the catalog of + /// `Visual/XSharp/Core/RuntimeCall.hpp`; the operands after it are + /// the arguments. + RuntimeCall, // Source-compatible names for pre-v3 native clients. They intentionally // alias the typed operations; width and signedness now come from diff --git a/Compiler/Headers/Visual/XSharp/Xmm/Wire.hpp b/Compiler/Headers/Visual/XSharp/Xmm/Wire.hpp index b69f11f4..48987d33 100644 --- a/Compiler/Headers/Visual/XSharp/Xmm/Wire.hpp +++ b/Compiler/Headers/Visual/XSharp/Xmm/Wire.hpp @@ -13,7 +13,7 @@ namespace Visual::XSharp::Xmm::Wire { /// Current Xmm wire schema version; decoders require an exact match. - inline constexpr std::uint16_t kCurrentVersion = 5U; + inline constexpr std::uint16_t kCurrentVersion = 7U; /// Shared resource ceilings for Xmm wire operations. using Limits = Artifact::Wire::Limits; /// Structured wire failure with byte offset and field context. diff --git a/Compiler/Headers/Visual/XSharp/Xpp/BUILD.bazel b/Compiler/Headers/Visual/XSharp/Xpp/BUILD.bazel index d6b77c6a..d0a31f34 100644 --- a/Compiler/Headers/Visual/XSharp/Xpp/BUILD.bazel +++ b/Compiler/Headers/Visual/XSharp/Xpp/BUILD.bazel @@ -6,6 +6,7 @@ cc_library( name = "xpp", hdrs = [ "IR.hpp", + "OwnershipPlacement.hpp", "OwnershipVerifier.hpp", "Verifier.hpp", "Wire.hpp", diff --git a/Compiler/Headers/Visual/XSharp/Xpp/IR.hpp b/Compiler/Headers/Visual/XSharp/Xpp/IR.hpp index 8a399635..59ba3235 100644 --- a/Compiler/Headers/Visual/XSharp/Xpp/IR.hpp +++ b/Compiler/Headers/Visual/XSharp/Xpp/IR.hpp @@ -58,7 +58,16 @@ namespace visual_xsharp::xpp BitwiseXor, ///< Bitwise exclusive disjunction. BitwiseOr, ///< Bitwise inclusive disjunction. BitwiseNot, ///< Bitwise complement. - TypeIs ///< Runtime type test. + TypeIs, ///< Runtime type test. + /// A callable that remembers its result. The operand is a callable + /// without parameters; the result calls it at most once, at its + /// own first call, and returns what it returned from then on. + Memoize, + /// A call of a function of the runtime. The first operand is an + /// integer literal, the identity of the function in the catalog of + /// `Visual/XSharp/Core/RuntimeCall.hpp`; the operands after it are + /// the arguments. + RuntimeCall }; /// Typed input to an Xpp instruction. diff --git a/Compiler/Headers/Visual/XSharp/Xpp/OwnershipPlacement.hpp b/Compiler/Headers/Visual/XSharp/Xpp/OwnershipPlacement.hpp new file mode 100644 index 00000000..a0b59236 --- /dev/null +++ b/Compiler/Headers/Visual/XSharp/Xpp/OwnershipPlacement.hpp @@ -0,0 +1,53 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include "Visual/XSharp/Xpp/IR.hpp" + +namespace Visual::XSharp::Xpp +{ + /** + * @brief Make the ownership of AARC values explicit in an Xpp module. + * + * CorePrep names values and does not say who releases them. This pass + * gives every AARC value one owner and writes the operations that keep + * the reference counts balanced, in the vocabulary the ownership + * verifiers of Xpp and Xmm check. + * + * The convention is the same in every function, so functions agree + * without looking at each other: + * + * - a parameter is borrowed: the caller keeps it alive for the call; + * - a result is owned: the caller receives one reference and releases + * it; + * - a local owns the reference it holds, from the instruction that + * defines it to its last use on each path, where it is released; + * - a copy of a value that is used again takes a reference of its own, + * and a copy of a value that is not used again takes over its + * reference; + * - a closure takes its own reference to each strong capture when it is + * created, which its destructor releases. + * + * A value is released after its last use rather than at the end of a + * source scope: Xpp has no scopes, and a value that is live at a point + * has been defined on every path to it, so the release needs no test of + * whether there is anything to release. Where a value dies on one edge + * of a branch and lives on the other, the release stands on the edge, + * in a block of its own when the target has other predecessors. + * + * A method that is used as a value, not called, is a closure without + * captures. The pass creates that closure where the value is used, + * because a method has no closure object of its own to point to. + * + * The pass runs once, on a module lowered from CorePrep. A module read + * from an Xpp artifact already carries its ownership operations. + * + * @param module Xpp module lowered from CorePrep, without ownership + * operations. + * @return The module with explicit retains, releases and closures of + * method values. + */ + [[nodiscard]] auto + PlaceOwnership(::visual_xsharp::xpp::Module module) + -> ::visual_xsharp::xpp::Module; +} // namespace Visual::XSharp::Xpp diff --git a/Compiler/Headers/Visual/XSharp/Xpp/Wire.hpp b/Compiler/Headers/Visual/XSharp/Xpp/Wire.hpp index 8f21ae39..22c5da63 100644 --- a/Compiler/Headers/Visual/XSharp/Xpp/Wire.hpp +++ b/Compiler/Headers/Visual/XSharp/Xpp/Wire.hpp @@ -13,7 +13,7 @@ namespace Visual::XSharp::Xpp::Wire { /// Current Xpp wire schema version; decoders require an exact match. - inline constexpr std::uint16_t kCurrentVersion = 5U; + inline constexpr std::uint16_t kCurrentVersion = 7U; /// Shared byte and collection ceilings for Xpp wire operations. using Limits = Artifact::Wire::Limits; /// Structured wire failure with byte offset and field context. diff --git a/Compiler/Linker/NativeLinker.cpp b/Compiler/Linker/NativeLinker.cpp index bdc78f85..d5598ab3 100644 --- a/Compiler/Linker/NativeLinker.cpp +++ b/Compiler/Linker/NativeLinker.cpp @@ -2,13 +2,21 @@ // SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 #include +#include #include #include #include "Compiler/Linker/NativeLinker.hpp" #ifdef _WIN32 +# ifndef WIN32_LEAN_AND_MEAN +# define WIN32_LEAN_AND_MEAN +# endif +# ifndef NOMINMAX +# define NOMINMAX +# endif # include +# include #endif namespace Visual::XSharp::Driver @@ -40,6 +48,143 @@ namespace Visual::XSharp::Driver } return {}; } + +#ifdef _WIN32 + /// Runs `lld-link` with the given arguments and waits for it. + /// + /// Owning strings remain separate from argv pointers, so that growth + /// of one vector cannot invalidate memory passed to `_wspawnvp`. + [[nodiscard]] auto + RunLinker(const std::vector &options) -> std::intptr_t + { + const std::wstring program = L"lld-link.exe"; + std::vector arguments; + arguments.reserve(options.size() + 2U); + arguments.push_back(program.c_str()); + for (const auto &option : options) + arguments.push_back(option.c_str()); + arguments.push_back(nullptr); + return _wspawnvp(_P_WAIT, program.c_str(), arguments.data()); + } + + /// The runtime library, when it stands beside the running program. + /// + /// `vxs-runtime.lib` is installed with the compiler, in the directory + /// of its executable, as the frontend library is. + [[nodiscard]] auto + RuntimeLibrary() -> std::optional + { + std::wstring buffer(32768U, L'\0'); + const auto length + = GetModuleFileNameW(nullptr, + buffer.data(), + static_cast(buffer.size())); + if (length == 0U || length >= buffer.size()) + return std::nullopt; + buffer.resize(length); + auto candidate = std::filesystem::path(buffer).parent_path() + / L"vxs-runtime.lib"; + std::error_code error; + if (!std::filesystem::is_regular_file(candidate, error) || error) + return std::nullopt; + return candidate; + } + + /// What the runtime library takes from the system: these functions of + /// kernel32 and nothing else. The list is the one the runtime + /// declares in `Compiler/Runtime/Text/Platform.cpp` and + /// `Compiler/Runtime/Freestanding/Freestanding.cpp`. + constexpr std::string_view kSystemImports = "LIBRARY KERNEL32.dll\n" + "EXPORTS\n" + " GetConsoleMode\n" + " GetProcessHeap\n" + " GetStdHandle\n" + " HeapAlloc\n" + " HeapFree\n" + " WriteConsoleW\n" + " WriteFile\n"; + + /// Removes the files of one link when it goes out of scope. + struct Scratch final + { + std::vector paths; + + Scratch() = default; + Scratch(const Scratch &) = delete; + Scratch(Scratch &&) = delete; + auto + operator=(const Scratch &) -> Scratch & = delete; + auto + operator=(Scratch &&) -> Scratch & = delete; + + ~Scratch() + { + for (const auto &path : paths) + { + std::error_code ignored; + std::filesystem::remove(path, ignored); + } + } + }; + + /// Writes an import library for the functions of kernel32 the + /// runtime uses, beside the output, and returns its path. + /// + /// An executable is linked without the libraries of a C runtime or + /// of a Windows SDK, so there is no `kernel32.lib` to name. The + /// linker makes an import library from a list of names, which is all + /// that is needed to call a function of a system library. + [[nodiscard]] auto + WriteSystemImports(const std::filesystem::path &output, + Scratch &scratch, + std::string &diagnostic) + -> std::optional + { + auto definition = output; + definition += L".imports.def"; + auto library = output; + library += L".imports.lib"; + scratch.paths.push_back(definition); + scratch.paths.push_back(library); + { + std::ofstream stream(definition, + std::ios::binary | std::ios::trunc); + stream.write( + kSystemImports.data(), + static_cast(kSystemImports.size())); + if (!stream) + { + diagnostic = "could not write the list of system imports " + "for the native link"; + return std::nullopt; + } + } +# if defined(_M_ARM64) + const std::wstring machine = L"/machine:arm64"; +# else + const std::wstring machine = L"/machine:x64"; +# endif + const auto status = RunLinker({ L"/lib", + L"/nologo", + machine, + L"/def:" + definition.wstring(), + L"/out:" + library.wstring() }); + if (status != 0) + { + diagnostic + = status == -1 + ? "could not start lld-link: " + + std::error_code(errno, + std::generic_category()) + .message() + : "lld-link could not make the import library of " + "the system functions; exit code " + + std::to_string(status); + return std::nullopt; + } + return library; + } +#endif } // namespace auto @@ -59,32 +204,42 @@ namespace Visual::XSharp::Driver "could not replace the existing native executable: " + removeError.message() }; - // The generated bridge is a freestanding PE entry and currently needs - // no CRT. Avoiding implicit default libraries keeps the first - // executable boundary small, deterministic, and independent from Visual - // Studio's compiler/linker binaries. - const std::wstring program = L"lld-link.exe"; - std::vector storage{ program, - L"/nologo", + // The generated bridge is a freestanding PE entry and needs no C + // runtime. Avoiding implicit default libraries keeps the executable + // small, deterministic, and independent from Visual Studio's + // compiler and linker binaries. What a program needs beyond its own + // code is the Visual X# runtime library, which is linked the same + // way, and through it a few functions of kernel32. + std::vector options{ L"/nologo", L"/entry:mainCRTStartup", L"/subsystem:console", L"/nodefaultlib", L"/out:" + request.outputPath.wstring() }; - storage.reserve(storage.size() + request.objectPaths.size()); + options.reserve(options.size() + request.objectPaths.size() + 2U); for (const auto &object : request.objectPaths) - storage.push_back(object.wstring()); - std::vector arguments; - // Owning strings remain separate from argv pointers. Reserving both - // vectors prevents growth from invalidating memory passed to - // `_wspawnvp`. - arguments.reserve(storage.size() + 1U); - for (const auto &argument : storage) - arguments.push_back(argument.c_str()); - arguments.push_back(nullptr); - - const auto status - = _wspawnvp(_P_WAIT, program.c_str(), arguments.data()); + options.push_back(object.wstring()); + + // A program that creates no closure, uses no string and writes + // nothing calls nothing of the runtime, and the linker takes + // nothing from a library that is not called. When the library is + // not installed, such a program still links; any other fails at an + // unresolved runtime symbol, which the diagnostic below explains. + Scratch scratch; + const auto runtime = RuntimeLibrary(); + if (runtime) + { + std::string importDiagnostic; + const auto imports = WriteSystemImports(request.outputPath, + scratch, + importDiagnostic); + if (!imports) + return { -1, std::move(importDiagnostic) }; + options.push_back(runtime->wstring()); + options.push_back(imports->wstring()); + } + + const auto status = RunLinker(options); // Waiting is deliberate: success cannot be reported, and `run` cannot // begin, until LLD has closed and finalized the PE image. if (status == -1) @@ -95,8 +250,15 @@ namespace Visual::XSharp::Driver }; if (status != 0) return { static_cast(status), - "lld-link failed with exit code " - + std::to_string(status) }; + "lld-link failed with exit code " + std::to_string(status) + + (runtime + ? std::string() + : std::string( + "; the Visual X# runtime library " + "vxs-runtime.lib was not found beside " + "the compiler, and a program that uses " + "strings, closures or the console " + "needs it")) }; std::error_code sizeError; const auto size diff --git a/Compiler/Linker/NativeLinker.hpp b/Compiler/Linker/NativeLinker.hpp index 8d0265d8..459a5f11 100644 --- a/Compiler/Linker/NativeLinker.hpp +++ b/Compiler/Linker/NativeLinker.hpp @@ -3,7 +3,9 @@ #pragma once #include +#include #include +#include #include #include "Visual/XSharp/Backend/LLVM.hpp" diff --git a/Compiler/Runtime/AARC/BUILD.bazel b/Compiler/Runtime/AARC/BUILD.bazel index 9b44f802..10a5f21d 100644 --- a/Compiler/Runtime/AARC/BUILD.bazel +++ b/Compiler/Runtime/AARC/BUILD.bazel @@ -6,5 +6,18 @@ cc_library( name = "aarc", srcs = ["Runtime.cpp"], hdrs = ["Internal.hpp"], + # Generated code calls the runtime by name while a program runs; see + # //Compiler/Runtime/Text. + alwayslink = True, deps = ["//Compiler/Headers/Visual/XSharp/Runtime:aarc_api"], ) + +# The sources as text, for the one translation unit of the freestanding +# runtime. +filegroup( + name = "sources", + srcs = [ + "Internal.hpp", + "Runtime.cpp", + ], +) diff --git a/Compiler/Runtime/AARC/Internal.hpp b/Compiler/Runtime/AARC/Internal.hpp index 5cc9ded1..0ac0acde 100644 --- a/Compiler/Runtime/AARC/Internal.hpp +++ b/Compiler/Runtime/AARC/Internal.hpp @@ -32,4 +32,9 @@ namespace Visual::XSharp::Runtime::Aarc std::atomic object{}; void *allocation{}; }; + + // Create a string from scalars of either spelling of the ABI. + template + [[nodiscard]] auto + MakeStringFrom(const Scalar *scalars, std::size_t count) noexcept -> void *; } // namespace Visual::XSharp::Runtime::Aarc diff --git a/Compiler/Runtime/AARC/Runtime.cpp b/Compiler/Runtime/AARC/Runtime.cpp index afcb06bd..5905475d 100644 --- a/Compiler/Runtime/AARC/Runtime.cpp +++ b/Compiler/Runtime/AARC/Runtime.cpp @@ -67,6 +67,11 @@ namespace Visual::XSharp::Runtime::Aarc return false; } + // Allocations made and not yet reclaimed. The count orders nothing: + // it is read only where no allocation is in flight. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables) + std::atomic liveAllocations{ 0U }; + void ReleaseControl(ObjectHeader *header) noexcept { @@ -74,6 +79,7 @@ namespace Visual::XSharp::Runtime::Aarc || header->weakCount.fetch_sub(1U, std::memory_order_acq_rel) != 1U) return; + liveAllocations.fetch_sub(1U, std::memory_order_relaxed); // The acquire side of the final decrement observes the destructor // and all preceding handle releases before reclaiming the combined @@ -104,6 +110,12 @@ namespace Visual::XSharp::Runtime::Aarc } } // namespace + auto + LiveAllocations() noexcept -> std::uint64_t + { + return liveAllocations.load(std::memory_order_relaxed); + } + auto Allocate(const TypeMetadata &metadata) noexcept -> void * { @@ -128,6 +140,7 @@ namespace Visual::XSharp::Runtime::Aarc auto *header = new (allocation) ObjectHeader{}; header->metadata = &metadata; header->allocation = allocation; + liveAllocations.fetch_add(1U, std::memory_order_relaxed); // Advance to the next multiple of the alignment without turning an // integer back into a pointer, which would discard provenance. @@ -239,6 +252,59 @@ namespace Visual::XSharp::Runtime::Aarc && header->metadata != nullptr && header->metadata->typeIdentity == typeIdentity; } + + auto + ViewString(const void *string) noexcept -> StringScalars + { + if (!IsExactType(string, kStringMetadata.typeIdentity)) + return {}; + const auto *object = static_cast(string); + return { object->scalars, static_cast(object->count) }; + } + + auto + MakeString(const char32_t *scalars, const std::size_t count) noexcept + -> void * + { + return MakeStringFrom(scalars, count); + } + + // The C side of the ABI spells a scalar `uint32_t` and this side + // `char32_t`. The two have one representation and are still two types, + // so the scalars are copied through whichever was given. + template + auto + MakeStringFrom(const Scalar *scalars, const std::size_t count) noexcept + -> void * + { + if ((scalars == nullptr && count != 0U) + || count == std::numeric_limits::max()) + return nullptr; + for (std::size_t index = 0; index < count; ++index) + if (scalars[index] > 0x10ffffU + || (scalars[index] >= 0xd800U && scalars[index] <= 0xdfffU)) + return nullptr; + auto *object = static_cast(Allocate(kStringMetadata)); + if (object == nullptr) + return nullptr; + object->scalars = new (std::nothrow) char32_t[count + 1U]; + if (object->scalars == nullptr) + { + ReleaseStrong(object); + return nullptr; + } + object->count = count; + for (std::size_t index = 0; index < count; ++index) + object->scalars[index] = static_cast(scalars[index]); + object->scalars[count] = U'\0'; + return object; + } + + template auto + MakeStringFrom(const char32_t *, std::size_t) noexcept -> void *; + template auto + MakeStringFrom(const std::uint32_t *, std::size_t) noexcept + -> void *; } // namespace Visual::XSharp::Runtime::Aarc extern "C" @@ -343,28 +409,7 @@ extern "C" vxs_aarc_string_literal(const std::uint32_t *scalars, std::size_t count) noexcept -> void * { - using namespace Visual::XSharp::Runtime::Aarc; - if ((scalars == nullptr && count != 0U) - || count == std::numeric_limits::max()) - return nullptr; - for (std::size_t index = 0; index < count; ++index) - if (scalars[index] > 0x10ffffU - || (scalars[index] >= 0xd800U && scalars[index] <= 0xdfffU)) - return nullptr; - auto *object = static_cast(Allocate(kStringMetadata)); - if (object == nullptr) - return nullptr; - object->scalars = new (std::nothrow) char32_t[count + 1U]; - if (object->scalars == nullptr) - { - ReleaseStrong(object); - return nullptr; - } - object->count = count; - for (std::size_t index = 0; index < count; ++index) - object->scalars[index] = static_cast(scalars[index]); - object->scalars[count] = U'\0'; - return object; + return Visual::XSharp::Runtime::Aarc::MakeStringFrom(scalars, count); } auto diff --git a/Compiler/Runtime/AARC/Tests/AARCRuntimeTests.cpp b/Compiler/Runtime/AARC/Tests/AARCRuntimeTests.cpp index 2085e288..840e04f9 100644 --- a/Compiler/Runtime/AARC/Tests/AARCRuntimeTests.cpp +++ b/Compiler/Runtime/AARC/Tests/AARCRuntimeTests.cpp @@ -196,6 +196,38 @@ TEST_CASE("strong references destroy the payload exactly once") CHECK(destructions.load() == 1U); } +TEST_CASE("the live allocation count follows storage, not strong lifetime") +{ + const auto before = Aarc::LiveAllocations(); + auto *first = static_cast(Aarc::Allocate(kMetadata)); + auto *second = static_cast(Aarc::Allocate(kMetadata)); + REQUIRE(first != nullptr); + REQUIRE(second != nullptr); + CHECK(Aarc::LiveAllocations() == before + 2U); + + // A retained object stays one allocation. + CHECK(Aarc::RetainStrong(first) == first); + CHECK(Aarc::LiveAllocations() == before + 2U); + Aarc::ReleaseStrong(first); + CHECK(Aarc::LiveAllocations() == before + 2U); + Aarc::ReleaseStrong(first); + CHECK(Aarc::LiveAllocations() == before + 1U); + + // A weak handle keeps the storage after the payload is destroyed. + const auto weak = Aarc::MakeWeak(second); + Aarc::ReleaseStrong(second); + CHECK(Aarc::LiveAllocations() == before + 1U); + Aarc::ReleaseWeak(weak); + CHECK(Aarc::LiveAllocations() == before); + + // A failed allocation and a null release count nothing. + auto invalid = kMetadata; + invalid.instanceSize = 0U; + CHECK(Aarc::Allocate(invalid) == nullptr); + Aarc::ReleaseStrong(nullptr); + CHECK(Aarc::LiveAllocations() == before); +} + TEST_CASE("exact type tests use stable metadata identity and reject null") { auto *payload = static_cast(Aarc::Allocate(kMetadata)); diff --git a/Compiler/Runtime/BUILD.bazel b/Compiler/Runtime/BUILD.bazel new file mode 100644 index 00000000..563a8bc6 --- /dev/null +++ b/Compiler/Runtime/BUILD.bazel @@ -0,0 +1,3 @@ +# The runtime libraries have packages of their own; this package holds what +# they share with the programs that link them. +exports_files(["host.bzl"]) diff --git a/Compiler/Runtime/Freestanding/BUILD.bazel b/Compiler/Runtime/Freestanding/BUILD.bazel new file mode 100644 index 00000000..1bad4f14 --- /dev/null +++ b/Compiler/Runtime/Freestanding/BUILD.bazel @@ -0,0 +1,33 @@ +load("@rules_cc//cc:cc_library.bzl", "cc_library") + +package(default_visibility = ["//visibility:public"]) + +# The runtime a native executable is linked with: the ownership runtime and +# the text and console runtime as one object that needs no C runtime. The +# development helper stages the archive beside `vxs` as `vxs-runtime.lib`, +# where the native linker looks for it. +# +# The native linker is the Windows one, so this target exists there only. +cc_library( + name = "vxs_runtime", + srcs = ["Freestanding.cpp"], + # No stack cookie, which needs runtime support; no default-library + # directive, since nothing of a C runtime is linked; and a stack probe + # threshold that no function of the runtime reaches, because the probe + # routine belongs to a C runtime too. + copts = [ + "/GS-", + "/Zl", + "/Gs1048576", + ], + linkstatic = True, + target_compatible_with = ["@platforms//os:windows"], + textual_hdrs = [ + "//Compiler/Runtime/AARC:sources", + "//Compiler/Runtime/Text:sources", + ], + deps = [ + "//Compiler/Headers/Visual/XSharp/Runtime:aarc_api", + "//Compiler/Headers/Visual/XSharp/Runtime:text_api", + ], +) diff --git a/Compiler/Runtime/Freestanding/Freestanding.cpp b/Compiler/Runtime/Freestanding/Freestanding.cpp new file mode 100644 index 00000000..9107ae1a --- /dev/null +++ b/Compiler/Runtime/Freestanding/Freestanding.cpp @@ -0,0 +1,131 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +// The Visual X# runtime as it is linked into a native executable. +// +// A native executable is linked without a C runtime. This translation unit +// is the whole runtime such an executable gets: the ownership runtime and +// the text and console runtime, from the same sources that a process +// hosting generated code links, followed by the few things those sources +// take from a C runtime when there is one. +// +// The sources are included rather than compiled on their own so that the +// library is one object built with one set of options. Those options turn +// off what needs runtime support the executable does not have: stack +// cookies, and instrumentation of any kind. A sanitizer build of the +// compiler therefore still produces this object uninstrumented, because the +// programs it is linked into are not sanitizer builds. +// +// What the executable imports from the system is kernel32 and nothing else: +// the process heap for memory and the standard handles for output. + +#include +#include + +#include "Compiler/Runtime/AARC/Runtime.cpp" +#include "Compiler/Runtime/Text/Console.cpp" +#include "Compiler/Runtime/Text/Format.cpp" +#include "Compiler/Runtime/Text/Platform.cpp" +#include "Compiler/Runtime/Text/Text.cpp" + +extern "C" +{ + __declspec(dllimport) void *__stdcall + GetProcessHeap(); + __declspec(dllimport) void *__stdcall + HeapAlloc(void *heap, unsigned long flags, std::size_t bytes); + __declspec(dllimport) int __stdcall + HeapFree(void *heap, unsigned long flags, void *block); + + // The linker looks for this symbol in any program that uses + // floating-point arithmetic; a C runtime defines it. + // NOLINTNEXTLINE(bugprone-reserved-identifier,cppcoreguidelines-avoid-non-const-global-variables) + int _fltused = 0x9875; + + // The compiler may turn a loop or an initialization into a call of + // these, and each is told not to turn its own loop into a call of + // itself. + __attribute__((no_builtin("memset"))) auto + memset(void *destination, int value, std::size_t count) -> void * + { + auto *bytes = static_cast(destination); + for (std::size_t index = 0U; index < count; ++index) + bytes[index] = static_cast(value); + return destination; + } + + __attribute__((no_builtin("memcpy"))) auto + memcpy(void *destination, const void *source, std::size_t count) -> void * + { + auto *to = static_cast(destination); + const auto *from = static_cast(source); + for (std::size_t index = 0U; index < count; ++index) + to[index] = from[index]; + return destination; + } + + __attribute__((no_builtin("memmove", "memcpy"))) auto + memmove(void *destination, const void *source, std::size_t count) -> void * + { + auto *to = static_cast(destination); + const auto *from = static_cast(source); + if (to < from) + for (std::size_t index = 0U; index < count; ++index) + to[index] = from[index]; + else + for (std::size_t index = count; index-- > 0U;) + to[index] = from[index]; + return destination; + } +} + +namespace std +{ + // The tag object of the non-throwing allocation functions, which a C++ + // runtime library defines. + const nothrow_t nothrow{}; +} // namespace std + +// Memory comes from the heap of the process. Only the non-throwing forms +// are defined: the runtime uses no other, and a use of a throwing form would +// fail to link rather than throw where nothing can catch. + +auto +operator new(std::size_t bytes, const std::nothrow_t &) noexcept -> void * +{ + return HeapAlloc(GetProcessHeap(), 0UL, bytes == 0U ? 1U : bytes); +} + +auto +operator new[](std::size_t bytes, const std::nothrow_t &) noexcept -> void * +{ + return HeapAlloc(GetProcessHeap(), 0UL, bytes == 0U ? 1U : bytes); +} + +void +operator delete(void *block) noexcept +{ + if (block != nullptr) + HeapFree(GetProcessHeap(), 0UL, block); +} + +void +operator delete[](void *block) noexcept +{ + if (block != nullptr) + HeapFree(GetProcessHeap(), 0UL, block); +} + +void +operator delete(void *block, std::size_t) noexcept +{ + if (block != nullptr) + HeapFree(GetProcessHeap(), 0UL, block); +} + +void +operator delete[](void *block, std::size_t) noexcept +{ + if (block != nullptr) + HeapFree(GetProcessHeap(), 0UL, block); +} diff --git a/Compiler/Runtime/Text/BUILD.bazel b/Compiler/Runtime/Text/BUILD.bazel new file mode 100644 index 00000000..1ae29e7a --- /dev/null +++ b/Compiler/Runtime/Text/BUILD.bazel @@ -0,0 +1,46 @@ +load("@rules_cc//cc:cc_library.bzl", "cc_library") + +package(default_visibility = ["//visibility:public"]) + +# String concatenation, the conversions of the output format grammar, and +# console output. A process that hosts generated code links this library and +# exports its entry points; a native executable gets the same sources through +# //Compiler/Runtime/Freestanding. +cc_library( + name = "text", + srcs = [ + "Console.cpp", + "Format.cpp", + "Platform.cpp", + "Text.cpp", + ], + hdrs = [ + "Format.hpp", + "Platform.hpp", + "Scalars.hpp", + ], + # Generated code is the only caller of most of these functions, and it + # finds them by name while a program runs. A linker that keeps only what + # the host itself calls would leave them out. + alwayslink = True, + deps = [ + "//Compiler/Headers/Visual/XSharp/Runtime:aarc_api", + "//Compiler/Headers/Visual/XSharp/Runtime:text_api", + "//Compiler/Runtime/AARC:aarc", + ], +) + +# The sources as text, for the one translation unit of the freestanding +# runtime. +filegroup( + name = "sources", + srcs = [ + "Console.cpp", + "Format.cpp", + "Format.hpp", + "Platform.cpp", + "Platform.hpp", + "Scalars.hpp", + "Text.cpp", + ], +) diff --git a/Compiler/Runtime/Text/Console.cpp b/Compiler/Runtime/Text/Console.cpp new file mode 100644 index 00000000..f98d7352 --- /dev/null +++ b/Compiler/Runtime/Text/Console.cpp @@ -0,0 +1,122 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include "Platform.hpp" +#include "Visual/XSharp/Runtime/AARC.hpp" +#include "Visual/XSharp/Runtime/Text.hpp" + +// Console output. A string is a sequence of scalar values; a stream takes +// bytes. The scalars are encoded as UTF-8 a block at a time, each block +// ending between two characters, and every block is handed to the platform +// before the write returns: the runtime keeps no buffer, so output appears +// in the order the program wrote it and nothing is lost when the program +// stops. + +namespace Visual::XSharp::Runtime::Console +{ + namespace + { + // The receiver a host installed, if any. A process that hosts + // generated code runs one program at a time on one thread. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables) + Sink installedSink = nullptr; + // NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables) + void *installedContext = nullptr; + + constexpr std::size_t kBlock = 512U; + + void + Deliver(const bool error, + const char *bytes, + const std::size_t count) noexcept + { + if (count == 0U) + return; + if (installedSink != nullptr) + installedSink(error ? VXS_CONSOLE_ERROR : VXS_CONSOLE_OUTPUT, + bytes, + count, + installedContext); + else + Platform::Write(error, bytes, count); + } + + /// Encodes one scalar value and returns the number of bytes. + [[nodiscard]] auto + Encode(const char32_t scalar, char *bytes) noexcept -> std::size_t + { + const auto value = static_cast(scalar); + if (value < 0x80U) + { + bytes[0] = static_cast(value); + return 1U; + } + if (value < 0x800U) + { + bytes[0] = static_cast(0xc0U | (value >> 6U)); + bytes[1] = static_cast(0x80U | (value & 0x3fU)); + return 2U; + } + if (value < 0x10000U) + { + bytes[0] = static_cast(0xe0U | (value >> 12U)); + bytes[1] = static_cast(0x80U | ((value >> 6U) & 0x3fU)); + bytes[2] = static_cast(0x80U | (value & 0x3fU)); + return 3U; + } + bytes[0] = static_cast(0xf0U | (value >> 18U)); + bytes[1] = static_cast(0x80U | ((value >> 12U) & 0x3fU)); + bytes[2] = static_cast(0x80U | ((value >> 6U) & 0x3fU)); + bytes[3] = static_cast(0x80U | (value & 0x3fU)); + return 4U; + } + } // namespace + + void + SetSink(const Sink sink, void *context) noexcept + { + installedSink = sink; + installedContext = context; + } +} // namespace Visual::XSharp::Runtime::Console + +extern "C" +{ + void + vxs_console_write(void *text, const std::int64_t target) noexcept + { + using namespace Visual::XSharp::Runtime; + const auto error + = target == VXS_CONSOLE_ERROR || target == VXS_CONSOLE_ERROR_LINE; + const auto line = target == VXS_CONSOLE_OUTPUT_LINE + || target == VXS_CONSOLE_ERROR_LINE; + const auto view = Aarc::ViewString(text); + + char block[Console::kBlock]; + std::size_t used = 0U; + for (std::size_t index = 0U; index < view.count; ++index) + { + // A character is at most four bytes; a block never ends inside + // one. + if (used + 4U > Console::kBlock) + { + Console::Deliver(error, block, used); + used = 0U; + } + used += Console::Encode(view.scalars[index], block + used); + } + if (line) + { + if (used + 2U > Console::kBlock) + { + Console::Deliver(error, block, used); + used = 0U; + } + for (const auto *terminator = Platform::LineTerminator(); + *terminator != '\0'; + ++terminator) + block[used++] = *terminator; + } + Console::Deliver(error, block, used); + } +} diff --git a/Compiler/Runtime/Text/Format.cpp b/Compiler/Runtime/Text/Format.cpp new file mode 100644 index 00000000..2ed63608 --- /dev/null +++ b/Compiler/Runtime/Text/Format.cpp @@ -0,0 +1,388 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include + +#include "Format.hpp" +#include "Visual/XSharp/Runtime/Text.h" + +namespace Visual::XSharp::Runtime::Text +{ + namespace + { + /// The field a conversion writes: what stands before the digits, + /// the body, and the padding that brings the two to the width. + /// + /// Padding stands at the left unless the conversion asks for the + /// value at the left. Zeros, which only a number may ask for, stand + /// between the prefix and the body, so that a sign stays in front. + void + AppendField(Scalars &output, + const char *prefix, + const Scalars &body, + const Conversion &conversion, + const bool zerosAllowed) noexcept + { + std::size_t prefixSize = 0U; + while (prefix[prefixSize] != '\0') + ++prefixSize; + const auto size = prefixSize + body.Size(); + // A width the value already fills, and a width that is absent + // or negative, add nothing. + const auto padding + = conversion.width > 0 + && static_cast(conversion.width) > size + ? static_cast( + static_cast(conversion.width) - size) + : std::size_t{ 0U }; + const auto left = conversion.Has(VXS_TEXT_FLAG_LEFT); + const auto zeros + = zerosAllowed && !left && conversion.Has(VXS_TEXT_FLAG_ZERO); + if (!left && !zeros) + output.Fill(U' ', padding); + output.AppendAscii(prefix); + if (zeros) + output.Fill(U'0', padding); + output.Append(body.Data(), body.Size()); + if (left) + output.Fill(U' ', padding); + } + + /// The sign a number is written with: a minus when it is negative, + /// and otherwise what the conversion asks for, if anything. + [[nodiscard]] auto + Sign(const bool negative, const Conversion &conversion) noexcept + -> const char * + { + if (negative) + return "-"; + if (conversion.Has(VXS_TEXT_FLAG_PLUS)) + return "+"; + if (conversion.Has(VXS_TEXT_FLAG_SPACE)) + return " "; + return ""; + } + + /// A natural number of bounded size, for the exact decimal + /// expansion of a floating-point value. + /// + /// A finite `double` is an integer of at most 53 bits times a power + /// of two between 2^-1074 and 2^971. Scaled by 10^1074 it is an + /// integer below 2^4592, which is what the capacity is chosen for; + /// nothing here rounds or estimates. + class Natural final + { + public: + explicit Natural(const std::uint64_t value) noexcept + { + limbs_[0] = static_cast(value & 0xffffffffULL); + limbs_[1] = static_cast(value >> 32U); + used_ = limbs_[1] != 0U ? 2U : limbs_[0] != 0U ? 1U : 0U; + } + + [[nodiscard]] auto + IsZero() const noexcept -> bool + { + return used_ == 0U; + } + + void + Multiply(const std::uint32_t factor) noexcept + { + std::uint64_t carry = 0U; + for (std::size_t index = 0U; index < used_; ++index) + { + const auto product + = static_cast(limbs_[index]) * factor + + carry; + limbs_[index] + = static_cast(product & 0xffffffffULL); + carry = product >> 32U; + } + if (carry != 0U) + Push(static_cast(carry)); + } + + void + ShiftLeft(const std::size_t bits) noexcept + { + if (used_ == 0U || bits == 0U) + return; + const auto whole = bits / 32U; + const auto part = static_cast(bits % 32U); + if (used_ + whole + 1U > kLimbs) + Platform::Fail(); + std::uint32_t carried = 0U; + if (part != 0U) + for (std::size_t index = 0U; index < used_; ++index) + { + const auto limb = limbs_[index]; + limbs_[index] = (limb << part) | carried; + carried = limb >> (32U - part); + } + if (carried != 0U) + limbs_[used_++] = carried; + if (whole != 0U) + { + for (std::size_t index = used_; index-- > 0U;) + limbs_[index + whole] = limbs_[index]; + for (std::size_t index = 0U; index < whole; ++index) + limbs_[index] = 0U; + used_ += whole; + } + } + + /// Divides by 2^bits and rounds to the nearest integer; a value + /// exactly between two integers goes to the even one. + void + ShiftRightToNearestEven(const std::size_t bits) noexcept + { + if (bits == 0U || used_ == 0U) + return; + const auto half = Bit(bits - 1U); + auto below = false; + for (std::size_t index = 0U; + !below && index < (bits - 1U) / 32U && index < used_; + ++index) + below = limbs_[index] != 0U; + if (!below && (bits - 1U) % 32U != 0U + && (bits - 1U) / 32U < used_) + below = (limbs_[(bits - 1U) / 32U] + & ((1U << ((bits - 1U) % 32U)) - 1U)) + != 0U; + + const auto whole = bits / 32U; + const auto part = static_cast(bits % 32U); + if (whole >= used_) + { + used_ = 0U; + } + else + { + for (std::size_t index = whole; index < used_; ++index) + { + auto limb = limbs_[index] >> part; + if (part != 0U && index + 1U < used_) + limb |= limbs_[index + 1U] << (32U - part); + limbs_[index - whole] = limb; + } + used_ -= whole; + while (used_ != 0U && limbs_[used_ - 1U] == 0U) + --used_; + } + const auto odd = used_ != 0U && (limbs_[0] & 1U) != 0U; + if (half && (below || odd)) + Increment(); + } + + /// Divides by a divisor and returns what remains. + [[nodiscard]] auto + Divide(const std::uint32_t divisor) noexcept -> std::uint32_t + { + std::uint64_t remainder = 0U; + for (std::size_t index = used_; index-- > 0U;) + { + const auto dividend = (remainder << 32U) | limbs_[index]; + limbs_[index] + = static_cast(dividend / divisor); + remainder = dividend % divisor; + } + while (used_ != 0U && limbs_[used_ - 1U] == 0U) + --used_; + return static_cast(remainder); + } + + private: + static constexpr std::size_t kLimbs = 160U; + + [[nodiscard]] auto + Bit(const std::size_t position) const noexcept -> bool + { + const auto limb = position / 32U; + return limb < used_ + && ((limbs_[limb] >> (position % 32U)) & 1U) != 0U; + } + + void + Push(const std::uint32_t limb) noexcept + { + if (used_ == kLimbs) + Platform::Fail(); + limbs_[used_++] = limb; + } + + void + Increment() noexcept + { + for (std::size_t index = 0U; index < used_; ++index) + if (++limbs_[index] != 0U) + return; + Push(1U); + } + + std::uint32_t limbs_[kLimbs]{}; + std::size_t used_{}; + }; + + /// The most digits after the point a finite `double` has: the + /// expansion of 2^-1074 ends at its 1074th. Every digit asked for + /// beyond that is a zero and is written without being computed. + constexpr std::int64_t kExactFractionDigits = 1074; + + /// Appends integer digits, least significant first, with a group + /// separator after every third when grouping is asked for. + class ReversedDigits final + { + public: + ReversedDigits(Scalars &digits, const bool grouped) noexcept + : digits_(digits) + , grouped_(grouped) + {} + + void + Append(const char32_t digit) noexcept + { + if (grouped_ && count_ != 0U && count_ % 3U == 0U) + digits_.Append(U'\''); + digits_.Append(digit); + ++count_; + } + + private: + Scalars &digits_; + bool grouped_; + std::size_t count_{}; + }; + } // namespace + + void + AppendInteger(Scalars &output, + const bool negative, + std::uint64_t magnitude, + const Conversion &conversion) noexcept + { + const auto hexadecimal = conversion.Has(VXS_TEXT_FLAG_HEXADECIMAL); + Scalars body; + { + ReversedDigits digits(body, + !hexadecimal + && conversion.Has(VXS_TEXT_FLAG_GROUP)); + const std::uint64_t base = hexadecimal ? 16U : 10U; + do + { + const auto digit = static_cast(magnitude % base); + digits.Append(static_cast( + digit < 10U ? U'0' + digit : U'a' + (digit - 10U))); + magnitude /= base; + } while (magnitude != 0U); + } + body.ReverseFrom(0U); + + // At most a sign and the two characters of the prefix. + char prefix[4]{}; + std::size_t used = 0U; + for (const auto *sign = Sign(negative, conversion); *sign != '\0'; + ++sign) + prefix[used++] = *sign; + if (hexadecimal && conversion.Has(VXS_TEXT_FLAG_ALTERNATE)) + { + prefix[used++] = '0'; + prefix[used++] = 'x'; + } + AppendField(output, prefix, body, conversion, true); + } + + void + AppendFloating(Scalars &output, + const double value, + const Conversion &conversion) noexcept + { + const auto bits = std::bit_cast(value); + const auto negative = (bits >> 63U) != 0U; + const auto exponentField + = static_cast((bits >> 52U) & 0x7ffU); + const auto fraction = bits & ((std::uint64_t{ 1U } << 52U) - 1U); + + if (exponentField == 0x7ffU) + { + // Not a number has no sign worth writing; an infinity has. + Scalars body; + body.AppendAscii(fraction != 0U ? "nan" : "inf"); + AppendField(output, + fraction != 0U ? "" : Sign(negative, conversion), + body, + conversion, + false); + return; + } + + const auto precision = conversion.precision < 0 ? std::int64_t{ 6 } + : conversion.precision; + const auto computed = precision < kExactFractionDigits + ? precision + : kExactFractionDigits; + + // The value is `significand * 2^exponent` exactly. + const auto significand = exponentField == 0U + ? fraction + : fraction | (std::uint64_t{ 1U } << 52U); + const int exponent = exponentField == 0U + ? -1074 + : static_cast(exponentField) - 1075; + + // `scaled` becomes the value times ten to the number of computed + // digits, rounded to the nearest integer. Its decimal digits are + // the digits of the result with the point left out. + Natural scaled(significand); + for (std::int64_t done = 0; done < computed;) + { + const auto step = computed - done >= 9 ? 9 : computed - done; + std::uint32_t power = 1U; + for (std::int64_t index = 0; index < step; ++index) + power *= 10U; + scaled.Multiply(power); + done += step; + } + if (exponent >= 0) + scaled.ShiftLeft(static_cast(exponent)); + else + scaled.ShiftRightToNearestEven(static_cast(-exponent)); + + // Fraction digits first, least significant first, then the integer + // digits, which are the ones grouping applies to. + Scalars body; + for (std::int64_t index = computed; index < precision; ++index) + body.Append(U'0'); + std::int64_t produced = 0; + const auto nextDigit = [&scaled]() noexcept -> char32_t { + return static_cast(U'0' + scaled.Divide(10U)); + }; + for (; produced < computed; ++produced) + body.Append(nextDigit()); + if (precision > 0) + body.Append(U'.'); + { + ReversedDigits integer(body, conversion.Has(VXS_TEXT_FLAG_GROUP)); + do + { + integer.Append(nextDigit()); + } while (!scaled.IsZero()); + } + body.ReverseFrom(0U); + AppendField(output, Sign(negative, conversion), body, conversion, true); + } + + void + AppendText(Scalars &output, + const char32_t *scalars, + std::size_t count, + const Conversion &conversion) noexcept + { + if (conversion.precision >= 0 + && static_cast(conversion.precision) < count) + count = static_cast(conversion.precision); + Scalars body; + body.Append(scalars, count); + AppendField(output, "", body, conversion, false); + } +} // namespace Visual::XSharp::Runtime::Text diff --git a/Compiler/Runtime/Text/Format.hpp b/Compiler/Runtime/Text/Format.hpp new file mode 100644 index 00000000..efc4c75e --- /dev/null +++ b/Compiler/Runtime/Text/Format.hpp @@ -0,0 +1,52 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include + +#include "Scalars.hpp" + +// The conversions of the output format grammar, as functions from a value +// and the parts of its conversion to scalars appended to a sequence. They +// do no I/O and allocate nothing but the sequence they append to. + +namespace Visual::XSharp::Runtime::Text +{ + /** The parts of one conversion: the flags written between `%` and the + * conversion letter, and its width and precision when it has them. */ + struct Conversion final + { + std::int64_t flags{}; + /// Negative when the conversion has no width. + std::int64_t width{ -1 }; + /// Negative when the conversion has no precision. + std::int64_t precision{ -1 }; + + [[nodiscard]] auto + Has(const std::int64_t flag) const noexcept -> bool + { + return (flags & flag) != 0; + } + }; + + /** An integer given as whether it is negative and its magnitude, so + * that the least signed value needs no case of its own. */ + void + AppendInteger(Scalars &output, + bool negative, + std::uint64_t magnitude, + const Conversion &conversion) noexcept; + + void + AppendFloating(Scalars &output, + double value, + const Conversion &conversion) noexcept; + + /** A run of scalars as the body of a `%s` or `%c` field. */ + void + AppendText(Scalars &output, + const char32_t *scalars, + std::size_t count, + const Conversion &conversion) noexcept; +} // namespace Visual::XSharp::Runtime::Text diff --git a/Compiler/Runtime/Text/Platform.cpp b/Compiler/Runtime/Text/Platform.cpp new file mode 100644 index 00000000..e8ea83cc --- /dev/null +++ b/Compiler/Runtime/Text/Platform.cpp @@ -0,0 +1,179 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include "Platform.hpp" + +#ifdef _WIN32 + +// The four functions of kernel32 the runtime needs are declared here rather +// than through : a native executable links against an import +// library that the compiler generates for exactly the functions the runtime +// names, and this list is that contract. `VxsKernel32.def.inc` repeats it +// for the linker. +extern "C" +{ + __declspec(dllimport) void *__stdcall + GetStdHandle(unsigned long standardHandle); + __declspec(dllimport) int __stdcall + GetConsoleMode(void *handle, unsigned long *mode); + __declspec(dllimport) int __stdcall + WriteConsoleW(void *handle, + const wchar_t *text, + unsigned long count, + unsigned long *written, + void *reserved); + __declspec(dllimport) int __stdcall + WriteFile(void *handle, + const void *bytes, + unsigned long count, + unsigned long *written, + void *overlapped); +} + +namespace Visual::XSharp::Runtime::Platform +{ + namespace + { + constexpr unsigned long kStandardOutput = 0xfffffff5UL; // -11 + constexpr unsigned long kStandardError = 0xfffffff4UL; // -12 + constexpr std::size_t kChunk = 512U; + + /// Write whole UTF-8 characters to a console as UTF-16. A console + /// shows what it is given as text in its own code page, so bytes + /// of UTF-8 written to it would show as the wrong characters. + void + WriteConsole(void *handle, + const char *bytes, + const std::size_t count) noexcept + { + wchar_t units[kChunk]; + std::size_t used = 0U; + std::size_t index = 0U; + while (index < count) + { + const auto lead = static_cast(bytes[index]); + const std::size_t length = lead < 0x80U ? 1U + : lead < 0xe0U ? 2U + : lead < 0xf0U ? 3U + : 4U; + if (index + length > count) + break; + std::uint32_t scalar = length == 1U ? lead + : length == 2U ? (lead & 0x1fU) + : length == 3U ? (lead & 0x0fU) + : (lead & 0x07U); + for (std::size_t offset = 1U; offset < length; ++offset) + scalar + = (scalar << 6U) + | (static_cast(bytes[index + offset]) + & 0x3fU); + index += length; + if (used + 2U > kChunk) + { + unsigned long written = 0UL; + WriteConsoleW(handle, + units, + static_cast(used), + &written, + nullptr); + used = 0U; + } + if (scalar < 0x10000U) + { + units[used++] = static_cast(scalar); + } + else + { + scalar -= 0x10000U; + units[used++] + = static_cast(0xd800U + (scalar >> 10U)); + units[used++] + = static_cast(0xdc00U + (scalar & 0x3ffU)); + } + } + if (used != 0U) + { + unsigned long written = 0UL; + WriteConsoleW(handle, + units, + static_cast(used), + &written, + nullptr); + } + } + } // namespace + + void + Fail() noexcept + { + __builtin_trap(); + } + + void + Write(const bool error, const char *bytes, const std::size_t count) noexcept + { + auto *handle = GetStdHandle(error ? kStandardError : kStandardOutput); + // A process without the stream has a null or an invalid handle. + if (handle == nullptr || reinterpret_cast(handle) == -1) + return; + unsigned long mode = 0UL; + if (GetConsoleMode(handle, &mode) != 0) + { + WriteConsole(handle, bytes, count); + return; + } + std::size_t done = 0U; + while (done < count) + { + unsigned long written = 0UL; + const auto part + = count - done > 0x40000000U ? 0x40000000U : count - done; + if (WriteFile(handle, + bytes + done, + static_cast(part), + &written, + nullptr) + == 0 + || written == 0UL) + return; + done += written; + } + } +} // namespace Visual::XSharp::Runtime::Platform + +#else + +# include +# include + +namespace Visual::XSharp::Runtime::Platform +{ + void + Fail() noexcept + { + __builtin_trap(); + } + + void + Write(const bool error, const char *bytes, const std::size_t count) noexcept + { + const int descriptor = error ? 2 : 1; + std::size_t done = 0U; + while (done < count) + { + const auto written + = ::write(descriptor, bytes + done, count - done); + if (written < 0) + { + if (errno == EINTR) + continue; + return; + } + if (written == 0) + return; + done += static_cast(written); + } + } +} // namespace Visual::XSharp::Runtime::Platform + +#endif diff --git a/Compiler/Runtime/Text/Platform.hpp b/Compiler/Runtime/Text/Platform.hpp new file mode 100644 index 00000000..f3570f9f --- /dev/null +++ b/Compiler/Runtime/Text/Platform.hpp @@ -0,0 +1,36 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include + +// What the text runtime asks of the system it runs on. Everything else in +// the runtime is arithmetic on memory it was given, so that the same source +// serves a process that hosts generated code and a native executable that +// is linked without a C runtime. + +namespace Visual::XSharp::Runtime::Platform +{ + /** Stop the program. The runtime calls this when it cannot go on: + * memory for a result is exhausted. It does not return. */ + [[noreturn]] void + Fail() noexcept; + + /** Write bytes of UTF-8 to standard output (`error` false) or standard + * error. The bytes are whole encoded characters. A stream that cannot + * be written to is ignored. */ + void + Write(bool error, const char *bytes, std::size_t count) noexcept; + + /** The line terminator of the platform, as ASCII. */ + [[nodiscard]] constexpr auto + LineTerminator() noexcept -> const char * + { +#ifdef _WIN32 + return "\r\n"; +#else + return "\n"; +#endif + } +} // namespace Visual::XSharp::Runtime::Platform diff --git a/Compiler/Runtime/Text/Scalars.hpp b/Compiler/Runtime/Text/Scalars.hpp new file mode 100644 index 00000000..091da31c --- /dev/null +++ b/Compiler/Runtime/Text/Scalars.hpp @@ -0,0 +1,140 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 +#pragma once + +#include +#include +#include + +#include "Platform.hpp" + +namespace Visual::XSharp::Runtime::Text +{ + /** A growing sequence of Unicode scalar values. + * + * The text routines build every result in one of these and turn it into + * a string at the end. It is not a standard container on purpose: the + * runtime is also compiled for executables that have no C runtime, and + * a standard container reports failure through functions such an + * executable does not have. Short results stay in the object; longer + * ones move to memory from the non-throwing allocator, and a result + * that cannot be allocated stops the program. + */ + class Scalars final + { + public: + Scalars() noexcept = default; + Scalars(const Scalars &) = delete; + Scalars(Scalars &&) = delete; + auto + operator=(const Scalars &) -> Scalars & = delete; + auto + operator=(Scalars &&) -> Scalars & = delete; + + ~Scalars() + { + if (data_ != inline_) + ::operator delete(data_); + } + + [[nodiscard]] auto + Size() const noexcept -> std::size_t + { + return size_; + } + + [[nodiscard]] auto + Data() const noexcept -> const char32_t * + { + return data_; + } + + void + Append(const char32_t scalar) noexcept + { + Reserve(1U); + data_[size_++] = scalar; + } + + void + Append(const char32_t *scalars, const std::size_t count) noexcept + { + Reserve(count); + for (std::size_t index = 0U; index < count; ++index) + data_[size_ + index] = scalars[index]; + size_ += count; + } + + /// Appends the characters of a NUL-terminated ASCII text. + void + AppendAscii(const char *text) noexcept + { + for (; *text != '\0'; ++text) + Append( + static_cast(static_cast(*text))); + } + + void + Fill(const char32_t scalar, const std::size_t count) noexcept + { + Reserve(count); + for (std::size_t index = 0U; index < count; ++index) + data_[size_ + index] = scalar; + size_ += count; + } + + /// Reverses the scalars from `first` to the end. + void + ReverseFrom(const std::size_t first) noexcept + { + if (size_ < first + 2U) + return; + for (std::size_t low = first, high = size_ - 1U; low < high; + ++low, --high) + { + const auto kept = data_[low]; + data_[low] = data_[high]; + data_[high] = kept; + } + } + + private: + static constexpr std::size_t kInline = 64U; + + void + Reserve(const std::size_t more) noexcept + { + if (more <= capacity_ - size_) + return; + // Twice what is held, or what is asked for when that is more. + // A size that does not fit the address space cannot be met. + constexpr auto kLimit + = static_cast(-1) / sizeof(char32_t) / 2U; + if (more > kLimit - size_) + Platform::Fail(); + auto capacity = capacity_ * 2U; + if (capacity < size_ + more) + capacity = size_ + more; + // The allocation function is called by name, as the ownership + // runtime calls it for an object: what comes back is storage or + // null, and null is the only report of failure there is in a + // runtime built without exceptions. `capacity` is at most twice + // `kLimit`, so the size in bytes does not wrap. + auto *grown = static_cast( + ::operator new(capacity * sizeof(char32_t), std::nothrow)); + if (grown == nullptr) + Platform::Fail(); + for (std::size_t index = 0U; index < size_; ++index) + grown[index] = data_[index]; + if (data_ != inline_) + ::operator delete(data_); + data_ = grown; + capacity_ = capacity; + } + + char32_t inline_[kInline]{}; + char32_t *data_{ inline_ }; + std::size_t size_{}; + std::size_t capacity_{ kInline }; + }; +} // namespace Visual::XSharp::Runtime::Text diff --git a/Compiler/Runtime/Text/Tests/BUILD.bazel b/Compiler/Runtime/Text/Tests/BUILD.bazel new file mode 100644 index 00000000..f50e47a3 --- /dev/null +++ b/Compiler/Runtime/Text/Tests/BUILD.bazel @@ -0,0 +1,16 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +package(default_visibility = ["//visibility:private"]) + +cc_binary( + name = "text_runtime_tests", + srcs = ["TextRuntimeTests.cpp"], + deps = [ + "//Compiler/Headers/Visual/XSharp/Runtime:aarc_api", + "//Compiler/Headers/Visual/XSharp/Runtime:text_api", + "//Compiler/Runtime/AARC:aarc", + "//Compiler/Runtime/Text:text", + "@catch3//:catch2_main", + "@catch3//src/Progmasoft:catch3", + ], +) diff --git a/Compiler/Runtime/Text/Tests/TextRuntimeTests.cpp b/Compiler/Runtime/Text/Tests/TextRuntimeTests.cpp new file mode 100644 index 00000000..5b96386d --- /dev/null +++ b/Compiler/Runtime/Text/Tests/TextRuntimeTests.cpp @@ -0,0 +1,887 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Runtime/AARC.hpp" +#include "Visual/XSharp/Runtime/Text.hpp" + +// The text and console runtime, called the way generated code calls it: +// through its C entry points, with strings that are objects of the ownership +// runtime. Each expected text is written by hand, or, for the digits of a +// floating-point number, taken from the C library of the host, which +// produces the exact decimal expansion too and shares no code with the +// runtime. +// +// Every case releases what it was given, and the last case of the file +// checks that the runtime holds no more objects than before the first. + +namespace +{ + namespace Aarc = Visual::XSharp::Runtime::Aarc; + namespace Console = Visual::XSharp::Runtime::Console; + + /// The scalars of a string as UTF-32, and the string released. + [[nodiscard]] auto + Take(void *string) -> std::u32string + { + REQUIRE(string != nullptr); + const auto view = Aarc::ViewString(string); + std::u32string scalars(view.scalars, view.count); + vxs_aarc_release_strong(string); + return scalars; + } + + /// An ASCII text as the UTF-32 the runtime works in. + [[nodiscard]] auto + Wide(std::string_view text) -> std::u32string + { + std::u32string scalars; + for (const auto character : text) + scalars.push_back( + static_cast(static_cast(character))); + return scalars; + } + + [[nodiscard]] auto + Make(std::u32string_view scalars) -> void * + { + auto *string = Aarc::MakeString(scalars.data(), scalars.size()); + REQUIRE(string != nullptr); + return string; + } + + [[nodiscard]] auto + Signed(std::int64_t value, + std::int64_t flags = 0, + std::int64_t width = VXS_TEXT_ABSENT) -> std::u32string + { + return Take( + vxs_text_format_signed(flags, width, VXS_TEXT_ABSENT, value)); + } + + [[nodiscard]] auto + Unsigned(std::uint64_t value, + std::int64_t flags = 0, + std::int64_t width = VXS_TEXT_ABSENT) -> std::u32string + { + return Take( + vxs_text_format_unsigned(flags, width, VXS_TEXT_ABSENT, value)); + } + + [[nodiscard]] auto + Floating(double value, + std::int64_t precision = VXS_TEXT_ABSENT, + std::int64_t flags = 0, + std::int64_t width = VXS_TEXT_ABSENT) -> std::u32string + { + return Take(vxs_text_format_floating(flags, width, precision, value)); + } + + [[nodiscard]] auto + Text(std::u32string_view value, + std::int64_t flags = 0, + std::int64_t width = VXS_TEXT_ABSENT, + std::int64_t precision = VXS_TEXT_ABSENT) -> std::u32string + { + auto *string = Make(value); + auto result + = Take(vxs_text_format_string(flags, width, precision, string)); + vxs_aarc_release_strong(string); + return result; + } + + /// What the C library of the host writes for `%.*f`. + [[nodiscard]] auto + HostFixed(double value, int precision) -> std::u32string + { + std::string buffer(2048U, '\0'); + const auto written = std::snprintf(buffer.data(), + buffer.size(), + "%.*f", + precision, + value); + REQUIRE(written > 0); + REQUIRE(static_cast(written) < buffer.size()); + buffer.resize(static_cast(written)); + return Wide(buffer); + } + + /// Writes one integer with the C library of the host: the buffer, its + /// size, a width and the value. + template + using HostPrint = int (*)(char *, std::size_t, int, Integer); + + /// A row of a comparison with the host: the flags the runtime is given + /// and the conversion of the C library that means the same. The format + /// is a literal where it is used, so the compiler checks it. +#define HOST_ROW(Integer, flags, format) \ + { (flags), \ + (format), \ + [](char *buffer, std::size_t size, int width, Integer value) { \ + return std::snprintf(buffer, size, (format), width, value); \ + } } + + template + struct HostRow final + { + std::int64_t flags; + const char *format; + HostPrint print; + }; + + /// What the C library of the host writes for one integer conversion. + template + [[nodiscard]] auto + HostInteger(HostPrint print, int width, Integer value) + -> std::u32string + { + std::string buffer(128U, '\0'); + const auto written = print(buffer.data(), buffer.size(), width, value); + REQUIRE(written > 0); + REQUIRE(static_cast(written) < buffer.size()); + buffer.resize(static_cast(written)); + return Wide(buffer); + } + + /// A sequence of 64-bit values that is the same on every run. + class Sequence final + { + public: + [[nodiscard]] auto + Next() noexcept -> std::uint64_t + { + state += 0x9e3779b97f4a7c15ULL; + auto value = state; + value = (value ^ (value >> 30U)) * 0xbf58476d1ce4e5b9ULL; + value = (value ^ (value >> 27U)) * 0x94d049bb133111ebULL; + return value ^ (value >> 31U); + } + + private: + std::uint64_t state{ 0x5eedULL }; + }; + + /// Values of every magnitude: each draw is cut to a number of bits that + /// is drawn as well, so that small numbers are as frequent as large. + [[nodiscard]] auto + Magnitudes(std::size_t count) -> std::vector + { + Sequence sequence; + std::vector values{ + 0U, + 1U, + 9U, + 10U, + 999U, + 1000U, + std::numeric_limits::max(), + static_cast( + std::numeric_limits::max()), + static_cast( + std::numeric_limits::min()), + }; + while (values.size() < count) + { + const auto bits = 1U + static_cast(sequence.Next() % 64U); + values.push_back(sequence.Next() >> (64U - bits)); + } + return values; + } + + /// The bytes one sink received, by stream. + struct Captured final + { + std::string output; + std::string error; + std::size_t deliveries{}; + }; + + void + Receive(std::int64_t stream, + const char *bytes, + std::size_t count, + void *context) noexcept + { + auto *captured = static_cast(context); + (stream == VXS_CONSOLE_ERROR ? captured->error : captured->output) + .append(bytes, count); + ++captured->deliveries; + } + + /// Writes a string to a target with the sink installed. + [[nodiscard]] auto + Write(std::u32string_view scalars, std::int64_t target) -> Captured + { + Captured captured; + Console::SetSink(Receive, &captured); + auto *string = Make(scalars); + vxs_console_write(string, target); + vxs_aarc_release_strong(string); + Console::SetSink(nullptr, nullptr); + return captured; + } + +#ifdef _WIN32 + constexpr std::string_view kLine = "\r\n"; +#else + constexpr std::string_view kLine = "\n"; +#endif + + // What the runtime held before the first case of this file ran. + // NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables) + const auto liveAtStart = Aarc::LiveAllocations(); +} // namespace + +TEST_CASE("an integer is written in decimal with its sign") +{ + CHECK(Take(vxs_text_from_signed(0)) == U"0"); + CHECK(Take(vxs_text_from_signed(42)) == U"42"); + CHECK(Take(vxs_text_from_signed(-42)) == U"-42"); + CHECK(Take(vxs_text_from_signed(std::numeric_limits::max())) + == U"9223372036854775807"); + // The least value has no positive counterpart; its magnitude is still + // written whole. + CHECK(Take(vxs_text_from_signed(std::numeric_limits::min())) + == U"-9223372036854775808"); + CHECK(Take(vxs_text_from_unsigned(0U)) == U"0"); + CHECK( + Take(vxs_text_from_unsigned(std::numeric_limits::max())) + == U"18446744073709551615"); +} + +TEST_CASE("a Boolean is written as a word and a character as itself") +{ + CHECK(Take(vxs_text_from_bool(true)) == U"true"); + CHECK(Take(vxs_text_from_bool(false)) == U"false"); + CHECK(Take(vxs_text_from_char(U'a')) == U"a"); + CHECK(Take(vxs_text_from_char(0x1f600U)) + == std::u32string(1U, U'\U0001f600')); + CHECK(Take(vxs_text_from_char(0U)) == std::u32string(1U, U'\0')); +} + +TEST_CASE("storage that is not a scalar value is written as the replacement " + "character") +{ + CHECK(Take(vxs_text_from_char(0xd800U)) == U"\xfffd"); + CHECK(Take(vxs_text_from_char(0xdfffU)) == U"\xfffd"); + CHECK(Take(vxs_text_from_char(0x110000U)) == U"\xfffd"); + CHECK(Take(vxs_text_format_char(0, 3, VXS_TEXT_ABSENT, 0xffffffffU)) + == U" \xfffd"); +} + +TEST_CASE("two strings are joined in order") +{ + auto *left = Make(U"Hello, "); + auto *right = Make(U"world"); + auto *empty = Make(U""); + CHECK(Take(vxs_text_concat(left, right)) == U"Hello, world"); + CHECK(Take(vxs_text_concat(right, left)) == U"worldHello, "); + CHECK(Take(vxs_text_concat(left, empty)) == U"Hello, "); + CHECK(Take(vxs_text_concat(empty, empty)).empty()); + // The operands are borrowed: they are what they were. + CHECK(Take(vxs_text_concat(left, left)) == U"Hello, Hello, "); + vxs_aarc_release_strong(left); + vxs_aarc_release_strong(right); + vxs_aarc_release_strong(empty); +} + +TEST_CASE("a string that is not there is the empty string") +{ + auto *text = Make(U"kept"); + CHECK(Take(vxs_text_concat(nullptr, text)) == U"kept"); + CHECK(Take(vxs_text_concat(text, nullptr)) == U"kept"); + CHECK(Take(vxs_text_concat(nullptr, nullptr)).empty()); + CHECK(vxs_text_equals(nullptr, nullptr)); + CHECK_FALSE(vxs_text_equals(nullptr, text)); + CHECK(Take(vxs_text_format_string(0, 4, VXS_TEXT_ABSENT, nullptr)) + == U" "); + vxs_aarc_release_strong(text); +} + +TEST_CASE("a long result leaves the buffer it started in") +{ + // Longer than what a result keeps in place before it allocates. + std::u32string longText(1000U, U'x'); + auto *left = Make(longText); + auto *right = Make(longText); + const auto joined = Take(vxs_text_concat(left, right)); + CHECK(joined.size() == 2000U); + CHECK(joined == longText + longText); + vxs_aarc_release_strong(left); + vxs_aarc_release_strong(right); + CHECK(Signed(7, 0, 500).size() == 500U); + CHECK(Signed(7, 0, 500).back() == U'7'); + CHECK(Signed(7, VXS_TEXT_FLAG_ZERO, 500) + == std::u32string(499U, U'0') + U"7"); +} + +TEST_CASE("strings are compared by the scalars they hold") +{ + auto *first = Make(U"same"); + auto *second = Make(U"same"); + auto *shorter = Make(U"sam"); + auto *other = Make(U"sane"); + CHECK(first != second); + CHECK(vxs_text_equals(first, second)); + CHECK(vxs_text_equals(first, first)); + CHECK_FALSE(vxs_text_equals(first, shorter)); + CHECK_FALSE(vxs_text_equals(shorter, first)); + CHECK_FALSE(vxs_text_equals(first, other)); + for (auto *string : { first, second, shorter, other }) + vxs_aarc_release_strong(string); +} + +TEST_CASE("%d pads to its width at the left, the right, or with zeros") +{ + CHECK(Signed(42, 0, 5) == U" 42"); + CHECK(Signed(42, VXS_TEXT_FLAG_LEFT, 5) == U"42 "); + CHECK(Signed(42, VXS_TEXT_FLAG_ZERO, 5) == U"00042"); + CHECK(Signed(-42, 0, 5) == U" -42"); + CHECK(Signed(-42, VXS_TEXT_FLAG_LEFT, 5) == U"-42 "); + // Zeros stand after the sign. + CHECK(Signed(-42, VXS_TEXT_FLAG_ZERO, 5) == U"-0042"); + // A width the number already fills adds nothing, and neither does one + // that is absent, zero or negative. + CHECK(Signed(12345, 0, 2) == U"12345"); + CHECK(Signed(42, 0, 0) == U"42"); + CHECK(Signed(42, 0, -7) == U"42"); + CHECK(Signed(42, VXS_TEXT_FLAG_ZERO, 2) == U"42"); +} + +TEST_CASE("%d writes a sign for a number that is not negative when asked") +{ + CHECK(Signed(42, VXS_TEXT_FLAG_PLUS) == U"+42"); + CHECK(Signed(0, VXS_TEXT_FLAG_PLUS) == U"+0"); + CHECK(Signed(-42, VXS_TEXT_FLAG_PLUS) == U"-42"); + CHECK(Signed(42, VXS_TEXT_FLAG_SPACE) == U" 42"); + CHECK(Signed(-42, VXS_TEXT_FLAG_SPACE) == U"-42"); + CHECK(Signed(42, VXS_TEXT_FLAG_PLUS | VXS_TEXT_FLAG_ZERO, 6) == U"+00042"); + CHECK(Signed(42, VXS_TEXT_FLAG_PLUS, 6) == U" +42"); +} + +TEST_CASE("the ' flag groups decimal digits in threes") +{ + CHECK(Signed(0, VXS_TEXT_FLAG_GROUP) == U"0"); + CHECK(Signed(123, VXS_TEXT_FLAG_GROUP) == U"123"); + CHECK(Signed(1234, VXS_TEXT_FLAG_GROUP) == U"1'234"); + CHECK(Signed(123456, VXS_TEXT_FLAG_GROUP) == U"123'456"); + CHECK(Signed(1234567, VXS_TEXT_FLAG_GROUP) == U"1'234'567"); + CHECK(Signed(-1234567, VXS_TEXT_FLAG_GROUP) == U"-1'234'567"); + CHECK(Signed(std::numeric_limits::min(), VXS_TEXT_FLAG_GROUP) + == U"-9'223'372'036'854'775'808"); + CHECK( + Unsigned(std::numeric_limits::max(), VXS_TEXT_FLAG_GROUP) + == U"18'446'744'073'709'551'615"); + CHECK(Signed(1234567, VXS_TEXT_FLAG_GROUP, 12) == U" 1'234'567"); +} + +TEST_CASE("%x writes a magnitude in lowercase hexadecimal") +{ + constexpr auto kHex = VXS_TEXT_FLAG_HEXADECIMAL; + CHECK(Signed(0, kHex) == U"0"); + CHECK(Signed(255, kHex) == U"ff"); + CHECK(Signed(48879, kHex) == U"beef"); + // A negative number is a sign and its magnitude, never the bit pattern + // of its representation. + CHECK(Signed(-255, kHex) == U"-ff"); + CHECK(Signed(-1, kHex) == U"-1"); + CHECK(Signed(std::numeric_limits::min(), kHex) + == U"-8000000000000000"); + CHECK(Unsigned(std::numeric_limits::max(), kHex) + == U"ffffffffffffffff"); + CHECK(Signed(255, kHex | VXS_TEXT_FLAG_ALTERNATE) == U"0xff"); + CHECK(Signed(-255, kHex | VXS_TEXT_FLAG_ALTERNATE) == U"-0xff"); + // Zeros stand after the sign and the prefix. + CHECK(Signed(255, kHex | VXS_TEXT_FLAG_ALTERNATE | VXS_TEXT_FLAG_ZERO, 8) + == U"0x0000ff"); + CHECK(Signed(255, kHex | VXS_TEXT_FLAG_ZERO, 8) == U"000000ff"); + CHECK(Signed(255, kHex, 6) == U" ff"); + // Grouping is a flag of the decimal conversions. + CHECK(Signed(0x1234567, kHex | VXS_TEXT_FLAG_GROUP) == U"1234567"); +} + +TEST_CASE("%s pads, and keeps at most its precision of characters") +{ + CHECK(Text(U"abc") == U"abc"); + CHECK(Text(U"abc", 0, 6) == U" abc"); + CHECK(Text(U"abc", VXS_TEXT_FLAG_LEFT, 6) == U"abc "); + CHECK(Text(U"abcdef", 0, VXS_TEXT_ABSENT, 2) == U"ab"); + CHECK(Text(U"abc", 0, VXS_TEXT_ABSENT, 5) == U"abc"); + CHECK(Text(U"abc", 0, VXS_TEXT_ABSENT, 0).empty()); + CHECK(Text(U"abcdef", 0, 5, 1) == U" a"); + // Zeros are for numbers; a string is padded with spaces whatever the + // flag says. + CHECK(Text(U"abc", VXS_TEXT_FLAG_ZERO, 6) == U" abc"); + CHECK(Text(U"", 0, 3) == U" "); +} + +TEST_CASE("width and precision count characters, not bytes") +{ + // Three characters that take two, three and four bytes in UTF-8. + const std::u32string mixed{ U'\xe9', U'\x20ac', U'\U0001f600' }; + CHECK(Text(mixed, 0, 5) == U" " + mixed); + CHECK(Text(mixed, 0, VXS_TEXT_ABSENT, 2) == mixed.substr(0U, 2U)); + CHECK(Take(vxs_text_format_char(0, 3, VXS_TEXT_ABSENT, 0x20acU)) + == std::u32string(U" ") + U'\x20ac'); + CHECK( + Take(vxs_text_format_char(VXS_TEXT_FLAG_LEFT, 3, VXS_TEXT_ABSENT, U'q')) + == U"q "); +} + +TEST_CASE("%f writes six digits after the point unless told otherwise") +{ + CHECK(Floating(0.0) == U"0.000000"); + CHECK(Floating(1.0) == U"1.000000"); + CHECK(Floating(12.5) == U"12.500000"); + CHECK(Floating(-12.5) == U"-12.500000"); + CHECK(Floating(12.5, 2) == U"12.50"); + CHECK(Floating(12.5, 0) == U"12"); + CHECK(Floating(3.14159, 3) == U"3.142"); + CHECK(Floating(1e6, 1) == U"1000000.0"); + CHECK(Floating(123456789.0, 0) == U"123456789"); +} + +TEST_CASE("%f rounds the exact value, a tie to the even digit") +{ + // These values are exact in binary, so the tie is real. + CHECK(Floating(0.5, 0) == U"0"); + CHECK(Floating(1.5, 0) == U"2"); + CHECK(Floating(2.5, 0) == U"2"); + CHECK(Floating(3.5, 0) == U"4"); + CHECK(Floating(0.125, 2) == U"0.12"); + CHECK(Floating(0.375, 2) == U"0.38"); + CHECK(Floating(0.25, 1) == U"0.2"); + CHECK(Floating(0.75, 1) == U"0.8"); + CHECK(Floating(-2.5, 0) == U"-2"); + // 0.1 is not exact: it lies above one tenth, which its digits show. + CHECK(Floating(0.1, 20) == U"0.10000000000000000555"); + CHECK(Floating(0.1, 1) == U"0.1"); + CHECK(Floating(0.15, 1) == U"0.1"); + CHECK(Floating(0.25, 0) == U"0"); + // Rounding carries into the integer part. + CHECK(Floating(9.999, 2) == U"10.00"); + CHECK(Floating(0.999, 0) == U"1"); + CHECK(Floating(99.5, 0) == U"100"); +} + +TEST_CASE("%f keeps the sign of a negative zero and of what rounds to zero") +{ + CHECK(Floating(-0.0) == U"-0.000000"); + CHECK(Floating(-0.0, 0) == U"-0"); + CHECK(Floating(-0.001, 1) == U"-0.0"); + CHECK(Floating(0.0, 0, VXS_TEXT_FLAG_PLUS) == U"+0"); +} + +TEST_CASE("%f pads, signs and groups like the integer conversions") +{ + CHECK(Floating(3.14159, 3, 0, 10) == U" 3.142"); + CHECK(Floating(3.14159, 3, VXS_TEXT_FLAG_LEFT, 10) == U"3.142 "); + CHECK(Floating(3.14159, 3, VXS_TEXT_FLAG_ZERO, 10) == U"000003.142"); + CHECK(Floating(-3.14159, 3, VXS_TEXT_FLAG_ZERO, 10) == U"-00003.142"); + CHECK(Floating(1.5, 1, VXS_TEXT_FLAG_PLUS) == U"+1.5"); + CHECK(Floating(1.5, 1, VXS_TEXT_FLAG_SPACE) == U" 1.5"); + // Only the integer digits are grouped. + CHECK(Floating(1234567.5, 2, VXS_TEXT_FLAG_GROUP) == U"1'234'567.50"); + CHECK(Floating(1234.56789, 5, VXS_TEXT_FLAG_GROUP) == U"1'234.56789"); + CHECK(Floating(999.5, 0, VXS_TEXT_FLAG_GROUP) == U"1'000"); +} + +TEST_CASE("%f writes what is not a finite number as a word") +{ + const auto infinity = std::numeric_limits::infinity(); + const auto notANumber = std::numeric_limits::quiet_NaN(); + CHECK(Floating(infinity) == U"inf"); + CHECK(Floating(-infinity) == U"-inf"); + CHECK(Floating(notANumber) == U"nan"); + CHECK(Floating(infinity, VXS_TEXT_ABSENT, VXS_TEXT_FLAG_PLUS) == U"+inf"); + // A word is padded with spaces, also where a number would get zeros. + CHECK(Floating(infinity, VXS_TEXT_ABSENT, VXS_TEXT_FLAG_ZERO, 6) + == U" inf"); + CHECK(Floating(notANumber, 2, VXS_TEXT_FLAG_LEFT, 5) == U"nan "); +} + +TEST_CASE("%f writes every digit of the extremes of the type") +{ + const auto least = std::numeric_limits::denorm_min(); + const auto greatest = std::numeric_limits::max(); + // The least positive value has 1074 digits after the point, the last of + // which is not zero; one more digit is a zero. + const auto digits = Floating(least, 1074); + CHECK(digits.size() == 1076U); + CHECK(digits.substr(0U, 10U) == U"0.00000000"); + CHECK(digits.back() == U'5'); + CHECK(Floating(least, 1075).back() == U'0'); + CHECK(Floating(least, 1075).substr(0U, 1076U) == digits); + CHECK(Floating(least, 1100).size() == 1102U); + CHECK(Floating(least) == U"0.000000"); + // The greatest value has 309 digits before the point and is an integer. + const auto whole = Floating(greatest, 0); + CHECK(whole.size() == 309U); + CHECK(whole.substr(0U, 17U) == U"17976931348623157"); + CHECK(Floating(greatest, 2).substr(309U) == U".00"); + CHECK(Floating(-greatest, 0) == U"-" + whole); +} + +TEST_CASE("%f agrees with the C library of the host") +{ + // The library of the host writes the correctly rounded expansion of the + // exact value as well; the two are independent implementations. + const std::vector values{ 0.0, + 1.0, + 0.1, + 0.2, + 0.3, + 1.0 / 3.0, + 2.0 / 3.0, + 1e-5, + 1e-10, + 1e-20, + 123.456, + 1e15, + 1e16, + 1e17, + 1e22, + 1e23, + 9.87654321e30, + 5e-324, + 2.2250738585072014e-308, + 1.7976931348623157e308, + 4.35, + 0.045, + 1.005, + 2.675, + 1234567.891, + 0.000123456789, + 8.5, + 1e100, + 3.0e-7, + 0.5e-6, + 1.9999999, + 65536.0, + 4294967296.5, + 0.999999999999999889 }; + for (const auto value : values) + for (const int precision : { 0, 1, 2, 3, 6, 9, 15, 17, 20, 40 }) + { + CAPTURE(value, precision); + CHECK(Floating(value, precision) == HostFixed(value, precision)); + CHECK(Floating(-value, precision) == HostFixed(-value, precision)); + } +} + +TEST_CASE("%f agrees with the host over a sweep of binary exponents") +{ + // One value in every tenth binade, with a significand that uses all of + // its bits, so that the scaling and the rounding meet every size of + // number. + for (int exponent = -1070; exponent <= 1020; exponent += 10) + { + const auto value = std::ldexp(1.2345678901234567, exponent); + for (const int precision : { 0, 3, 12, 30 }) + { + CAPTURE(exponent, precision); + CHECK(Floating(value, precision) == HostFixed(value, precision)); + } + } +} + +TEST_CASE("%d agrees with the C library of the host over many values") +{ + const HostRow kRows[] = { + HOST_ROW(long long, 0, "%*lld"), + HOST_ROW(long long, VXS_TEXT_FLAG_LEFT, "%-*lld"), + HOST_ROW(long long, VXS_TEXT_FLAG_ZERO, "%0*lld"), + HOST_ROW(long long, VXS_TEXT_FLAG_PLUS, "%+*lld"), + HOST_ROW(long long, VXS_TEXT_FLAG_SPACE, "% *lld"), + HOST_ROW(long long, VXS_TEXT_FLAG_PLUS | VXS_TEXT_FLAG_ZERO, "%+0*lld"), + HOST_ROW(long long, + VXS_TEXT_FLAG_SPACE | VXS_TEXT_FLAG_ZERO, + "% 0*lld"), + HOST_ROW(long long, VXS_TEXT_FLAG_PLUS | VXS_TEXT_FLAG_LEFT, "%-+*lld"), + HOST_ROW(long long, + VXS_TEXT_FLAG_SPACE | VXS_TEXT_FLAG_LEFT, + "%- *lld"), + }; + std::size_t compared = 0U; + for (const auto magnitude : Magnitudes(400U)) + { + const auto value = static_cast(magnitude); + for (const auto &row : kRows) + { + for (const int width : { 0, 1, 7, 19, 20, 21, 40 }) + { + const auto written = Signed(value, row.flags, width); + const auto expected + = HostInteger(row.print, + width, + static_cast(value)); + if (written != expected) + { + CAPTURE(value); + CAPTURE(row.format); + CAPTURE(width); + CHECK(written == expected); + } + ++compared; + } + } + } + CHECK(compared == std::size_t{ 25200U }); +} + +TEST_CASE("%u and %x agree with the C library of the host over many values") +{ + const HostRow kRows[] = { + HOST_ROW(unsigned long long, 0, "%*llu"), + HOST_ROW(unsigned long long, VXS_TEXT_FLAG_LEFT, "%-*llu"), + HOST_ROW(unsigned long long, VXS_TEXT_FLAG_ZERO, "%0*llu"), + HOST_ROW(unsigned long long, VXS_TEXT_FLAG_HEXADECIMAL, "%*llx"), + HOST_ROW(unsigned long long, + VXS_TEXT_FLAG_HEXADECIMAL | VXS_TEXT_FLAG_LEFT, + "%-*llx"), + HOST_ROW(unsigned long long, + VXS_TEXT_FLAG_HEXADECIMAL | VXS_TEXT_FLAG_ZERO, + "%0*llx"), + }; + std::size_t compared = 0U; + for (const auto value : Magnitudes(400U)) + { + for (const auto &row : kRows) + { + for (const int width : { 0, 1, 7, 16, 17, 20, 21, 40 }) + { + const auto written = Unsigned(value, row.flags, width); + const auto expected + = HostInteger(row.print, + width, + static_cast(value)); + if (written != expected) + { + CAPTURE(value); + CAPTURE(row.format); + CAPTURE(width); + CHECK(written == expected); + } + ++compared; + } + } + } + CHECK(compared == std::size_t{ 19200U }); +} + +TEST_CASE("grouping only adds apostrophes, every third digit from the right") +{ + for (const auto value : Magnitudes(400U)) + { + const auto plain = Unsigned(value); + const auto grouped = Unsigned(value, VXS_TEXT_FLAG_GROUP); + std::u32string rebuilt; + for (std::size_t index = 0U; index < plain.size(); ++index) + { + if (index != 0U && (plain.size() - index) % 3U == 0U) + rebuilt.push_back(U'\''); + rebuilt.push_back(plain[index]); + } + if (grouped != rebuilt) + { + CAPTURE(value); + CHECK(grouped == rebuilt); + } + // A sign stands before the first group and is not counted in it. + const auto negative = -static_cast(value >> 1U); + if (negative != 0) + { + const auto signedGrouped = Signed(negative, VXS_TEXT_FLAG_GROUP); + const auto magnitude = Unsigned(value >> 1U, VXS_TEXT_FLAG_GROUP); + if (signedGrouped != U"-" + magnitude) + { + CAPTURE(negative); + CHECK(signedGrouped == U"-" + magnitude); + } + } + } +} + +TEST_CASE("%f agrees with the host over numbers drawn from every exponent") +{ + // Any bit pattern that is a finite number: the sign, the exponent and + // the fraction are all drawn, so subnormal numbers, numbers with + // hundreds of digits before the point and numbers with hundreds of + // zeros after it all occur. + Sequence sequence; + std::size_t compared = 0U; + while (compared < 3000U) + { + const auto bits = sequence.Next(); + double value = 0.0; + static_assert(sizeof(value) == sizeof(bits)); + std::memcpy(&value, &bits, sizeof(value)); + if (!std::isfinite(value)) + continue; + const auto precision = static_cast(sequence.Next() % 25U); + const auto written = Floating(value, precision); + const auto expected = HostFixed(value, precision); + if (written != expected) + { + CAPTURE(bits); + CAPTURE(precision); + CHECK(written == expected); + } + ++compared; + } + CHECK(compared == 3000U); +} + +TEST_CASE("%f agrees with the host on numbers a digit decides the rounding " + "of") +{ + // Multiples of a power of two small enough to be exact: half of them + // lie exactly between two numbers of the precision that is asked for. + std::size_t compared = 0U; + for (int numerator = -2000; numerator <= 2000; ++numerator) + { + for (const double denominator : { 2.0, 8.0, 32.0, 1024.0 }) + { + const auto value = static_cast(numerator) / denominator; + for (int precision = 0; precision <= 4; ++precision) + { + const auto written = Floating(value, precision); + const auto expected = HostFixed(value, precision); + if (written != expected) + { + CAPTURE(numerator); + CAPTURE(denominator); + CAPTURE(precision); + CHECK(written == expected); + } + ++compared; + } + } + } + CHECK(compared == std::size_t{ 80020U }); +} + +TEST_CASE("a field is never shorter than its width and never loses a digit") +{ + for (const auto magnitude : Magnitudes(200U)) + { + const auto value = static_cast(magnitude); + const auto plain = Signed(value); + for (std::int64_t width = 0; width <= 48; width += 3) + { + for (const auto flags : + { INT64_C(0), VXS_TEXT_FLAG_LEFT, VXS_TEXT_FLAG_ZERO }) + { + const auto field = Signed(value, flags, width); + const auto least = static_cast(width); + if (field.size() + != (plain.size() > least ? plain.size() : least)) + { + CAPTURE(value); + CAPTURE(width); + CAPTURE(flags); + CHECK(field.size() == least); + } + } + } + } +} + +TEST_CASE("%n is the line terminator of the platform") +{ + CHECK(Take(vxs_text_newline()) == Wide(kLine)); +} + +TEST_CASE("the console receives a string as UTF-8") +{ + CHECK(Write(U"plain", VXS_CONSOLE_OUTPUT).output == "plain"); + CHECK(Write(U"", VXS_CONSOLE_OUTPUT).output.empty()); + // One character of each encoded length. + const std::u32string mixed{ U'a', U'\xe9', U'\x20ac', U'\U0001f600' }; + CHECK(Write(mixed, VXS_CONSOLE_OUTPUT).output + == "a\xc3\xa9\xe2\x82\xac\xf0\x9f\x98\x80"); + // The first and the last scalar of each length. + CHECK(Write(std::u32string{ U'\x7f', U'\x80' }, VXS_CONSOLE_OUTPUT).output + == "\x7f\xc2\x80"); + CHECK(Write(std::u32string{ U'\x7ff', U'\x800' }, VXS_CONSOLE_OUTPUT).output + == "\xdf\xbf\xe0\xa0\x80"); + CHECK(Write(std::u32string{ U'\xffff', U'\U00010000' }, VXS_CONSOLE_OUTPUT) + .output + == "\xef\xbf\xbf\xf0\x90\x80\x80"); + CHECK(Write(std::u32string{ U'\U0010ffff' }, VXS_CONSOLE_OUTPUT).output + == "\xf4\x8f\xbf\xbf"); +} + +TEST_CASE("the target of a write selects the stream and the line ending") +{ + const auto output = Write(U"a", VXS_CONSOLE_OUTPUT); + CHECK(output.output == "a"); + CHECK(output.error.empty()); + + const auto outputLine = Write(U"a", VXS_CONSOLE_OUTPUT_LINE); + CHECK(outputLine.output == "a" + std::string(kLine)); + CHECK(outputLine.error.empty()); + + const auto error = Write(U"a", VXS_CONSOLE_ERROR); + CHECK(error.output.empty()); + CHECK(error.error == "a"); + + const auto errorLine = Write(U"a", VXS_CONSOLE_ERROR_LINE); + CHECK(errorLine.output.empty()); + CHECK(errorLine.error == "a" + std::string(kLine)); + + // An empty line is its terminator. + CHECK(Write(U"", VXS_CONSOLE_OUTPUT_LINE).output == kLine); + // Nothing is delivered for nothing. + CHECK(Write(U"", VXS_CONSOLE_OUTPUT).deliveries == 0U); +} + +TEST_CASE("a long write is delivered whole, in blocks that end between " + "characters") +{ + // Four-byte characters, enough of them for several blocks, offset so + // that a block boundary would fall inside a character if the encoder + // let it. + std::u32string scalars(U"x"); + scalars.append(1000U, U'\U0001f600'); + const auto captured = Write(scalars, VXS_CONSOLE_OUTPUT_LINE); + std::string expected = "x"; + for (int index = 0; index < 1000; ++index) + expected += "\xf0\x9f\x98\x80"; + expected += kLine; + CHECK(captured.output == expected); + CHECK(captured.deliveries > 1U); +} + +TEST_CASE("writes reach the sink in the order they were made") +{ + Captured captured; + Console::SetSink(Receive, &captured); + for (const auto *text : { U"one", U"two", U"three" }) + { + auto *string = Make(text); + vxs_console_write(string, VXS_CONSOLE_OUTPUT); + vxs_console_write(string, VXS_CONSOLE_ERROR_LINE); + vxs_aarc_release_strong(string); + } + Console::SetSink(nullptr, nullptr); + CHECK(captured.output == "onetwothree"); + CHECK(captured.error + == "one" + std::string(kLine) + "two" + std::string(kLine) + "three" + + std::string(kLine)); +} + +TEST_CASE("the text runtime leaves no object behind") +{ + // Every case above released what it was given and what it made. + CHECK(Aarc::LiveAllocations() == liveAtStart); +} diff --git a/Compiler/Runtime/Text/Text.cpp b/Compiler/Runtime/Text/Text.cpp new file mode 100644 index 00000000..e8ccf0eb --- /dev/null +++ b/Compiler/Runtime/Text/Text.cpp @@ -0,0 +1,214 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include "Format.hpp" +#include "Platform.hpp" +#include "Scalars.hpp" +#include "Visual/XSharp/Runtime/AARC.hpp" +#include "Visual/XSharp/Runtime/Text.h" + +// The C entry points of the text runtime. Each builds its result in a +// sequence of scalars and hands it to the ownership runtime as a string. + +namespace Visual::XSharp::Runtime::Text +{ + namespace + { + constexpr char32_t kReplacement = U'\xfffd'; + + /// The string that holds the scalars, owned by the caller. The + /// program stops when the string cannot be allocated: a text + /// routine has no way to report failure, and a null result would + /// only move the failure to whoever reads it. + [[nodiscard]] auto + Finish(const Scalars &scalars) noexcept -> void * + { + auto *string = Aarc::MakeString(scalars.Data(), scalars.Size()); + if (string == nullptr) + Platform::Fail(); + return string; + } + + [[nodiscard]] auto + ConversionOf(const std::int64_t flags, + const std::int64_t width, + const std::int64_t precision) noexcept -> Conversion + { + return { flags, width, precision }; + } + + /// A `char` holds a Unicode scalar value. Storage that holds + /// something else is written as the replacement character rather + /// than put into a string, which admits scalar values only. + [[nodiscard]] auto + Scalar(const std::uint32_t value) noexcept -> char32_t + { + return value > 0x10ffffU || (value >= 0xd800U && value <= 0xdfffU) + ? kReplacement + : static_cast(value); + } + + [[nodiscard]] auto + Magnitude(const std::int64_t value) noexcept -> std::uint64_t + { + // Negating in unsigned arithmetic gives the magnitude of the + // least value as well, which has no positive counterpart. + return value < 0 + ? std::uint64_t{ 0U } - static_cast(value) + : static_cast(value); + } + } // namespace +} // namespace Visual::XSharp::Runtime::Text + +extern "C" +{ + auto + vxs_text_equals(void *left, void *right) noexcept -> bool + { + using namespace Visual::XSharp::Runtime; + const auto first = Aarc::ViewString(left); + const auto second = Aarc::ViewString(right); + if (first.count != second.count) + return false; + for (std::size_t index = 0U; index < first.count; ++index) + if (first.scalars[index] != second.scalars[index]) + return false; + return true; + } + + auto + vxs_text_concat(void *left, void *right) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + const auto first = Aarc::ViewString(left); + const auto second = Aarc::ViewString(right); + Text::Scalars scalars; + scalars.Append(first.scalars, first.count); + scalars.Append(second.scalars, second.count); + return Text::Finish(scalars); + } + + auto + vxs_text_from_signed(const std::int64_t value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + Text::AppendInteger(scalars, value < 0, Text::Magnitude(value), {}); + return Text::Finish(scalars); + } + + auto + vxs_text_from_unsigned(const std::uint64_t value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + Text::AppendInteger(scalars, false, value, {}); + return Text::Finish(scalars); + } + + auto + vxs_text_from_bool(const bool value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + scalars.AppendAscii(value ? "true" : "false"); + return Text::Finish(scalars); + } + + auto + vxs_text_from_char(const std::uint32_t value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + scalars.Append(Text::Scalar(value)); + return Text::Finish(scalars); + } + + auto + vxs_text_format_signed(const std::int64_t flags, + const std::int64_t width, + const std::int64_t precision, + const std::int64_t value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + Text::AppendInteger(scalars, + value < 0, + Text::Magnitude(value), + Text::ConversionOf(flags, width, precision)); + return Text::Finish(scalars); + } + + auto + vxs_text_format_unsigned(const std::int64_t flags, + const std::int64_t width, + const std::int64_t precision, + const std::uint64_t value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + Text::AppendInteger(scalars, + false, + value, + Text::ConversionOf(flags, width, precision)); + return Text::Finish(scalars); + } + + auto + vxs_text_format_floating(const std::int64_t flags, + const std::int64_t width, + const std::int64_t precision, + const double value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + Text::AppendFloating(scalars, + value, + Text::ConversionOf(flags, width, precision)); + return Text::Finish(scalars); + } + + auto + vxs_text_format_string(const std::int64_t flags, + const std::int64_t width, + const std::int64_t precision, + void *value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + const auto view = Aarc::ViewString(value); + Text::Scalars scalars; + Text::AppendText(scalars, + view.scalars, + view.count, + Text::ConversionOf(flags, width, precision)); + return Text::Finish(scalars); + } + + auto + vxs_text_format_char(const std::int64_t flags, + const std::int64_t width, + const std::int64_t precision, + const std::uint32_t value) noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + const auto scalar = Text::Scalar(value); + Text::Scalars scalars; + // A character has no precision; the argument is there so that every + // conversion has one signature. + static_cast(precision); + Text::AppendText(scalars, + &scalar, + 1U, + Text::ConversionOf(flags, width, VXS_TEXT_ABSENT)); + return Text::Finish(scalars); + } + + auto + vxs_text_newline() noexcept -> void * + { + using namespace Visual::XSharp::Runtime; + Text::Scalars scalars; + scalars.AppendAscii(Platform::LineTerminator()); + return Text::Finish(scalars); + } +} diff --git a/Compiler/Runtime/host.bzl b/Compiler/Runtime/host.bzl new file mode 100644 index 00000000..f8e5177f --- /dev/null +++ b/Compiler/Runtime/host.bzl @@ -0,0 +1,56 @@ +"""Link options of a program that hosts generated code. + +A compiled program calls the Visual X# runtime: a closure allocates through +it, an argument that is passed by need does, a string is one of its objects, +and console output is one of its functions. The JIT resolves a runtime symbol +in the process that hosts it, so every program that runs compiled programs +links the runtime libraries and exports their entry points. + +The list is the C ABI of `Visual/XSharp/Runtime/AARC.h` and +`Visual/XSharp/Runtime/Text.h`. A function added to either header is added +here, or a program that calls it fails to resolve it under the JIT on +Windows, where nothing is exported that is not named. +""" + +RUNTIME_HOST_DEPS = [ + "//Compiler/Headers/Visual/XSharp/Runtime:aarc_api", + "//Compiler/Headers/Visual/XSharp/Runtime:text_api", + "//Compiler/Runtime/AARC:aarc", + "//Compiler/Runtime/Text:text", +] + +_ENTRY_POINTS = [ + "vxs_aarc_abi_version", + "vxs_aarc_allocate", + "vxs_aarc_copy_unowned", + "vxs_aarc_copy_weak", + "vxs_aarc_is_exact_type", + "vxs_aarc_load_unowned", + "vxs_aarc_lock_weak", + "vxs_aarc_make_unowned", + "vxs_aarc_make_weak", + "vxs_aarc_release_strong", + "vxs_aarc_release_unowned", + "vxs_aarc_release_weak", + "vxs_aarc_retain_strong", + "vxs_aarc_string_literal", + "vxs_console_write", + "vxs_text_concat", + "vxs_text_equals", + "vxs_text_format_char", + "vxs_text_format_floating", + "vxs_text_format_signed", + "vxs_text_format_string", + "vxs_text_format_unsigned", + "vxs_text_from_bool", + "vxs_text_from_char", + "vxs_text_from_signed", + "vxs_text_from_unsigned", + "vxs_text_newline", +] + +RUNTIME_HOST_LINKOPTS = select({ + "@platforms//os:windows": ["/EXPORT:" + name for name in _ENTRY_POINTS], + "@platforms//os:macos": [], + "//conditions:default": ["-rdynamic"], +}) diff --git a/Compiler/Support/BUILD.bazel b/Compiler/Support/BUILD.bazel new file mode 100644 index 00000000..ad8d8e0e --- /dev/null +++ b/Compiler/Support/BUILD.bazel @@ -0,0 +1,14 @@ +load("@rules_cc//cc:cc_library.bzl", "cc_library") + +package(default_visibility = ["//visibility:public"]) + +# The thread the compiler runs on. Creating it needs the platform's own +# thread interface: the stack must be reserved, not committed. +cc_library( + name = "compiler_stack", + srcs = ["CompilerStack.cpp"], + deps = [ + "//Compiler/Headers/Visual/XSharp/Support:compiler_stack_api", + "@llvm//:llvm", + ], +) diff --git a/Compiler/Support/CompilerStack.cpp b/Compiler/Support/CompilerStack.cpp new file mode 100644 index 00000000..22c1b1d0 --- /dev/null +++ b/Compiler/Support/CompilerStack.cpp @@ -0,0 +1,161 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include + +#include "Visual/XSharp/Support/CompilerStack.hpp" + +#if defined(_WIN32) +# ifndef WIN32_LEAN_AND_MEAN +# define WIN32_LEAN_AND_MEAN +# endif +# ifndef NOMINMAX +# define NOMINMAX +# endif +# include +# include +#else +# include +# include +# include +# include +#endif + +namespace Visual::XSharp::Support +{ + namespace + { + struct Start final + { + void (*function)(void *); + void *argument; + }; + +#if defined(_WIN32) + unsigned __stdcall + Enter(void *context) + { + const auto *start = static_cast(context); + start->function(start->argument); + return 0U; + } +#else + void * + Enter(void *context) + { + const auto *start = static_cast(context); + start->function(start->argument); + return nullptr; + } +#endif + } // namespace + + void + RunOnCompilerStackRaw(void (*function)(void *), void *argument) + { + RunOnStackRaw(kCompilerStackBytes, function, argument); + } + + void + RunOnStackRaw(std::size_t bytes, void (*function)(void *), void *argument) + { + Start start{ function, argument }; +#if defined(_WIN32) + // Without the reservation flag the size would be committed up + // front: every run would charge the whole stack against the + // system's commit limit and pay for mapping it. + const auto handle = _beginthreadex(nullptr, + static_cast(bytes), + Enter, + &start, + STACK_SIZE_PARAM_IS_A_RESERVATION, + nullptr); + if (handle == 0U) + llvm::report_fatal_error("could not start the compiler thread"); + // _beginthreadex returns the thread handle as an integer; this cast + // is how its documentation says to recover the handle. + // NOLINTNEXTLINE(performance-no-int-to-ptr) + const auto thread = reinterpret_cast(handle); + if (WaitForSingleObject(thread, INFINITE) != WAIT_OBJECT_0) + llvm::report_fatal_error("could not wait for the compiler thread"); + CloseHandle(thread); +#else + // A pthread stack is mapped lazily, so the size is a reservation. + pthread_attr_t attributes; + if (pthread_attr_init(&attributes) != 0 + || pthread_attr_setstacksize(&attributes, bytes) != 0) + llvm::report_fatal_error( + "could not size the stack of the compiler thread"); + pthread_t thread; + if (pthread_create(&thread, &attributes, Enter, &start) != 0) + llvm::report_fatal_error("could not start the compiler thread"); + pthread_attr_destroy(&attributes); + if (pthread_join(thread, nullptr) != 0) + llvm::report_fatal_error("could not wait for the compiler thread"); +#endif + } + + auto + CommittedStackBytes() -> std::size_t + { +#if defined(_WIN32) + MEMORY_BASIC_INFORMATION information{}; + const char here = 0; + if (VirtualQuery(&here, &information, sizeof(information)) == 0U) + return 0U; + // A thread stack is one allocation: reserved pages at the bottom, + // then the guard page, then the committed pages up to the top. + const auto *const allocation = information.AllocationBase; + std::size_t committed = 0U; + // NOLINTBEGIN(cppcoreguidelines-pro-bounds-pointer-arithmetic) + const auto *cursor = static_cast(allocation); + while (VirtualQuery(cursor, &information, sizeof(information)) != 0U + && information.AllocationBase == allocation) + { + if (information.State == MEM_COMMIT) + committed += information.RegionSize; + cursor += information.RegionSize; + } + // NOLINTEND(cppcoreguidelines-pro-bounds-pointer-arithmetic) + return committed; +#elif defined(__APPLE__) || defined(__linux__) + // A thread stack is mapped lazily, so the pages that are resident + // are the pages the thread has touched. The stack of the initial + // thread is not one mapping; for it the query fails and nothing is + // reported. + void *low = nullptr; + std::size_t size = 0U; +# if defined(__APPLE__) + const auto self = pthread_self(); + size = pthread_get_stacksize_np(self); + // The address the system reports is the high end of the stack. + // NOLINTNEXTLINE(cppcoreguidelines-pro-bounds-pointer-arithmetic) + low = static_cast(pthread_get_stackaddr_np(self)) - size; + using PageState = char; +# else + pthread_attr_t attributes; + if (pthread_getattr_np(pthread_self(), &attributes) != 0) + return 0U; + const auto found = pthread_attr_getstack(&attributes, &low, &size); + pthread_attr_destroy(&attributes); + if (found != 0) + return 0U; + using PageState = unsigned char; +# endif + const auto pageSize = sysconf(_SC_PAGESIZE); + if (pageSize <= 0 || size == 0U || low == nullptr) + return 0U; + const auto page = static_cast(pageSize); + std::vector states((size + page - 1U) / page); + if (mincore(low, size, states.data()) != 0) + return 0U; + std::size_t committed = 0U; + for (const auto state : states) + if ((static_cast(state) & 1U) != 0U) + committed += page; + return committed; +#else + return 0U; +#endif + } +} // namespace Visual::XSharp::Support diff --git a/Compiler/Support/Tests/BUILD.bazel b/Compiler/Support/Tests/BUILD.bazel new file mode 100644 index 00000000..33749ec4 --- /dev/null +++ b/Compiler/Support/Tests/BUILD.bazel @@ -0,0 +1,19 @@ +load("@rules_cc//cc:cc_binary.bzl", "cc_binary") + +package(default_visibility = ["//visibility:public"]) + +# A measuring instrument, not a test: it runs one native stage on a stack of +# a chosen size, so that the stack a stage needs per level of nesting can be +# found by searching for the smallest size that completes. +cc_binary( + name = "stack_probe", + srcs = ["StackProbe.cpp"], + deps = [ + "//Compiler/Core:core", + "//Compiler/Driver:pipeline", + "//Compiler/Headers/Visual/XSharp:pipeline_api", + "//Compiler/Headers/Visual/XSharp/Core:scalar", + "//Compiler/Support:compiler_stack", + "@llvm//:llvm", + ], +) diff --git a/Compiler/Support/Tests/StackProbe.cpp b/Compiler/Support/Tests/StackProbe.cpp new file mode 100644 index 00000000..f48e0f7d --- /dev/null +++ b/Compiler/Support/Tests/StackProbe.cpp @@ -0,0 +1,251 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +#include +#include +#include +#include +#include +#include +#include + +#include "Visual/XSharp/Core/CorePrep/Prepare.hpp" +#include "Visual/XSharp/Core/Scalar.hpp" +#include "Visual/XSharp/Core/Verifier.hpp" +#include "Visual/XSharp/Core/Wire.hpp" +#include "Visual/XSharp/Pipeline.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" + +// Measures how much stack a native stage needs for a given nesting depth. +// +// stack_probe +// +// The shapes are `statements`, an `if` nested in an `if`; `expressions`, a +// chain of additions that nests in the first operand; `operands`, additions +// that nest in the second operand; and `chain`, an `else if` chain. +// +// builds a Core module of the given shape and depth on the compiler stack, +// then runs one stage on a thread whose stack has the given size. The +// process exits with 0 when the stage completes and is terminated by the +// operating system when the stack is too small. Searching for the smallest +// size that completes gives the stack the stage uses at that depth, in the +// build that is measured, which is how the nesting limits are justified for +// ordinary and for sanitizer builds. The program is a measuring instrument, +// not a test: a run that overflows is an expected outcome. + +namespace +{ + namespace Core = Visual::XSharp::Core; + namespace Wire = Visual::XSharp::Core::Wire; + + constexpr std::uint64_t kValue = 2U; + constexpr std::uint64_t kTotal = 3U; + + [[nodiscard]] auto + Integer(std::int64_t value) -> Core::Expression + { + return Core::Expression::Constant(value, Core::Type::int64()); + } + + [[nodiscard]] auto + Value() -> Core::Expression + { + return Core::Expression::Variable({ kValue, U"value" }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Total() -> Core::Expression + { + return Core::Expression::Variable({ kTotal, U"total" }, + Core::Type::int64()); + } + + [[nodiscard]] auto + Module(std::vector statements) -> Core::Module + { + std::vector body; + body.push_back( + Core::Statement::Bind(Core::Binding{ { kTotal, U"total" }, + Core::Type::int64(), + true, + Integer(0) })); + for (auto &statement : statements) + body.push_back(std::move(statement)); + body.push_back(Core::Statement::Return(Total())); + return { { U"Probe" }, + { Core::Function{ + { 1U, U"Pick" }, + { { { kValue, U"value" }, Core::Type::int64() } }, + Core::Type::int64(), + std::move(body) } } }; + } + + /// `if (value > 0) { if (value > 1) { ... total = value; } }` + [[nodiscard]] auto + Statements(std::size_t depth) -> Core::Module + { + auto statement = Core::Statement::Assign({ kTotal, U"total" }, Value()); + for (std::size_t index = depth; index-- > 0U;) + { + std::vector body; + body.push_back(std::move(statement)); + statement = Core::Statement::If( + Core::Expression::InvokePrimitive( + Core::Primitive::GreaterThan, + { Value(), Integer(static_cast(index)) }, + Core::Type::boolean()), + std::move(body), + {}); + } + std::vector statements; + statements.push_back(std::move(statement)); + return Module(std::move(statements)); + } + + /// `total = (((value + 1) + 1) + ...)`: the first operand nests. + [[nodiscard]] auto + Expressions(std::size_t depth) -> Core::Module + { + auto expression = Value(); + for (std::size_t index = 0U; index < depth; ++index) + { + std::vector operands; + operands.push_back(std::move(expression)); + operands.push_back(Integer(1)); + expression = Core::Expression::InvokePrimitive(Core::Primitive::Add, + std::move(operands), + Core::Type::int64()); + } + std::vector statements; + statements.push_back(Core::Statement::Assign({ kTotal, U"total" }, + std::move(expression))); + return Module(std::move(statements)); + } + + /// `total = (value + (value + (...)))`: the second operand nests, + /// which is nesting that no stage walks in a loop. + [[nodiscard]] auto + Operands(std::size_t depth) -> Core::Module + { + auto expression = Value(); + for (std::size_t index = 0U; index < depth; ++index) + { + std::vector operands; + operands.push_back(Value()); + operands.push_back(std::move(expression)); + expression = Core::Expression::InvokePrimitive(Core::Primitive::Add, + std::move(operands), + Core::Type::int64()); + } + std::vector statements; + statements.push_back(Core::Statement::Assign({ kTotal, U"total" }, + std::move(expression))); + return Module(std::move(statements)); + } + + /// `if (value == 0) { total = value; } else if (value == 1) { ... }` + [[nodiscard]] auto + Chain(std::size_t links) -> Core::Module + { + std::vector rest; + for (std::size_t index = links; index-- > 0U;) + { + std::vector body; + body.push_back( + Core::Statement::Assign({ kTotal, U"total" }, Value())); + auto link = Core::Statement::If( + Core::Expression::InvokePrimitive( + Core::Primitive::Equal, + { Value(), Integer(static_cast(index)) }, + Core::Type::boolean()), + std::move(body), + std::move(rest)); + rest.clear(); + rest.push_back(std::move(link)); + } + return Module(std::move(rest)); + } + + [[nodiscard]] auto + Usage() -> int + { + llvm::errs() + << "usage: stack_probe " + " " + "\n"; + return 2; + } + + [[nodiscard]] auto + Number(const char *text) -> std::size_t + { + return static_cast(std::strtoull(text, nullptr, 10)); + } +} // namespace + +int +main(int argc, char **argv) +{ + if (argc != 5) + return Usage(); + // The arguments of a measuring tool, read once at the start. + // NOLINTBEGIN(cppcoreguidelines-pro-bounds-pointer-arithmetic) + const std::string_view shape = argv[1]; + const auto depth = Number(argv[2]); + const std::string_view stage = argv[3]; + const auto stack = Number(argv[4]) * 1024U; + // NOLINTEND(cppcoreguidelines-pro-bounds-pointer-arithmetic) + if (depth == 0U || stack == 0U) + return Usage(); + + // Everything that recurses along the nesting, including building and + // destroying the module, happens on the compiler stack; only the stage + // under measurement runs on the probed one. + return Visual::XSharp::Support::RunOnCompilerStack([&]() -> int { + Wire::Limits limits; + limits.maximumStatementDepth = depth + 16U; + limits.maximumExpressionDepth = depth + 16U; + const auto module = shape == "statements" ? Statements(depth) + : shape == "expressions" ? Expressions(depth) + : shape == "chain" ? Chain(depth) + : shape == "operands" ? Operands(depth) + : Core::Module{}; + if (module.functions.empty()) + return Usage(); + const auto encoded = Wire::Encode(module, limits); + if (!encoded) + { + llvm::errs() << "stack_probe: the module could not be encoded\n"; + return 3; + } + int outcome = 0; + std::size_t committed = 0U; + Visual::XSharp::Support::RunOnStack(stack, [&] { + if (stage == "encode") + outcome = Wire::Encode(module, limits) ? 0 : 4; + else if (stage == "decode") + outcome = Wire::Decode(encoded.bytes, limits) ? 0 : 4; + else if (stage == "verify") + outcome = Core::Verify(module).empty() ? 0 : 4; + else if (stage == "prepare") + outcome + = Core::CorePrep::Prepare(module).functions.empty() ? 4 : 0; + else if (stage == "pipeline") + // The whole native route from Core bytes to LLVM, as the + // driver runs it, under the default wire limits. + outcome = Visual::XSharp::Pipeline::ConsumeCore(encoded.bytes) + .succeeded + ? 0 + : 4; + else + outcome = 2; + committed = Visual::XSharp::Support::CommittedStackBytes(); + }); + // The size on the command line is only reserved. What the stage + // touched is committed: an upper bound on what it used, to the page. + if (outcome == 0) + llvm::outs() << "ok committed-kib " << committed / 1024U << '\n'; + return outcome; + }); +} diff --git a/Documents/AARC-ABI.md b/Documents/AARC-ABI.md index ef1ee047..53693ca9 100644 --- a/Documents/AARC-ABI.md +++ b/Documents/AARC-ABI.md @@ -115,7 +115,7 @@ reference. ## Xpp, Xmm, and LLVM -Xpp and Xmm wire version 5 preserve `RetainStrong`, `ReleaseStrong`, `MakeWeak`, +Xpp and Xmm wire version 7 preserve `RetainStrong`, `ReleaseStrong`, `MakeWeak`, `LockWeak`, `ReleaseWeak`, `MakeUnowned`, `LoadUnowned`, and `ReleaseUnowned`. Producing operations preserve the operand's language type. Release operations have no destination and carry `Unit` as the result marker. Both stage verifiers @@ -130,17 +130,65 @@ for the duration of a call because the closure keeps them alive. Weak and unowned captures are upgraded to temporary strong references and released after the lifted call. The generated destructor balances every owning/control slot. +`Memoize` creates an object of the same kind for a callable that remembers +its result. Its payload is an invoke-thunk pointer, a byte that says whether +the result is known, the result, and the computation: + +```text +{ ptr invoke, i8 known, T result, ptr computation } +``` + +The thunk has the signature of a closure without parameters, so the object is +called exactly as a closure is and a caller cannot tell the two apart. It +returns the result when the byte is set; otherwise it calls the computation +through that object's own thunk, stores the result, sets the byte and returns. +The byte is set after the computation returns. The object takes a strong reference of its own to the +computation when it is created, and its destructor releases it. The result +slot holds a `bool` or a number and owns nothing. + String constants keep `i32` Unicode-scalar storage and call `vxs_aarc_string_literal`, which creates a `System.String` AARC object without introducing UTF-8 storage. ## Current boundary -This slice does not yet insert whole-program retain/release placement for every -source binding, package the runtime into every final native link, or collect -cycles. It establishes the checked IR vocabulary, concrete object ABI, runtime -primitives, String and first-class closure invocation lowering, and regression -coverage that those later passes target. The future concurrent Bacon–Rajan plus +Retains and releases are placed for every AARC value of a function by the Xpp +ownership placement pass, described in [Ownership flow](OWNERSHIP-FLOW.md). +This slice does not collect cycles. A native executable is linked with the +runtime library; see "Native executables" below. +`Visual::XSharp::Runtime::Aarc::LiveAllocations` counts the +allocations whose storage has not been reclaimed; it is a C++ entry point for +tests and is not part of the C ABI. The future concurrent Bacon–Rajan plus trial-deletion collector remains opt-in with `-Cycle-Collector true`. The ordinary acyclic path must not pay its cost when disabled, and no trial begins when no candidate exists or the program has already broken the candidate cycle. + +## Native executables + +A native executable is linked without a C runtime. What its code calls of the +runtime, it finds in `vxs-runtime.lib`, which the compiler links from the +directory of its own executable: the ownership runtime of this document and +the text and console runtime, as one object. + +That object is built from the sources a host process links, +`Compiler/Runtime/AARC` and `Compiler/Runtime/Text`, in one translation unit +with the few things those sources take from a C runtime when there is one: +the non-throwing allocation functions, over the heap of the process, the +memory functions a compiler may call for a loop or an initialization, and the +tag object of the non-throwing forms. It is compiled without stack cookies +and without instrumentation of any kind, in a sanitizer build of the compiler +as well, because the programs it is linked into are not sanitizer builds. + +An executable therefore has the whole ownership ABI: strong, weak and unowned +references, strings and type tests, counted with the same atomic operations +as in a host. It imports seven functions of kernel32 and nothing else. The +linker makes the import library for them from a list of names, so that +linking a program needs neither the libraries of a C runtime nor those of a +Windows SDK. + +A program that calls nothing of the runtime takes nothing from the library. +When the library is not installed beside the compiler, such a program still +links, and any other fails with a diagnostic that names the library. + +The native linker is the Windows one; executables on Linux and macOS are +pending, and with them the same library for those systems. diff --git a/Documents/ARCHITECTURE.md b/Documents/ARCHITECTURE.md index 620b60db..912d23d4 100644 --- a/Documents/ARCHITECTURE.md +++ b/Documents/ARCHITECTURE.md @@ -164,7 +164,7 @@ The current public artifact names are: Normal compilation keeps these representations in memory. Haskell writes real `.core` artifacts and C++20 consumes them through the full verified pipeline. Explicit `.ll`, `.bc`, `.o`, and `.asm` emission is available after source or Core input. Binary emission adds the platform entry bridge, writes a temporary object, links one `.vxse`, and removes the -temporary object. Bounded Xpp/Xmm v5 readers and writers support verified forward pipeline resumption. +temporary object. Bounded Xpp/Xmm v7 readers and writers support verified forward pipeline resumption. ## Process and temporary-file model diff --git a/Documents/ARTIFACT-WIRE.md b/Documents/ARTIFACT-WIRE.md index ec07a19a..c64b0dc0 100644 --- a/Documents/ARTIFACT-WIRE.md +++ b/Documents/ARTIFACT-WIRE.md @@ -11,10 +11,10 @@ compiler artifacts rather than source formats. Core, Xpp, and Xmm are public | Contract | Magic | Current version | Producer | Consumer | | --- | --- | ---: | --- | --- | -| Core | `VXCR` | 7 | Haskell frontend | native Core reader | -| CorePrep | `VXCP` | 6 | CorePrep adapter | native pipeline tools | -| Xpp | `VXPP` | 5 | verified CorePrep-to-Xpp lowering | Xmm lowering or artifact tools | -| Xmm | `VXMM` | 5 | verified Xpp-to-Xmm lowering | LLVM backend or artifact tools | +| Core | `VXCR` | 10 | Haskell frontend | native Core reader | +| CorePrep | `VXCP` | 8 | CorePrep adapter | native pipeline tools | +| Xpp | `VXPP` | 7 | verified CorePrep-to-Xpp lowering | Xmm lowering or artifact tools | +| Xmm | `VXMM` | 7 | verified Xpp-to-Xmm lowering | LLVM backend or artifact tools | The contracts have related scalar encodings but separate structural schemas. Their magic values must never be treated as aliases. @@ -91,7 +91,7 @@ the bit pattern is identical. ### Core scalar tags (introduced in v5) -The native and Haskell Core codecs retain these assignments in version 8. This +The native and Haskell Core codecs retain these assignments in version 10. This table is an implementation-maintenance aid, not a user extension API. | Tag | Type | Tag | Type | @@ -115,7 +115,7 @@ declared scalar type. ### CorePrep scalar tags (introduced in v5) CorePrep retains historical `int` and `long` positions before the extended -catalog. Version 6 adds provenance without changing these tags; assignments +catalog. Versions 6 and 7 change none of these tags; assignments must therefore not be copied blindly from Core: | Tag | Type | Tag | Type | @@ -228,20 +228,99 @@ payload follows in the order shown. Core wire v8 added tag 6. Its three children have fixed positions, so the record carries no count: a shorter payload is a truncation, never a smaller -conditional. Each child counts one level against the expression depth limit, -exactly like a primitive operand. The reader does not check that the arms +conditional. Each child counts one level against the expression depth limit. +Among the operands of a primitive only those after the first do: the first +operand is at the level of the primitive, so a chain of operators, which +nests in that operand as deep as it is long, is one level, and the readers +and the writers walk it in a loop. The reader does not check that the arms agree with the result type; that is the Core verifier's rule and runs on every decoded module. +The native reader and writer also bound the nesting of statement bodies, with +the same default of 4096 levels. A function body is level 1, and a branch, a +loop body and a closure body are each one level below the statement or +expression that holds them. A false branch that holds exactly one conditional +statement is an `else if`: the reader and the writer walk such a chain in a +loop, and its links share one level. An empty body costs no level. Input that +nests deeper is rejected with a limit error before it is walked. The limit is +a property of the reader, not of the format: it changes no byte of a valid +document, and the Haskell codec, whose stack grows on demand, does not need +it. + CorePrep, Xpp, and Xmm have no conditional-expression record. The expression is lowered to blocks, a branch, and assignments to one slot before CorePrep is serialized, so their versions did not change. +### The remembering operation + +Core v9, CorePrep v7, Xpp v6 and Xmm v6 add one operation: a callable that +remembers its result, `Memoize`. It is not a new record. It is one more value +of the operation field every stage already writes, with one operand and a +result, and it takes the next free tag of each catalog: + +| Contract | Field | Tag | +| --- | --- | ---: | +| Core | primitive | 24 | +| CorePrep | operation | 27 | +| Xpp | opcode | next after the type test | +| Xmm | opcode | next after the type test | + +A reader of an earlier version does not know the tag, which is why all four +versions changed together: a document that holds the operation must not be +read as one that cannot. The readers check only that the tag is known. That +the operand is a callable without parameters whose result is `bool` or +numeric, and that the result has the operand's type, is the rule of each +stage's verifier, which runs on every decoded module. + +### The runtime call + +Core v10, CorePrep v8, Xpp v7 and Xmm v7 add one more operation: a call of a +function of the runtime. Like the remembering operation it is a value of the +operation field every stage already writes, and it takes the next free tag: + +| Contract | Field | Tag | +| --- | --- | ---: | +| Core | primitive | 25 | +| CorePrep | operation | 28 | +| Xpp | opcode | next after the remembering operation | +| Xmm | opcode | next after the remembering operation | + +The operation has no field of its own. Its first operand is an integer +literal of type `int`, the identity of the function; the operands after it +are the arguments. The identities are those of the runtime catalog: + +| Identity | Function | Arguments | Result | +| ---: | --- | --- | --- | +| 1 | join two strings | `String`, `String` | `String` | +| 2 | a signed integer as text | signed integer | `String` | +| 3 | an unsigned integer as text | unsigned integer | `String` | +| 4 | a Boolean as text | `bool` | `String` | +| 5 | a character as text | `char` | `String` | +| 6 | `%d` and `%x` of a signed integer | flags, width, precision, signed integer | `String` | +| 7 | `%u` and `%x` of an unsigned integer | flags, width, precision, unsigned integer | `String` | +| 8 | `%f` | flags, width, precision, floating-point number | `String` | +| 9 | `%s` | flags, width, precision, `String` | `String` | +| 10 | `%c` | flags, width, precision, `char` | `String` | +| 11 | the line terminator of the platform | none | `String` | +| 12 | write to the console | `String`, target | none | +| 13 | whether two strings are equal | `String`, `String` | `bool` | + +Flags, a width, a precision and a target are of type `int`. A signed or an +unsigned integer argument is of any width up to 64 bits, and a floating-point +argument of any up to 64. An identity is never reused or renumbered: a new +function takes the next number, and the document versions change with it, +because an older reader does not know the function. + +The readers check that the tag is known and decode the operands as they +decode any others. That the first operand names a function, and that the +arguments and the result are the ones that function has, is the rule of each +stage's verifier. + ### Version transition Versions are strict, not feature-negotiated. Core readers accept only version -8, CorePrep readers accept only version 6, and Xpp/Xmm readers accept only -version 5. Every older or future version fails at the version field before +10, CorePrep readers accept only version 8, and Xpp/Xmm readers accept only +version 7. Every older or future version fails at the version field before body decoding. The compiler does not guess whether a document happens to contain only fields from an older schema. Recompile the owning source or regenerate the intermediate artifact with the @@ -382,7 +461,7 @@ when written; decoding never recreates a host-width alternative. ## Xpp document order -An Xpp v5 document contains: +An Xpp v7 document contains: 1. `VXPP`, version, and zero reserved flags; 2. qualified module name; @@ -404,7 +483,7 @@ dedicated symbol field rather than an untyped extra operand. ## Xmm document order -An Xmm v5 document contains: +An Xmm v7 document contains: 1. `VXMM`, version, and zero reserved flags; 2. qualified module name; @@ -449,11 +528,14 @@ verified again before serialization or forward lowering. The version field describes the entire schema. Core v6 and CorePrep v6 added project source catalogs and per-function ownership; Core v7 additionally added structured `while`, `do/while`, classic `for`, `break`, and `continue` records, -and Core v8 adds the conditional expression record. +Core v8 adds the conditional expression record, and Core v9 with CorePrep v7 +adds the remembering operation. Xpp/Xmm began independently at -version 1; their current version 5 retains the explicit ownership operations, +version 1; version 5 retains the explicit ownership operations, template values, and type-test operation, and adds source catalogs and function -owners. The intermediate versions remain strict historical contracts; their +owners, version 6 adds the remembering operation, and their current +version 7, with Core v10 and CorePrep v8, adds the runtime call. +The intermediate versions remain strict historical contracts; their documents are not guessed or accepted by the current readers. Every current reader rejects earlier and future versions for its own magic. diff --git a/Documents/BRANCHING.md b/Documents/BRANCHING.md index 9b12a847..1af8c775 100644 --- a/Documents/BRANCHING.md +++ b/Documents/BRANCHING.md @@ -6,9 +6,13 @@ This page describes what the compiler implements today for `match`, for `if` used as an expression, for `guard` and for a block written as a statement: what is accepted, in what order things are evaluated, and where the -implemented subset ends. The language design is in `Spec/Language/Decls.vxs`, -sections 31 and 32; the diagnostics are listed in -[Diagnostics](DIAGNOSTICS.md); the lowering is described in +implemented subset ends. It describes an implementation and is not the +language contract: the language is defined by `Spec/`, here +`Spec/Language/Decls.vxs`, sections 31 and 32, and where this page and the +specification differ the specification is right and the compiler is wrong. +Where the specification is silent, what the compiler does today is a state of +the implementation and not a decision about the language. The diagnostics are +listed in [Diagnostics](DIAGNOSTICS.md); the lowering is described in [Core IR](CORE-IR.md). ## Match @@ -31,27 +35,31 @@ int kind = match (code), (strict) { its arm accept. A guard that is false passes the subjects on to the arms after it. A guard is `bool` or numeric; a numeric guard holds when it is not zero. -- The body is an expression or a block. The comma after a block body is - optional; after an expression body it is required unless the arm is the - last one. +- The body is an expression or a block. The comma after an arm is optional, + as in the grammar. An expression body is parsed like every expression, so + a parenthesized pattern that follows it without a comma is read as the + argument list of a call of the body; the arrow that then follows is + reported as `VXP0038`, which says what was read. ### Patterns | Pattern | Accepts | Notes | | --- | --- | --- | -| a literal, `1`, `true`, `'a'` | the value equal to it | typed from its subject; has no sign | +| a literal, `1`, `-1`, `true`, `'a'` | the value equal to it | typed from its subject and checked against its range; a `-` may precede a numeric literal only | | `_` | every value | | -| `Type name` | every value of the subject's type | binds the subject as an immutable local | +| `Type name` | every value of the subject's type | binds the value of the subject as a local of its arm | | `Type _` | every value of the subject's type | binds nothing | -A binding is in scope in the guard and the body of its own arm only. Two arms -may bind the same name; one arm may not bind a name twice, and a binding may +A binding is in scope in the guard and the body of its own arm only. It is an +ordinary local: it may be assigned, and assigning it does not change the +subject. Two arms may bind the same name; one arm may not bind a name twice, and a binding may not reuse a name that is already in scope. A bare name is not a pattern: a binding always states its type. -`null`, enum case patterns such as `.Ready`, and type patterns that name -another type than their subject's are parsed and rejected, because reference -subjects, enum declarations and class hierarchies are not implemented. +`null` and type patterns that name another type than their subject's are +parsed and rejected, because reference subjects and class hierarchies are +not implemented. Enum case patterns such as `.Ready` are implemented; see +"Limits of the implemented subset" below. ### Statement and expression @@ -61,7 +69,8 @@ may `return`, and inside a loop they may `break` and `continue`. An expression body is evaluated for its effect and must have one. When no arm accepts, nothing happens. -Anywhere else a `match` is an expression: +Anywhere else a `match` is an expression, and so is a `match` that is the +last item of a block used as a value: - every arm yields a value, and all arms have one type; - an arm made only of untyped numeric literals takes its type from the place @@ -86,15 +95,73 @@ An `if` in operand position is an expression. Both blocks are required, the `else` branch is a block and not another `if`, and each block ends with an expression that has no semicolon; that expression is the value of the block. Only the selected block runs. The two blocks have one type. An `if` at the -start of a statement is the `if` statement, as before. +start of a statement is the `if` statement, as before, with one exception: as +the last item of a block used as a value, an `if` whose two blocks both end +with a value is the value of that block. + +```vxs +int sign = if (value < 0) { 0 - 1 } else { if (value > 0) { 1 } else { 0 } }; +``` ## Blocks used as values The blocks of an `if` expression and the block bodies of the arms of a match expression are blocks used as values. Statements before the final expression -run in order, and names declared in the block end with it. `return`, `break` -and `continue` cannot leave such a block; a loop inside the block may still be -left with `break`. +run in order, and names declared in the block end with it. + +A block may leave instead of yielding a value: + +```vxs +int size = if (count > 0) { count * 2 } else { return 0; }; + +while (index < limit) { + index += 1; + total += match (index) { 3 -> { continue; }, 7 -> { break; }, int n -> n }; +} +``` + +`return` leaves the enclosing method and carries its return type, also from +inside a loop used as an expression. Where the return type of a callable is +inferred, the returns in its value blocks count like the others; the returns +of a nested callable are its own. `break` and `continue` target the nearest +loop around the expression and need one. A `break` may carry a value to a +loop used as an expression, exactly as a `break` statement in its body does. +A block that cannot complete normally has no final expression and no value; +the expression has the type of the blocks that complete. Such a block is +lowered as its statements alone: nothing is stored for it. + +A value block may also stand in the condition or the update clause of a +loop. Both belong to their loop, as examples 79 to 83 of +`Spec/Language/Iteration.vxs` state. A `break` in either leaves that loop, +with the effects of the condition or the update up to it. A `continue` in +the condition abandons the rest of the condition and evaluates it again, +without running the body or, in a `for`, the update; a condition that always +continues is an endless loop. A `continue` in the update clause ends the +update, and the condition is tested next. A callable is not inside the loops +around the place that creates it. + +```vxs +while (if (index >= limit) { break; } else { true }) { index += 1; } + +for (int i = 0; i < 6; i += if (skip) { skip = false; continue; } else { 1 }) { } +``` + +An expression none of whose blocks completes is valid and never yields a +value: + +```vxs +int result = if (known) { return code; } else { return 0; }; +``` + +Whether an expression completes is a fact about control flow that the +compiler keeps apart from types (`Visual.XSharp.Completion`); there is no +type for it in the language. What would have received the value, here the +binding, is not held to a type and is not lowered: the statement becomes the +conditional over the two `return` statements, with no result slot and no +placeholder, and the statements after it, which are never reached, are +checked but not lowered. The same holds for such an expression as an +operand, an argument, a condition or a returned value: the operands that are +evaluated before it keep their effects, and nothing after it is evaluated. ## Guard @@ -104,10 +171,13 @@ guard (count > 0) else { } ``` -The block runs when the condition is false. It must not complete normally: -its last statement is `return`, `break` or `continue`, an `if` whose two -branches both end that way, or a nested block that does. The statements after -the guard therefore run only when the condition held. +The block runs when the condition is false. It must not complete normally on +any path, so the statements after the guard run only when the condition held. +That is decided from the control flow of the block, by the rule given under +`VXT0061` in [Diagnostics](DIAGNOSTICS.md): `return`, `break` and `continue` +leave, and so do an `if` both of whose blocks leave, a loop that cannot end, +and a statement `match` that always selects an arm and all of whose arms +leave. A call is assumed to return. ## Nested blocks @@ -122,15 +192,27 @@ name that is in scope around it. `bool` or numeric. Other types need storage rules for the result slot that the backend does not have yet. A value of a template type parameter is rejected for the same reason. -- A binding in the condition of an `if`, a `guard` or a `while`, such as - `if (auto user = Find())`, is recognized and rejected: it requires optional - values. +- A `match` over a classic enum uses case patterns, `.Member`. The pattern + accepts the value of the member, so two members with one value are one + case and the second arm for it can never be selected (`VXT0053`). The + match accepts every value, and needs no `_` arm, when its arms without + guards name every value of the enum; with several subjects, every + combination of the values of enums and `bool` subjects, up to 256 + combinations. +- Parts of these forms that the specification has and the compiler does not + implement yet are listed, with their diagnostics, under "Pending branching + and loop forms" in [Implementation status](IMPLEMENTATION.md). +- A type pattern over a scalar subject names the type of the subject itself; + no numeric conversion is applied, so `long n` does not match an `int`. - `match` and `guard` are reserved words. -- A match may have any number of arms: the lowering groups them, so its - nesting does not grow with the number of arms. Statements nested by the - programmer, such as an `if` inside the first branch of an `if`, are still - limited by the stack of the native stages after Core; see the known - limitations in the changelog. +- A match may have any number of arms: it is lowered to one chain of + conditionals, the shape of an `else if` chain, which every stage walks in + a loop. The body of an arm is one statement level below its match however + many arms the match has. See the nesting limits in + [Diagnostics](DIAGNOSTICS.md). +- In the body of a method, a closure or a property, which may also end with + an expression, an `if` or a `match` in last position is still the + statement form. ## Where it is implemented and tested @@ -140,7 +222,7 @@ name that is in scope around it. | scoping of pattern bindings and nested blocks | `Visual.XSharp.Resolver.Renamer` | | typing rules | `Visual.XSharp.TypeChecker.Branching` | | lowering to Core | `Visual.XSharp.Desugarer.Branching` | -| grammar, typing, lowered shapes, evaluation | `BranchingTests.hs` | +| grammar, typing, lowered shapes, evaluation | `BranchingTests.hs`, with the programs of `BranchingEvaluationCases.hs` | | match against the `if` chain it stands for | `BranchingOracleTests.hs` | | diagnostic positions, damaged input, templates | `BranchingDiagnosticTests.hs` | | native execution in both pipeline modes | `Compiler/Fuzzing/BranchingExecutionCases.cpp` | diff --git a/Documents/BUILDING.md b/Documents/BUILDING.md index 62c14656..3c185fba 100644 --- a/Documents/BUILDING.md +++ b/Documents/BUILDING.md @@ -112,7 +112,7 @@ suite set on macOS. Before creating release artifacts, validate the exact cross-build-system version: ```powershell -go run ./helpers/cmd/develop version 0.4.1 +go run ./helpers/cmd/develop version 0.5.0 ``` The check compares `MODULE.bazel`, the changelog heading, the Haskell package, and the Kotlin project runtime using exact diff --git a/Documents/CALLABLE-ABI.md b/Documents/CALLABLE-ABI.md index 35a971c1..9e7c09f3 100644 --- a/Documents/CALLABLE-ABI.md +++ b/Documents/CALLABLE-ABI.md @@ -156,9 +156,12 @@ with opaque pointers. ## Current limits This slice provides native indirect invocation for closure values already -present in CorePrep. It does not complete whole-program escape analysis, -retain/release placement for every local, cross-module callable ABI stability, -exception cleanup around upgraded captures, or the future cycle collector. +present in CorePrep, and for methods used as values, which become closures +without captures in Xpp. A callable parameter is borrowed and a callable +result is owned; the placement pass of [Ownership flow](OWNERSHIP-FLOW.md) +writes the retains and releases. It does not complete whole-program escape +analysis, cross-module callable ABI stability, exception cleanup around +upgraded captures, or the future cycle collector. The planned cycle collector combines concurrent Bacon–Rajan processing with trial deletion and remains disabled by default. Callable environments preserve diff --git a/Documents/CLI.md b/Documents/CLI.md index 08d67f6f..8f8e76be 100644 --- a/Documents/CLI.md +++ b/Documents/CLI.md @@ -255,7 +255,7 @@ vxs build -Build core -Emit llvmll -File module.core vxs build -Build core -Emit llvmbc -File module.core ``` -The native C++20 route reads the Haskell `VXCR` v6 contract with byte, collection, text, type-depth, and expression-depth +The native C++20 route reads the Haskell `VXCR` v10 contract with byte, collection, text, type-depth, and expression-depth limits. It verifies Core semantics before adapting nested expressions and source control flow to CorePrep, then runs the existing verified CorePrep → Xpp → Xmm → LLVM pipeline entirely in memory. `check` writes nothing. The two `build` examples write a sibling `.ll` or `.bc` file. A Core build can also write a sibling `.o` or `.asm`, or link a `.vxse`; binary is the diff --git a/Documents/CLOSURE-PIPELINE.md b/Documents/CLOSURE-PIPELINE.md index e86c4f57..6c14d56b 100644 --- a/Documents/CLOSURE-PIPELINE.md +++ b/Documents/CLOSURE-PIPELINE.md @@ -84,11 +84,11 @@ symbol validity, parameter uniqueness, nested expressions, and return behavior. Optimization recursively folds capture initializers and closure bodies without reordering captures. -Core wire version 8 serializes ownership, captures, parameters, return type, +Core wire version 10 serializes ownership, captures, parameters, return type, nested statements, and structured loop statements. Existing byte, count, type-depth, and expression-depth limits also apply to closures. -The native VXCR v8 reader and writer carry the same closure expression tag and +The native VXCR v10 reader and writer carry the same closure expression tag and field order as the Haskell frontend. The C++ Core verifier validates capture ownership, callable shape, nested body returns, and capture mutation before CorePrep lifting. This keeps callable-containing `.vxs` input on the ordinary @@ -106,14 +106,14 @@ CorePrep converts each closure by: 6. processing that queue until nested closures are also lifted. CorePrep verification checks callable result type, lifted target, capture atom -types, symbol validity, and non-owning restrictions. CorePrep wire v6 preserves +types, symbol validity, and non-owning restrictions. CorePrep wire v8 preserves this lifted closure metadata. Xpp wire v3 gives closure creation a dedicated operation tag; a function symbol is never encoded as a fake data operand. Xpp and -Xmm wire v5 also retain source catalog and function ownership metadata. +Xmm wire v7 also retain source catalog and function ownership metadata. ## Native C++ stages -The C++20 Xpp/Xmm decoders consume their version 5 contracts. Xpp retains the +The C++20 Xpp/Xmm decoders consume their version 7 contracts. Xpp retains the lifted symbol, ordered operands, ownership vector, and callable result. Its verifier checks the target's hidden parameter prefix against captures. @@ -129,6 +129,11 @@ allocation, capture initialization, and a destructor that balances strong, weak, and unowned slots. Indirect invocation through the resulting closure pointer is the remaining callable boundary; construction and destruction are connected. +A callable that remembers its result, `Memoize`, is a second producer of the +same kind of object: an invoke thunk first, and a destructor named by its +metadata. It is called and released through the same instructions as a +closure. `AARC-ABI.md` has its payload. + ## Verification coverage Tests cover delimiter and parameter forms, expression/block bodies, empty and diff --git a/Documents/COMPILER-PIPELINE.md b/Documents/COMPILER-PIPELINE.md index 9e776a23..eee922da 100644 --- a/Documents/COMPILER-PIPELINE.md +++ b/Documents/COMPILER-PIPELINE.md @@ -311,8 +311,8 @@ Artifact ownership is explicit: `check` writes no artifact. Binary emission creates the required entry bridge, writes a temporary object, invokes LLD with a typed argument vector rather than a shell string, validates the resulting executable, and removes its temporary object. -Project binary builds produce one executable. The entry namespace's Core v8 -preserves source ownership through CorePrep v6, Xpp v5, and Xmm v5. Project +Project binary builds produce one executable. The entry namespace's Core v10 +preserves source ownership through CorePrep v8, Xpp v7, and Xmm v7. Project object and assembly emission lowers each source in that selected namespace's source catalog into a separate `.o` or `.asm` in the selected output directory. A source with no declarations still receives an output; functions defined by diff --git a/Documents/CONSOLE-IO.md b/Documents/CONSOLE-IO.md new file mode 100644 index 00000000..39d282c3 --- /dev/null +++ b/Documents/CONSOLE-IO.md @@ -0,0 +1,202 @@ + + + +# Console output and strings + +A Visual X# program writes to the console with `System.Console`, and +`System` is imported implicitly, so `Console.Println("Hello")` is a complete +statement. The console is specified in +`Spec/StandardLibrary/IO/ConsoleIO.vxs`, and the two operations on strings +that this document covers, `+` and `==`, in `Spec/Language/Operators.vxs`. + +This document says how much of that the compiler implements and how. It is an +implementation reference, not the definition of the language. + +## What is implemented + +| Form | Meaning | +| --- | --- | +| `Console.Print(value)` | writes the value to standard output | +| `Console.Println(value)` | the same, and then ends the line | +| `Console.Printf(format, ...)` | writes the format with its conversions applied | +| `Console.Printfn(format, ...)` | the same, and then ends the line | +| `Console.Error`, `Errorln`, `Errorf`, `Errorfn` | the same four, to standard error | +| `Console.Format(format, ...)` | applies a format and returns the `String`; writes nothing | +| `left + right` with a `String` on either side | the two as text, joined | +| `text += value` | appends to a `String` variable | +| `left == right`, `left \= right` on two `String`s | whether they hold the same characters | + +`System.Console` may be written in place of `Console`. A class of the program +named `Console` is found first and is the program's own. + +A value that is not a `String` is written as text by `Print` and by `+`: an +integer of at most 64 bits in decimal, a `bool` as `true` or `false`, a `char` +as the character. + +## The output format grammar + +A format is a string literal. It is read when the program is compiled, and +everything about it is checked then: a format that is wrong is an error of +the program, never of a run. + +A conversion is `%`, flags, a width, a precision after a point, and a letter. + +| Conversion | Argument | Flags | Precision | +| --- | --- | --- | --- | +| `%d` | signed integer | `-` `0` `+` space `'` | no | +| `%u` | unsigned integer | `-` `0` `'` | no | +| `%x` | signed or unsigned integer | `-` `0` `#` | no | +| `%f` | `sfloat`, `lfloat` or `float` | `-` `0` `+` space `'` | digits after the point, six by default | +| `%s` | `String` | `-` | the most characters written | +| `%c` | `char` | `-` | no | +| `%b` | `bool` | `-` | no | +| `%n` | none | none | no | +| `%%` | none | none | no | + +- `-` and `0` exclude each other, and so do `+` and the space. +- A width or a precision written as `*` is an `int` argument that stands + before the value. +- `%x` writes a negative number as a minus sign and its magnitude, and `#` + puts `0x` between the sign and the digits. +- `'` groups the digits before the point in threes with apostrophes. +- `%f` writes the digits of the exact binary value, correctly rounded, a tie + to the even digit. `nan`, `inf` and `-inf` are written as those words. +- `%n`, like `Println`, writes the line terminator of the platform: a carriage + return and a line feed on Windows, and a line feed elsewhere. +- An argument is never converted to fit a conversion: `%d` of a `uint` and + `%f` of an `int` are errors. + +## Diagnostics + +| Code | Reported when | +| --- | --- | +| `VXT0071` | `Console` has no method of that name | +| `VXT0072` | `Print` or one of its siblings is given no value or several, or a format method is given no format | +| `VXT0073` | a value of a type that has no text form is written or joined | +| `VXT0074` | the format is not a string literal | +| `VXT0075` | the format has a `%` that begins no conversion, or ends inside one | +| `VXT0076` | a conversion has a flag or a precision it does not take, or flags that exclude each other | +| `VXT0077` | the number of arguments is not the number the format takes | +| `VXT0078` | an argument does not have the type its conversion takes | +| `VXT0079` | the form is specified and not implemented yet; see below | + +## How a console call is compiled + +`Console` is not declared in any source file. The renamer declares `System` +and `Console` for every program, outside the program's own names, and the +type checker recognizes a call on one of them. + +The type checker rewrites the call. It reads the format, checks each argument +against its conversion, and leaves in the typed tree the conversions as calls +of the runtime, joined, and one write. `Console.Printfn("%s: %d", name, n)` +becomes, in effect: + +```text +write(concat(concat(name, ": "), format_signed(flags, width, precision, n)), output line) +``` + +No stage after the type checker knows what a format is. + +Each of those calls is a *runtime call*: one operation of Core, CorePrep, Xpp +and Xmm whose first operand is a literal that names a function of the runtime +catalog and whose remaining operands are the arguments. The catalog is one +table, written in `Visual/XSharp/Core/RuntimeCall.hpp` for the native stages +and in `Visual.XSharp.RuntimeCall` for the frontend, and every stage verifies +a call against it. `CORE-IR.md` has the operation and `ARTIFACT-WIRE.md` its +encoding. + +A conversion takes its flags, its width and its precision before the value. +That is the order of a format's arguments, and the operands of a call are +evaluated in order, so a width written as `*` is evaluated before the value +it applies to. + +LLVM lowers a runtime call to a call of the function's symbol. An integer +argument is widened to 64 bits, sign-extended or zero-extended by its type, +and a floating-point argument to a `double`; neither changes a value. + +## Output is an effect + +Evaluation is by need, and an effect is not: it happens where it is written. +Writing to the console is the first effect of the language that the +expression that has it does not show: `int x = Log(1);` writes only because of +what `Log` does. + +The frontend therefore finds the methods a call of which may write, from the +bodies of all methods of the program together, the specializations of +templates included. An expression that contains such a call is evaluated +where it stands: its binding is not deferred, and as an argument it is +computed at the call and not suspended. A value that only computes is by need +as before, in the same program. + +The answer errs on the side of an effect. A method that creates a callable +that writes is taken for writing, and a call through a callable value is taken +to write when the body of any callable of the program writes or calls a +method that does, or when a method that writes is used as a value. A program +in which no callable writes keeps its calls through callables by need, and a +program without console output is unaffected altogether. + +The Core optimizer treats a console write like a call of a function it knows +nothing about: it is kept, kept once, and kept where it stands. The other +runtime calls compute a string and nothing else; removing one that nothing +uses changes nothing a program can observe. + +## The runtime + +`Compiler/Runtime/Text` implements the functions: joining, the conversions +and the write. It is written without a standard container and without a +C runtime function, because it is compiled twice. + +A process that hosts generated code, such as the interactive shell and the +test programs, links it as a library and exports its entry points to the JIT. +Such a host may install a sink that receives what a program writes in place +of the standard streams; that is how the tests read output. + +A native executable is linked with `vxs-runtime.lib`, which stands beside the +compiler. It is one object, built from the same sources as the libraries of +a host together with the ownership runtime, and it needs no C runtime: memory +comes from the process heap and output goes to the standard handles, through +seven functions of kernel32. The linker makes the import library for those +seven from a list of names, so linking a program needs neither a C runtime +nor a Windows SDK. `AARC-ABI.md` describes the ownership side. + +The runtime keeps no buffer. Every write reaches the stream before the call +returns, so output appears in the order it was written and nothing is lost +when a program stops. To a Windows console a string is written as UTF-16; to +a file or a pipe, as UTF-8. + +## What is pending + +| Pending | Today | +| --- | --- | +| `%A` and `%O`, and the `IDebug` and `IDisplay` contracts | `VXT0079`; they need the object model | +| a text form of a floating-point number for `Print` and `+` | `VXT0079`; `%f` writes one with a precision | +| `longint`, `ulongint` and `double` as text | `VXT0079`; the runtime functions take 64 bits | +| `Console.Stdout()`, `Stderr()`, `Stdin()` and everything on them: `Flush`, `Read`, `Readln`, `Readf`, `Prompt` | `VXT0079`; they need stream objects | +| a format given by a compile-time directive, `#define FORMAT "%d"` | the format must be a literal | +| formatted string interpolation, `"%d$count"` | not parsed as interpolation | +| `-Type-Safe-Format false` | the option is accepted and has no effect: formats are always checked strictly | +| other operations on strings: length, indexing, slicing, ordering | not implemented | +| native executables on Linux and macOS | the native linker is the Windows one | + +## Verification + +`ConsoleTests.hs` runs some 180 programs in the reference Core evaluator, on +the Core the frontend lowered and on the Core the optimizer left, and compares +what each wrote with text written by hand; it holds about a hundred programs +that must be rejected, each with its code. The text functions of that +evaluator are written a second time, in `RuntimeText.hs`, with unbounded +integers and exact rationals, and share nothing with the runtime library. + +`TextRuntimeTests.cpp` calls the runtime library the way generated code does. +The digits of `%f` are compared with those the C library of the host writes, +which is a third implementation, over values chosen for their rounding and +over a sweep of binary exponents. + +`RuntimeCallPipelineTests.cpp` pins the catalog row by row and breaks a +runtime call in each way it can be broken at each native stage. + +`source_console_smoke` compiles programs from source through LLVM and the JIT +in both pipeline modes, reads what they write through a sink, and requires +each to leave no object of the runtime behind. `executable_run_tests` builds +programs into native executables, runs them as processes and reads their +standard output and standard error. diff --git a/Documents/CORE-IR.md b/Documents/CORE-IR.md index a67c1716..4e7d0e08 100644 --- a/Documents/CORE-IR.md +++ b/Documents/CORE-IR.md @@ -244,9 +244,28 @@ update list, including inside a `CoreIf` there, is rejected with `VXC1066`: it has no later point of the same iteration to reach, and lowering it would jump back to the start of the update without testing the condition. A loop nested inside an update list has its own body and continuation point, so -`CoreContinue` is valid again inside it. Source code cannot produce this form -because a `for` update is a list of expressions; the rule protects Core built -or transformed by other means. +`CoreContinue` is valid again inside it. A source `continue` in an update +clause ends the update, and the frontend lowers it without `CoreContinue`: +an update clause that holds one becomes a loop that runs once, its +statements followed by `CoreBreak`, and the `continue` becomes a `CoreBreak` +of that loop. A source `break` in an update clause is the Core `CoreBreak` +there; in an update clause that is wrapped it sets a flag bound before the +loop, and the loop is left after the wrapper when the flag is set. A source +`continue` in the condition of a `for` must not run the update, which the +Core `CoreContinue` of the loop would: such a condition is evaluated in a +loop of its own, which the `continue` repeats and which is left once the +condition has a result. In a `while` or `do`/`while` loop the condition +already stands at the top of the Core loop body, where `CoreContinue` does +what the source asks. The rule protects Core built or transformed by other +means. + +A function that returns a value must not fall off the end of its body +(`VXC1005`). A body does not when every path returns or runs into a loop +that cannot be left: one whose condition is the literal `true` and that no +`CoreBreak` of its own body or update list leaves. A `CoreBreak` in a loop +nested inside it leaves only that loop. The Haskell and the native verifier +apply the same rule, and the native one decides it without recursing along +an `else if` chain. Optimizer passes preserve the explicit loop form unless their rewrite proves the replacement semantics, including effects and transfer edges. In @@ -349,21 +368,25 @@ The statements run where the source evaluates the expression: conjunction of `subject == literal` comparisons, the first branch is the body, and the second branch holds the arms after it. A pattern that accepts every value contributes no comparison, so a catch-all arm is its body - without a test and ends the chain. A name bound by a type pattern is an - immutable local initialized from its subject before the test of its arm. -- A guard on an arm with comparisons is decided in a `$accepted` slot by the - rule of `&&` with a storing right operand: the guard and its statements run - only when the comparisons hold. A guard on an arm without comparisons is - the test itself. + without a test and ends the chain. A name bound by a type pattern is a + mutable local initialized from its subject. All such locals are bound + after the subjects and before the first test: binding an evaluated scalar + has no effect of its own, every local has its own symbol, and the chain + then holds nothing but tests and bodies. +- A guard is the last operand of the short-circuit conjunction that tests + its arm, so it is evaluated only when the comparisons hold. A guard that + stores into a local needs statements of its own; it is decided in a + `$accepted` slot by the rule of `&&` with a storing right operand, and its + statements stand before the conditional of its arm, inside the false + branch of the arm before it. - A match used as an expression binds a mutable `$matched` slot and every body assigns it. The statement form has no slot; its block bodies are statement blocks of the enclosing body, so a `break value;` in them stores into the slot of the enclosing loop expression. -- A match with more than 16 arms is lowered in groups of 16. The groups - follow each other in one statement sequence; a mutable `$taken` slot is - set by every body before it runs, and each group after the first is the - else branch of a test of that slot. The nesting of the lowered statements - is therefore bounded by one group, whatever the number of arms. +- The arms form one chain: each arm is the false branch of the arm before + it, which is the shape of an `else if` chain. Every stage walks that shape + in a loop, so the lowering of a match is as deep as one arm, whatever the + number of arms. - `guard (condition) else { ... }` is `if (condition) { } else { ... }`. - A block statement has no Core form: its statements join the enclosing sequence. Every local has its own symbol, so the names of the block cannot @@ -377,8 +400,42 @@ links in a loop instead of recursing, because recursion would use stack in proportion to the length of the chain. The encoding, the verifier's checks and the block numbering of CorePrep are those of the nested formulation; the Haskell CorePrep lowering and the native adapter are compared on such chains -like on any other program. Other deep nesting is still walked recursively, -and the wire format has no statement depth limit yet. +like on any other program. + +Three shapes of expression nest as deep as an expression is long: a chain +of operators nests in the first operand of each primitive, a chain of +conditional expressions in each false arm, and a sequence of bindings in +each let body. The native wire reader walks all three in a loop; the Core +verifier and the adapter walk operator chains in a loop, and the adapter +also let bodies and conditional chains. The symbols, the blocks and the +checks are those of the nested formulation. + +Other nesting is walked recursively, one level of recursion per level of +nesting, in the wire codec, the verifier and the adapter. The functions on +those paths are written to keep their frames small: a statement holds two +expressions by value and is large, so it is read into its place and built +by functions that return before the next level is entered, and diagnostics +and instructions are built outside the functions that recurse. Releasing a +module does not recurse along a chain either: the destructors of an +expression and of a statement move their operands and nested statements to a +list and release them from there. Copying a module still recurses once per +level; the pipeline copies only the bodies of closures. + +Two bounds keep the recursion within the stack. The frontend rejects a +function body that nests statements more than 256 levels or expressions more +than 1024 levels deep, with a source position, before Core exists. The +native wire reader and writer bound statement bodies and expressions at 4096 +levels each, so Core from a file is bounded as well. `vxs`, `vxsi` and the +fuzz programs run the pipeline on a thread with 256 MiB of reserved stack, +`Visual/XSharp/Support/CompilerStack.hpp`, instead of the stack the +operating system gives the process, which is one megabyte on Windows. A +program that hosts the pipeline on another thread must give it enough stack +or accept a lower depth. The stack each stage uses per level is measured +with `//Compiler/Fuzzing:source_stack_probe`, which compiles a source file +and reports the stack the compilation committed, and +with `//Compiler/Support/Tests:stack_probe`, which runs one stage on a stack +of a chosen size; the measurements are recorded in +`Benchmarks/2026-10-04-Nesting-And-Chains.md`. `Visual.XSharp.Desugarer.Sequencing` and `Visual.XSharp.Desugarer.Branching` hold these rules. Result slots are @@ -388,7 +445,7 @@ initialized with a neutral literal of their type that no path can observe. `CoreInterpreter.hs`, on the unoptimized and on the optimized Core against hand-written results. `BranchingOracleTests.hs` additionally compares generated matches with the `if` chains they stand for. The same programs run -through CorePrep, Xpp, Xmm, LLVM and the ORC JIT in `source_fuzz_smoke`. +through CorePrep, Xpp, Xmm, LLVM and the ORC JIT in `source_execution_smoke`. ## Expressions @@ -436,6 +493,61 @@ Arithmetic operands must be numeric and use the same type. Comparisons return `bool`. Logical operands accept bool or numeric context and return `bool`. Unary primitives take one operand; other primitives take two. +An integer quotient or remainder by zero, and a shift by an amount that is +negative or not less than the width of its left operand, have no value. Core +says nothing about them; the optimizer does not fold them, and generated code +stops when it reaches one. `EVALUATION.md` has the details. + +#### Memoize + +`CoreMemoize` is a unary primitive whose operand is a callable. It is how a +value by need is handed from one function to another; `EVALUATION.md` says +when the frontend produces it. + +- The operand is a callable without parameters whose result is `bool` or + numeric. +- The result is a callable of the same type. The first call of it calls the + operand and remembers what it returned; every later call returns the + remembered value and does not call the operand again. +- The result is a value like any other callable: it may be bound, passed, + captured and returned, and every copy shares the one remembered value. + +#### Runtime call + +`CoreRuntimeCall` is a call of a function of the runtime: joining two +strings, a conversion of the output format grammar, a console write. It is +the one primitive whose arity depends on its first operand. + +- The first operand is an integer literal of type `int`, the identity of the + function in the runtime catalog. It is never a value that is computed: what + a program calls is fixed when the program is compiled. +- The operands after it are the arguments, in the order the function takes + them, and they are evaluated in that order. +- The type of the expression is the one the function returns: `String`, + `bool` or no value. + +The catalog is `Visual.XSharp.RuntimeCall` in the frontend and +`Visual/XSharp/Core/RuntimeCall.hpp` in the native stages; `ARTIFACT-WIRE.md` +lists its rows. A call that names no function, has the wrong number of +arguments, an argument of a type its function does not take, or a type other +than the function's result is rejected: `VXC1075` by the Core verifiers, +`VXC0026` by the Haskell and `VXC1076` by the native CorePrep verifier, +`VXP1048` by Xpp and `VXL1054` by Xmm. + +A console write is an effect. The optimizer treats it as a call of a function +it knows nothing about: it is not removed when nothing uses its result, not +repeated and not moved. The other runtime calls compute a string and nothing +else. The frontend produces the +primitive for `System.Console` and for `+`, `==` and `\=` on strings; +`CONSOLE-IO.md` says how. + +The Core verifiers reject any other operand of a remembering callable with `VXC1073`. The native +CorePrep verifier reports `VXC1074` and the Haskell one `VXC0025`, the Xpp +verifier `VXP1047` and the Xmm verifier `VXL1053`; each of the last four also +requires the result to have the operand's type. The optimizer does not fold +the primitive, and `MemoizeTests.hs` pins that it neither repeats a remembered +computation nor turns one remembering callable into two. + ### Let `CoreLet` binds one immutable symbol to a value and evaluates a body with that diff --git a/Documents/DIAGNOSTICS.md b/Documents/DIAGNOSTICS.md index c7a5db46..2266c0ad 100644 --- a/Documents/DIAGNOSTICS.md +++ b/Documents/DIAGNOSTICS.md @@ -218,7 +218,7 @@ The type checker reports: | `VXT0036` | a conditional test is neither `bool` nor numeric | | `VXT0037` | the two results of a conditional have different types | | `VXT0038` | the two operands of truthy coalescing have different types | -| `VXT0039` | the result of a conditional form is neither `bool` nor numeric; other result types are not lowered yet | +| `VXT0039` | the result of a conditional form is not `bool`, numeric, an enum or, for `? :`, a `String`; other result types are not lowered yet | A compound assignment otherwise reuses the assignment and operator diagnostics: `VXT0003` for an immutable target and `VXT0012` for operands the @@ -258,8 +258,11 @@ reports: | `VXT0041` | the condition of a loop used as an expression is not the constant `true` (or, for `for`, absent), so the loop could end without a value | | `VXT0042` | a loop used as an expression has no `break` that carries a value | | `VXT0043` | the `break` values of one loop have different types | -| `VXT0044` | the loop value is neither `bool` nor numeric; other result types are not lowered yet | -| `VXT0045` | `return` inside a loop used as an expression, which is not supported yet | +| `VXT0044` | the loop value is neither `bool`, numeric nor an enum; other result types are not lowered yet | + +A `return` inside a loop used as an expression leaves the method. A loop that +no `break` leaves and that returns never yields a value, which is valid; a +loop that neither breaks nor returns is `VXT0042`. A `break` always leaves the innermost loop, so a value-carrying `break` inside a loop statement nested in a loop expression is still `VXT0026`. The value of @@ -293,15 +296,19 @@ The parser reports: | `VXP0033` | an `if` used as an expression has no `else` branch | | `VXP0034` | the `else` branch of an `if` used as an expression is another `if` instead of a block | | `VXP0035` | the condition of an `if`, `guard` or `while` is a binding such as `auto user = Find()`, which requires optional values | -| `VXP0036` | a match arm does not start with a pattern: a literal, `_`, `null`, `.Case`, or a type followed by a name or `_` | +| `VXP0036` | a match arm does not start with a pattern: a literal, a `-` and a numeric literal, `_`, `null`, `.Case`, or a type followed by a name or `_`; or a `-` in a pattern is not followed by a numeric literal | | `VXP0037` | `guard (condition)` is not followed by `else` | -| `VXP0038` | a match arm with an expression body is followed by neither `,` nor `}` | +| `VXP0038` | the pattern of a match arm was read as part of the expression body of the arm before it | A bare name is not a pattern, so `value -> ...` is `VXP0036`: a binding always -states its type, as in `int value -> ...`. A literal pattern has no sign. -The comma after a block body is optional; after an expression body it is -required unless the arm is the last one, because the parenthesized pattern of -the next arm would otherwise continue the expression as a call. An unterminated +states its type, as in `int value -> ...`. A numeric literal may be preceded +by `-`, which makes the constant negative; no other expression is a pattern. +The constant is checked against the type of its subject like every literal +(`VXT0016`), so `-128` is a pattern for a `byte` and `-129`, `128` and any +negative constant for an unsigned subject are not. +The comma after an arm is optional. Without it, a parenthesized pattern after +an expression body continues that expression as a call, and the `->` that +follows cannot; `VXP0038` is reported at that arrow. An unterminated match reports `VXP0002`. The renamer reports `VXR0008` when a pattern binds a name that is already in @@ -313,8 +320,7 @@ The type checker reports: | Code | Meaning | | --- | --- | -| `VXT0046` | a block used as a value does not end with an expression that has no semicolon | -| `VXT0047` | `return` inside a block used as a value, which is not supported yet | +| `VXT0046` | a block used as a value can complete normally and does not end with an expression that has no semicolon | | `VXT0048` | a match arm does not have exactly one pattern for each subject | | `VXT0049` | a match guard is neither `bool` nor numeric | | `VXT0050` | the arms of a match used as an expression have different types | @@ -323,12 +329,56 @@ The type checker reports: | `VXT0053` | a match arm can never be selected because an earlier arm accepts everything it accepts | | `VXT0054` | a literal pattern cannot be compared with its subject | | `VXT0055` | a `null` pattern; reference subjects are not supported in `match` yet | -| `VXT0056` | an enum case pattern such as `.Ready`; enum declarations are not implemented | +| `VXT0056` | an enum case pattern such as `.Ready` for a subject that is not of an enum type | | `VXT0057` | a type pattern names another type than its subject's; class hierarchies are not implemented | | `VXT0058` | a match subject is neither `bool` nor numeric; other subject types are not lowered yet | -| `VXT0059` | `break` or `continue` would leave a block that is used as a value | | `VXT0060` | a guard condition is neither `bool` nor numeric | | `VXT0061` | the `else` block of a guard can complete normally instead of leaving the enclosing scope | +| `VXT0062` | the `return` statements of a method or callable whose result type is inferred carry values of different types | +| `VXT0063` | the return type of a method declared with `auto` cannot be inferred: every result is a call that depends on the method itself | +| `VXT0064` | an enum has no member of the given name, in `Enum.Member` or in a case pattern | +| `VXT0065` | an operation on a value of an enum other than `==` or `\=` with a value of the same enum | +| `VXT0066` | the underlying type of an enum is not an integer type | +| `VXT0067` | an enum names a member twice | +| `VXT0068` | the value of an enum member does not fit the underlying type of the enum | +| `VXT0069` | a target-typed `.Member` stands where no enum type is expected: the type of the place is inferred, or is not an enum | +| `VXT0070` | the value written for an enum member is not a constant integer expression, names something that is not an earlier member of the same enum, or has no value, as a division by zero has none | + +A block used as a value may leave instead of yielding a value: `return` +leaves the enclosing method and is checked against its return type +(`VXT0005`), also from inside a loop used as an expression, and `break` and +`continue` target the nearest loop around the expression and need one +(`VXT0025`, `VXT0027`). A `break` follows the rules of that loop: it carries +a value to a loop used as an expression (`VXT0040` without one) and none to a +loop statement (`VXT0026`). The condition and the update clause of a loop +belong to the loop: a `break` in a block used as a value there leaves that +loop, a `continue` in the condition evaluates the condition again, and a +`continue` in the update clause of a `for` ends the update. A callable is +not inside the loops around the place that creates it, so a `break` or +`continue` in its body is `VXT0025` or `VXT0027` unless a loop of its own +encloses it. The returns of a callable whose +result type is inferred are collected through expressions as well and must +agree (`VXT0062`); the returns of a nested callable are its own. A method +declared with `auto` is inferred the same way before its callers are +checked, wherever it is declared; calls of methods that are not inferred yet +take no part, so a recursive method is inferred from its base case, and a +method all of whose results depend on itself is `VXT0063`. A block that cannot complete normally +needs no final expression and gives its expression no type; the blocks that +complete do. When no block completes, the expression never yields a value. +That is valid: the place that would have received the value is not held to a +type, because it is never reached, and nothing is stored for it. The +statements after it are still checked. + +Whether a block can complete normally, for `VXT0046` and `VXT0061`, is decided +from its control flow. A statement cannot complete when it is a `return`, a +`break` or a `continue`; an `if` with an `else` whose two blocks both cannot; +a nested block that cannot; a loop whose condition is the literal `true`, or +absent in a `for`, and that no `break` leaves, also from a block used as a +value; a statement `match` one arm of which always matches and all arms of +which are blocks that cannot; or any statement an expression of which is +always evaluated and never yields a value. A block cannot complete when any +of its statements cannot. A call is assumed to return: calls whose result is `never` +are not recognized yet. A match used as an expression is complete, so that `VXT0052` is not reported, when an arm without a guard has only `_` and type patterns, or when every @@ -341,7 +391,7 @@ literal. An arm made only of untyped numeric literals takes its type from the context that receives the match, and without one from the first arm that has a type. A literal pattern is typed as the other operand of a comparison with its -subject. A pattern binding is immutable, so assigning it is `VXT0003`. An +subject. A pattern binding is an ordinary local and may be assigned. An expression body in a statement match must have an effect, like any expression statement; a pure one is `VXT0013`. @@ -358,6 +408,48 @@ of its own: an unterminated block is `VXP0002`, a name it declares is unknown after it, and declaring a name that is already in scope is `VXR0003`, as for any local. Two blocks side by side may declare the same name. +### Nesting limits + +The stages after Core recurse once per level of real nesting, so the +frontend bounds how deep a function body may nest and reports the place where +it becomes too deep. The check runs before any analysis of the body. The two +values are resource limits of this implementation, chosen against the +measured cost of a level in every native stage; the specification states no +nesting limit, and they are not language rules. They are the limits this +version of the compiler ships with, together with a compiler stack +reservation of 256 MiB. A program at the expression limit commits about +2.5 MiB of that stack in an ordinary build on Windows, Linux and macOS, and +at most 5.5 MiB in a sanitizer build; the measurements are in +`Benchmarks/2026-10-04-Nesting-And-Chains.md`: + +| Code | Meaning | +| --- | --- | +| `VXP0039` | a statement is nested more than 256 levels deep in other statements | +| `VXP0040` | an expression is nested more than 1024 levels deep in other expressions | + +`VXP0041` is no longer reported. It refused a value of an enum member that +was not an integer literal; a member value is now any constant integer +expression, and `VXT0070` reports one that is not. + +The statements of a function body are at level 1, and an expression that is +not an operand is at level 1. A block, a branch, a loop body, a `guard` +block, a block used as a value and a closure body are each one statement +level below what holds them. The links of an `else if` chain are all at the +level of the first `if`, and so are the `if` statements of `else { if ... }`. +The body of a `match` arm is one level below its match, however many arms +the match has. Expressions inside a statement that is itself inside +an expression keep counting from that expression. Each function reports each +code at most once, at the first node in source order that is one level +beyond the limit, and nothing below that node is examined. + +A chain of a binary operator is not nesting either. The left operand of a +binary operator is at the level of the operator, so `a + b + c + ...` and +`a && b && c && ...` are at one level however long they are: every stage +walks such a chain in a loop, and a sum of 50000 operands compiles. A right +operand, a call argument, a conditional result and the operand of a unary +operator are one level below the expression that holds them, so +`a + (b + (c + ...))` nests one level per addition. + ### Supplied token streams Embedding clients may supply a token list through `ParserInput`. An empty list @@ -440,6 +532,40 @@ Xpp and Xmm have bounded public readers. Malformed artifacts report framing, ver at decode. Structurally valid but semantically invalid artifacts report the owning Xpp or Xmm verifier failure before optimization or lowering. +A callable that remembers its result is checked by every stage that carries it. Each stage reports under its own code +that the operand is not a callable without parameters whose result is `bool` or numeric, or that the result does not +have the operand's type: + +| Code | Stage | +| --- | --- | +| `VXC1073` | Core, in the Haskell and the native verifier | +| `VXC0025` | CorePrep, in the Haskell verifier | +| `VXC1074` | CorePrep, in the native verifier | +| `VXP1047` | Xpp | +| `VXL1053` | Xmm | + +None of them can be reached from source: the frontend produces the operation only in a form that passes. They report a +malformed artifact or a defect of a stage. + +A call of a runtime function is checked the same way, by every stage, against the runtime catalog: that its first operand +is a literal that names a function, that it has that function's number of arguments, that each argument has a type the +function takes, and that the call has the type the function returns. + +| Code | Stage | +| --- | --- | +| `VXC1075` | Core, in the Haskell and the native verifier | +| `VXC0026` | CorePrep, in the Haskell verifier | +| `VXC1076` | CorePrep, in the native verifier | +| `VXP1048` | Xpp | +| `VXL1054` | Xmm | + +These cannot be reached from source either. What a program can get wrong about the console and about strings is reported +by the type checker, with `VXT0071` to `VXT0079`; [Console output and strings](CONSOLE-IO.md) lists them. + +A method reference `Type::Method` reports `VXT0080` when the method has several overloads and the place does not expect +the type of exactly one of them, and `VXT0081` when its receiver is a value and not a type; +[Static member resolution](STATIC-MEMBER-RESOLUTION.md) lists the forms. + ## Safe output behavior Failure must not make an old file look newly built. diff --git a/Documents/EVALUATION.md b/Documents/EVALUATION.md new file mode 100644 index 00000000..3d51ceec --- /dev/null +++ b/Documents/EVALUATION.md @@ -0,0 +1,242 @@ + + + +# Evaluation by need + +Visual X# is a lazy language with call-by-need evaluation. A value is computed +when it is first needed and at most once, and a value that is never needed is +never computed. Effects are not lazy: they happen where they are written. The +source has no notation for either; the compiler tells the two apart. The rules +are in `Spec/Language/Evaluation.vxs`. + +This document says how much of that the compiler implements and how. It is an +implementation reference, not the definition of the language. + +## What is implemented + +The value of a local binding is computed by need when all of the following +hold: + +- its type is `bool`, a numeric type or an enum; +- the binding is never assigned after it is declared; +- no closure captures it; +- its initializer is built from names, literals, operators, tests, + conditionals and calls, with no store, no transfer of control and no block. + +That holds in the body of a method and in the body of a callable alike. + +Such a value is computed by the first read that is reached, and by no later +one. A binding whose initializer is never read computes nothing, so a division +by zero or a call that never returns in it does nothing. + +An argument of a call that names a method directly is passed by need when all +of the following hold: + +- the type of the parameter is `bool`, a numeric type or an enum; +- the method is not certain to need the parameter before it does anything + else that can be observed; +- the argument has no effect, and it may fail, may run without end, or reads a + value that is itself computed by need. + +Such an argument is computed when the method first needs it, at most once +however often the method reads it and however many methods it is handed +through, and not at all when no method needs it. When the caller needs the +same value, whoever needs it first computes it for both. + +Everything else is computed where it is written, as before. That is correct +for an expression with an effect and a restriction of the implementation for +the rest; see "What is pending". + +## How a binding is deferred + +The frontend lowers a deferred binding without any new form of Core. The +binding declares three kinds of local: + +- the local itself, with a neutral value; +- a flag that is false until the value has been computed; +- a copy of each variable the initializer reads that is assigned anywhere in + the function, taken where the binding stands. + +Every read of the local is lowered to a test of the flag that, when the flag +is false, evaluates the initializer, stores the value and sets the flag, +followed by a read of the local. The initializer is lowered at each read, over +the copies, so it means what its variables held at the binding however late it +runs. The stages after the frontend see ordinary locals, stores and branches. + +The body of a method is lowered twice. The first lowering evaluates every +binding in place and is kept only for what it shows: which locals are assigned +after their binding and which a closure captures. The second lowering defers. + +An expression written as a statement of its own, and a value assigned to the +discard, are evaluated: the statement is the need. That is the rule of the +language, example 12 of the specification file, not a restriction. + +## Output is an effect + +A store shows in the expression that makes it. Console output does not: +`int x = Log(1);` writes only because of what the body of `Log` does. The +frontend therefore finds the methods a call of which may write, from all +methods of the program together, and an expression that contains such a call +is evaluated where it stands: its binding computes its value in place, and as +an argument it is computed at the call. Examples 15 and 16 of the +specification file state the rule, and `CONSOLE-IO.md` describes the analysis +and what it errs on. + +## How an argument is passed by need + +A value that one function hands to another cannot live in a flag and a slot of +a frame: the other function cannot reach the frame. It lives in a *suspended +computation*, a callable without parameters that computes the value the first +time it is called and returns the remembered value every time after. Core has +one primitive for it, `Memoize`: its operand is a callable without parameters +whose result is `bool` or numeric, and its result is a callable of the same +type that calls the operand at most once. Every copy of the result shares the +one remembered value. `CORE-IR.md` has the operation and `AARC-ABI.md` the +object it becomes. + +The frontend decides per method which parameters are passed by need. A +parameter of a type that can be suspended is passed by need unless the first +thing the method does that can be observed needs it. Statements that only +bind values without effects, or evaluate expressions that can neither fail nor +run without end, are passed over on the way to that first thing; a call does +not need an argument it passes by need itself, so the answer for one method +depends on the others and is computed for all of them together until it no +longer changes. + +A method keeps the function it always had, with every parameter a value. A +method with a parameter by need gets a second function beside it, named after +the method with `$need` and a number, that takes a suspended computation in +the place of each such parameter. It is lowered from the same body, and a read +of such a parameter calls its computation. It is lowered only when some call +asks for it. + +A call uses the second function when at least one argument is worth +suspending. For each parameter that is passed by need it passes: + +- the suspended computation of the argument, created at the call, when the + argument is handed on by need. It takes the values of the variables the + argument reads as they are at the call, so the argument means what it meant + there however late it is computed; +- the computation itself, when the argument is the name of a value that is + already suspended. That is how a value handed through several methods is + still computed once; +- a callable that returns the value, computed at the call, for any other + argument. + +Every other call uses the function the method always had and computes its +arguments at the call, which costs nothing and cannot be told apart. + +A local whose value may be handed on by need is suspended in a computation of +its own instead of a flag and a slot, and every read of the local calls it. +That is what lets the caller and the method share one computation. + +A suspended computation is not free. It is two objects of the runtime, a +closure over what the argument reads and the callable that remembers its +result, and in generated code a function for the argument and an entry the +closure is called through. A program whose calls pass many arguments that are +themselves calls is several times larger than it was when arguments were +values, and takes correspondingly longer to compile. One object and no +function of its own for a computation that only calls a method is possible +and pending. + +## Where laziness cannot be observed + +Deferring a value costs a flag and a test at each read. The compiler computes +a value where its binding stands, with neither, in two cases in which no +program can tell: + +- the initializer cannot fail and cannot run without end, and reads no value + that is itself deferred: it does not call and does not divide; +- the statement after the binding is certain to read the value: the + expression that statement evaluates first reads it outside the right side of + `&&` and `||` and outside the branches of a conditional. + +In the second case one thing differs: when the value and an operand evaluated +before it in that statement both fail, the value fails first. The language +leaves that open. Example 13 of `Spec/Language/Evaluation.vxs` says that +whether a program fails is determined and which of several failures one +statement meets is not, and that a value which runs without end counts as one +that cannot be computed. The order of statements and of effects is +determined. The same rule covers a value that an expression is certain to +read more than once, which is computed ahead of that expression. + +A call needs the arguments it passes as values and does not need the ones it +passes by need, so the second case does not count a read that stands in an +argument passed by need. + +The same reasoning decides which parameters are passed as values: one that the +method is certain to need first is computed at the call. + +## What is pending + +These are parts of the language the compiler does not implement yet. None is a +restriction of the language. + +| Pending | Today | +| --- | --- | +| an argument of a call through a callable | computed at the call: a callable takes values | +| an argument that a closure of the method captures | computed when the method is entered, because a closure takes the values of its captures when it is created | +| arguments of other types by need: `String`, callables, objects | computed at the call | +| arguments of a method of a template | computed at the call | +| results returned by need | a returned value is computed before the method returns | +| values of other types by need: `String`, callables, objects | computed where they are written | +| a variable that is assigned again | the binding and every assignment are computed in place | +| a binding that a closure captures | computed in place | +| every value by need as a suspended computation | a value that stays in one function is a flag and a slot in its frame; only a value that may be handed on is a suspended computation | +| the computation of a value by need stated once | every read carries it; see below | +| effects other than stores, transfers of control and console output | none exist in the implemented subset: it has no input and no shared state | + +A read of a value by need carries the computation of that value, guarded by +its flag, because the frame of one function is all a value by need has +today. An expression that is certain to read such a value more than once +computes it once ahead of itself, so a chain of values of which each reads +the one before it several times grows with its length. Reads that stand in +different branches of one expression are not certain and each carries the +computation: `int b = (c ? a : 0) + (d ? a : 1);` holds the computation of +`a` twice, and a chain of such bindings doubles with every link. That is a +cost in compile time and code size, never in what a program computes. A value +that is suspended does not have it: a read of it is one call. + +## Values that cannot be computed + +Example 14 of the specification file names the values that have none: an +integer quotient or remainder by zero, with `/`, `//` and `%`, and a shift by +an amount that is negative or not less than the width of the shifted value. A +program that needs one stops. + +LLVM gives the instructions for these operations no meaning on such operands. +Without a check an optimizer is free to delete the computation and what +depends on it, and it did: a program that divided by zero and needed the +quotient ran on as if it had not. The backend therefore precedes each such +instruction with a check that stops the program with a trap. The check is a +small internal function that is always inlined, so it costs one comparison +where the operand is unknown and nothing where it is known. + +The least value of a signed type divided by minus one is not such a value. +Its quotient does not fit the type, which generated code treats as it treats a +sum or a product that does not fit: the result wraps, to the least value +again, with remainder zero. The processor's own division instruction would +stop the program, so the backend carries that one division out with a divisor +of one. The language has not fixed what a result that does not fit is; this +is the behaviour of the backend, pinned by `ComputabilityExecutionTests.cpp`. +The reference Core evaluator of the tests computes with unbounded integers and +does not wrap. + +A program that stops this way ends with the status the operating system gives +a process that executed an instruction the processor refuses. A diagnostic +that names the failure is pending. + +## Verification + +`LazyEvaluationTests.hs` observes laziness through what tells a computed value +from one that was not: a division by zero, a call that never returns, and the +number of steps a program takes in the reference Core evaluator, which is how +"at most once" is checked, for a value the caller and a method share as well. +`MemoizeTests.hs` has the rules of the remembering callable and its evaluation. +`source_feature_smoke` runs the same programs through LLVM in both pipeline +modes, where a division by zero that was computed would end the process and a +call that never returns would never end, and checks after each program that no +object of the runtime is left behind. `executable_run_tests` builds programs +into native executables and runs them as processes: a program that passes +arguments by need, one that needs a quotient by zero and stops, and one that +creates and releases two million suspended computations. diff --git a/Documents/FUZZING.md b/Documents/FUZZING.md index 386c2c2e..9d9ccec6 100644 --- a/Documents/FUZZING.md +++ b/Documents/FUZZING.md @@ -189,17 +189,122 @@ documents. It runs independently of libFuzzer and does not claim guided coverage. `source_fuzz_smoke` checks valid-source lowering and then runs the differential oracle on every generated program shape at trip counts 0 through 11, once with a zero and once with a nonzero generated expression, before -mutation campaigns begin. It also runs the programs of -`ExpressionExecutionCases.cpp` and `BranchingExecutionCases.cpp`: assignments, +mutation campaigns begin. `source_execution_smoke` runs the programs of +`BranchingExecutionCases.cpp`, and `source_expression_smoke` those of +`ExpressionExecutionCases.cpp` and +`LeavingExecutionCases.cpp`: assignments, increments and loops used as values, and `match`, `if` expressions and `guard`, each with a hand-written result that both native pipeline modes must return. Those tables are transcribed from the evaluation tables of `AssignmentExpressionTests.hs`, `LoopExpressionTests.hs` and `BranchingTests.hs`, where the same programs are checked against a reference Core evaluator. `BranchingExecutionCases.cpp` also runs a match of 200 arms on -subjects at the start, at both sides of a group boundary of the lowering, deep -in later groups and in the catch-all, and an `else if` chain of 300 links, -which the native stages after Core walk in a loop. +seven subjects, from its first arm to the catch-all, an `else if` chain of +300 links, which the native stages after Core walk in a loop, on five +subjects, and programs at the nesting limits of the frontend: 255 nested +`if` statements, entered and not entered, 1023 calls nested in each other's +arguments, and a sum of 1024 operands. The nested calls are the shape that +costs most compiler stack for each level. `source_execution_smoke` ends by +printing `compiler stack committed: N KiB`, the stack its compiler thread +committed for all of its programs, so that the figure can be read from the +log of any platform that reports it: Windows, Linux and macOS. + +A body that several runs share is compiled once, and up to eight small +bodies share a program. `ExecutionCases.cpp` puts each body in a method of +its own and calls it once for each run from one further method, which +compares every result with its expected value and returns a distinct bit for +each run that differs; the program must return zero from both +pipeline modes, and a result that is not zero names the runs that failed. +Compiling dominates the cost of these cases under sanitizers, and the smoke +program has a process watchdog, so the runs of a body are not worth a +compilation each. + +`source_feature_smoke` runs the tables that are written by hand for single +features: methods with inferred return types, evaluation by need, arguments +passed by need and classic enums. A program that passes an argument by need +creates objects of the runtime, so each of those runs alone under the +allocation check described below. It also compiles programs that own +closures while control leaves through a block used as a value, so that the +ownership verifiers of Xpp and Xmm see those paths, and runs a hand-written +table of closures: created, called, nested, returned and alive across loop +transfers. A closure calls the AARC runtime, and the JIT resolves a runtime +symbol in the process that hosts it, so this program links the runtime and +exports its entry points. The other fuzz programs do not link it. Each +closure program runs alone, and the runtime must hold no more allocations +after it than before it, so a closure or a capture that is not released +fails the program that leaked on every platform, not only where a leak +sanitizer runs. + +The branching and leaving tables are generated. Their cases are written in +`Compiler/Fuzzing/Cases/Selection.cases` and `Leaving.cases`: a body, its +runs and the value each run must return, written by hand from the language +rules. `go -C helpers run ./cmd/execution-cases generate` writes the rows +under `Compiler/Fuzzing/Generated` and the Haskell module +`BranchingEvaluationCases.hs` from them, and `check` fails when a committed +table differs; the helper tests and CI run that check. One source keeps the +two tables equal. It does not make them independent: a wrong expectation in +a case file is wrong in both. The independent checks are the hand-written +tables that do not come from these files, `ExpressionExecutionCases.cpp` and +the inferred-return table of `SourceFeatureSmoke.cpp`, the oracle tests of +`BranchingOracleTests.hs`, which compare each `match` with the `if` chain it +stands for, and the differential generator with its host model. + +The tables are a program of their own because each smoke program is one +deterministic check under one process watchdog. As one program, in the +fuzzing configuration, the tables took 107 seconds and the differential +sweep 107 seconds of a 221 second run, which left no margin under the 240 +second watchdog on a loaded machine; a passing second attempt was not a +fix. The watchdog is unchanged, and no case was removed. Measured on the +same Windows machine in the same configuration after the split, three runs +each: `source_execution_smoke` takes 77 to 110 seconds, of which in the run +that was broken down 23 were the expression table, 35 the branching table, +17 the leaving table and 2 the ownership programs; `source_fuzz_smoke` takes +100 to 105 seconds, of which 98 are the differential sweep. Other work ran +on the machine during these runs, so the spread is not the programs' own. +In an ordinary build each takes about 7 seconds. + +The feature tables became a third program for the same reason. With the +tables for evaluation by need and enums, `source_execution_smoke` ran past +the watchdog in the fuzzing configuration. Apart, on the same Windows +machine in that configuration with nothing else running, +`source_execution_smoke` takes 91 seconds, of which 24 are the expression +table, 45 the branching table and 22 the leaving table, and +`source_feature_smoke` takes 45 seconds, of which 28 are the closure +programs. The watchdog is unchanged, and no case was removed. + +`source_console_smoke` runs programs that write: console output, the +conversions of a format, and the joining and comparing of strings. Each is +compiled from source in both pipeline modes and run under the JIT with the +runtime library this program links, whose output goes to a sink instead of +the streams of the process. What a program wrote is compared with text +written by hand, and the runtime must hold no more allocations after a +program than before it, because every string is one of its objects. + +The expression and leaving tables became a fourth program, +`source_expression_smoke`, when arguments came to be passed by need. Every +run of a body passes its inputs as calls, so that no stage can fold them, and +each such argument is now a suspended computation: two objects, a function +for its body and an entry it is called through. The programs of the three +tables grew with that. In an ordinary build they took 27 seconds together +where they had taken 7, and 344 seconds in the fuzzing configuration, past +the watchdog with every case passing. Three changes brought the ordinary +build to 21 seconds: a module has one entry, one destructor and one metadata +record for the callables that remember a result of one type, where it had +them for each such callable; a closure that owns nothing has no destructor; +and the native Core verifier no longer copies every visible definition for +each branch and each closure, which had made its time grow with the square +of a function's size. That was not enough for the fuzzing configuration, so +the tables are two programs. The watchdog is unchanged, and no case was +removed. What an argument by need costs in generated code is a cost of the +current lowering, listed in `EVALUATION.md`. + +The source fuzz targets and the source smoke programs run the compiler on the +compiler stack, as `vxs` does, because an input nested up to the frontend's +limits does not fit on the default stack of a process. `Corpus/source` has +permanent seeds for deep nesting, for nesting at and one level beyond each +limit, for long `else if` and operator chains, which are not nesting, for +negative constant patterns, and for blocks used as values that leave with +`return`, `break` and `continue`. The differential generator selects one of fourteen program shapes from the byte after its generated expression, modulo the shape count. Adding a shape diff --git a/Documents/IMPLEMENTATION.md b/Documents/IMPLEMENTATION.md index 25c1f4f1..39d38e83 100644 --- a/Documents/IMPLEMENTATION.md +++ b/Documents/IMPLEMENTATION.md @@ -46,9 +46,10 @@ is no decrement operator; `--` always starts a comment. A `while` or classic `fo with literal, wildcard and binding patterns and guards; `if` is also an expression over two value blocks; `guard (condition) else { ... }` runs its block when the condition is false; and a `{ ... }` at the start of a statement is a nested block with its own scope. Null coalescing `??` and `??=`, storage -targets other than a named local, loop, conditional and match values that are not `bool` or numeric, `return` inside a -loop expression, `return`, `break` and `continue` out of a block used as a value, the `null` and enum case patterns, type -patterns over class hierarchies, and bindings in conditions are not implemented. +targets other than a named local, loop, conditional and match values that are not `bool`, numeric or an enum, the +`null` pattern, type patterns over class hierarchies, and bindings in conditions are not implemented. `return`, +`break` and `continue` out of a block used as a value are implemented, in loop bodies, loop headers and loops used as +expressions; "Pending branching and loop forms" below lists what is still owed. It does not yet implement the complete language catalog in `Spec/`. Core optimization is connected, verifier-guarded, and fixed-point driven. It performs immutable literal propagation, @@ -79,6 +80,39 @@ The connected frontend is strongest around scalar expressions, local control flo CorePrep, namespace merging, and entry validation. Fixed-width integer/radix/separator behavior, character packing, numeric boolean context, source `void`, stable `SymbolId` identity, and constant range checks are represented before Core emission. +### Pending branching and loop forms + +These are parts of the specified language that the frontend recognizes and +rejects with a diagnostic that says so, or that fail in a later stage. They +are owed work, not rules: none of them is a restriction of the language, and +the specification is not changed to match them. + +| Pending | Specified by | Today | Needs | +| --- | --- | --- | --- | +| a binding in the condition of `if`, `guard` or `while`, as in `guard (auto user = Find()) else { return; }` | `Spec/Language/Decls.vxs`, examples 190 to 192 | `VXP0035` | optional values | +| evaluation by need beyond local bindings of scalar type: arguments, results, other types, assigned and captured variables | `Spec/Language/Evaluation.vxs` | computed where they are written; see [Evaluation by need](EVALUATION.md) | thunks as values in the IR and the runtime | +| a call that does not return as a way of leaving a `guard` block or a block used as a value | example 297 | every call is assumed to return, so the block is taken to complete: `VXT0061` or `VXT0046` | a way to know that a call does not return; how that is expressed in the language is not decided here | +| a `match` over a subject that may be null | example 304 | `VXT0055` | nullable subjects | +| an enum declared inside a class | the nested declarations of section 9 | `VXP0006` | qualified type names | +| `enum class`, the enum with payloads | section 8 | not parsed | the object model | +| type patterns over class hierarchies | examples 198 to 203 | `VXT0057` | class hierarchies | + +Implemented and verified through native execution, unoptimized and +optimized: `return`, `break` and `continue` out of a block used as a value, +in loop bodies; `break` and `continue` in a loop condition; `break` and +`continue` in a `for` update clause; a `break` that carries a value out of a +block used as a value to a loop used as an expression; `return` out of a +loop used as an expression, directly and from a block used as a value; an +`if` or `match` expression none of whose branches completes; classic enums, +with their numbering, member values computed from earlier members, their +comparison and `match` over them, complete +without a catch-all arm when every value is named; calls of +methods whose return type is inferred, in the same class, in another class, +through chains of such methods and through mutual recursion; the inference +of a callable's return type from returns inside its expressions; and +callables created inside callables, called, returned and kept, with the +AARC runtime that owns them. + The full `Spec/` catalog is not implemented. Object/value layout, the complete standard-library surface, cross-namespace imports, template declaration cloning and constraint selection, exception lowering, ownership runtime operations, generators, FFI, assembly, and many advanced declaration forms require additional semantic and native work. Unsupported forms must produce frontend or @@ -99,6 +133,7 @@ The repository contains: - Xpp control-flow, self-copy, and liveness-based dead `Define Copy` optimization; - shared directional worklist scheduling, dense definite-initialization facts, and packed AARC ownership states; - an Xpp-owned verifier for module/function identity, storage declarations, typed operands, and CFG targets; +- Xpp ownership placement: explicit retains and releases for every AARC value, and closures for methods used as values; - Xpp-to-Xmm lowering; and - Xmm virtual-register move and dead materialization optimization; - an Xmm-owned verifier with register, signature, call, operand, result, and control-flow diagnostics, exposed through the @@ -111,7 +146,7 @@ The repository contains: - in-memory LLVM IR and bitcode serialization with explicit `.ll`/`.bc` writers. The production frontend boundary uses public `VXCR` Core. The internal `VXCP` codec remains tested for in-process and golden -contract coverage, but the CLI does not expose CorePrep. Bounded `VXPP` and `VXMM` v5 codecs now own public Xpp/Xmm disk +contract coverage, but the CLI does not expose CorePrep. Bounded `VXPP` and `VXMM` v7 codecs now own public Xpp/Xmm disk artifacts and forward-only pipeline resumption. LLVM target-machine emission and typed C++20 LLD invocation produce `.o`, `.asm`, and `.vxse` artifacts. Project object and assembly requests produce one flattened output per source in the selected entry namespace; each owner boundary verifies the source catalog, and the driver replaces the set through a recoverable @@ -122,7 +157,7 @@ unit. | Capability | Status | Boundary | | --- | --- | --- | -| bounded VXCR v6 decode | connected | C++20 Core reader, closure records, template arguments, source ownership, and scalar payload validation | +| bounded VXCR v10 decode | connected | C++20 Core reader, closure records, template arguments, source ownership, and scalar payload validation | | native Core semantic verification | connected | `Compiler/Core` | | Core-to-CorePrep atomization/CFG | connected | dedicated adapter | | CorePrep structural/semantic verification | connected | native CorePrep verifier | @@ -133,8 +168,13 @@ unit. | target object and assembly output | connected for supported values | target machine | | `.vxse` link | connected for supported values | entry bridge plus typed LLD driver | | closure object ABI | connected | Xpp/Xmm, LLVM, and AARC runtime boundary | +| callable that remembers its result | connected | Core, CorePrep, Xpp, Xmm and LLVM carry `Memoize`; the frontend produces it for arguments passed by need | +| runtime calls | connected | one operation with a catalog of functions in Core, CorePrep, Xpp, Xmm and LLVM | +| console output, string concatenation and equality | connected | `Console.Print`, `Println`, `Printf`, `Printfn`, the four `Error` forms and `Format`; see [Console output and strings](CONSOLE-IO.md) | +| the runtime in a native executable | connected | `vxs-runtime.lib`, the ownership and text runtime without a C runtime, linked with every executable | +| integer division by zero and shifts outside the width | connected | a check before the instruction stops the program | | recursive constructed-type classification | Haskell/native semantic models complete, process connection pending | frontend and Core nominal catalogs | -| Xpp/Xmm disk codecs | connected | bounded v5 `VXPP`/`VXMM` readers and writers | +| Xpp/Xmm disk codecs | connected | bounded v7 `VXPP`/`VXMM` readers and writers | | project per-source object/assembly emission | connected for the selected namespace | source ownership through CorePrep, Xpp, and Xmm | | VXCI `-Header` | registered and rejected explicitly | export/ABI semantics and a header writer are not connected | diff --git a/Documents/MONOMORPHIZATION.md b/Documents/MONOMORPHIZATION.md index a1a2003d..98f221ed 100644 --- a/Documents/MONOMORPHIZATION.md +++ b/Documents/MONOMORPHIZATION.md @@ -26,7 +26,7 @@ The implemented slice includes: - unambiguous literal value arguments such as `Buffer<32>`; - exact integer, character, Boolean, unary, and binary constant evaluation; - an ordered `TemplateArgument` sum in typed AST, Core, CorePrep, Xpp, and Xmm; -- strict Core v8, CorePrep v6, and Xpp/Xmm v5 codecs; +- strict Core v10, CorePrep v8, and Xpp/Xmm v7 codecs; - recursive structural validation and parameter collection; - independent type-parameter and value-parameter substitution; - deterministic structural identity rendering; diff --git a/Documents/OWNERSHIP-FLOW.md b/Documents/OWNERSHIP-FLOW.md index f41fd5d5..787f7d59 100644 --- a/Documents/OWNERSHIP-FLOW.md +++ b/Documents/OWNERSHIP-FLOW.md @@ -114,16 +114,70 @@ Before LLVM lowering, a successful Xpp/Xmm ownership check guarantees: - a join cannot hide a release performed on only some paths; and - block serialization order cannot select an ownership outcome. -The pass does not yet prove global leak freedom, infer missing retain/release -operations, or decide whether a source-level object graph contains a cycle. -Those are separate responsibilities. Automatic retain/release insertion must -eventually produce explicit Xpp operations that satisfy this verifier. +The verifier does not prove leak freedom or decide whether a source-level +object graph contains a cycle. Writing the retains and releases is the job of +ownership placement, described next, whose output this verifier checks. Concurrent Bacon-Rajan with trial deletion remains a future optional cycle collector and does not weaken deterministic AARC verification. +## Ownership placement + +CorePrep names values and does not say who releases them. +`Visual::XSharp::Xpp::PlaceOwnership` runs once, on the Xpp form lowered from +CorePrep and after the Xpp optimizer when that is enabled, and makes ownership +explicit. An Xpp artifact read from disk already carries the result and is not +placed again. + +The convention is local to a function, so functions agree without looking at +each other: + +| Value | Owner | +| --- | --- | +| parameter | the caller, for the duration of the call; the callee never releases it | +| result | the caller, which receives one reference and releases it | +| local | the function, from the definition to the last use on each path | +| strong capture | the closure, which takes its own reference when it is created and releases it in its destructor | +| computation of a callable that remembers its result | that callable, which takes its own reference when it is created and releases it in its destructor | + +From that follow the operations the pass writes: + +- a value is released after the instruction that uses it last, and directly + after the instruction that defines it when nothing uses it; +- a value that dies on one edge of a branch and lives on another is released + on the edge: at the top of the target when the target has no other + predecessor, otherwise in a block of its own on that edge; +- a copy of a value that is used again becomes `RetainStrong`; a copy of a + value that is not used again stays a copy and takes over the reference; +- a returned local leaves with its reference; a returned parameter is retained + first, because the caller receives an owned result; +- a call whose owned result is discarded defines a symbol, so that the result + can be released; +- an instruction that reads the symbol it writes reads the old value from a + symbol of its own, which is released after the new value is stored; +- a parameter that the body assigns is copied into a local at a new entry + block, with a reference of its own, so that a symbol is either borrowed or + owned for the whole function. + +A value is released after its last use rather than at the end of a source +scope. Xpp has no scopes, and a value that is live at a point has been defined +on every path to that point, so no release has to test whether there is +anything to release. Liveness is computed for the owned symbols alone, with +the analysis the optimizer uses. + +A method that is named where a value is expected, rather than called, has no +closure object to point to. The pass creates a closure without captures at +that use; the callee of a direct call stays a method. Before this pass such a +value was the address of the method's code, and calling it read an invoke +pointer out of that code. + +What the pass does not do: it does not break reference cycles, it keeps a +value alive until its last use and no longer, which is earlier than the end of +its source scope, and it knows no weak or unowned locals, because the frontend +produces those only as capture modes of a closure. + ## Testing -Three layers protect the contract: +Five layers protect the contract: 1. Compiler/Analysis/Tests/OwnershipFlowTests.cpp tests the reusable state lattice, fixed point, malformed CFG handling, loops, joins, and deterministic @@ -134,6 +188,18 @@ Three layers protect the contract: 3. Compiler/Codegen/Xmm/Tests/OwnershipVerifierTests.cpp repeats the boundary cases for typed virtual registers and confirms the main Xmm verifier publishes ownership failures. +4. Compiler/Codegen/Xpp/Tests/OwnershipPlacementTests.cpp runs the placed + function on a reference-count model that knows nothing of the pass: along + every path, taking each loop around once more than it must be entered, each + reference must be released exactly once and nothing may be used after its + last release. The model is first shown to reject a leak, a double release + and a use after release. +5. `source_feature_smoke` runs closure programs through LLVM with the AARC + runtime, one program at a time, and requires the runtime to hold no more + allocations after a program than before it, in both pipeline modes. Without + the placement pass that check fails on the first program. The programs + that pass arguments by need are held to the same check: each suspended + computation is two objects, one of which owns the other. Tests construct a complete operation sequence. Merely asserting that an opcode exists does not prove its handle precondition, result representation, or diff --git a/Documents/PROJECT-ARTIFACTS.md b/Documents/PROJECT-ARTIFACTS.md index 818a6c92..25aa5966 100644 --- a/Documents/PROJECT-ARTIFACTS.md +++ b/Documents/PROJECT-ARTIFACTS.md @@ -85,7 +85,7 @@ explicit single-file module may omit project ownership metadata. A project object/assembly build requires a complete, valid catalog and owners for every function in the selected module. -Current wire versions are Core v8, CorePrep v6, Xpp v5, and Xmm v5. Older +Current wire versions are Core v10, CorePrep v8, Xpp v7, and Xmm v7. Older versions are rejected by their owning reader. The new fields are part of the schema and are not silently defaulted when reading an older project artifact. See [Artifact wire contracts](ARTIFACT-WIRE.md) for field order, decoding @@ -233,7 +233,7 @@ The relevant tests are kept with their owners: - `Compiler/Artifact/Tests/SourcePathTests.cpp` covers canonical source identity and portable path rejection. -- `Compiler/Core/Tests/CorePipelineTests.cpp` covers the non-empty Haskell v8 +- `Compiler/Core/Tests/CorePipelineTests.cpp` covers the non-empty Haskell v10 golden, metadata propagation through Xpp/Xmm, per-source object and assembly output, empty source units, and collision preservation. - `Compiler/Driver/Tests/ProjectArtifactTests.cpp` covers flattening, diff --git a/Documents/README.md b/Documents/README.md index 7535c4d5..ffb515c9 100644 --- a/Documents/README.md +++ b/Documents/README.md @@ -69,8 +69,12 @@ following vocabulary: - [Template monomorphization](MONOMORPHIZATION.md) defines concrete specialization identity, placement, demand discovery, and current limits. - [Numeric types](NUMERIC-TYPES.md) records fixed scalar widths and the gap between frontend semantics and the current Core transport. -- [Scalar pipeline](SCALAR-PIPELINE.md) follows fixed-width values through verification, Core wire v8, Xpp, Xmm, and LLVM. +- [Scalar pipeline](SCALAR-PIPELINE.md) follows fixed-width values through verification, Core wire v10, Xpp, Xmm, and LLVM. - [Artifact wire](ARTIFACT-WIRE.md) defines the bounded internal Core and CorePrep transport contracts. +- [Console output and strings](CONSOLE-IO.md) covers `System.Console`, the output format grammar, the runtime calls + it compiles to and the runtime library a native executable is linked with. +- [Evaluation by need](EVALUATION.md) says how much of the lazy evaluation of the language the compiler implements, + how a value is deferred, and what is pending. - [Match, if expressions, guard and nested blocks](BRANCHING.md) describes the implemented subset of those forms, their evaluation order and their limits. - [Control-flow safety](CONTROL-FLOW-SAFETY.md) defines short-circuit lowering and the shared Xpp/Xmm diff --git a/Documents/SCALAR-PIPELINE.md b/Documents/SCALAR-PIPELINE.md index f2f561c1..d7c73fee 100644 --- a/Documents/SCALAR-PIPELINE.md +++ b/Documents/SCALAR-PIPELINE.md @@ -107,8 +107,8 @@ canonical. These rules make the representation independent of endianness and make equality deterministic. Compatibility alternatives for the historical signed 32-bit and 64-bit -payloads remain readable in the in-memory variant. Current Core v8 and CorePrep -v6 producers use the structured representation for the complete scalar +payloads remain readable in the in-memory variant. Current Core v10 and CorePrep +v8 producers use the structured representation for the complete scalar catalog. `FloatingLiteral` contains a validated ASCII spelling. Accepted finite forms @@ -137,9 +137,9 @@ numeric types; CorePrep compares each numeric operand with a same-typed zero so Xpp, Xmm, and LLVM receive canonical booleans. Numeric branch conditions use the same conversion. -## Core wire v8 +## Core wire v10 -Core wire v8 writes a distinct type tag for every catalog member and retains +Core wire v10 writes a distinct type tag for every catalog member and retains the project source ownership introduced in Core v6. It adds structured loop statement tags without changing scalar payloads. Integer payloads contain: diff --git a/Documents/SPECIFICATION.md b/Documents/SPECIFICATION.md index 192f43cf..a33034b4 100644 --- a/Documents/SPECIFICATION.md +++ b/Documents/SPECIFICATION.md @@ -24,6 +24,7 @@ The complete linked catalog is maintained in [`Spec/README.md`](../Spec/README.m | --- | --- | | declarations and type forms | `Language/Decls.vxs` | | attributes | `Language/Attributes.vxs` | +| evaluation by need and effects | `Language/Evaluation.vxs` | | operators and precedence | `Language/Operators.vxs` | | optionals and null behavior | `Language/Optional.vxs` | | strings, scalars, escapes, and source text | `Language/String.vxs` | diff --git a/Documents/STATIC-MEMBER-RESOLUTION.md b/Documents/STATIC-MEMBER-RESOLUTION.md index f8ec5196..e8e70a7d 100644 --- a/Documents/STATIC-MEMBER-RESOLUTION.md +++ b/Documents/STATIC-MEMBER-RESOLUTION.md @@ -61,8 +61,12 @@ owned by the current type. The following forms are deliberately not implied by t | `Current()` inside `Counter` | Supported for the current type's method family | The lexical method spelling is associated with its declaring type before overload selection. | | `counter.Current()` | Rejected | The receiver denotes a value; instance-member dispatch and its ABI are not part of this vertical slice. | | `Counter.Create().Current()` | Rejected | A call-result receiver is a value expression, not a type name. | -| `Counter.Current` | Rejected | A bare method group does not yet have a first-class member-group type. | -| `System.Math.Math.Sqrt(x)` | Not connected | Cross-namespace type lookup and qualified namespace paths are a separate frontend seam. | +| `Counter.Current` | Rejected | A selector with a dot is not a method reference; the reference is written `Counter::Current`. | +| `Counter::Current` | Supported | The static method as a callable value. With overloads, the type the place expects selects one. | +| `counter::Current` | Rejected | A reference through a value is a bound method, which needs instance members. | +| `Demo.Counter.Current()` inside namespace `Demo` | Supported | A declaration of the namespace being compiled may be named through that namespace, as a receiver of `.` and of `::`. | +| `Demo.Counter value` | Not connected | A type written through its namespace in a type position is not yet the same type as its bare name. | +| `System.Math.Math.Sqrt(x)` | Not connected | Cross-namespace type lookup is a separate frontend seam; only the namespace being compiled is known. | | `Counter.Nested.Current()` | Not connected | Nested-type catalog lookup is not part of this top-level type catalog. | | `counter.Field` or `counter.Property` | Not connected | Field/property resolution and instance layout are separate from static method binding. | | Extension member lookup | Not connected | Extension discovery and real-member precedence remain separate work. | @@ -219,6 +223,8 @@ All diagnostics below originate in the Type Checker unless otherwise noted. | `VXT0032` | The selector receiver is not a declared type name. | | `VXT0033` | Matching candidates exist but are not accessible from the call site. | | `VXT0034` | A member selector is used without a supported direct method call. | +| `VXT0080` | A method reference names a method with several overloads, and the place does not expect the type of exactly one of them. | +| `VXT0081` | A method reference whose receiver is not a type: bound method references need instance members. | The checker preserves a source span on member-call diagnostics. An unresolved receiver remains a Name Resolution failure instead of being replaced by a fabricated Type Checker error. This makes the phase boundary visible to the CLI, analyzer, diff --git a/Documents/TEST-OWNERSHIP.md b/Documents/TEST-OWNERSHIP.md index 3d412d63..6d60181d 100644 --- a/Documents/TEST-OWNERSHIP.md +++ b/Documents/TEST-OWNERSHIP.md @@ -77,6 +77,8 @@ Representative labels are: //Compiler/Cli/Tests:cli_parser_tests //Compiler/Cli/Commands/Tests:execution_status_tests //Compiler/Cli/Commands/Tests:cli_command_tests +//Compiler/Cli/Commands/Tests:executable_run_tests +//Compiler/Runtime/Text/Tests:text_runtime_tests //Compiler/Core/Tests:core_pipeline_tests //Compiler/Driver/Tests:closure_pipeline_tests //Compiler/Driver/Tests:project_artifact_tests diff --git a/Formatter/README.md b/Formatter/README.md index fad3885d..20ab2c6b 100644 --- a/Formatter/README.md +++ b/Formatter/README.md @@ -8,6 +8,10 @@ changing it. It normalizes line endings, trailing horizontal whitespace, the fin indentation. The compiler's lossless source-fragment model distinguishes code from comments, character literals, normal strings, and raw strings, so structural characters inside protected text never affect indentation. +The formatter accepts what the compiler it is built with accepts. Version 0.1.1 is built with the compiler of +Visual X# 0.5.0, so a source may use method references (`Type::Method`), names written through their namespace, +conditionals over strings, `match`, enums and console formats; none of them changes how a line is laid out. + ```text vfmt Program.vxs vfmt -In-Place Program.vxs Library.vxs diff --git a/Formatter/build.gradle.kts b/Formatter/build.gradle.kts index 5a65453f..c5af40a6 100644 --- a/Formatter/build.gradle.kts +++ b/Formatter/build.gradle.kts @@ -14,7 +14,7 @@ plugins { } group = "com.progmasoft.visual.formatter" -version = "0.1.0" +version = "0.1.1" repositories { mavenCentral() } diff --git a/Formatter/sources/test/haskell/Main.hs b/Formatter/sources/test/haskell/Main.hs index 97fcadb1..4d656e88 100644 --- a/Formatter/sources/test/haskell/Main.hs +++ b/Formatter/sources/test/haskell/Main.hs @@ -33,6 +33,10 @@ main = do check "block formatting reaches a fixed point" formattingIsIdempotent check "pattern combinators remain intact while their block is indented" formatsPatternCombinators check "match arms, value blocks and nested blocks are indented by their braces" formatsBranchingForms + check "method references and names through the namespace are kept as written" formatsMethodReferences + check "a conditional over strings keeps its strings and its colon" formatsStringConditional + check "console formats keep their percent signs and their braces" formatsConsoleFormats + check "a source with the forms of 0.5.0 reaches a fixed point" currentFormsAreIdempotent checkIO "encoding conversion follows explicit input and output settings" encodingRoundTrip checkIO "UTF-8 input rejects malformed byte sequences" rejectsMalformedUtf8 @@ -312,3 +316,118 @@ rejectsOptions :: FormatOptions -> Bool rejectsOptions options = case formatSource options (CompilerInput "Program.vxs" "class Program {}") of Left [problem] -> diagnosticCode problem == "VXF0001" _ -> False + +-- `::` is one token of the language; the formatter must neither split it +-- nor take a part of a qualified name for something to lay out. +formatsMethodReferences :: Bool +formatsMethodReferences = + formats + defaultFormatOptions + ( concat + [ "namespace Demo;\n" + , "public class Program {\n" + , "public static int Log(_ int v) { return v; }\n" + , "public static int Run(_ (int) -> int f) {\n" + , "return f(3);\n" + , "}\n" + , "public static void Main() {\n" + , "auto held = Program::Log;\n" + , "int first = Run(Demo.Program::Log);\n" + , "int second = Demo.Program.Log(1);\n" + , "}\n" + , "}\n" + ] + ) + ( concat + [ "namespace Demo;\n" + , "public class Program {\n" + , " public static int Log(_ int v) { return v; }\n" + , " public static int Run(_ (int) -> int f) {\n" + , " return f(3);\n" + , " }\n" + , " public static void Main() {\n" + , " auto held = Program::Log;\n" + , " int first = Run(Demo.Program::Log);\n" + , " int second = Demo.Program.Log(1);\n" + , " }\n" + , "}\n" + ] + ) + +formatsStringConditional :: Bool +formatsStringConditional = + formats + defaultFormatOptions + ( concat + [ "class Program {\n" + , "static String Kind(_ int count) {\n" + , "String kind = count > 0 ? \"some { \" : \"none } \";\n" + , "return count > 9\n" + , "? kind + \"!\"\n" + , ": kind;\n" + , "}\n" + , "}\n" + ] + ) + ( concat + [ "class Program {\n" + , " static String Kind(_ int count) {\n" + , " String kind = count > 0 ? \"some { \" : \"none } \";\n" + , " return count > 9\n" + , " ? kind + \"!\"\n" + , " : kind;\n" + , " }\n" + , "}\n" + ] + ) + +formatsConsoleFormats :: Bool +formatsConsoleFormats = + formats + defaultFormatOptions + ( concat + [ "class Program {\n" + , "static void Main() {\n" + , "Console.Printfn(\"{%05d} %'d %-8s| %.2f %%\", 42, 1234567, \"ab\", 3.14159);\n" + , "String made = Console.Format(\"%*d}\", 5, 42);\n" + , "Console.Errorln(made + \"{\");\n" + , "}\n" + , "}\n" + ] + ) + ( concat + [ "class Program {\n" + , " static void Main() {\n" + , " Console.Printfn(\"{%05d} %'d %-8s| %.2f %%\", 42, 1234567, \"ab\", 3.14159);\n" + , " String made = Console.Format(\"%*d}\", 5, 42);\n" + , " Console.Errorln(made + \"{\");\n" + , " }\n" + , "}\n" + ] + ) + +currentFormsAreIdempotent :: Bool +currentFormsAreIdempotent = + case formatSource defaultFormatOptions (CompilerInput "Program.vxs" source) of + Right first -> case formatSource defaultFormatOptions (CompilerInput "Program.vxs" (formattedSource first)) of + Right second -> formattingChanged first && not (formattingChanged second) + Left _ -> False + Left _ -> False + where + source = + concat + [ "namespace Demo;\n" + , "enum Color { Red, Green }\n" + , "public class Program {\n" + , "public static int Log(_ int v) { Console.Println(v); return v; }\n" + , "public static void Main() {\n" + , "auto held = Demo.Program::Log;\n" + , "String kind = held(1) > 0 ? \"some\" : \"none\";\n" + , "Color chosen = match (held(2)) {\n" + , "0 -> Demo.Color.Red,\n" + , "_ -> .Green\n" + , "};\n" + , "Console.Printfn(\"%s %b\", kind, chosen == Color.Green);\n" + , "}\n" + , "}\n" + ] diff --git a/Formatter/visual-formatter.cabal b/Formatter/visual-formatter.cabal index fe99c70a..f2900008 100644 --- a/Formatter/visual-formatter.cabal +++ b/Formatter/visual-formatter.cabal @@ -2,7 +2,7 @@ cabal-version: 3.12 -- SPDX-FileCopyrightText: 2026 Progmasoft -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 name: visual-formatter -version: 0.1.0 +version: 0.1.1 synopsis: Syntax-aware source formatter for Visual X# description: Provides safe source layout normalization and the vfmt command-line tool @@ -25,7 +25,7 @@ library base >=4.20 && <4.23, bytestring >=0.12 && <0.13, text >=2.1 && <2.3, - visual-xsharp-compiler >=0.4.0 && <0.5 + visual-xsharp-compiler >=0.5.0 && <0.6 default-language: GHC2024 executable vfmt diff --git a/Grammar/visual-xsharp.ebnf b/Grammar/visual-xsharp.ebnf index 88f68278..80c6c619 100644 --- a/Grammar/visual-xsharp.ebnf +++ b/Grammar/visual-xsharp.ebnf @@ -231,8 +231,8 @@ match-expression = "match", match-targets, "{", { match-arm }, "}" ; match-targets = "(", expression, ")", { ",", "(", expression, ")" } ; match-arm = match-patterns, [ "if", expression ], "->", ( block | expression ), [ "," ] ; match-patterns = match-pattern, { ",", match-pattern } ; -match-pattern = [ "(" ], ( literal | "null" | "_" | ".", identifier - | type, ( identifier | "_" ) ), [ ")" ] ; +match-pattern = [ "(" ], ( literal | "-", number-literal | "null" | "_" + | ".", identifier | type, ( identifier | "_" ) ), [ ")" ] ; while-statement = "while", "(", condition, ")", block ; do-while-statement = "do", block, "while", "(", expression, ")", ";" ; diff --git a/Interactive/BUILD.bazel b/Interactive/BUILD.bazel index d50878e4..82e6c80d 100644 --- a/Interactive/BUILD.bazel +++ b/Interactive/BUILD.bazel @@ -1,3 +1,4 @@ +load("//Compiler/Runtime:host.bzl", "RUNTIME_HOST_DEPS", "RUNTIME_HOST_LINKOPTS") load("@rules_cc//cc:cc_binary.bzl", "cc_binary") load("@rules_cc//cc:cc_library.bzl", "cc_library") @@ -7,6 +8,9 @@ cc_library( name = "interactive_runtime", srcs = ["Runtime/History.cpp", "Runtime/ReplInput.cpp", "Runtime/Session.cpp", "Runtime/Source.cpp", "Runtime/Value.cpp"], hdrs = ["Runtime/Source.hpp", "Runtime/Value.hpp"], + # A cell is compiled and run in this process, which therefore holds the + # runtime its code calls. + linkopts = RUNTIME_HOST_LINKOPTS, deps = [ "//Compiler/Backend/LLVM:llvm_backend", "//Compiler/Cli/Commands:frontend", @@ -15,7 +19,7 @@ cc_library( "//Interactive/Headers/Visual/XSharp/Interactive:repl_input", "//Compiler/Headers/Visual/XSharp/Core:scalar", "@fmt//:fmt", - ], + ] + RUNTIME_HOST_DEPS, ) cc_library( @@ -30,5 +34,10 @@ cc_library( cc_binary( name = "vxsi", srcs = ["Main.cpp"], - deps = [":interactive_arguments", ":interactive_runtime", "@fmt//:fmt"], + deps = [ + ":interactive_arguments", + ":interactive_runtime", + "//Compiler/Support:compiler_stack", + "@fmt//:fmt", + ], ) diff --git a/Interactive/Main.cpp b/Interactive/Main.cpp index 95e62036..1dd4cd4d 100644 --- a/Interactive/Main.cpp +++ b/Interactive/Main.cpp @@ -9,6 +9,7 @@ #include "Visual/XSharp/Interactive/Arguments.hpp" #include "Visual/XSharp/Interactive/ReplInput.hpp" #include "Visual/XSharp/Interactive/Session.hpp" +#include "Visual/XSharp/Support/CompilerStack.hpp" namespace { @@ -104,34 +105,52 @@ namespace } } // namespace +namespace +{ + auto + RunMain(int argc, char **argv) -> int; +} // namespace + auto main(int argc, char **argv) -> int { - using namespace Visual::XSharp::Interactive; - const auto request = ParseArguments(argc, argv); - switch (request.kind) + // Cells are compiled on the compiler stack, as sources are in `vxs`. + return Visual::XSharp::Support::RunOnCompilerStack([argc, argv] { + return RunMain(argc, argv); + }); +} + +namespace +{ + auto + RunMain(int argc, char **argv) -> int { - case RequestKind::Repl: - return Repl(); - case RequestKind::Help: - PrintHelp(); - return 0; - case RequestKind::Error: - fmt::print(stderr, "vxsi: {}\n", request.diagnostic); - return 2; - case RequestKind::Evaluate: + using namespace Visual::XSharp::Interactive; + const auto request = ParseArguments(argc, argv); + switch (request.kind) { - Session session; - const auto result = session.Evaluate(request.expression); - if (result.status != CellStatus::Value - && result.status != CellStatus::Void) + case RequestKind::Repl: + return Repl(); + case RequestKind::Help: + PrintHelp(); + return 0; + case RequestKind::Error: + fmt::print(stderr, "vxsi: {}\n", request.diagnostic); + return 2; + case RequestKind::Evaluate: { - fmt::print(stderr, "vxsi: {}\n", result.text); - return 1; + Session session; + const auto result = session.Evaluate(request.expression); + if (result.status != CellStatus::Value + && result.status != CellStatus::Void) + { + fmt::print(stderr, "vxsi: {}\n", result.text); + return 1; + } + fmt::print("{}\n", result.text); + return 0; } - fmt::print("{}\n", result.text); - return 0; } + return 2; } - return 2; -} +} // namespace diff --git a/Linter/README.md b/Linter/README.md index 444a70c1..1c7feb0f 100644 --- a/Linter/README.md +++ b/Linter/README.md @@ -13,6 +13,12 @@ vlint -List-Checks vlint -Help ``` +The linter reports what the compiler it is built with reports. Version 0.1.1 is built with the compiler of +Visual X# 0.5.0: a source may use method references (`Type::Method`), names written through their namespace, +conditionals over strings and console output, and their diagnostics arrive under the `compiler` check, for example +`compiler.VXT0080` for a method reference that does not select one overload and `compiler.VXT0076` for a format the +compiler rejects. + `-Fix` applies only safe physical-source fixes and never rewrites compiler or semantic diagnostics. `-List-Checks` prints the stable rule identifiers currently implemented by the binary. diff --git a/Linter/build.gradle.kts b/Linter/build.gradle.kts index 4e3fd99a..79b97d58 100644 --- a/Linter/build.gradle.kts +++ b/Linter/build.gradle.kts @@ -13,7 +13,7 @@ plugins { } group = "com.progmasoft.visual.linter" -version = "0.1.0" +version = "0.1.1" repositories { mavenCentral() } diff --git a/Linter/sources/test/haskell/Main.hs b/Linter/sources/test/haskell/Main.hs index 58c9e39b..25c610a9 100644 --- a/Linter/sources/test/haskell/Main.hs +++ b/Linter/sources/test/haskell/Main.hs @@ -16,6 +16,12 @@ main = do check "safe fixes reach an idempotent result" safeFixesAreIdempotent check "malformed source remains a compiler diagnostic after a failed fix" malformedSourceRemainsVisible check "check catalog exposes stable command output" checkCatalog + check "a source with the forms of 0.5.0 is clean" currentFormsAreClean + check "a selector with a dot and no call is a compiler diagnostic" dottedSelectorIsReported + check "a method reference the place does not select is a compiler diagnostic" ambiguousReferenceIsReported + check "a method reference through a value is a compiler diagnostic" boundReferenceIsReported + check "a format that does not match its arguments is a compiler diagnostic" formatMismatchIsReported + check "safe fixes keep a method reference as written" safeFixesKeepReferences check :: String -> Bool -> IO () check label passed = if passed then putStrLn ("PASS: " ++ label) else putStrLn ("FAIL: " ++ label) >> exitFailure @@ -64,3 +70,60 @@ checkCatalog = , "format.mixedLineEndings" , "format.missingFinalNewline" ] + +-- The body of @Main@ in a program that declares what the cases below use. +currentProgram :: String -> String +currentProgram body = + unlines + [ "namespace Demo;" + , "enum Color { Red, Green }" + , "public class Other {" + , " public static int Pick(_ int v) { return v + 10; }" + , " public static int Pick(_ int v, _ int w) { return v + w; }" + , "}" + , "public class Program {" + , " public static int Log(_ int v) { Console.Println(v); return v; }" + , " public static int Run(_ (int) -> int f) { return f(3); }" + , " public static void Main() {" + , " " ++ body + , " }" + , "}" + ] + +ruleIdsOf :: String -> [String] +ruleIdsOf body = map lintRuleId (lintSource (CompilerInput "Program.vxs" (currentProgram body))) + +currentFormsAreClean :: Bool +currentFormsAreClean = + null + ( ruleIdsOf + ( concat + [ "auto held = Demo.Program::Log; " + , "int first = Run(Program::Log) + Run(Other::Pick) + Demo.Program.Log(1); " + , "String kind = held(first) > 0 ? \"some\" : \"none\"; " + , "Color chosen = Demo.Color.Green; " + , "Console.Printfn(\"%s %b %05d\", kind, chosen == Color.Green, first);" + ] + ) + ) + +dottedSelectorIsReported :: Bool +dottedSelectorIsReported = "compiler.VXT0034" `elem` ruleIdsOf "auto held = Program.Log;" + +ambiguousReferenceIsReported :: Bool +ambiguousReferenceIsReported = "compiler.VXT0080" `elem` ruleIdsOf "auto held = Other::Pick;" + +boundReferenceIsReported :: Bool +boundReferenceIsReported = "compiler.VXT0081" `elem` ruleIdsOf "int value = 1; auto held = value::Pick;" + +formatMismatchIsReported :: Bool +formatMismatchIsReported = any (`elem` ruleIdsOf "Console.Printf(\"%d\", \"text\");") ["compiler.VXT0077", "compiler.VXT0078"] + +safeFixesKeepReferences :: Bool +safeFixesKeepReferences = + case applySafeFixes (CompilerInput "Program.vxs" (currentProgram "auto held = Demo.Program::Log; ")) of + Right result -> + formattingChanged result + && lines (formattedSource result) !! 10 == " auto held = Demo.Program::Log;" + && null (lintSource (CompilerInput "Program.vxs" (formattedSource result))) + Left _ -> False diff --git a/Linter/visual-linter.cabal b/Linter/visual-linter.cabal index ca7bd8ff..66946fd8 100644 --- a/Linter/visual-linter.cabal +++ b/Linter/visual-linter.cabal @@ -2,7 +2,7 @@ cabal-version: 3.12 -- SPDX-FileCopyrightText: 2026 Progmasoft -- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 name: visual-linter -version: 0.1.0 +version: 0.1.1 synopsis: Compiler-backed static analysis for Visual X# description: Provides compiler diagnostics, source hygiene checks, safe fixes, and the @@ -20,7 +20,7 @@ library build-depends: base >=4.20 && <4.23, visual-formatter >=0.1 && <0.2, - visual-xsharp-compiler >=0.4.0 && <0.5 + visual-xsharp-compiler >=0.5.0 && <0.6 default-language: GHC2024 executable vlint diff --git a/MODULE.bazel b/MODULE.bazel index 50992e54..d42b3561 100644 --- a/MODULE.bazel +++ b/MODULE.bazel @@ -1,6 +1,6 @@ module( name = "visual_xsharp", - version = "0.4.1", + version = "0.5.0", ) bazel_dep(name = "rules_cc", version = "0.2.22") diff --git a/MODULE.bazel.lock b/MODULE.bazel.lock index de8a1a46..2845a4ce 100644 --- a/MODULE.bazel.lock +++ b/MODULE.bazel.lock @@ -179,7 +179,7 @@ "//Compiler/Build/Bazel:llvm.bzl%llvm": { "general": { "bzlTransitiveDigest": "d5n0dYsk+AgFJkyvMPDSSOLLwOjjLGJVSC3ypF4EO2w=", - "usagesDigest": "5QIuEN/lYLjf042ZQO/KdKKWvGnTPsAGBn1+/SzUhdY=", + "usagesDigest": "yQ8pWIvt5KAf7Xsvm+irG1mf9Qylc76nx5kN0mOvNmw=", "recordedInputs": [], "generatedRepoSpecs": { "llvm": { diff --git a/ProjectSystem/Visual.XSharp.kts b/ProjectSystem/Visual.XSharp.kts index 1a90587e..21d7e5d5 100644 --- a/ProjectSystem/Visual.XSharp.kts +++ b/ProjectSystem/Visual.XSharp.kts @@ -13,7 +13,7 @@ project { } compiler { - version = "0.4.1" + version = "0.5.0" standard = "26" backend = Backend.LLVM buildMode = BuildMode.RELEASE diff --git a/ProjectSystem/build.gradle.kts b/ProjectSystem/build.gradle.kts index c3612aa1..3c507687 100644 --- a/ProjectSystem/build.gradle.kts +++ b/ProjectSystem/build.gradle.kts @@ -15,7 +15,7 @@ plugins { } group = "com.progmasoft.visual.xsharp" -version = "0.4.1" +version = "0.5.0" repositories { mavenCentral() diff --git a/README.md b/README.md index 41a4ff92..85b11b29 100644 --- a/README.md +++ b/README.md @@ -82,7 +82,7 @@ the normal workflow: ```powershell go run ./helpers/cmd/develop doctor -go run ./helpers/cmd/develop version 0.4.1 +go run ./helpers/cmd/develop version 0.5.0 go run ./helpers/cmd/develop build go run ./helpers/cmd/develop test go run ./helpers/cmd/develop benchmark diff --git a/Spec/Language/Decls.vxs b/Spec/Language/Decls.vxs index d4aeac3e..8df7d9f6 100644 --- a/Spec/Language/Decls.vxs +++ b/Spec/Language/Decls.vxs @@ -369,6 +369,37 @@ enum Value = int { FOURTH, -- 11 } +-- ============================================================================ +-- Example 316: 7.2 Member Values +-- Demonstrates: The value of a member is a constant integer expression: integer literals and earlier members +-- of the same enum, joined by the arithmetic, shift and bitwise operators. Inside the declaration of its +-- enum the name of a member stands for its number; this is where the numbers behind the names are defined. +-- The expression is computed in the underlying type, and a member without a value follows the computed one. + +enum Access = ubyte { + READ = 1, + WRITE = READ << 1, -- 2 + EXECUTE = WRITE * 2, -- 4 + ALL = READ | WRITE | EXECUTE, -- 7 + NEXT, -- 8 + REST = !(EXECUTE + 1), -- 250: the complement of 5 in ubyte +} + +-- ============================================================================ +-- Example 317: 7.2 Member Values +-- Demonstrates: A member value names earlier members only, of its own enum only, and is nothing but integer +-- arithmetic. Outside the declaration a member is a value of the enum and never a number. +-- Expected result: compile-time rejection; each INVALID marker is normative. + +enum Order { + FIRST = SECOND, -- INVALID: SECOND is declared later + SECOND = SECOND + 1, -- INVALID: a member does not name itself + THIRD = Access.READ, -- INVALID: a member of another enum is not a number here + FOURTH = Compute(), -- INVALID: a call is not a constant + FIFTH = 4 / 0, -- INVALID: a constant division by zero + SIXTH = 1 < 2, -- INVALID: a comparison is not an integer +} + -- ============================================================================ -- Example 41: 7.3 Duplicate Numeric Values -- Demonstrates: Different enum members may have the same numeric value. @@ -397,6 +428,50 @@ match (status) { .READY -> { ... }, } +-- ============================================================================ +-- Example 311: 7.4 Operations on Enum Values +-- Demonstrates: The values of a classic enum are compared with == and \=, with values of the same enum. +-- Two members with the same numeric value are equal. + +Status.READY == status -- bool +status \= Status.NONE -- bool + +-- ============================================================================ +-- Example 312: 7.4 Operations on Enum Values +-- Demonstrates: A classic enum has no order and no arithmetic. Its members are names for values, and the +-- numeric value behind a name is not a quantity. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +Status.NONE < Status.READY -- INVALID: enum values are not ordered +Status.NONE + Status.READY -- INVALID: enum values are not added +Status.NONE == Level.LOW -- INVALID: values of different enums + +-- ============================================================================ +-- Example 313: 7.5 Enums and Integers +-- Demonstrates: A classic enum is a type of its own. There is no implicit conversion between an enum and an +-- integer in either direction, whatever its underlying type is. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +int code = Status.READY; -- INVALID: an enum value is not an integer +Status status = 1; -- INVALID: an integer is not an enum value +Status.NONE == 0 -- INVALID: an enum value is not compared with an integer + +-- ============================================================================ +-- Example 314: 7.6 Access to Members +-- Demonstrates: The target-typed .Member spelling is the same for a classic enum as for an enum class: it +-- names a member of the enum that the context expects, in a match pattern and in an expression whose +-- target type is known. + +Status first = Status.READY; +Status second = .READY; + +-- ============================================================================ +-- Example 315: 7.6 Access to Members +-- Demonstrates: Target-typed .Member syntax requires a known target type, for a classic enum as well. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +auto status = .READY; -- INVALID + -- ============================================================================ -- Example 44: 8. enum class -- Demonstrates: An enum class is an AARC reference type whose cases are singleton instances. @@ -1860,6 +1935,142 @@ else { 2 }; +-- ============================================================================ +-- Example 295: 31. if and guard Bindings +-- Demonstrates: A guard may also test a plain condition. The else block runs when the condition is false. + +guard (count > 0) else { + return 0; +} + +Use(count); + +-- ============================================================================ +-- Example 296: 31. if and guard Bindings +-- Demonstrates: The else block of a guard must leave the enclosing scope on every path, so the statements +-- after the guard run only when the condition held. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +guard (ready) else { + Log("not ready"); -- INVALID: the block can complete normally +} + +guard (ready) else { + if (strict) { + return; + } + -- INVALID: when strict is false the block completes normally +} + +-- ============================================================================ +-- Example 297: 31. if and guard Bindings +-- Demonstrates: Whether the else block leaves is decided by its control flow, not by the spelling of its last +-- statement. return leaves the method; break and continue leave or continue the enclosing loop and are valid +-- only where they have one; a loop that cannot end, a match all of whose arms leave and one of which always +-- matches, and a call that is known never to return also do not complete. +-- Context note: Fail is known never to return. + +while (HasNext()) { + auto item = Next(); + + guard (item.valid) else { + continue; + } + + guard (item.size < limit) else { + break; + } + + guard (item.kind \= 0) else { + if (strict) { + return; + } + else { + Fail("unknown kind"); + } + } + + Use(item); +} + +-- ============================================================================ +-- Example 298: 31.1 if Expressions +-- Demonstrates: A block of an if expression may leave instead of producing a value. return leaves the +-- enclosing method and break and continue target the enclosing loop, under the rules those statements have +-- anywhere else. A block that leaves produces no value; the expression has the type of the block that +-- completes. + +auto size = if (item.valid) { + item.size +} +else { + return 0; +}; + +while (HasNext()) { + total += if (Next().skip) { + continue; + } + else { + 1 + }; +} + +-- ============================================================================ +-- Example 310: 31.1 if Expressions +-- Demonstrates: An if or match expression none of whose branches completes normally is valid. It never +-- produces a value, so whatever would have received the value is never evaluated and the statements after +-- it are not reached. Each return is still checked against the return type of the method, and break and +-- continue still need a loop. + +int Classify(bool known, int code) { + int result = if (known) { + return code; + } + else { + return 0; + }; + + return result; -- never reached +} + +-- ============================================================================ +-- Example 299: 31.1 if Expressions +-- Demonstrates: A block that can complete normally must end with its value. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +auto value = if (condition) { + Log("first"); -- INVALID: the block completes without a value +} +else { + 2 +}; + +auto other = if (condition) { + break; -- INVALID outside a loop: break needs a loop to leave +} +else { + 2 +}; + +-- ============================================================================ +-- Example 300: 31.1 if Expressions +-- Demonstrates: Where a block ends with an expression that is its value, an if with an else or a match in +-- that position is that value; no parentheses are needed. This applies to blocks used as values. It does +-- not give a method body an implicit return: the return rules of methods are unchanged. + +auto sign = if (value < 0) { + -1 +} +else { + if (value > 0) { + 1 + } + else { + 0 + } +}; + -- ============================================================================ -- Example 194: 32. match -- Demonstrates: Single-target form: @@ -1894,6 +2105,165 @@ match (a), (b) { }, } +-- ============================================================================ +-- Example 301: 32. match +-- Demonstrates: When no arm of a match statement matches, execution continues with the statement after it, +-- like an if without an else. + +match (number) { + 1 -> { + HandleOne(); + }, +} + +Continue(); -- reached for every number, also when it is not 1 + +-- ============================================================================ +-- Example 302: 32. match +-- Demonstrates: A match used as an expression must be exhaustive: some arm must match whatever the targets +-- are. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +auto text = match (number) { -- INVALID: no arm for other numbers + 1 -> "one", + 2 -> "two", +}; + +-- ============================================================================ +-- Example 303: 32. match +-- Demonstrates: Exhaustiveness is established by a wildcard, by a type pattern that matches every possible +-- value of its target, or by arms that cover both values of a bool target. An arm with a guard never +-- counts, because its guard may be false. +-- Expected result: the last match is rejected; the INVALID marker is normative. + +auto a = match (number) { + 1 -> 10, + _ -> 0, +}; + +auto b = match (number) { + int n -> n * 2, -- every value of number is an int +}; + +auto c = match (flag) { + true -> 1, + false -> 0, +}; + +auto d = match (number) { -- INVALID: the only catch-all arm has a guard + _ if enabled -> 1, +}; + +-- ============================================================================ +-- Example 304: 32. match +-- Demonstrates: A match over an enum is exhaustive when its arms cover every member; no wildcard is needed and +-- the enum does not have to be marked [Exhaustive]. When the target may be null, null must be covered as +-- well. +-- Context note: Status has exactly the members NONE and READY. +-- Expected result: the last match is rejected; the INVALID marker is normative. + +auto a = match (status) { + .NONE -> 0, + .READY -> 1, +}; + +auto b = match (maybeStatus) { + .NONE -> 0, + .READY -> 1, + null -> 2, +}; + +auto c = match (maybeStatus) { -- INVALID: null is not covered + .NONE -> 0, + .READY -> 1, +}; + +-- ============================================================================ +-- Example 305: 32. match +-- Demonstrates: The comma after an arm is optional. An expression body is read like every expression, so a +-- parenthesized pattern that follows it without a comma is read as the argument list of a call of the body. +-- The compiler reports that reading and asks for the comma; it does not require commas elsewhere. +-- Expected result: the second match is rejected; the INVALID marker is normative. + +auto a = match (number) { + 0 -> 10 + 1 -> 20 + _ -> 30 +}; + +auto b = match (number) { + 0 -> Twice + (1) -> 20 -- INVALID: read as Twice(1) -> 20; write `0 -> Twice,` + _ -> 30 +}; + +-- ============================================================================ +-- Example 306: 32. match +-- Demonstrates: The block of an arm of a match expression may leave instead of producing a value, like a +-- block of an if expression. The match has the type of the arms that complete. + +auto size = match (kind) { + 0 -> { + return 0; + }, + int other -> other * 2, +}; + +-- ============================================================================ +-- Example 307: 32.1 Pattern Forms +-- Demonstrates: A numeric constant pattern may be negative: a minus sign directly before a numeric literal is +-- part of the constant. The constant must be a value of the type of its target, so the most negative value +-- of a signed type is a valid pattern. +-- Context note: small is a byte and code is a ubyte. + +match (number) { + -1 -> { ... }, + 0 -> { ... }, +} + +match (small) { + -128 -> { ... }, -- the least byte + 127 -> { ... }, +} + +-- ============================================================================ +-- Example 308: 32.1 Pattern Forms +-- Demonstrates: The minus sign belongs to a numeric literal only. It is not an operator, and no other +-- expression is a pattern. +-- Context note: small is a byte and code is a ubyte. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +match (number) { + -limit -> { ... }, -- INVALID: a name is not a numeric literal + 1 + 1 -> { ... }, -- INVALID: an expression is not a pattern + -(1) -> { ... }, -- INVALID +} + +match (small) { + -129 -> { ... }, -- INVALID: not a value of byte + 128 -> { ... }, -- INVALID: not a value of byte +} + +match (code) { + -1 -> { ... }, -- INVALID: not a value of ubyte +} + +-- ============================================================================ +-- Example 309: 32.1 Pattern Forms +-- Demonstrates: A type pattern applies no implicit numeric conversion: over a scalar target it matches only +-- the type of the target itself. This does not restrict type patterns over reference types, which match +-- the dynamic type of the target and its subtypes. +-- Context note: number is an int. +-- Expected result: the second match is rejected; the INVALID marker is normative. + +match (number) { + int n -> { Use(n); }, +} + +match (number) { + long n -> { Use(n); }, -- INVALID: an int is not matched as a long +} + -- ============================================================================ -- Example 196: 32.1 Pattern Forms -- Demonstrates: Constant pattern: diff --git a/Spec/Language/Evaluation.vxs b/Spec/Language/Evaluation.vxs new file mode 100644 index 00000000..2cd2f1d1 --- /dev/null +++ b/Spec/Language/Evaluation.vxs @@ -0,0 +1,186 @@ +-- SPDX-FileCopyrightText: 2026 Progmasoft +-- SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +-- Visual X# language design examples: Evaluation +-- +-- This file is a topic-oriented language design example suite. +-- Each fragment states its intent explicitly; fragments marked INVALID are expected to be rejected. +-- Examples record intended behavior; they do not claim complete compiler support. + +-- ============================================================================ +-- Example 1: 1. Evaluation by Need +-- Demonstrates: Visual X# is a lazy language. A value is computed when it is first needed, not where it is +-- written. Binding a name to an expression computes nothing. + +int total = Sum(values); -- nothing is computed here + +if (verbose) { + Print(total); -- Sum(values) is computed here, if this is reached +} + +-- ============================================================================ +-- Example 2: 1. Evaluation by Need +-- Demonstrates: A value that is never needed is never computed. Whatever computing it would have done does +-- not happen: it does not fail, and it does not run without end. + +int quotient = left / right; -- right may be zero +int endless = Forever(); -- Forever never returns + +return 5; -- neither value is needed: the result is 5 + +-- ============================================================================ +-- Example 3: 1. Evaluation by Need +-- Demonstrates: Evaluation is call by need: a value is computed at most once. Every later use reads the value +-- that the first use computed. + +int cost = Expensive(input); + +return cost + cost + cost; -- Expensive(input) is computed once + +-- ============================================================================ +-- Example 4: 1. Evaluation by Need +-- Demonstrates: A value is needed where the program cannot go on without it: in a condition, as the subject of +-- a match, as an operand of an arithmetic or comparison operator whose result is needed, and where it leaves +-- the program. A value computed from another needs that other only when it is itself needed. + +int quotient = left / right; +int next = quotient + 1; -- needs quotient only if next is needed + +return right > 0 ? next : 0; -- when right is zero, nothing divides + +-- ============================================================================ +-- Example 12: 1. Evaluation by Need +-- Demonstrates: An expression written as a statement of its own is evaluated, and so is a value assigned to +-- the discard. A statement exists to be carried out; nothing else could ever need its value. + +Validate(input); -- evaluated: the statement is the need +_ = left / right; -- evaluated: if right is zero, the failure is here + +-- ============================================================================ +-- Example 5: 1. Evaluation by Need +-- Demonstrates: An argument is passed by need. The method computes it if it needs it and at most once, however +-- many times its parameter is read. + +static int Choose(_ bool first, _ int one, _ int other) { + return first ? one : other; +} + +Choose(true, 1, left / right); -- right may be zero: the third argument is not needed + +-- ============================================================================ +-- Example 6: 2. Failure +-- Demonstrates: A value that cannot be computed fails where it is needed, not where it is written. A program +-- that needs such a value fails; a program that does not need it does not. + +int quotient = left / right; -- no failure here, whatever right is + +Use(quotient); -- if right is zero, the failure is here + +-- ============================================================================ +-- Example 13: 2. Failure +-- Demonstrates: Whether a program fails is determined: it fails when it needs a value that cannot be computed. +-- Which failure it meets is not, when one statement needs several such values. The values a statement needs +-- have no order among themselves that a program may rely on, and a value that runs without end is one that +-- cannot be computed: a statement that needs it and a value that fails may fail or may never end. +-- Statements are still carried out in the order they are written, and so are effects. + +int first = left / zero; -- cannot be computed +int second = Parse(text); -- cannot be computed either, for this text + +return first + second; -- the program fails; which of the two failures it reports is not determined + +Use(first); -- as two statements the order is determined: +Use(second); -- the failure is that of first, and this statement is never reached + +-- ============================================================================ +-- Example 14: 2. Failure +-- Demonstrates: Some values cannot be computed at all. An integer quotient or remainder by zero has no value, +-- with /, with // and with %. A shift has no value when its amount is negative or is not less than the width +-- of the shifted value. A program that needs such a value stops there; it does not continue with some other +-- value in its place, and nothing after that point happens. + +int quotient = total / count; -- no value when count is zero +int rest = total % count; -- no value when count is zero +int bit = 1 << position; -- no value when position is negative, or 64 or more for an int + +Use(quotient); -- when count is zero the program stops here + +-- ============================================================================ +-- Example 7: 3. Suspended Computations +-- Demonstrates: A value that has not been computed yet is a suspended computation, a thunk. Thunks are not +-- part of the source language: there is no syntax that creates, tests or forces one, and the type of a name +-- is the type of its value, computed or not. + +int total = Sum(values); -- total is an int, before and after it is computed + +-- ============================================================================ +-- Example 8: 4. Effects +-- Demonstrates: Effects are not lazy. A store into a variable, and anything else that changes what the rest of +-- the program observes, happens where it is written and in the order in which it is written, whether or not +-- the value of the expression that contains it is ever needed. + +int count = 0; +int unused = (count += 1) + Expensive(input); -- the store happens here + +return count; -- 1 + +-- ============================================================================ +-- Example 9: 4. Effects +-- Demonstrates: The language has an effect system, and it is not visible in the source: no annotation says +-- that an expression has an effect, and none is needed to make one happen. The compiler tells an expression +-- that only computes a value, which is evaluated by need, from one that has an effect, which is evaluated in +-- place. + +int quotient = left / right; -- computes a value: evaluated by need +count += 1; -- has an effect: happens here + +-- ============================================================================ +-- Example 15: 4. Effects +-- Demonstrates: Output is an effect. An expression that writes to the console, itself or through a method it +-- calls, is evaluated where it is written and in the order it is written, whether or not its value is +-- ever needed. A value that only computes is still computed by need beside it. + +static int Log(_ int value) { + Console.Println(value); + return value; +} + +int unused = Log(1); -- writes 1 here +int skipped = left / zero; -- computes nothing: the value is never needed + +Console.Println(2); -- 1 has already been written + +-- ============================================================================ +-- Example 16: 4. Effects +-- Demonstrates: An argument that has an effect is evaluated at the call, in the order the arguments are +-- written, whether or not the method reads its parameter. + +Choose(true, Log(1), Log(2)); -- writes 1 and then 2; the method reads only its second parameter + +-- ============================================================================ +-- Example 10: 4. Effects +-- Demonstrates: A value means what its variables held where it was written, however late it is computed. A +-- later store into a variable does not change a value that was bound before the store. + +int width = 8; +int area = width * Height(); -- the value of width here is 8 + +width = 100; + +return area; -- 8 * Height(), not 100 * Height() + +-- ============================================================================ +-- Example 11: 5. Strictness in BLINQ +-- Demonstrates: .Strict and .AllStrict of BLINQ make a query eager: the first the chain above it, the second +-- the whole query expression. They belong to BLINQ. They do not make any other part of a program strict, and +-- the language has no construct that does. +-- Context note: values is a sequence; the BLINQ rules are in StandardLibrary/Query/BLINQ.vxs. + +auto ready = + values + .Filter { + $0 > 10 + } + .Strict; -- the chain above is eager + +int quotient = left / right; -- still evaluated by need diff --git a/Spec/Language/Iteration.vxs b/Spec/Language/Iteration.vxs index 301b91de..58b641e9 100644 --- a/Spec/Language/Iteration.vxs +++ b/Spec/Language/Iteration.vxs @@ -453,6 +453,60 @@ for (auto value : CountTo(5)) { Use(value); } +-- ============================================================================ +-- Example 79: 14. break and continue +-- Demonstrates: The condition of a loop belongs to that loop. A break in a block used as a value in +-- the condition leaves the loop whose condition it is, not a loop around it. What the condition +-- evaluated before the break has happened, once; the body does not run again. + +while (Outer()) { + while (if (Done()) { break; } else { true }) { + Step(); + } + + AfterInner(); -- reached: the break left the inner loop only +} + +-- ============================================================================ +-- Example 80: 14. break and continue +-- Demonstrates: A continue in the condition of a loop abandons the rest of that evaluation and +-- evaluates the condition of the same loop again. The body does not run in between, and in a +-- for loop neither does the update. The same holds for the condition of a do/while loop. +-- A condition that continues on every evaluation never ends: that is an endless loop, as written. + +for (int index = 0; if (Busy()) { continue; } else { index < limit }; index += 1) { + Use(index); -- index advances only after a completed condition and body +} + +-- ============================================================================ +-- Example 81: 14. break and continue +-- Demonstrates: A continue in the update of a for loop skips the rest of the update and goes on to +-- the condition. It does not start the update again, and what the update did before it stays done. + +for (int index = 0; index < limit; index += 1, total += if (Odd(index)) { continue; } else { index }) { + Use(index); +} + +-- ============================================================================ +-- Example 82: 14. break and continue +-- Demonstrates: A break in the update of a for loop leaves that loop. The rest of the update and +-- the condition are not evaluated. In a loop used as an expression it carries the value of the +-- loop, like any other break of that loop. + +for (int index = 0; index < limit; index += if (Full()) { break; } else { 1 }) { + Use(index); +} + +-- ============================================================================ +-- Example 83: 14. break and continue +-- Demonstrates: A callable is not inside the loops around the place that creates it. break and +-- continue in its body, also in a block used as a value there, do not reach such a loop. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +while (Ready()) { + auto stop = \() -> { break; }; -- INVALID: no loop encloses the body of the callable +} + -- ============================================================================ -- Example 44: 15. Value-Carrying break -- Demonstrates: Visual X# supports: diff --git a/Spec/Language/Operators.vxs b/Spec/Language/Operators.vxs index 09cb6740..eb186f65 100644 --- a/Spec/Language/Operators.vxs +++ b/Spec/Language/Operators.vxs @@ -103,6 +103,36 @@ a ? b : c ? d : e a ? b : (c ? d : e) +-- ============================================================================ +-- Example 87: 4. Addition and String Concatenation + +-- Demonstrates: + joins two Strings. When one operand is a String and the other is not, the other is written +-- as text first and the result is a String: an integer in decimal, a bool as true or false, a char as the +-- character itself. + associates to the left. + +"Hello from " + language + "!" -- one String +"Age: " + age -- Age: 36 +"" + true -- true +"a" + 1 + 2 -- a12 +1 + 2 + "a" -- 3a + +-- ============================================================================ +-- Example 88: 4. Addition and String Concatenation + +-- Demonstrates: += appends to a String variable, with the same rule for a value that is not a String. + +String line = "x"; + +line += "y"; +line += 3; -- xy3 + +-- ============================================================================ +-- Example 89: 4. Addition and String Concatenation + +-- Demonstrates: A String has no other arithmetic operator and no ordering operator. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +"a" - "b" -- INVALID +"a" * 2 -- INVALID +"a" < "b" -- INVALID + -- ============================================================================ -- Example 17: 5. Rounded Integer Division // -- Demonstrates: Its result is always an integer. @@ -235,6 +265,18 @@ a < b && b < c -- VALID a == b == c -- INVALID +-- ============================================================================ +-- Example 86: 12. Equality Operators +-- Demonstrates: Two Strings are equal when they hold the same characters in the same order. Where each is +-- kept does not matter: a String is compared by what it holds. + +String first = "ab"; +String second = "a" + "b"; + +first == second -- true +first \= second -- false +"a" == "ab" -- false + -- ============================================================================ -- Example 33: 14. Null Coalescing ?? -- Demonstrates: ?? evaluates and returns its right operand only when the left operand is null. diff --git a/Spec/README.md b/Spec/README.md index 9ec7c56d..4967b063 100644 --- a/Spec/README.md +++ b/Spec/README.md @@ -24,6 +24,7 @@ See [the specification guide](../Documents/SPECIFICATION.md) for the topic map a ### Language - [Declarations](Language/Decls.vxs) and [attributes](Language/Attributes.vxs) +- [Evaluation](Language/Evaluation.vxs): evaluation by need and effects - [Operators](Language/Operators.vxs), [iteration](Language/Iteration.vxs), and [unsafe behavior](Language/Unsafe.vxs) - [String](Language/String.vxs), [Optional](Language/Optional.vxs), and [exceptions](Language/Exceptions.vxs) diff --git a/Spec/StandardLibrary/IO/ConsoleIO.vxs b/Spec/StandardLibrary/IO/ConsoleIO.vxs index ca47f6b9..e62ccc4e 100644 --- a/Spec/StandardLibrary/IO/ConsoleIO.vxs +++ b/Spec/StandardLibrary/IO/ConsoleIO.vxs @@ -542,3 +542,171 @@ Console.Stdin().Readf("%.2f", value); Object value; Console.Stdin().Readf("%A", value); + +-- ============================================================================ +-- Example 69: 2.1 Plain output +-- Demonstrates: Print, Println, Error and Errorln take one value. A value that is not a String is written as +-- text: an integer in decimal with a minus sign when it is negative, a bool as true or false, and a char +-- as the character itself. This is the text the + operator joins to a String. + +Console.Println(42); -- 42 +Console.Println(-7); -- -7 +Console.Println(true); -- true +Console.Println('x'); -- x + +-- ============================================================================ +-- Example 70: 2.2 Line-output forms +-- Demonstrates: Println, Printfn, Errorln and Errorfn end the line with the line terminator of the platform, +-- which is what %n writes: a carriage return and a line feed on Windows, and a line feed elsewhere. An +-- escape written in a String is the character it names on every platform. + +Console.Println("Hello"); -- Hello, and the line terminator of the platform +Console.Print("Hello\n"); -- Hello, and a line feed + +-- ============================================================================ +-- Example 71: 4.1 Conversions +-- Demonstrates: The conversions of the output format grammar. Each takes one argument, except %n and %%, +-- which take none. %A and %O take an object. + +Console.Printf("%d", count); -- a signed integer in decimal +Console.Printf("%u", size); -- an unsigned integer in decimal +Console.Printf("%x", mask); -- an integer in hexadecimal, with lowercase digits +Console.Printf("%f", ratio); -- a floating-point number with a fixed number of digits after the point +Console.Printf("%s", name); -- a String +Console.Printf("%c", initial); -- a char +Console.Printf("%b", ready); -- a bool, as true or false +Console.Printf("%n"); -- the line terminator of the platform +Console.Printf("100%%"); -- a percent sign: 100% + +-- ============================================================================ +-- Example 72: 4.1 Conversions +-- Demonstrates: A sequence that begins with % and is not a conversion is a compile-time error, and so is a +-- format that ends inside one. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +Console.Printf("%q", value); -- INVALID: no such conversion +Console.Printf("100%"); -- INVALID: the format ends inside a conversion + +-- ============================================================================ +-- Example 73: 4.1 Conversions +-- Demonstrates: The arguments of a format are the ones its conversions take, in the order of the +-- conversions, no more and no fewer. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +Console.Printf("%d and %d", first); -- INVALID: an argument is missing +Console.Printf("%d", first, second); -- INVALID: an argument is left over + +-- ============================================================================ +-- Example 74: 4.1 Conversions +-- Demonstrates: An argument has the type its conversion takes. %s takes a String, %c a char and %b a bool; +-- %x takes a signed or an unsigned integer. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +Console.Printf("%s", 42); -- INVALID: %s takes a String +Console.Printf("%c", 65); -- INVALID: %c takes a char +Console.Printf("%b", 1); -- INVALID: %b takes a bool +Console.Printf("%x", 1.5); -- INVALID: %x takes an integer + +-- ============================================================================ +-- Example 75: 4.3 Floating-point values +-- Demonstrates: %f writes six digits after the point unless a precision says otherwise. The digits are those +-- of the exact value of the number, rounded to the nearest; a value exactly between two rounds to the one +-- whose last digit is even. + +Console.Printf("%f", 12.5); -- 12.500000 +Console.Printf("%.2f", 12.5); -- 12.50 +Console.Printf("%.0f", 2.5); -- 2 +Console.Printf("%.0f", 3.5); -- 4 + +-- ============================================================================ +-- Example 76: 4.3 Floating-point values +-- Demonstrates: A value that is not a finite number is written as a word: nan, inf or -inf. + +Console.Printf("%f", ratio); -- inf when ratio is positive infinity + +-- ============================================================================ +-- Example 77: 5.1 Flags +-- Demonstrates: The flags, and the conversions that take each. - puts the value at the left of its field and +-- is taken by every conversion that has a width. 0 pads a number with zeros after its sign. + writes a +-- plus sign before a number that is not negative, and a space writes a space there. # writes 0x before a +-- hexadecimal number. ' groups the integer digits of a decimal number. + +Console.Printf("%-8d|", 42); -- 42 | +Console.Printf("%08d", -42); -- -0000042 +Console.Printf("%+d", 42); -- +42 +Console.Printf("% d", 42); -- 42 +Console.Printf("%#x", 255); -- 0xff +Console.Printf("%'d", 1234567); -- 1'234'567 + +-- ============================================================================ +-- Example 78: 5.1 Flags +-- Demonstrates: 0 is taken by %d, %u, %x and %f. + and the space are taken by %d and %f. # is taken by %x. +-- ' is taken by %d, %u and %f. A flag written twice is an error. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +Console.Printf("%+u", size); -- INVALID: an unsigned number has no sign +Console.Printf("%0c", initial); -- INVALID +Console.Printf("%#f", ratio); -- INVALID +Console.Printf("%++d", count); -- INVALID: the flag is written twice + +-- ============================================================================ +-- Example 79: 4.2 Integers +-- Demonstrates: %x writes the magnitude of the number. A negative number has a minus sign before its +-- magnitude, and the # prefix stands between the sign and the digits. + +Console.Printf("%x", 255); -- ff +Console.Printf("%x", -255); -- -ff +Console.Printf("%#x", -255); -- -0xff + +-- ============================================================================ +-- Example 80: 6. Width +-- Demonstrates: A width is the least number of characters a conversion writes; a value that needs more is +-- written whole. A width and a precision count characters, not bytes. %n and %% take neither. +-- Expected result: the last two lines are rejected at compile time; the INVALID marker is normative. + +Console.Printf("%2d", 12345); -- 12345 +Console.Printf("%5%"); -- INVALID +Console.Printf("%-n"); -- INVALID + +-- ============================================================================ +-- Example 81: 6. Width +-- Demonstrates: A width or a precision written as * is an int argument that stands before the value. The +-- arguments of a format are evaluated in the order they are written. + +Console.Printf("%*d", width, value); -- width is evaluated first, then value +Console.Printf("%*.*f", width, precision, value); -- width, then precision, then value + +-- ============================================================================ +-- Example 82: 6. Width +-- Demonstrates: The argument of a * is an int. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +uint width = 8; + +Console.Printf("%*d", width, value); -- INVALID + +-- ============================================================================ +-- Example 83: 7. Precision +-- Demonstrates: A precision is taken by %f, where it is the number of digits after the point, by %s, where +-- it is the most characters written, and by %A. Any other conversion with a precision is an error. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +Console.Printf("%.2d", count); -- INVALID +Console.Printf("%.1c", initial); -- INVALID + +-- ============================================================================ +-- Example 84: 5.2 Digit grouping +-- Demonstrates: Grouping applies to the digits before the point. Zeros that pad a number are not grouped. + +Console.Printf("%'.2f", 1234567.891); -- 1'234'567.89 +Console.Printf("%'d", 123); -- 123 + +-- ============================================================================ +-- Example 85: 3.1 Compile-time format requirement +-- Demonstrates: A format is known when the program is compiled. A String that is computed is not a format. +-- Expected result: compile-time rejection; the INVALID marker is normative. + +String format = "%d"; + +Console.Printf(format, value); -- INVALID +Console.Printf("%d" + suffix, value); -- INVALID diff --git a/helpers/cmd/execution-cases/main.go b/helpers/cmd/execution-cases/main.go new file mode 100644 index 00000000..215c49bc --- /dev/null +++ b/helpers/cmd/execution-cases/main.go @@ -0,0 +1,93 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +// execution-cases writes the shared execution test tables from their case +// files and checks that the committed tables are current. +package main + +import ( + "errors" + "fmt" + "io" + "os" + + "github.com/Progmasoft/visual-xsharp/helpers/internal/executioncases" + "github.com/spf13/cobra" +) + +// newCommand builds the command line. Results go to the first writer and the +// names of stale tables to the second. +func newCommand(output io.Writer, problems io.Writer) *cobra.Command { + var root string + command := &cobra.Command{ + Use: "execution-cases", + Short: "Generate and check the execution test tables", + Long: `The cases under Compiler/Fuzzing/Cases own the programs that the Haskell +frontend tests run in a reference evaluator and that source_execution_smoke +runs through LLVM. This command writes both tables from those files.`, + SilenceUsage: true, + SilenceErrors: true, + } + command.SetOut(output) + command.SetErr(problems) + command.PersistentFlags().StringVar(&root, "root", "", "repository root (default: found above the working directory)") + resolve := func() (string, error) { + if root != "" { + return root, nil + } + return executioncases.FindRoot(".") + } + command.AddCommand(&cobra.Command{ + Use: "generate", + Short: "Write every table that differs from what the case files generate", + Args: cobra.NoArgs, + RunE: func(*cobra.Command, []string) error { + directory, err := resolve() + if err != nil { + return err + } + written, err := executioncases.Write(directory) + if err != nil { + return err + } + for _, path := range written { + fmt.Fprintln(output, "wrote", path) + } + if len(written) == 0 { + fmt.Fprintln(output, "Every generated table is current.") + } + return nil + }, + }) + command.AddCommand(&cobra.Command{ + Use: "check", + Short: "Fail when a committed table is not what the case files generate", + Args: cobra.NoArgs, + RunE: func(*cobra.Command, []string) error { + directory, err := resolve() + if err != nil { + return err + } + stale, err := executioncases.Stale(directory) + if err != nil { + return err + } + for _, path := range stale { + fmt.Fprintln(problems, "stale:", path) + } + if len(stale) != 0 { + return errors.New("generated tables are stale; run `go -C helpers run ./cmd/execution-cases generate`") + } + fmt.Fprintln(output, "Every generated table is current.") + return nil + }, + }) + return command +} + +func main() { + if err := newCommand(os.Stdout, os.Stderr).Execute(); err != nil { + fmt.Fprintln(os.Stderr, "execution-cases:", err) + os.Exit(1) + } +} diff --git a/helpers/cmd/execution-cases/main_test.go b/helpers/cmd/execution-cases/main_test.go new file mode 100644 index 00000000..b9645f0d --- /dev/null +++ b/helpers/cmd/execution-cases/main_test.go @@ -0,0 +1,93 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package main + +import ( + "bytes" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/Progmasoft/visual-xsharp/helpers/internal/executioncases" +) + +// repository creates a root that holds case files and nothing generated. +func repository(t *testing.T) string { + t.Helper() + root := t.TempDir() + for _, name := range []string{"Selection.cases", "Leaving.cases"} { + path := filepath.Join(root, "Compiler", "Fuzzing", "Cases", name) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte("body: return left;\nrun: plain 3 0 -> 3\n"), 0o644); err != nil { + t.Fatal(err) + } + } + return root +} + +func run(t *testing.T, arguments ...string) (string, string, error) { + t.Helper() + var output, problems bytes.Buffer + command := newCommand(&output, &problems) + command.SetArgs(arguments) + err := command.Execute() + return output.String(), problems.String(), err +} + +func TestCheckFailsUntilGenerateHasRun(t *testing.T) { + root := repository(t) + _, problems, err := run(t, "check", "--root", root) + if err == nil { + t.Fatal("check passed without generated tables") + } + if !strings.Contains(problems, "stale: "+executioncases.HaskellModule) { + t.Errorf("the missing module is not named: %q", problems) + } + + output, _, err := run(t, "generate", "--root", root) + if err != nil { + t.Fatal(err) + } + if strings.Count(output, "wrote ") != 3 { + t.Errorf("generate reported %q", output) + } + if output, _, err = run(t, "check", "--root", root); err != nil || !strings.Contains(output, "current") { + t.Fatalf("check after generate: %q, %v", output, err) + } + if output, _, err = run(t, "generate", "--root", root); err != nil || strings.Contains(output, "wrote ") { + t.Fatalf("a second generate wrote again: %q, %v", output, err) + } +} + +func TestCheckReportsAnEditedTableAndABrokenCaseFile(t *testing.T) { + root := repository(t) + if _, _, err := run(t, "generate", "--root", root); err != nil { + t.Fatal(err) + } + include := filepath.Join(root, "Compiler", "Fuzzing", "Generated", "LeavingCases.inc") + if err := os.WriteFile(include, []byte("// edited by hand\n"), 0o644); err != nil { + t.Fatal(err) + } + _, problems, err := run(t, "check", "--root", root) + if err == nil || !strings.Contains(problems, "LeavingCases.inc") || strings.Contains(problems, "SelectionCases.inc") { + t.Fatalf("the edited table alone must be reported: %q, %v", problems, err) + } + + cases := filepath.Join(root, "Compiler", "Fuzzing", "Cases", "Leaving.cases") + if err := os.WriteFile(cases, []byte("body: return left;\n"), 0o644); err != nil { + t.Fatal(err) + } + if _, _, err := run(t, "generate", "--root", root); err == nil || !strings.Contains(err.Error(), "has no run") { + t.Fatalf("a body without a run must stop generation: %v", err) + } +} + +func TestPositionalArgumentsAreRejected(t *testing.T) { + if _, _, err := run(t, "check", "extra"); err == nil { + t.Fatal("check accepted a positional argument") + } +} diff --git a/helpers/internal/development/build.go b/helpers/internal/development/build.go index 8992d289..59c3b97c 100644 --- a/helpers/internal/development/build.go +++ b/helpers/internal/development/build.go @@ -21,6 +21,7 @@ var nativeTargets = []string{ "//Compiler/Cli/Tests:cli_parser_tests", "//Compiler/Cli/Commands/Tests:execution_status_tests", "//Compiler/Cli/Commands/Tests:cli_command_tests", + "//Compiler/Cli/Commands/Tests:executable_run_tests", "//Compiler/Core/Tests:callable_contract_tests", "//Compiler/Core/Tests:core_pipeline_tests", "//Compiler/Diagnostic/Tests:diagnostic_protocol_tests", @@ -32,7 +33,12 @@ var nativeTargets = []string{ "//Compiler/Codegen/Xpp/Tests:xpp_verifier_tests", "//Compiler/Runtime/AARC/Tests:aarc_runtime_tests", "//Compiler/Runtime/AARC/Tests:aarc_c_abi_tests", + "//Compiler/Runtime/Text/Tests:text_runtime_tests", "//Compiler/Fuzzing:source_fuzz_smoke", + "//Compiler/Fuzzing:source_execution_smoke", + "//Compiler/Fuzzing:source_expression_smoke", + "//Compiler/Fuzzing:source_feature_smoke", + "//Compiler/Fuzzing:source_console_smoke", "//Compiler/Fuzzing/Tests:coreprep_parity_tests", "//Compiler/ProjectSystem/Bridge/Tests:project_registry_tests", "//Interactive/Tests:interactive_tests", @@ -78,6 +84,9 @@ func buildTargets(repository string, runner commandRunner, config string, extra arguments = append(arguments, "//Compiler/Cli:vxs") arguments = append(arguments, "//Interactive:vxsi") arguments = append(arguments, nativeTargets...) + if runtimeLibraryTarget != "" { + arguments = append(arguments, runtimeLibraryTarget) + } arguments = append(arguments, extra...) fmt.Printf("Building compiler and %d native suites...\n", len(nativeTargets)) if err := runner.Run(repository, nil, bazel, cachedBuild(arguments)...); err != nil { @@ -86,6 +95,44 @@ func buildTargets(repository string, runner commandRunner, config string, extra if err := stageFrontendForBuildOutputs(repository, frontendLibrary); err != nil { return err } + return stageRuntimeForBuildOutputs(repository) +} + +// The runtime library a native executable is linked with. The compiler looks +// for it beside its own executable under the name it is installed by. Native +// executables are linked on Windows only, so the library exists there only. +const runtimeLibraryName = "vxs-runtime.lib" + +var runtimeLibraryTarget = func() string { + if runtime.GOOS == "windows" { + return "//Compiler/Runtime/Freestanding:vxs_runtime" + } + return "" +}() + +// builtRuntimeLibrary is where Bazel leaves the archive of runtimeLibraryTarget. +func builtRuntimeLibrary(repository string) string { + return filepath.Join(repository, "bazel-bin", "Compiler", "Runtime", "Freestanding", "vxs_runtime.lib") +} + +// stageRuntimeForBuildOutputs puts the runtime library beside every program +// that may link a native executable: the compiler, and the suites that drive +// it in their own process. +func stageRuntimeForBuildOutputs(repository string) error { + if runtimeLibraryTarget == "" { + return nil + } + library := builtRuntimeLibrary(repository) + if _, err := os.Stat(library); err != nil { + return fmt.Errorf("Bazel did not produce the runtime library: %w", err) + } + executablePaths := append([]string{"Compiler/Cli/vxs", "Interactive/vxsi"}, nativePrograms...) + for _, relative := range executablePaths { + directory := filepath.Dir(filepath.Join(repository, "bazel-bin", filepath.FromSlash(relative))) + if err := copyFile(library, filepath.Join(directory, runtimeLibraryName), 0o644); err != nil { + return fmt.Errorf("cannot stage the runtime library beside %s: %w", relative, err) + } + } return nil } diff --git a/helpers/internal/development/bundle.go b/helpers/internal/development/bundle.go index 3fa51131..b8b46d09 100644 --- a/helpers/internal/development/bundle.go +++ b/helpers/internal/development/bundle.go @@ -77,6 +77,14 @@ func buildBundle(repository string, currentHost host, runner commandRunner, baze if err := copyFile(frontend, filepath.Join(stagingDirectory, frontendLibraryName), 0o755); err != nil { return fmt.Errorf("cannot stage the Haskell frontend shared library: %w", err) } + // A program the bundled compiler links needs the runtime library beside + // the compiler. + if runtimeLibraryTarget != "" { + if err := copyFile(builtRuntimeLibrary(repository), filepath.Join(stagingDirectory, runtimeLibraryName), 0o644); err != nil { + return fmt.Errorf("cannot stage the runtime library: %w", err) + } + stagedFiles = append(stagedFiles, runtimeLibraryName) + } for _, relative := range bundleFiles { if err := copyFile(filepath.Join(repository, filepath.FromSlash(relative)), filepath.Join(stagingDirectory, filepath.FromSlash(relative)), 0o644); err != nil { return fmt.Errorf("cannot stage %s: %w", relative, err) diff --git a/helpers/internal/development/fuzz.go b/helpers/internal/development/fuzz.go index 3c2cc46a..bfa4864a 100644 --- a/helpers/internal/development/fuzz.go +++ b/helpers/internal/development/fuzz.go @@ -53,7 +53,7 @@ func fuzzBuildArguments(configuration, sanitizerConfiguration, macRuntime string if macRuntime != "" { campaign = append(campaign, "--linkopt="+macRuntime) } - smoke = append(smoke, "//Compiler/Fuzzing:wire_fuzz_smoke", "//Compiler/Fuzzing:source_fuzz_smoke") + smoke = append(smoke, "//Compiler/Fuzzing:wire_fuzz_smoke", "//Compiler/Fuzzing:source_fuzz_smoke", "//Compiler/Fuzzing:source_execution_smoke", "//Compiler/Fuzzing:source_expression_smoke", "//Compiler/Fuzzing:source_feature_smoke", "//Compiler/Fuzzing:source_console_smoke") for _, target := range nativeFuzzTargets() { campaign = append(campaign, target.label) } @@ -167,6 +167,24 @@ func runFuzzCampaign(repository string, currentHost host, runner commandRunner, if err := runner.Run(repository, selectedEnvironment, smokeSource); err != nil { return fmt.Errorf("source-to-LLVM differential smoke failed; preserved %q: %w", work, err) } + // The programs with hand-written results stand in the same directory, + // beside the frontend library staged above. + smokeExecution := filepath.Join(filepath.Dir(smokeSource), "source_execution_smoke"+currentHost.executable) + if err := runner.Run(repository, selectedEnvironment, smokeExecution); err != nil { + return fmt.Errorf("source execution smoke failed; preserved %q: %w", work, err) + } + smokeExpression := filepath.Join(filepath.Dir(smokeSource), "source_expression_smoke"+currentHost.executable) + if err := runner.Run(repository, selectedEnvironment, smokeExpression); err != nil { + return fmt.Errorf("source expression smoke failed; preserved %q: %w", work, err) + } + smokeFeature := filepath.Join(filepath.Dir(smokeSource), "source_feature_smoke"+currentHost.executable) + if err := runner.Run(repository, selectedEnvironment, smokeFeature); err != nil { + return fmt.Errorf("source feature smoke failed; preserved %q: %w", work, err) + } + smokeConsole := filepath.Join(filepath.Dir(smokeSource), "source_console_smoke"+currentHost.executable) + if err := runner.Run(repository, selectedEnvironment, smokeConsole); err != nil { + return fmt.Errorf("source console smoke failed; preserved %q: %w", work, err) + } // Run smoke tests before changing Bazel's instrumentation configuration and // staging the campaign binaries into the same host output tree. if err := runner.Run(repository, nil, bazel, cachedBuild(campaignArguments)...); err != nil { diff --git a/helpers/internal/development/fuzz_haskell_test.go b/helpers/internal/development/fuzz_haskell_test.go index ae48ae68..28d40464 100644 --- a/helpers/internal/development/fuzz_haskell_test.go +++ b/helpers/internal/development/fuzz_haskell_test.go @@ -58,6 +58,10 @@ func TestFuzzProcessWatchdogDoesNotLimitBuildTools(t *testing.T) { seconds int }{ {"source_fuzz_smoke.exe", nil, 240}, + {"source_execution_smoke.exe", nil, 240}, + {"source_expression_smoke.exe", nil, 240}, + {"source_feature_smoke.exe", nil, 240}, + {"source_console_smoke.exe", nil, 240}, {"wire_fuzzer", []string{"-max_total_time=30"}, 120}, {"frontend-fuzz.exe", []string{"parser", "900"}, 1200}, {"cabal", []string{"build", "all"}, 0}, diff --git a/helpers/internal/development/process.go b/helpers/internal/development/process.go index 6e7e75cb..f130f777 100644 --- a/helpers/internal/development/process.go +++ b/helpers/internal/development/process.go @@ -92,12 +92,19 @@ func watchedCommandWithin(seconds int, name string, arguments []string) (*exec.C // an ordinary build and about 78 seconds under AddressSanitizer and // UndefinedBehaviorSanitizer on a hosted Windows runner. The former bound of 90 // seconds left that run 12 seconds of headroom, and a slower machine exceeded it -// with every case passing. +// with every case passing. The execution tables are a program of their own, +// source_execution_smoke: with them, source_fuzz_smoke ran 221 seconds in the +// fuzzing configuration on a developer machine. Apart, each ran 77 to 110 +// seconds there, so the bound did not have to grow with the tables. The tables +// of single features are source_feature_smoke for the same reason, and so are +// the expression and leaving tables, source_expression_smoke: once arguments +// were passed by need, the programs of the three tables together ran 344 +// seconds under the sanitizers there. const smokeProcessSeconds = 240 func fuzzProcessSeconds(name string, arguments []string) int { binary := strings.TrimSuffix(filepath.Base(name), ".exe") - if binary == "source_fuzz_smoke" || binary == "wire_fuzz_smoke" { + if binary == "source_fuzz_smoke" || binary == "source_execution_smoke" || binary == "source_expression_smoke" || binary == "source_feature_smoke" || binary == "source_console_smoke" || binary == "wire_fuzz_smoke" { return smokeProcessSeconds } if binary == "frontend-fuzz" && len(arguments) >= 2 { diff --git a/helpers/internal/executioncases/cases.go b/helpers/internal/executioncases/cases.go new file mode 100644 index 00000000..0691a9ff --- /dev/null +++ b/helpers/internal/executioncases/cases.go @@ -0,0 +1,137 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +// Package executioncases writes the test tables that the Haskell frontend +// tests and the native smoke program share from the case files that own them. +// +// The case files under Compiler/Fuzzing/Cases are the only place a case is +// written. The Haskell module runs each case in a reference Core evaluator and +// the C++ tables run it through LLVM; both must return the expected value the +// case file states. Generating the two tables from one file keeps them equal. +// It is not a second opinion on that value: an expected value that is wrong in +// the case file is wrong in both tables. The values are therefore written by +// hand from the language rules, and tests whose expectations do not come from +// these files stay beside them. +package executioncases + +import ( + "bufio" + "fmt" + "strconv" + "strings" +) + +// Run is one call of a case body and the value it must return. +type Run struct { + Flag bool + Other bool + Left int64 + Right int64 + Expected int64 +} + +// Case is one body of `int Run(bool flag, bool other, int left, int right)` +// with the comment written above it and its runs in file order. +type Case struct { + Comment []string + Body string + Runs []Run +} + +// Parse reads a case file. The name is used in error messages only. +func Parse(name string, text string) ([]Case, error) { + var cases []Case + var comment []string + scanner := bufio.NewScanner(strings.NewReader(text)) + scanner.Buffer(make([]byte, 0, 64*1024), 4*1024*1024) + line := 0 + fail := func(format string, arguments ...any) ([]Case, error) { + return nil, fmt.Errorf("%s:%d: %s", name, line, fmt.Sprintf(format, arguments...)) + } + for scanner.Scan() { + line++ + current := strings.TrimRight(scanner.Text(), " \t\r") + switch { + case current == "": + // A comment belongs to the body directly below it. + comment = nil + case strings.HasPrefix(current, "#"): + comment = append(comment, strings.TrimPrefix(strings.TrimPrefix(current, "#"), " ")) + case strings.HasPrefix(current, "body: "): + body := strings.TrimPrefix(current, "body: ") + if strings.TrimSpace(body) == "" { + return fail("a body is empty") + } + for _, earlier := range cases { + if earlier.Body == body { + return fail("this body is already a case; add the run to it") + } + } + cases = append(cases, Case{Comment: comment, Body: body}) + comment = nil + case strings.HasPrefix(current, "run: "): + if len(cases) == 0 || comment != nil { + return fail("a run must follow the body it runs") + } + run, err := parseRun(strings.TrimPrefix(current, "run: ")) + if err != nil { + return fail("%v", err) + } + last := &cases[len(cases)-1] + last.Runs = append(last.Runs, run) + default: + return fail("expected a comment, `body: ` or `run: `") + } + } + if err := scanner.Err(); err != nil { + return nil, fmt.Errorf("%s: %w", name, err) + } + for _, parsed := range cases { + if len(parsed.Runs) == 0 { + return nil, fmt.Errorf("%s: the body %q has no run", name, parsed.Body) + } + } + if len(cases) == 0 { + return nil, fmt.Errorf("%s: the file has no case", name) + } + return cases, nil +} + +// parseRun reads `plain L R -> E` or `flags F O L R -> E`. +func parseRun(text string) (Run, error) { + arguments, expected, found := strings.Cut(text, " -> ") + if !found { + return Run{}, fmt.Errorf("a run needs ` -> ` and its expected value") + } + fields := strings.Fields(arguments) + var run Run + var numbers []string + switch { + case len(fields) == 3 && fields[0] == "plain": + numbers = fields[1:] + case len(fields) == 5 && fields[0] == "flags": + flag, err := strconv.ParseBool(fields[1]) + if err != nil || (fields[1] != "true" && fields[1] != "false") { + return Run{}, fmt.Errorf("flag must be true or false, not %q", fields[1]) + } + other, err := strconv.ParseBool(fields[2]) + if err != nil || (fields[2] != "true" && fields[2] != "false") { + return Run{}, fmt.Errorf("other must be true or false, not %q", fields[2]) + } + run.Flag, run.Other = flag, other + numbers = fields[3:] + default: + return Run{}, fmt.Errorf("a run is `plain ` or `flags `") + } + values := make([]int64, 0, 3) + for _, number := range append(numbers, strings.TrimSpace(expected)) { + // The arguments are `int` in the language and in both harnesses. + value, err := strconv.ParseInt(number, 10, 32) + if err != nil { + return Run{}, fmt.Errorf("%q is not a 32-bit integer", number) + } + values = append(values, value) + } + run.Left, run.Right, run.Expected = values[0], values[1], values[2] + return run, nil +} diff --git a/helpers/internal/executioncases/cases_test.go b/helpers/internal/executioncases/cases_test.go new file mode 100644 index 00000000..fd46d705 --- /dev/null +++ b/helpers/internal/executioncases/cases_test.go @@ -0,0 +1,159 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package executioncases + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +const sample = `# A header that belongs to no case. + +# First line. +# Second line. +body: return left + "\("; +run: plain 1 -2 -> 3 +run: flags true false 0 0 -> -4 + +body: return right; +run: plain 0 7 -> 7 +` + +func TestParseReadsCommentsBodiesAndRuns(t *testing.T) { + cases, err := Parse("sample", sample) + if err != nil { + t.Fatal(err) + } + if len(cases) != 2 { + t.Fatalf("got %d cases", len(cases)) + } + first := cases[0] + if strings.Join(first.Comment, "|") != "First line.|Second line." { + t.Errorf("comment: %q", first.Comment) + } + if first.Body != `return left + "\(";` { + t.Errorf("body: %q", first.Body) + } + want := []Run{{Left: 1, Right: -2, Expected: 3}, {Flag: true, Expected: -4}} + if len(first.Runs) != 2 || first.Runs[0] != want[0] || first.Runs[1] != want[1] { + t.Errorf("runs: %+v", first.Runs) + } + if len(cases[1].Comment) != 0 { + t.Errorf("the second case has no comment, got %q", cases[1].Comment) + } +} + +func TestParseRejectsMalformedFiles(t *testing.T) { + for name, text := range map[string]string{ + "empty file": "# nothing\n", + "run before a body": "run: plain 0 0 -> 0\n", + "body without a run": "body: return 0;\n", + "unknown line": "body: return 0;\nrun: plain 0 0 -> 0\nnote\n", + "missing expectation": "body: return 0;\nrun: plain 0 0\n", + "wrong arity": "body: return 0;\nrun: plain 0 -> 0\n", + "flag spelled wrong": "body: return 0;\nrun: flags True false 0 0 -> 0\n", + "not a number": "body: return 0;\nrun: plain x 0 -> 0\n", + "out of range": "body: return 0;\nrun: plain 0 0 -> 4294967296\n", + "repeated body": "body: return 0;\nrun: plain 0 0 -> 0\n\nbody: return 0;\nrun: plain 1 0 -> 0\n", + "comment before run": "body: return 0;\n# stray\nrun: plain 0 0 -> 0\n", + "empty body": "body: \nrun: plain 0 0 -> 0\n", + } { + if _, err := Parse(name, text); err == nil { + t.Errorf("%s: accepted", name) + } + } +} + +func TestRenderingEscapesAndIsDeterministic(t *testing.T) { + cases, err := Parse("sample", sample) + if err != nil { + t.Fatal(err) + } + table := Table{Source: "Cases/Sample.cases", Include: "Sample.inc", Binding: "sampleCases", Summary: "A sample.", Cases: cases} + include := RenderInclude(table) + for _, expected := range []string{ + "{ false, false, 1, -2, 3,\n \"return left + \\\"\\\\(\\\";\" },\n", + "{ true, false, 0, 0, -4,\n", + "// First line.\n// Second line.\n", + "Generated from Cases/Sample.cases", + } { + if !strings.Contains(include, expected) { + t.Errorf("the include lacks %q:\n%s", expected, include) + } + } + haskell := RenderHaskell([]Table{table}) + for _, expected := range []string{ + " [ -- First line.\n -- Second line.\n ( \"return left + \\\"\\\\(\\\";\"\n", + " , [((False, False, 1, -2), 3), ((True, False, 0, 0), -4)]\n", + "evaluationCases = sampleCases\n", + " ,\n ( \"return right;\"\n", + } { + if !strings.Contains(haskell, expected) { + t.Errorf("the module lacks %q:\n%s", expected, haskell) + } + } + if include != RenderInclude(table) || haskell != RenderHaskell([]Table{table}) { + t.Error("rendering the same table twice gave different text") + } +} + +// The committed tables must be what the committed case files generate. This +// is the check that keeps a table from being edited by hand or left behind. +func TestCommittedTablesAreCurrent(t *testing.T) { + root, err := FindRoot(".") + if err != nil { + t.Fatal(err) + } + stale, err := Stale(root) + if err != nil { + t.Fatal(err) + } + if len(stale) != 0 { + t.Fatalf("stale generated tables %v; run `go -C helpers run ./cmd/execution-cases generate`", stale) + } +} + +func TestWriteLeavesCurrentFilesAndRepairsStaleOnes(t *testing.T) { + root := t.TempDir() + for _, table := range tables() { + path := filepath.Join(root, filepath.FromSlash(table.Source)) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte("body: return left;\nrun: plain 3 0 -> 3\n"), 0o644); err != nil { + t.Fatal(err) + } + } + written, err := Write(root) + if err != nil { + t.Fatal(err) + } + if len(written) != len(tables())+1 { + t.Fatalf("the first write wrote %v", written) + } + if again, err := Write(root); err != nil || len(again) != 0 { + t.Fatalf("a second write wrote %v, %v", again, err) + } + // A checkout with the other line ending is still current. + module := filepath.Join(root, filepath.FromSlash(HaskellModule)) + text, err := os.ReadFile(module) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(module, []byte(strings.ReplaceAll(string(text), "\n", "\r\n")), 0o644); err != nil { + t.Fatal(err) + } + if stale, err := Stale(root); err != nil || len(stale) != 0 { + t.Fatalf("a CRLF checkout is reported stale: %v, %v", stale, err) + } + if err := os.WriteFile(module, append(text, []byte("-- edited\n")...), 0o644); err != nil { + t.Fatal(err) + } + stale, err := Stale(root) + if err != nil || len(stale) != 1 || stale[0] != HaskellModule { + t.Fatalf("an edited table is not reported: %v, %v", stale, err) + } +} diff --git a/helpers/internal/executioncases/generate.go b/helpers/internal/executioncases/generate.go new file mode 100644 index 00000000..d26f5f6e --- /dev/null +++ b/helpers/internal/executioncases/generate.go @@ -0,0 +1,128 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package executioncases + +import ( + "bytes" + "fmt" + "os" + "path/filepath" + "sort" + "strings" +) + +// HaskellModule is the generated Haskell module, relative to the repository +// root. +const HaskellModule = "Compiler/Haskell/Driver/test/BranchingEvaluationCases.hs" + +// tables lists the case files in the order the Haskell tests report them. +func tables() []Table { + return []Table{ + { + Source: "Compiler/Fuzzing/Cases/Selection.cases", + Include: "Compiler/Fuzzing/Generated/SelectionCases.inc", + Binding: "selectionCases", + Summary: "Which arm, block or body is selected, and how often each part runs.", + }, + { + Source: "Compiler/Fuzzing/Cases/Leaving.cases", + Include: "Compiler/Fuzzing/Generated/LeavingCases.inc", + Binding: "leavingCases", + Summary: "Expressions that leave instead of yielding a value.", + }, + } +} + +// Outputs reads every case file under the repository root and returns the +// generated files by their path relative to that root. The result depends on +// the case files alone. +func Outputs(root string) (map[string]string, error) { + loaded := tables() + outputs := make(map[string]string, len(loaded)+1) + for index := range loaded { + table := &loaded[index] + text, err := os.ReadFile(filepath.Join(root, filepath.FromSlash(table.Source))) + if err != nil { + return nil, fmt.Errorf("read case file: %w", err) + } + table.Cases, err = Parse(table.Source, string(text)) + if err != nil { + return nil, err + } + outputs[table.Include] = RenderInclude(*table) + } + outputs[HaskellModule] = RenderHaskell(loaded) + return outputs, nil +} + +// sameText compares generated text with a file as checked out: the working +// tree may hold either line ending. +func sameText(generated string, onDisk []byte) bool { + return generated == string(bytes.ReplaceAll(onDisk, []byte("\r\n"), []byte("\n"))) +} + +// Stale returns the generated files that are missing or differ from what the +// case files generate, sorted. +func Stale(root string) ([]string, error) { + outputs, err := Outputs(root) + if err != nil { + return nil, err + } + var stale []string + for path, text := range outputs { + onDisk, err := os.ReadFile(filepath.Join(root, filepath.FromSlash(path))) + if err != nil && !os.IsNotExist(err) { + return nil, fmt.Errorf("read generated file: %w", err) + } + if err != nil || !sameText(text, onDisk) { + stale = append(stale, path) + } + } + sort.Strings(stale) + return stale, nil +} + +// Write regenerates every file that is stale and returns the files it wrote. +// A file that is current is left untouched, so its line endings and its +// modification time stay as they are. +func Write(root string) ([]string, error) { + outputs, err := Outputs(root) + if err != nil { + return nil, err + } + stale, err := Stale(root) + if err != nil { + return nil, err + } + for _, path := range stale { + target := filepath.Join(root, filepath.FromSlash(path)) + if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil { + return nil, fmt.Errorf("create directory: %w", err) + } + if err := os.WriteFile(target, []byte(outputs[path]), 0o644); err != nil { + return nil, fmt.Errorf("write generated file: %w", err) + } + } + return stale, nil +} + +// FindRoot walks up from a directory to the repository root, which is the +// directory that holds the case files. +func FindRoot(start string) (string, error) { + directory, err := filepath.Abs(start) + if err != nil { + return "", fmt.Errorf("resolve directory: %w", err) + } + marker := filepath.FromSlash(tables()[0].Source) + for { + if _, err := os.Stat(filepath.Join(directory, marker)); err == nil { + return directory, nil + } + parent := filepath.Dir(directory) + if parent == directory { + return "", fmt.Errorf("no repository root with %s above %s", strings.ReplaceAll(marker, `\`, "/"), start) + } + directory = parent + } +} diff --git a/helpers/internal/executioncases/render.go b/helpers/internal/executioncases/render.go new file mode 100644 index 00000000..9e3d38a3 --- /dev/null +++ b/helpers/internal/executioncases/render.go @@ -0,0 +1,136 @@ +// SPDX-FileCopyrightText: 2026 Progmasoft +// SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1 + +package executioncases + +import ( + "fmt" + "strings" +) + +const license = `SPDX-FileCopyrightText: 2026 Progmasoft +SPDX-License-Identifier: MPL-2.0 WITH AdditionRef-Progmasoft-Exception-1.1` + +// A Table is one case file and the names its cases have in the generated +// sources. +type Table struct { + // Source is the case file, relative to the repository root. + Source string + // Include is the generated C++ table, relative to the repository root. + Include string + // Binding is the name of the list in the generated Haskell module. + Binding string + // Summary is the Haddock comment of that list. + Summary string + Cases []Case +} + +// quote writes a string literal that Haskell and C++ read the same way: a +// case body is one line of printable source, so only the backslash and the +// quotation mark need an escape. +func quote(text string) string { + return `"` + strings.NewReplacer(`\`, `\\`, `"`, `\"`).Replace(text) + `"` +} + +// RenderInclude writes the rows of a C++ array of `ExecutionCase`, one row +// for each run, in file order. +func RenderInclude(table Table) string { + var out strings.Builder + for _, line := range strings.Split(license, "\n") { + out.WriteString("// " + line + "\n") + } + out.WriteString("//\n") + out.WriteString("// Generated from " + table.Source + " by\n") + out.WriteString("// `go -C helpers run ./cmd/execution-cases generate`. Do not edit.\n") + out.WriteString("//\n") + out.WriteString("// clang-format off\n") + for _, current := range table.Cases { + for _, comment := range current.Comment { + out.WriteString(strings.TrimRight("// "+comment, " ") + "\n") + } + for _, run := range current.Runs { + fmt.Fprintf(&out, "{ %t, %t, %d, %d, %d,\n %s },\n", + run.Flag, run.Other, run.Left, run.Right, run.Expected, quote(current.Body)) + } + } + out.WriteString("// clang-format on\n") + return out.String() +} + +// RenderHaskell writes the module that holds every table as a list. +func RenderHaskell(tables []Table) string { + var out strings.Builder + for _, line := range strings.Split(license, "\n") { + out.WriteString("-- " + line + "\n") + } + out.WriteString(` +{- | The programs that "BranchingTests" runs, with the value each must return. + +Generated by @go -C helpers run ./cmd/execution-cases generate@. Do not edit: +the cases are written in the files under @Compiler/Fuzzing/Cases@, which also +generate the tables the native smoke program @source_execution_smoke@ runs +through LLVM. + +A case is the body of @int Run(bool flag, bool other, int left, int right)@ +and the runs of that body. The expected values are written by hand from the +language rules. +-} +module BranchingEvaluationCases + ( EvaluationCase + , evaluationCases +`) + for _, table := range tables { + out.WriteString(" , " + table.Binding + "\n") + } + out.WriteString(` ) where + +-- | A body, and for each run its arguments and the value it must return. +type EvaluationCase = (String, [((Bool, Bool, Integer, Integer), Integer)]) + +-- | Every case, in the order the tests report them. +evaluationCases :: [EvaluationCase] +evaluationCases = `) + names := make([]string, 0, len(tables)) + for _, table := range tables { + names = append(names, table.Binding) + } + out.WriteString(strings.Join(names, " ++ ") + "\n") + for _, table := range tables { + out.WriteString("\n-- | " + table.Summary + " From @" + table.Source + "@.\n") + out.WriteString(table.Binding + " :: [EvaluationCase]\n") + out.WriteString(table.Binding + " =\n") + for index, current := range table.Cases { + opener := " , " + if index == 0 { + opener = " [ " + } + if len(current.Comment) == 0 { + out.WriteString(strings.TrimRight(opener, " ") + "\n") + } + for line, comment := range current.Comment { + prefix := " " + if line == 0 { + prefix = opener + } + out.WriteString(strings.TrimRight(prefix+"-- "+comment, " ") + "\n") + } + out.WriteString(" ( " + quote(current.Body) + "\n") + runs := make([]string, 0, len(current.Runs)) + for _, run := range current.Runs { + runs = append(runs, fmt.Sprintf("((%s, %s, %d, %d), %d)", + haskellBool(run.Flag), haskellBool(run.Other), run.Left, run.Right, run.Expected)) + } + out.WriteString(" , [" + strings.Join(runs, ", ") + "]\n") + out.WriteString(" )\n") + } + out.WriteString(" ]\n") + } + return out.String() +} + +func haskellBool(value bool) string { + if value { + return "True" + } + return "False" +}